@cspeach/cli 0.9.0 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (195) hide show
  1. package/README.md +1 -1
  2. package/dist/agent/intent-system-prompt.js +1 -1
  3. package/dist/agent/loop.js +228 -26
  4. package/dist/agent/providers/license-gate.js +44 -0
  5. package/dist/agent/skill-checkpoint.js +1 -1
  6. package/dist/agent/tool-dispatch.js +15 -0
  7. package/dist/approvals/canonical.js +91 -0
  8. package/dist/approvals/jwt.js +39 -2
  9. package/dist/approvals/op-labels.js +124 -0
  10. package/dist/approvals/render.js +42 -36
  11. package/dist/auth/org-anthropic-key.js +25 -0
  12. package/dist/classifier/client.js +18 -3
  13. package/dist/cli.js +15 -0
  14. package/dist/commands/compact.js +28 -2
  15. package/dist/commands/config-set.js +284 -0
  16. package/dist/commands/config-show.js +20 -0
  17. package/dist/commands/export-audit.js +43 -0
  18. package/dist/commands/help.js +5 -0
  19. package/dist/commands/login.js +31 -14
  20. package/dist/commands/plan-audit-evidence.js +266 -0
  21. package/dist/commands/plan-audit.js +692 -0
  22. package/dist/commands/plan-chain.js +671 -0
  23. package/dist/commands/plan-continue.js +179 -0
  24. package/dist/commands/plan-gate.js +154 -0
  25. package/dist/commands/plan-model-tier.js +83 -0
  26. package/dist/commands/plan-resume.js +728 -46
  27. package/dist/config/loader.js +223 -5
  28. package/dist/config/model-defaults.js +14 -0
  29. package/dist/cost/pricing.js +27 -1
  30. package/dist/doctor/checks/_http-probe.js +1 -0
  31. package/dist/doctor/checks/cert.js +14 -3
  32. package/dist/doctor/checks/sap.js +30 -8
  33. package/dist/doctor/checks/system-roles.js +41 -0
  34. package/dist/doctor/checks/zcspeach.js +19 -4
  35. package/dist/doctor/run.js +2 -0
  36. package/dist/models/resolve.js +61 -0
  37. package/dist/models/server-config.js +155 -0
  38. package/dist/one-shot.js +76 -6
  39. package/dist/projects/answer-blockers.js +137 -0
  40. package/dist/projects/extract-cca.js +111 -17
  41. package/dist/projects/extract-modernize.js +4 -2
  42. package/dist/projects/extract-plan.js +184 -37
  43. package/dist/projects/extract-spec-gap.js +34 -7
  44. package/dist/projects/extract-test-coverage.js +4 -2
  45. package/dist/projects/extract-upgrade.js +116 -23
  46. package/dist/projects/handover-md.js +195 -0
  47. package/dist/projects/index.js +5 -2
  48. package/dist/projects/merge-cca.js +292 -0
  49. package/dist/projects/merge-upgrade.js +173 -0
  50. package/dist/projects/migration.js +103 -1
  51. package/dist/projects/output-paths.js +27 -0
  52. package/dist/projects/plan-run.js +285 -27
  53. package/dist/projects/plan-schema.js +136 -3
  54. package/dist/projects/promote-command.js +25 -2
  55. package/dist/projects/promote.js +128 -0
  56. package/dist/projects/run-lease.js +157 -0
  57. package/dist/projects/save-command.js +259 -21
  58. package/dist/projects/status.js +3 -1
  59. package/dist/projects/validate.js +1 -1
  60. package/dist/projects/workspace.js +164 -20
  61. package/dist/renderer/notices.js +64 -0
  62. package/dist/renderer/progress-chatter.js +8 -0
  63. package/dist/renderer/status-footer.js +22 -12
  64. package/dist/renderer/thinking-heartbeat.js +64 -8
  65. package/dist/renderer/todo-block.js +51 -0
  66. package/dist/renderer/tool-widget.js +55 -4
  67. package/dist/renderer/tty.js +43 -4
  68. package/dist/renderer/verify-chain.js +77 -0
  69. package/dist/repl/at-picker.js +60 -7
  70. package/dist/repl/bracketed-paste.js +28 -19
  71. package/dist/repl/builtin-commands.js +42 -0
  72. package/dist/repl/current-transport.js +10 -0
  73. package/dist/repl/early-line-buffer.js +68 -0
  74. package/dist/repl/history.js +86 -0
  75. package/dist/repl/ink-stdin-guard.js +64 -0
  76. package/dist/repl/inquirer-guard.js +70 -5
  77. package/dist/repl/mode-ceiling.js +16 -0
  78. package/dist/repl/mode-cycle.js +104 -0
  79. package/dist/repl/numbered-menu.js +131 -0
  80. package/dist/repl/post-turn-status.js +26 -6
  81. package/dist/repl/rule8-detector.js +17 -2
  82. package/dist/repl/safety-confirm.js +111 -2
  83. package/dist/repl/safety-mode-state.js +19 -3
  84. package/dist/repl/slash-completer.js +5 -0
  85. package/dist/repl/slash-picker.js +10 -15
  86. package/dist/repl.js +1232 -95
  87. package/dist/rewind/candidates.js +194 -0
  88. package/dist/rewind/cli.js +137 -0
  89. package/dist/rewind/format.js +27 -0
  90. package/dist/rewind/restore.js +245 -0
  91. package/dist/router/classifier.js +150 -6
  92. package/dist/sap/capability-matrix.js +20 -0
  93. package/dist/sap/capability-matrix.json +11236 -0
  94. package/dist/sap/capability.js +146 -0
  95. package/dist/sap/connection-manager.js +19 -1
  96. package/dist/sap/onboarding.js +42 -4
  97. package/dist/session/audit-export.js +459 -0
  98. package/dist/session/context-report.js +163 -0
  99. package/dist/session/pending.js +27 -0
  100. package/dist/session/recap.js +160 -0
  101. package/dist/skill-catalog.js +51 -40
  102. package/dist/skills/bundled-skills.js +272 -1
  103. package/dist/skills/promotion-dispatch.js +23 -0
  104. package/dist/tools/_command-shared.js +36 -12
  105. package/dist/tools/_filesystem-shared.js +139 -4
  106. package/dist/tools/_flag.js +25 -0
  107. package/dist/tools/approval.js +177 -26
  108. package/dist/tools/ask-question.js +400 -7
  109. package/dist/tools/capability/tool.js +74 -0
  110. package/dist/tools/dispatch-skill.js +22 -1
  111. package/dist/tools/extend-model/anchored-insert.js +1414 -0
  112. package/dist/tools/extend-model/tool.js +340 -0
  113. package/dist/tools/filesystem/extract-document.js +57 -0
  114. package/dist/tools/filesystem/file-edit.js +12 -2
  115. package/dist/tools/filesystem/file-read.js +2 -2
  116. package/dist/tools/filesystem/file-write.js +11 -2
  117. package/dist/tools/filesystem/glob.js +11 -0
  118. package/dist/tools/filesystem/grep.js +10 -0
  119. package/dist/tools/filesystem/read-document.js +107 -0
  120. package/dist/tools/fiori/apply.js +50 -0
  121. package/dist/tools/fiori/bin.js +3 -0
  122. package/dist/tools/fiori/catalog/index.js +27 -0
  123. package/dist/tools/fiori/catalog/value-help.js +230 -0
  124. package/dist/tools/fiori/catalog/viz-chart.js +177 -0
  125. package/dist/tools/fiori/cli.js +71 -0
  126. package/dist/tools/fiori/deploy-config.js +73 -0
  127. package/dist/tools/fiori/fe-extend.js +76 -0
  128. package/dist/tools/fiori/fe-scaffold.js +71 -0
  129. package/dist/tools/fiori/floorplan-map.js +19 -0
  130. package/dist/tools/fiori/i18n.js +39 -0
  131. package/dist/tools/fiori/manifest.js +70 -0
  132. package/dist/tools/fiori/render.js +77 -0
  133. package/dist/tools/fiori/samples/data/index.json +13602 -0
  134. package/dist/tools/fiori/samples/data/sources.generated.js +808 -0
  135. package/dist/tools/fiori/samples/loader.js +248 -0
  136. package/dist/tools/fiori/samples/search.js +63 -0
  137. package/dist/tools/fiori/samples/types.js +2 -0
  138. package/dist/tools/fiori/scaffold.js +39 -0
  139. package/dist/tools/fiori/smoke/assertions.js +74 -0
  140. package/dist/tools/fiori/smoke/browser.js +52 -0
  141. package/dist/tools/fiori/smoke/driver.js +89 -0
  142. package/dist/tools/fiori/smoke/freestyle-spec.js +317 -0
  143. package/dist/tools/fiori/smoke/run-smoke.js +149 -0
  144. package/dist/tools/fiori/tools.js +681 -0
  145. package/dist/tools/fiori/types.js +1 -0
  146. package/dist/tools/local-build.js +86 -0
  147. package/dist/tools/local-files.js +31 -0
  148. package/dist/tools/project/_merge-shared.js +68 -0
  149. package/dist/tools/project/cca_merge.js +164 -0
  150. package/dist/tools/project/playbook_get.js +1 -1
  151. package/dist/tools/project/upgrade_merge_progress.js +206 -0
  152. package/dist/tools/sap-read.js +132 -20
  153. package/dist/tools/sap-write.js +550 -21
  154. package/dist/tools/shell/shell_exec.js +41 -6
  155. package/dist/tools/snapshot.js +63 -14
  156. package/dist/tools/subagent/agent_run.js +27 -3
  157. package/dist/tools/subagent/background_run.js +17 -1
  158. package/dist/tools/todo.js +144 -0
  159. package/dist/tools/transport-resolution.js +86 -0
  160. package/dist/tools/transport.js +224 -5
  161. package/dist/tools/write-mode.js +4 -0
  162. package/dist/ui/app.js +378 -21
  163. package/dist/ui/approval-modal.js +49 -16
  164. package/dist/ui/ask-question-emitter.js +14 -0
  165. package/dist/ui/body.js +13 -0
  166. package/dist/ui/context-grid.js +108 -0
  167. package/dist/ui/footer.js +120 -27
  168. package/dist/ui/header.js +7 -0
  169. package/dist/ui/line-resolution.js +35 -8
  170. package/dist/ui/rewind-emitter.js +10 -0
  171. package/dist/ui/rewind-panel.js +81 -0
  172. package/dist/ui/sap-state-store.js +1 -0
  173. package/dist/ui/session-timeline.js +1 -0
  174. package/dist/ui/status-line.js +43 -0
  175. package/dist/ui/text-input.js +214 -0
  176. package/dist/ui/todo-emitter.js +25 -0
  177. package/dist/ui/todo-panel.js +64 -0
  178. package/dist/ui/turn-status-emitter.js +50 -4
  179. package/dist/ui/turn-status.js +18 -3
  180. package/dist/ui/widgets/ask-form.js +242 -0
  181. package/dist/ui/widgets/ask-question-modal.js +21 -8
  182. package/package.json +22 -3
  183. package/bench/README.md +0 -78
  184. package/bench/prompts/abap-document-cds.md +0 -44
  185. package/bench/prompts/abap-explain-bdef-handler.md +0 -57
  186. package/bench/prompts/abap-test-method.md +0 -42
  187. package/bench/results/abap-document-cds/claude-haiku-4-5.md +0 -189
  188. package/bench/results/abap-document-cds/claude-opus-4-7.md +0 -120
  189. package/bench/results/abap-document-cds/claude-sonnet-4-6.md +0 -151
  190. package/bench/results/abap-explain-bdef-handler/claude-haiku-4-5.md +0 -112
  191. package/bench/results/abap-explain-bdef-handler/claude-opus-4-7.md +0 -101
  192. package/bench/results/abap-explain-bdef-handler/claude-sonnet-4-6.md +0 -101
  193. package/bench/results/abap-test-method/claude-haiku-4-5.md +0 -186
  194. package/bench/results/abap-test-method/claude-opus-4-7.md +0 -193
  195. package/bench/results/abap-test-method/claude-sonnet-4-6.md +0 -234
@@ -16,7 +16,20 @@
16
16
  // precede a closing ask_question, see turn-assistant-text.ts) →
17
17
  // extract the LAST csforge:plan-manifest block → build the version
18
18
  // N+1 revision (same id, history appended) → save → print the
19
- // updated tracker + next-step hint.
19
+ // updated tracker + next-step hint → return the saved path + the
20
+ // HARNESS-built resume command for the next phase (A1, 2026-06-10).
21
+ //
22
+ // A1 (defect D23) — true bounded phases: phase end = turn end.
23
+ // The model's resume turn ends at the manifest block. It must NOT ask a
24
+ // continuation question, NOT call dispatch_skill, and NOT execute a
25
+ // second phase in-turn (the old contract allowed in-session "continue",
26
+ // which snowballed a 6-phase run into one 3.24M-token-context session).
27
+ // Continuation is harness-owned: repl.tsx calls offerNextPhaseAutoRun
28
+ // with the PlanResumeOutcome; on consent it queues the EXACT command
29
+ // buildResumeCommand produced from the just-saved path (the model once
30
+ // hallucinated a wrong base when it authored this text itself), and
31
+ // every plan-resume turn starts from reset session messages
32
+ // (resetContextForPlanResume).
20
33
  //
21
34
  // The save hook inside runTurn is suppressed for resume turns
22
35
  // (RunTurnParams.suppressSaveHook) — it would otherwise offer to save a
@@ -25,17 +38,125 @@
25
38
  // that survives the session, so losing the write-back breaks the plan.
26
39
  import { readFileSync, readdirSync, statSync } from 'node:fs';
27
40
  import { basename, dirname, join } from 'node:path';
41
+ import chalk from 'chalk';
28
42
  import { readProjectFile } from '../projects/status.js';
29
43
  import { resolveAtTokenAsync, formatProjectFileList, ensureWorkspace } from '../projects/workspace.js';
30
- import { parsePlanContent } from '../projects/plan-schema.js';
31
- import { statusesFromItems, computeNextPhase, renderPlanTracker, buildPlanRevision, } from '../projects/plan-run.js';
44
+ import { parsePlanContent, phaseWrites } from '../projects/plan-schema.js';
45
+ import { statusesFromItems, computeNextPhase, renderPlanTracker, buildPlanRevision, isPhaseSatisfied, isAuditResolved, planCompletionLines, findServiceBinding, } from '../projects/plan-run.js';
32
46
  import { extractPlan } from '../projects/extract-plan.js';
33
47
  import { saveProject } from '../projects/save.js';
48
+ import { writePlanHandover } from '../projects/handover-md.js';
49
+ import { acquireRunLease, releaseRunLease } from '../projects/run-lease.js';
34
50
  import { validateEnvelope } from '../projects/validate.js';
35
51
  import { collectTurnAssistantText } from '../agent/turn-assistant-text.js';
36
52
  import { getAuthorIdentity } from '../agent/loop.js';
53
+ import { FLOORPLAN_TO_TEMPLATE } from '../tools/fiori/floorplan-map.js';
54
+ import { getCurrentTransport, setCurrentTransport } from '../repl/current-transport.js';
55
+ /**
56
+ * Parse and STRIP the per-run chaining-mode (`--guarded` / `--step`) and
57
+ * audit off-switch (`--audit` / `--no-audit`) flags from a plan-resume body.
58
+ * Pure; every other token (including `--resume` and `@` tokens) passes through
59
+ * untouched, so downstream parsing is unchanged.
60
+ *
61
+ * Conflict resolution — the SAFER option wins, always:
62
+ * - mode: BOTH `--guarded`+`--step` present ⇒ `--step` (more prompting).
63
+ * - audit: BOTH `--audit`+`--no-audit` present ⇒ `--audit`/'on' (more
64
+ * checking — a conflicting toggle must never silently disable the
65
+ * auditor; mirrors the --step-beats---guarded precedent).
66
+ */
67
+ export function parsePlanResumeFlags(body) {
68
+ let guarded = false;
69
+ let step = false;
70
+ let auditOn = false;
71
+ let auditOff = false;
72
+ const rest = [];
73
+ for (const tok of body.split(/\s+/)) {
74
+ if (tok === '--guarded') {
75
+ guarded = true;
76
+ continue;
77
+ }
78
+ if (tok === '--step') {
79
+ step = true;
80
+ continue;
81
+ }
82
+ if (tok === '--audit') {
83
+ auditOn = true;
84
+ continue;
85
+ }
86
+ if (tok === '--no-audit') {
87
+ auditOff = true;
88
+ continue;
89
+ }
90
+ if (tok.length > 0)
91
+ rest.push(tok);
92
+ }
93
+ const mode = step ? 'step' : guarded ? 'guarded' : undefined;
94
+ // 'on' wins the conflict (safer — never let a conflicting flag turn audits off).
95
+ const audit = auditOn ? 'on' : auditOff ? 'off' : undefined;
96
+ return { mode, audit, body: rest.join(' ') };
97
+ }
98
+ /**
99
+ * A1 (defect D23) — the harness-owned continuation contract, stated to the
100
+ * model verbatim in every resume prompt: the manifest ends the turn; the
101
+ * model must not call dispatch_skill or ask a continuation question — the
102
+ * CLI dispatches the next phase itself. Exported as a named constant so
103
+ * tests pin the prompt to this exact block instead of brittle prose
104
+ * regexes (the old contract let the model keep executing phases in-turn —
105
+ * a 6-phase run once snowballed to a 3.24M-token context).
106
+ */
107
+ export const PLAN_RESUME_HARNESS_OVERRIDE = 'HARNESS OVERRIDE of the skill\'s Mode 2 step 8: do NOT ask a continuation question, do NOT call dispatch_skill, and NEVER start another phase in this turn. After your manifest is persisted, the CLI itself asks the user whether to run the next phase and dispatches it with the exact saved file in a fresh bounded context. Phase end = turn end.';
108
+ /** Harness-owned ui-phase context (Track 1). Same principle as
109
+ * buildResumeCommand (D23): the model must never have to recall the
110
+ * binding name, flavor, or app dir from a prior session — the envelope
111
+ * knows. app/service are recorded by ui.build but consumed by ui.deploy,
112
+ * so resolution scans ALL ui phases, last-written wins. */
113
+ export function buildUiPhaseContext(phases, next) {
114
+ if (next.layer !== 'ui')
115
+ return [];
116
+ const binding = findServiceBinding(phases);
117
+ const uiPhases = phases.filter(p => p.layer === 'ui');
118
+ const appDir = [...uiPhases].reverse().map(p => p.work?.app?.dir).find(Boolean);
119
+ const service = [...uiPhases].reverse().map(p => p.work?.service).find(s => s && (s.url || s.path));
120
+ const lines = ['UI phase context (harness-provided — authoritative, do not re-derive):'];
121
+ lines.push(binding
122
+ ? `- Published service binding: ${binding}`
123
+ : '- Published service binding: NOT RECORDED in any service phase work.binding — resolve it live (sap_search_object) before building, and record it in this phase\'s work.');
124
+ if (next.ui) {
125
+ lines.push(`- Flavor: ${next.ui.flavor}`);
126
+ if (next.ui.floorplan)
127
+ lines.push(`- Floorplan: ${next.ui.floorplan} (tool template: ${FLOORPLAN_TO_TEMPLATE[next.ui.floorplan]})`);
128
+ if (next.ui.appId)
129
+ lines.push(`- App id: ${next.ui.appId}`);
130
+ if (next.ui.appTitle)
131
+ lines.push(`- App title: ${next.ui.appTitle}`);
132
+ }
133
+ if (appDir)
134
+ lines.push(`- App directory (recorded by ui.build): ${appDir}`);
135
+ if (service)
136
+ lines.push(`- Service URL: ${service.url ?? ''}${service.path ?? ''} (OData ${service.version ?? '?'})`);
137
+ return lines;
138
+ }
139
+ /** Compact-example work keys for ui phases — the write-back contract
140
+ * (review C3). Returns an OBJECT fragment because compactExample renders
141
+ * via JSON.stringify: it is spread into the `work` object exactly like the
142
+ * existing service-phase `binding` conditional (a string-array spread there
143
+ * would render numeric keys — cycle-2 N-1). */
144
+ export function buildUiWorkExample(next) {
145
+ if (next.layer !== 'ui')
146
+ return {};
147
+ return {
148
+ app: { dir: '<app directory ui.build created — ui.build only, omit on deploy>' },
149
+ service: { url: '<https://host:port — ui.build only>', path: '</sap/opu/odata4/...>', version: '4.0' },
150
+ deployedUrl: '<BSP index.html URL — ui.deploy only>',
151
+ };
152
+ }
37
153
  export async function preparePlanResume(args) {
38
- const tokens = args.body
154
+ // Agentic-flow: pull the per-run --guarded/--step mode override out of the
155
+ // body BEFORE the @-token filter so the flags never collide with token
156
+ // resolution (they wouldn't match the @ filter anyway, but stripping keeps
157
+ // the invariant explicit and the flags out of any downstream body use).
158
+ const flags = parsePlanResumeFlags(args.body);
159
+ const tokens = flags.body
39
160
  .split(/\s+/)
40
161
  .filter((tok) => tok.startsWith('@'))
41
162
  .map((tok) => tok.slice(1))
@@ -103,29 +224,211 @@ export async function preparePlanResume(args) {
103
224
  args.log(` ${e}`);
104
225
  return null;
105
226
  }
106
- const content = pc.content;
107
- const statuses = statusesFromItems(envelope.interaction.items, content.phases);
108
- const next = computeNextPhase(content.phases, statuses);
227
+ let content = pc.content;
228
+ // Transport display on resume (2026-07-09) — adopt the plan's dedicated
229
+ // transport into the session so the `tr <TRKORR>` status segment shows it.
230
+ // A fresh process resuming mid-plan runs no sap_transport_create (the
231
+ // transport already exists), so setCurrentTransport was never called and the
232
+ // status read `tr none` even while the phase resolves + writes to the
233
+ // recorded transport. Runs on EVERY resume, BEFORE the phase (or re-audit)
234
+ // executes. No-clobber (an explicit /transport or a create earlier THIS
235
+ // session wins) and null-safe (no recorded transport yet ⇒ leave it,
236
+ // `tr none` is then correct) are both handled inside the helper. Rule 9
237
+ // bonus: the tool-handler 3-tier fallback can now use it too.
238
+ adoptPlanTransportOnResume(content.phases);
239
+ // Task 9 — run lease: refuse a SECOND session resuming the same plan
240
+ // family concurrently (double-resume forks sibling envelope versions
241
+ // silently: saveProject appends collision suffixes and the newest-version
242
+ // redirect above tie-breaks on mtime). Taken HERE — after the redirect and
243
+ // validation — so the lease targets the file actually resumed; keyed on
244
+ // the FAMILY base (leasePathFor strips the -vN suffix) so the v3→v4
245
+ // rotation every phase save performs keeps the lease. Same-session
246
+ // re-acquire is idempotent, so the queueDispatch re-entry on every chained
247
+ // phase never blocks itself. Best-effort: fs errors inside acquireRunLease
248
+ // log one line and proceed as acquired (courtesy guard, availability over
249
+ // strictness). Only a LIVE other-session holder stops us — and even then
250
+ // the user may take over interactively.
251
+ if (args.sessionId) {
252
+ const lease = acquireRunLease(path, args.sessionId, { log: (line) => args.log(line) });
253
+ if (!lease.acquired && lease.holder) {
254
+ const holder = lease.holder;
255
+ args.log(`⚠ another session is already resuming this plan — session ${holder.sessionId} (pid ${holder.pid}${holder.at ? `, since ${holder.at}` : ''}).`);
256
+ const answer = args.prompt
257
+ ? (await args.prompt('Take over its run lease and resume here anyway? [y/N]: ')).trim().toLowerCase()
258
+ : '';
259
+ if (answer !== 'y' && answer !== 'yes') {
260
+ args.log('Resume refused — the other session keeps the lease. Re-run when it finishes (a dead session\'s lease is taken over automatically).');
261
+ return null;
262
+ }
263
+ acquireRunLease(path, args.sessionId, { force: true, log: (line) => args.log(line) });
264
+ args.log('Run lease taken over — resuming in this session.');
265
+ }
266
+ }
267
+ const effectiveMode = flags.mode ?? args.configMode ?? 'step';
268
+ // Audit off-switch (2026-07-06, owner): per-run flag > config default > 'on'.
269
+ const effectiveAudit = flags.audit ?? args.configAudit ?? 'on';
270
+ // Explicit --no-audit is a strong, deliberate signal — it is the ONLY thing
271
+ // that WAIVES persisted unresolved audits (below). A config-level off never
272
+ // waives: a pending/infra record was created under ON and gets its honest
273
+ // verdict re-run (config-off governs only NEW phases).
274
+ const auditsExplicitlyOff = flags.audit === 'off';
275
+ let statuses = statusesFromItems(envelope.interaction.items, content.phases);
276
+ // Audit off-switch — explicit `--no-audit` cleanup. A pending/infra_failed
277
+ // (or died-at-menu 'failed'-on-satisfied) record left by an earlier ON run
278
+ // would block eligibility forever if left unresolved and never re-audited.
279
+ // Under an explicit --no-audit the user has said "no audits now": resolve
280
+ // them as waived('audits turned off') — NO audit turn runs — so the plan
281
+ // can proceed. (Config-level off deliberately does NOT do this; see above.)
282
+ if (auditsExplicitlyOff) {
283
+ const unresolved = content.phases.filter((p) => {
284
+ const st = p.audit?.state;
285
+ return st === 'pending' || st === 'infra_failed'
286
+ || (st === 'failed' && isPhaseSatisfied(statuses[p.id]));
287
+ });
288
+ if (unresolved.length > 0) {
289
+ for (const p of unresolved) {
290
+ path = await applyAuditResult(path, p.id, { state: 'waived', waivedReason: 'audits turned off' }, { log: (...lines) => args.log(...lines) });
291
+ }
292
+ envelope = readProjectFile(path);
293
+ const rc = parsePlanContent(envelope.content);
294
+ if (rc.ok)
295
+ content = rc.content;
296
+ statuses = statusesFromItems(envelope.interaction.items, content.phases);
297
+ // M5 — name every waived phase, and flag a FAILED-with-findings record
298
+ // loudly: waiving a recorded failure is a bigger call than waiving a
299
+ // never-judged pending, and the user must see which one they made.
300
+ const names = unresolved.map((p) => p.audit?.state === 'failed'
301
+ ? `${p.id} (FAILED audit — ${p.audit.findings?.length ?? 0} finding${(p.audit.findings?.length ?? 0) === 1 ? '' : 's'} waived)`
302
+ : `${p.id} (${p.audit?.state ?? 'unresolved'})`);
303
+ args.log(`↪ audits off (--no-audit): waived ${unresolved.length} unresolved audit${unresolved.length === 1 ? '' : 's'} from an earlier run — ${names.join(', ')} (reason: audits turned off).`);
304
+ }
305
+ }
306
+ // Task 6 — hand-edit detection: the newest history entry was NOT authored
307
+ // by this harness (getAuthorIdentity), so its "validated" claims were never
308
+ // earned through the audited flow. Downgrade satisfied phases whose audit
309
+ // is absent-or-passed to not_audited_hand_edited IN MEMORY ONLY — the file
310
+ // is never rewritten, and eligibility (computeNextPhase below) still uses
311
+ // the REAL content, or every hand-touched legacy plan would deadlock.
312
+ // Step mode: the tracker glyph is the whole effect. Guarded mode: the chain
313
+ // stops to re-audit EVERY hand-touched satisfied phase (I-2 pass, below).
314
+ const lastHistory = envelope.history[envelope.history.length - 1];
315
+ const handEdited = lastHistory !== undefined && lastHistory.by?.name !== getAuthorIdentity().name;
316
+ const handEditDowngraded = (p) => isPhaseSatisfied(statuses[p.id]) && (p.audit === undefined || p.audit.state === 'passed');
317
+ let trackerContent = content;
318
+ if (handEdited) {
319
+ trackerContent = {
320
+ ...content,
321
+ phases: content.phases.map((p) => handEditDowngraded(p)
322
+ ? { ...p, audit: { ...(p.audit ?? {}), state: 'not_audited_hand_edited' } }
323
+ : p),
324
+ };
325
+ args.log(`⚠ envelope was last edited outside this harness (by ${lastHistory?.by?.name ?? 'unknown'}) — satisfied phases without a harness audit are shown as hand-edited (file not rewritten).`);
326
+ }
327
+ // Task 6 — re-audit-on-resume: an unresolved persisted audit (a crash
328
+ // between phase save and verdict save leaves 'pending'; an auditor outage
329
+ // leaves 'infra_failed') is re-run FIRST — no new phase executes this
330
+ // dispatch. Scan newest-last so the most recent execution is re-judged.
331
+ // Fix-wave (item 10a): a persisted 'failed' on a phase that still CLAIMS
332
+ // satisfied means the session died AT the failure menu (the verdict wrote,
333
+ // the resolution never did) — re-audit it and re-offer the menu instead of
334
+ // dead-ending in the "resolve by hand" message. A failed audit on a
335
+ // non-satisfied phase (forced block / user chose re-run) was already
336
+ // resolved and is NOT picked up.
337
+ // Audit off-switch: under an explicit --no-audit the persisted unresolved
338
+ // records were already waived above, so this scan finds nothing anyway; the
339
+ // guard keeps the intent explicit. Under config-off the persisted scan STILL
340
+ // runs (a stuck pending must resolve — config-off governs only NEW phases),
341
+ // but the guarded HAND-EDIT re-verification (itself an audit pass) does not.
342
+ let reauditPhaseIds = [];
343
+ if (!auditsExplicitlyOff) {
344
+ for (let i = content.phases.length - 1; i >= 0; i--) {
345
+ const p = content.phases[i];
346
+ const st = p.audit?.state;
347
+ // audit-infra-continue (2026-07-09): 'infra_failed' is NO LONGER scanned
348
+ // for re-audit — it is a RESOLVED state (the audit could not run; the
349
+ // chain already continued past it). Re-auditing it here would wall an
350
+ // already-continued phase on a later resume. Only a crash-left 'pending'
351
+ // or a died-at-menu 'failed' on a still-satisfied phase re-audits.
352
+ if (st === 'pending'
353
+ || (st === 'failed' && isPhaseSatisfied(statuses[p.id]))) {
354
+ reauditPhaseIds = [p.id];
355
+ break;
356
+ }
357
+ }
358
+ // Guarded hand-edit re-verification (review I-2): no persisted unresolved
359
+ // audit, but the envelope was hand-touched — re-audit EVERY satisfied phase
360
+ // the hand edit invalidated (plan order: dependencies first), in one pass
361
+ // via runReauditPass, before chaining anything. Skipped when audits are off
362
+ // (config or flag): a hand-edit re-verification IS an audit pass.
363
+ if (reauditPhaseIds.length === 0 && handEdited && effectiveMode === 'guarded'
364
+ && effectiveAudit === 'on') {
365
+ reauditPhaseIds = content.phases.filter(handEditDowngraded).map((p) => p.id);
366
+ }
367
+ }
368
+ const reauditPhaseId = reauditPhaseIds[0];
369
+ const next = reauditPhaseId ? null : computeNextPhase(content.phases, statuses);
109
370
  args.log('');
110
371
  args.log(renderPlanTracker({
111
372
  title: envelope.title,
112
373
  version: envelope.version,
113
- content,
374
+ content: trackerContent,
114
375
  statuses,
115
- currentId: next?.id ?? null,
376
+ currentId: reauditPhaseId ?? next?.id ?? null,
377
+ auditsOff: effectiveAudit === 'off',
116
378
  }));
117
379
  args.log('');
380
+ if (reauditPhaseId) {
381
+ const reauditPhase = content.phases.find((p) => p.id === reauditPhaseId);
382
+ const st = reauditPhase.audit?.state;
383
+ args.log(st === 'failed'
384
+ ? `Phase ${reauditPhaseId} has a failed audit with no recorded resolution — re-running the audit, then offering re-run / waive / block.`
385
+ : st === 'pending'
386
+ ? `Phase ${reauditPhaseId} has an unresolved audit (${st}) — re-running the audit before executing anything.`
387
+ : `Guarded mode: re-auditing ${reauditPhaseIds.length} hand-edited phase${reauditPhaseIds.length === 1 ? '' : 's'} (${reauditPhaseIds.join(', ')}) before chaining.`);
388
+ return {
389
+ path, envelope, statuses,
390
+ nextPhaseId: reauditPhaseId,
391
+ nextDelegateTo: reauditPhase.delegateTo,
392
+ llmPrompt: '',
393
+ mode: flags.mode,
394
+ effectiveMode,
395
+ auditFlag: flags.audit,
396
+ auditMode: effectiveAudit,
397
+ reauditPhaseId,
398
+ reauditPhaseIds,
399
+ };
400
+ }
118
401
  if (!next) {
119
- const allValidated = content.phases.every((p) => statuses[p.id] === 'validated');
402
+ // C1 (D30 waiver): 'validated-with-waiver' is satisfied — a plan whose
403
+ // last gate was explicitly waived is complete, not stuck.
404
+ const allValidated = content.phases.every((p) => isPhaseSatisfied(statuses[p.id], p.audit));
120
405
  if (allValidated) {
121
- args.log('Plan complete — every phase is validated.');
122
- args.log('Consider /abap-preflight on the produced transport(s) before release.');
406
+ // C2 (D24): a finished UI-less backend stack chains to /abap-fiori-build
407
+ // with the exact binding name — the marketed idea→app story must not
408
+ // silently end at the service binding.
409
+ args.log(...planCompletionLines(content.phases));
123
410
  }
124
411
  else {
125
412
  const blocked = content.phases.filter((p) => statuses[p.id] === 'blocked').map((p) => p.id);
126
- args.log(`No eligible phase. Blocked: ${blocked.join(', ') || '(none)'} — remaining phases wait on them.`);
127
- args.log('Unblock (fix + edit the envelope status back to todo) and resume again.');
413
+ // Task 2 review carry-over: a validated-but-audit-unresolved phase used
414
+ // to surface as "Blocked: (none)" — name the real block reason. Rarely
415
+ // reached since the fix-wave: pending/infra_failed AND satisfied-but-
416
+ // failed audits are all re-audited above; what remains is a hand-
417
+ // authored state (e.g. a user writing 'not_audited_hand_edited').
418
+ const unresolved = content.phases
419
+ .filter((p) => isPhaseSatisfied(statuses[p.id]) && !isAuditResolved(p.audit))
420
+ .map((p) => `${p.id} (audit ${p.audit?.state ?? 'unknown'})`);
421
+ const unresolvedNote = unresolved.length > 0 ? ` Unresolved audits: ${unresolved.join(', ')}.` : '';
422
+ args.log(`No eligible phase. Blocked: ${blocked.join(', ') || '(none)'}.${unresolvedNote} Remaining phases wait on them.`);
423
+ args.log(unresolved.length > 0
424
+ ? 'Resolve the unresolved audit (waive it or set the phase status back to todo for a re-run) and resume again.'
425
+ : 'Unblock (fix + edit the envelope status back to todo) and resume again.');
128
426
  }
427
+ // Task 9 — terminal state: nothing will run this dispatch, so don't
428
+ // leave a lease behind (a completed/deadlocked plan must be freely
429
+ // resumable by any session).
430
+ if (args.sessionId)
431
+ releaseRunLease(path);
129
432
  return null;
130
433
  }
131
434
  // Inline the phase's declared rule files (coarse v1: whole files).
@@ -143,10 +446,49 @@ export async function preparePlanResume(args) {
143
446
  }
144
447
  }
145
448
  const planState = JSON.stringify({ title: envelope.title, version: envelope.version, statuses, content }, null, 2);
449
+ // A2 (defects D23/D30) — compact write-back: the model re-emitting the
450
+ // full plan content every phase cost ~6–10k output tokens/phase at Opus
451
+ // pricing for data the envelope already holds. The resume turn emits
452
+ // statuses + a `changed` entry for the executed phase only; extractPlan
453
+ // merges it into the prior content. The example is built with the REAL
454
+ // phase ids and current statuses so the model copies, not reconstructs.
455
+ // Task 11 (2026-07-03): deliberately NO `writes` here — the guarded-mode
456
+ // `writes` flag is authored once at plan CREATION (abap-plan SKILL.md
457
+ // Mode 1, item 7a); resume's `changed.work` merge never re-authors it and
458
+ // extractPlan preserves existing phase fields.
459
+ const compactExample = [
460
+ '<!-- csforge:plan-manifest',
461
+ JSON.stringify({
462
+ title: envelope.title,
463
+ statuses: { ...statuses, [next.id]: '<validated | validated-with-waiver | blocked>' },
464
+ changed: {
465
+ [next.id]: {
466
+ work: {
467
+ generated: ['<object names created/changed>'],
468
+ transport: '<transport number — omit the key if none>',
469
+ // C2 (D24): the service phase records the published SRVB name in
470
+ // work.binding — the plan-complete message hands it to
471
+ // /abap-fiori-build. Only shown when this turn IS the service phase.
472
+ ...(next.layer === 'service'
473
+ ? { binding: '<published service binding name, e.g. ZUI_MAINTREQ_O4>' }
474
+ : {}),
475
+ ...buildUiWorkExample(next),
476
+ notes: '<decisions, substitutions, blocker details worth keeping — omit if none>',
477
+ },
478
+ },
479
+ },
480
+ }, null, 2),
481
+ '-->',
482
+ ].join('\n');
483
+ // Track 1 (cosmetic F4): hoist the single evaluation — the double call plus
484
+ // the leading '' below produced a DOUBLE blank line on ui turns (line 520's
485
+ // '' already ends the prior element). Spread the blank AFTER the block so
486
+ // context keeps single-blank separation on both sides.
487
+ const uiContext = buildUiPhaseContext(content.phases, next);
146
488
  const llmPrompt = [
147
489
  `Resume execution of the project plan "${envelope.title}" (envelope v${envelope.version}).`,
148
490
  '',
149
- `Execute Mode 2 of the abap-plan skill for phase "${next.id}" ONLY — it is the computed next eligible phase. Do not execute any other phase unless the user explicitly picks "continue" at the phase-end question.`,
491
+ `Execute Mode 2 of the abap-plan skill for phase "${next.id}" ONLY — it is the computed next eligible phase. Never execute any other phase in this turn; the CLI itself offers and dispatches the next phase (in a fresh bounded context) after this one is persisted.`,
150
492
  '',
151
493
  '<plan_state>',
152
494
  planState,
@@ -156,13 +498,70 @@ export async function preparePlanResume(args) {
156
498
  ...ruleBlocks,
157
499
  '</phase_rules>',
158
500
  '',
159
- // 2026-06-06 live-smoke lesson #3: the model asked the continuation
160
- // question FIRST, then ended the turn on the user's "exit" answer with
161
- // a one-line acknowledgment — and the whole phase result was lost. The
162
- // ordering must be explicit and the consequence named.
163
- 'CRITICAL — write-back ordering: emit the COMPLETE updated <!-- csforge:plan-manifest --> block (full phase list, statuses entry for every phase, this phase\'s work filled in — including work.notes with the compact decision register when the phase produced decisions rather than SAP objects) BEFORE the phase-end continuation question. The manifest block is how the result is persisted; a turn that ends without it LOSES the phase. After the user answers "exit", reply with at most one short line — the manifest must already be in the transcript by then.',
501
+ // A1 (2026-06-10, defect D23): the old contract let the model ask a
502
+ // continuation question and keep executing phases in-turn — a 6-phase
503
+ // run snowballed to a 3.24M-token context because the turn never
504
+ // ended. The manifest now ENDS the turn; the harness owns continuation
505
+ // (offerNextPhaseAutoRun) and dispatches the next phase itself with
506
+ // the exact saved path, in reset context.
507
+ 'CRITICAL — the write-back ENDS the turn: emit ONE COMPACT <!-- csforge:plan-manifest --> block as the LAST thing in your output, then END THE TURN. Compact shape = "title" + a "statuses" entry for EVERY phase + a "changed" map carrying ONLY the phase(s) you touched this turn (normally exactly this one). Each "changed" entry holds the phase\'s COMPLETE updated "work" (generated / transport / snapshot / notes — plus, on ui phases: app.dir + service{url,path,version} (ui.build) and deployedUrl (ui.deploy) — put the compact decision register in work.notes when the phase produced decisions rather than SAP objects); it replaces that phase\'s prior work wholesale. Do NOT re-emit "content" or the full phase list — the CLI already holds the full plan, merges your "changed" entries into it, and recomputes the summary. Exact shape for this turn:',
508
+ '',
509
+ compactExample,
510
+ '',
511
+ // C1 (D30 waiver) — the waiver status exists so an explicitly-waived exit
512
+ // gate is recorded as what it is, instead of being laundered into a
513
+ // "validated" the gate never earned (the c1.test AUnit-gap incident).
514
+ 'Status rules: "validated" ONLY when the exit gate actually held and was verified. If the gate could NOT be met but the user EXPLICITLY waived it this turn (e.g. "skip the test gate, continue anyway"), use "validated-with-waiver" and record what was waived and why in work.notes — never mark an unmet gate "validated", and never use the waiver status without an explicit user waiver. Otherwise the phase is "blocked".',
515
+ '',
516
+ 'Only if the plan itself must change structurally (a phase added, removed, or re-sequenced) fall back to the full shape with "content" — the CLI accepts both. The manifest block is how the result is persisted; a turn that ends without it LOSES the phase.',
517
+ '',
518
+ ...(uiContext.length ? [...uiContext, ''] : []),
519
+ PLAN_RESUME_HARNESS_OVERRIDE,
164
520
  ].join('\n');
165
- return { path, envelope, statuses, nextPhaseId: next.id, llmPrompt };
521
+ return {
522
+ path, envelope, statuses,
523
+ nextPhaseId: next.id,
524
+ nextDelegateTo: next.delegateTo,
525
+ // Task 10 — the executing phase's writes declaration for the deviation
526
+ // backstop (fail-safe: undeclared ⇒ true, backstop never fires).
527
+ nextPhaseWrites: phaseWrites(next),
528
+ llmPrompt, mode: flags.mode, effectiveMode,
529
+ auditFlag: flags.audit, auditMode: effectiveAudit,
530
+ };
531
+ }
532
+ /**
533
+ * Pick the plan's dedicated session transport from the phases' recorded work.
534
+ * All phases share the ONE dedicated transport (Rule 9), so latest-wins — the
535
+ * last phase in plan order carrying a non-empty `work.transport` — reflects
536
+ * what the NEXT write will use and settles the rare divergent case. Returns
537
+ * null when no phase has recorded one yet (the design phase that creates the
538
+ * transport hasn't run), in which case `tr none` is correct. Pure.
539
+ */
540
+ export function recordedPlanTransport(phases) {
541
+ let found = null;
542
+ for (const p of phases) {
543
+ const t = p.work?.transport;
544
+ if (typeof t === 'string' && t.trim().length > 0)
545
+ found = t;
546
+ }
547
+ return found;
548
+ }
549
+ /**
550
+ * On a plan resume, adopt the envelope's recorded dedicated transport into the
551
+ * session so the `tr <TRKORR>` status segment shows it (setCurrentTransport
552
+ * pushes to globalStore.transport, which the statusline reads). No-clobber: an
553
+ * explicit `/transport` or a `sap_transport_create` earlier THIS session set a
554
+ * transport already and must win, so adopt ONLY when the session has none.
555
+ * Null-safe: when the plan has recorded no transport yet, leave the session
556
+ * untouched (`tr none` is then the honest state). Does not touch
557
+ * current-transport's setter semantics — it is purely a new caller.
558
+ */
559
+ export function adoptPlanTransportOnResume(phases) {
560
+ if (getCurrentTransport() !== null)
561
+ return; // explicit/session transport wins
562
+ const recorded = recordedPlanTransport(phases);
563
+ if (recorded)
564
+ setCurrentTransport(recorded);
166
565
  }
167
566
  /**
168
567
  * Given a set of candidate file paths (e.g. the matches from an ambiguous
@@ -193,14 +592,62 @@ export function pickNewestPlanVersion(paths) {
193
592
  return items[0].path;
194
593
  }
195
594
  /**
196
- * True when a queued dispatch command is a `/abap-plan --resume …`. The REPL
197
- * uses this to clear session.messages before re-entering, so the auto-fired
198
- * next phase runs in fresh, bounded context (the cheap path).
595
+ * True when a queued dispatch command is a `/abap-plan --resume …`.
596
+ * Consumer: dispatch_skill REJECTS model-authored plan-resume dispatches
597
+ * (A1 — the harness owns that command; the model once hallucinated a wrong
598
+ * base token). The transcript clear for resume turns is owned solely by
599
+ * resetContextForPlanResume, called from the REPL's plan-resume prepare
600
+ * branch after preparePlanResume succeeds.
199
601
  */
200
602
  export function isPlanResumeCommand(cmd) {
201
603
  const c = cmd.trim();
202
604
  return /^\/abap-plan\b/.test(c) && /(^|\s)--resume(\s|$)/.test(c);
203
605
  }
606
+ /**
607
+ * A1 (defect D23) — the HARNESS builds the next-phase resume command from
608
+ * the exact path finishPlanResume just saved. Single construction site:
609
+ * the model never authors this text (it once invented `@…-plan-c1` when
610
+ * the real file was `…-plan-c9b3-v14.cspeach.json`). The basename keeps
611
+ * the version suffix — it resolves to that exact file, and the
612
+ * newest-version redirect in preparePlanResume still protects against a
613
+ * concurrent later save.
614
+ */
615
+ export function buildResumeCommand(savedPath) {
616
+ return `/abap-plan --resume @${basename(savedPath)}`;
617
+ }
618
+ /**
619
+ * A1 — every plan-resume turn starts from RESET session messages: the
620
+ * envelope + the bounded preparePlanResume prompt carry all needed state,
621
+ * so prior-phase (or prior-chat) transcript is pure cost. Called by the
622
+ * REPL right after preparePlanResume succeeds, which covers BOTH the
623
+ * harness-dispatched auto-run path and a manually typed `--resume`.
624
+ * Returns true when messages were actually cleared (caller logs a notice).
625
+ * Property reassignment (not splice) — session.messages consumers always
626
+ * re-read the property.
627
+ */
628
+ export function resetContextForPlanResume(session) {
629
+ if (session.messages.length === 0)
630
+ return false;
631
+ session.messages = [];
632
+ return true;
633
+ }
634
+ export async function offerNextPhaseAutoRun(args) {
635
+ if (!args.outcome.nextPhaseId)
636
+ return false;
637
+ const delegate = args.outcome.nextDelegateTo ? ` (${args.outcome.nextDelegateTo})` : '';
638
+ if (args.gate?.action === 'auto') {
639
+ args.log(`▶ auto-running next phase ${args.outcome.nextPhaseId}${delegate} — guarded mode: declared non-write phase, prior audit clean.`);
640
+ args.queueDispatch(args.outcome.resumeCommand);
641
+ return true;
642
+ }
643
+ const answer = (await args.prompt(`Run next phase ${args.outcome.nextPhaseId}${delegate} now in a fresh context? [y/N]: `)).trim().toLowerCase();
644
+ if (answer !== 'y' && answer !== 'yes') {
645
+ args.log(`Exiting — resume later with: ${args.outcome.resumeCommand}`);
646
+ return false;
647
+ }
648
+ args.queueDispatch(args.outcome.resumeCommand);
649
+ return true;
650
+ }
204
651
  /**
205
652
  * Find the newest sibling version of the envelope at `path` — same
206
653
  * filename family (`<slug>-<shortid>-vN[...]`) AND same envelope id (the
@@ -250,11 +697,18 @@ export async function finishPlanResume(args) {
250
697
  args.log('');
251
698
  args.log('[plan] turn produced no assistant output — plan envelope UNCHANGED.');
252
699
  args.log(`[plan] re-run: /abap-plan --resume @${basename(args.prepared.path)}`);
253
- return;
700
+ return null;
254
701
  }
255
702
  let extract;
256
703
  try {
257
- extract = extractPlan(text);
704
+ // A2 — pass the prior content so a COMPACT manifest (statuses + changed
705
+ // map) can be merged into it. prepared.envelope.content already passed
706
+ // parsePlanContent in preparePlanResume; re-parsing here hands extractPlan
707
+ // the normalised PlanContent without widening PreparedPlanResume. If the
708
+ // parse somehow fails, prior stays undefined and a compact manifest fails
709
+ // loudly below (full manifests are unaffected).
710
+ const prior = parsePlanContent(args.prepared.envelope.content);
711
+ extract = extractPlan(text, prior.ok ? prior.content : undefined);
258
712
  }
259
713
  catch (e) {
260
714
  // Loud by design: the phase may have built real SAP objects, but the
@@ -265,14 +719,58 @@ export async function finishPlanResume(args) {
265
719
  args.log('[plan] The plan envelope is unchanged. Check what the phase actually built');
266
720
  args.log('[plan] (transport, activated objects), update the envelope statuses by hand or');
267
721
  args.log(`[plan] re-run: /abap-plan --resume @${basename(args.prepared.path)}`);
268
- return;
722
+ return null;
723
+ }
724
+ const nowIso = new Date().toISOString();
725
+ const statuses = extract.statuses;
726
+ const executedId = args.prepared.nextPhaseId;
727
+ // Task 6 — the provisional next phase is computed BEFORE the pending-audit
728
+ // injection below (as-if the audit will resolve): the injected 'pending'
729
+ // makes the executed phase unsatisfied on disk (crash-safety), but the
730
+ // post-turn messages and the outcome's nextPhaseId describe the happy path;
731
+ // the post-phase chain (plan-chain.ts) recomputes from the real verdict.
732
+ const provisionalPhases = extract.content.phases.map((p) => p.id === executedId ? { ...p, audit: undefined } : p);
733
+ const next = computeNextPhase(provisionalPhases, statuses);
734
+ // Task 6 — persist the executed phase WITH audit:'pending' in the SAME
735
+ // revision that records its result. Ordering is the crash-safety design:
736
+ // a crash between this save and the verdict save leaves 'pending', which
737
+ // eligibility treats as unsatisfied — resume re-offers the audit, never
738
+ // skips it. Only success claims are audited: a phase that ended 'blocked'
739
+ // (or died in progress) claims nothing to verify.
740
+ //
741
+ // Audit off-switch (2026-07-06, owner): when auditMode is 'off' we inject NO
742
+ // pending audit — the phase ends with no audit field, which isPhaseSatisfied
743
+ // treats as legacy-satisfied. runPostPhaseChain then skips the audit step and
744
+ // the board renders straight below (no "audit pending…" holdback). This is
745
+ // the ONLY behavioural change when off; the ON path is byte-identical.
746
+ const auditEnabled = (args.auditMode ?? 'on') !== 'off';
747
+ const executedStatus = statuses[executedId];
748
+ if (auditEnabled && (executedStatus === 'validated' || executedStatus === 'validated-with-waiver')) {
749
+ const executed = extract.content.phases.find((p) => p.id === executedId);
750
+ if (executed) {
751
+ const carried = executed.audit; // prior audit from the merged content
752
+ executed.audit = {
753
+ state: 'pending',
754
+ at: nowIso,
755
+ ...(args.sessionId ? { sessionId: args.sessionId } : {}),
756
+ // D-B item 1 — the attempt window for re-audits (fresh-session
757
+ // re-audit of a multi-attempt JSONL must see only THIS attempt).
758
+ ...(args.attemptStartedAt ? { attemptStartedAt: args.attemptStartedAt } : {}),
759
+ // A re-run after a failed audit carries the failure counter through
760
+ // 'pending' so the NEXT failed verdict counts as consecutive.
761
+ ...(carried?.state === 'failed' && carried.consecutiveFailures
762
+ ? { consecutiveFailures: carried.consecutiveFailures }
763
+ : {}),
764
+ };
765
+ }
269
766
  }
270
- const revision = buildPlanRevision(args.prepared.envelope, extract, getAuthorIdentity(), new Date().toISOString());
767
+ const revision = buildPlanRevision(args.prepared.envelope, extract, getAuthorIdentity(), nowIso);
768
+ // JSON round-trip simulates disk serialization for the validator.
271
769
  const check = validateEnvelope(JSON.parse(JSON.stringify(revision)));
272
770
  if (!check.ok) {
273
771
  args.log('');
274
772
  args.log(`[plan] PHASE RESULT NOT PERSISTED — revision failed validation: ${check.error.message}`);
275
- return;
773
+ return null;
276
774
  }
277
775
  let outDir;
278
776
  try {
@@ -282,27 +780,211 @@ export async function finishPlanResume(args) {
282
780
  outDir = process.cwd();
283
781
  }
284
782
  const savedPath = await saveProject(revision, { cwd: outDir });
285
- const statuses = extract.statuses;
286
- const next = computeNextPhase(extract.content.phases, statuses);
783
+ // Task 8 — regenerate the write-only markdown handover projection beside
784
+ // the envelope. Best-effort: a failed projection write logs one dim line
785
+ // and never blocks the run (the envelope is the source of truth).
786
+ writePlanHandover(savedPath, args.log);
287
787
  args.log('');
288
788
  args.log(`Plan updated: ${savedPath}`);
289
- args.log('');
290
- args.log(renderPlanTracker({
291
- title: revision.title,
292
- version: revision.version,
293
- content: extract.content,
294
- statuses,
295
- currentId: null,
789
+ // D-E — DON'T render the progress board / tally / "Next:" line here when an
790
+ // audit is pending. Doing so used the PROVISIONAL (audit-erased) view, which
791
+ // counted the just-run phase as done ("2/3") and printed "Next: …" as if the
792
+ // gate had held — moments before the auditor could FAIL and block it, which
793
+ // the product owner watched happen. The referee rules first: print only a
794
+ // neutral one-liner; the post-phase chain (plan-chain.ts) renders the board
795
+ // with an audit-AWARE tally after the verdict lands (a validated phase with a
796
+ // pending audit is NOT satisfied, so the honest tally is "1/3", not "2/3").
797
+ // Off-switch: with no pending injected, there is no verdict to contradict a
798
+ // board — so an off-run renders the board here exactly like a blocked/no-audit
799
+ // phase (honest: the phase IS satisfied, no audit is coming).
800
+ const auditPending = auditEnabled
801
+ && (executedStatus === 'validated' || executedStatus === 'validated-with-waiver');
802
+ if (auditPending) {
803
+ args.log(chalk.dim('Phase result recorded — audit pending…'));
804
+ }
805
+ else {
806
+ // A blocked / no-audit phase claims nothing to verify, so no verdict will
807
+ // contradict a board here — render it (and the next-step / completion hint).
808
+ args.log('');
809
+ args.log(renderPlanTracker({
810
+ title: revision.title,
811
+ version: revision.version,
812
+ content: { ...extract.content, phases: provisionalPhases },
813
+ statuses,
814
+ currentId: null,
815
+ auditsOff: !auditEnabled,
816
+ }));
817
+ args.log('');
818
+ if (next) {
819
+ args.log(`Next: ${next.id} (${next.delegateTo}) — run /abap-plan --resume @${basename(savedPath)} in a fresh session.`);
820
+ }
821
+ else if (provisionalPhases.every((p) => isPhaseSatisfied(statuses[p.id], p.audit))) {
822
+ // C2 (D24): same chain as preparePlanResume — single source in plan-run.ts.
823
+ args.log(...planCompletionLines(provisionalPhases));
824
+ }
825
+ else {
826
+ const blocked = provisionalPhases.filter((p) => statuses[p.id] === 'blocked').map((p) => p.id);
827
+ args.log(`No eligible next phase. Blocked: ${blocked.join(', ')}. Unblock, then resume again.`);
828
+ }
829
+ }
830
+ return {
831
+ savedPath,
832
+ resumeCommand: buildResumeCommand(savedPath),
833
+ nextPhaseId: next?.id ?? null,
834
+ nextDelegateTo: next?.delegateTo ?? null,
835
+ };
836
+ }
837
+ /**
838
+ * Shared revision writer for audit verdicts and chain-stop records: load the
839
+ * envelope, patch ONE phase (audit and/or status) + content.lastStop,
840
+ * recompute the summary audit-aware, and save version N+1. Throws on
841
+ * structural problems (invalid envelope, unknown phase, failed validation) —
842
+ * callers surface the error; the unresolved on-disk audit state ('pending')
843
+ * then guarantees a re-audit on the next resume.
844
+ */
845
+ async function savePlanPatch(envelopePath, phaseId, patch, log) {
846
+ const envelope = readProjectFile(envelopePath);
847
+ if (envelope.artefactType !== 'plan') {
848
+ throw new Error(`expected a plan envelope, got '${envelope.artefactType}'`);
849
+ }
850
+ const pc = parsePlanContent(envelope.content);
851
+ if (!pc.ok) {
852
+ throw new Error(`plan content failed validation: ${pc.errors.join('; ')}`);
853
+ }
854
+ // Deep clone — never mutate the parsed content of the source envelope.
855
+ const content = JSON.parse(JSON.stringify(pc.content));
856
+ const phase = content.phases.find((p) => p.id === phaseId);
857
+ if (!phase) {
858
+ throw new Error(`phase '${phaseId}' not found in plan envelope '${envelope.title}'`);
859
+ }
860
+ if (patch.audit === 'clear')
861
+ delete phase.audit;
862
+ else if (patch.audit !== 'keep')
863
+ phase.audit = patch.audit;
864
+ const statuses = statusesFromItems(envelope.interaction.items, content.phases);
865
+ if (patch.status)
866
+ statuses[phaseId] = patch.status;
867
+ if (patch.lastStop === 'clear')
868
+ delete content.lastStop;
869
+ else if (patch.lastStop)
870
+ content.lastStop = patch.lastStop;
871
+ // Recompute the summary audit-aware (same predicates as the DAG).
872
+ const nextPhase = computeNextPhase(content.phases, statuses);
873
+ content.summary = {
874
+ total: content.phases.length,
875
+ validated: content.phases.filter((p) => isPhaseSatisfied(statuses[p.id], p.audit)).length,
876
+ blocked: content.phases.filter((p) => statuses[p.id] === 'blocked').length,
877
+ next: nextPhase?.id ?? null,
878
+ };
879
+ const items = content.phases.map((p) => ({
880
+ id: p.id,
881
+ status: statuses[p.id],
882
+ answer: null,
883
+ answeredBy: null,
884
+ answeredAt: null,
885
+ comments: [],
296
886
  }));
297
- args.log('');
298
- if (next) {
299
- args.log(`Next: ${next.id} (${next.delegateTo}) — run /abap-plan --resume @${basename(savedPath)} in a fresh session.`);
887
+ const revision = buildPlanRevision(envelope, { title: envelope.title, content, items, statuses }, getAuthorIdentity(), new Date().toISOString(), patch.summary);
888
+ const check = validateEnvelope(JSON.parse(JSON.stringify(revision)));
889
+ if (!check.ok) {
890
+ throw new Error(`audit revision failed validation: ${check.error.message}`);
891
+ }
892
+ let outDir;
893
+ try {
894
+ outDir = await ensureWorkspace();
300
895
  }
301
- else if (extract.content.phases.every((p) => statuses[p.id] === 'validated')) {
302
- args.log('Plan complete — every phase validated. Consider /abap-preflight before release.');
896
+ catch {
897
+ outDir = dirname(envelopePath);
303
898
  }
304
- else {
305
- const blocked = extract.content.phases.filter((p) => statuses[p.id] === 'blocked').map((p) => p.id);
306
- args.log(`No eligible next phase. Blocked: ${blocked.join(', ')}. Unblock, then resume again.`);
899
+ const saved = await saveProject(revision, { cwd: outDir });
900
+ // Task 8 — every envelope save in the plan flow regenerates the markdown
901
+ // handover. This is the single chokepoint for ALL chain writes:
902
+ // applyAuditResult and applyPhaseResolution both route through here.
903
+ // Fix-wave (item 4): the caller's log fn is threaded through so the
904
+ // best-effort failure notice lands in the REPL stream, not bare console.
905
+ writePlanHandover(saved, log);
906
+ return saved;
907
+ }
908
+ /**
909
+ * Persist an audit verdict (or a user waiver) for one phase as revision N+1.
910
+ * Returns the saved path. Counter semantics (documented choice — dedicated
911
+ * field over findings-history parsing): 'failed' increments
912
+ * audit.consecutiveFailures (prior value survives the re-run's 'pending'
913
+ * state via finishPlanResume's carry-over); 'passed'/'warn'/'waived' clear it
914
+ * (a warn is NOT a failure — it neither increments nor carries a streak);
915
+ * 'infra_failed' carries it unchanged (an auditor outage is not a code
916
+ * verdict and must not break the consecutive count in either direction).
917
+ */
918
+ export async function applyAuditResult(envelopePath, phaseId, verdict, opts) {
919
+ const envelope = readProjectFile(envelopePath);
920
+ const pc = parsePlanContent(envelope.content);
921
+ const prior = pc.ok
922
+ ? pc.content.phases.find((p) => p.id === phaseId)?.audit
923
+ : undefined;
924
+ const audit = { state: verdict.state, at: new Date().toISOString() };
925
+ if (verdict.findings && verdict.findings.length > 0)
926
+ audit.findings = verdict.findings;
927
+ if (verdict.state === 'waived') {
928
+ audit.waivedReason = verdict.waivedReason ?? 'waived by user';
929
+ // §10 Q2 — EVERY waive lands a category so the FP-rate query has a complete
930
+ // denominator. Conservative default: a waive with no explicit category
931
+ // (headless, off-switch bulk-waive, legacy caller) is 'accepted-risk',
932
+ // never a 'false-positive' the human didn't assert.
933
+ audit.waiveCategory = verdict.waiveCategory ?? 'accepted-risk';
934
+ }
935
+ if (prior?.sessionId)
936
+ audit.sessionId = prior.sessionId;
937
+ // D-B item 1 — carry the attempt window alongside the evidence pointer: a
938
+ // died-at-menu 'failed' (or an interleaved infra_failed) is re-audited
939
+ // against the SAME attempt, so its window must survive the verdict write.
940
+ if (prior?.attemptStartedAt)
941
+ audit.attemptStartedAt = prior.attemptStartedAt;
942
+ if (verdict.state === 'failed') {
943
+ // Fix-wave (item 10a): a re-confirmed failure (re-audit of an already-
944
+ // 'failed' phase after a menu-death resume) holds the counter — the
945
+ // phase never re-ran, so this is the SAME failure re-judged, not a new
946
+ // consecutive one. Floor 1: a failed verdict always counts at least once.
947
+ audit.consecutiveFailures = opts?.reconfirmedFailure
948
+ ? Math.max(prior?.consecutiveFailures ?? 0, 1)
949
+ : (prior?.consecutiveFailures ?? 0) + 1;
950
+ }
951
+ else if (verdict.state === 'infra_failed' && prior?.consecutiveFailures) {
952
+ audit.consecutiveFailures = prior.consecutiveFailures;
307
953
  }
954
+ const findingsNote = audit.findings?.length
955
+ ? ` (${audit.findings.length} finding${audit.findings.length === 1 ? '' : 's'})`
956
+ : '';
957
+ const statusNote = opts?.statusOverride ? `; status → ${opts.statusOverride}` : '';
958
+ // Review M-5: a RESOLVED verdict means the chain continues — any recorded
959
+ // stop (audit-failed from the earlier verdict, audit-infra from an outage, a
960
+ // write-phase consent) no longer describes an outstanding stop, so clear it
961
+ // instead of leaving a stale record. 'warn' is non-blocking, so it clears a
962
+ // stale stop exactly like a pass. 'infra_failed' (audit-infra-continue,
963
+ // 2026-07-09) is also non-blocking now — the audit could not run but the
964
+ // chain continues, so it too clears any stale stop.
965
+ const lastStop = opts?.lastStop
966
+ ?? (audit.state === 'passed' || audit.state === 'waived'
967
+ || audit.state === 'warn' || audit.state === 'infra_failed'
968
+ ? 'clear'
969
+ : undefined);
970
+ return savePlanPatch(envelopePath, phaseId, {
971
+ audit,
972
+ status: opts?.statusOverride,
973
+ lastStop,
974
+ summary: `audit ${phaseId}: ${audit.state}${findingsNote}${statusNote}`,
975
+ }, opts?.log);
976
+ }
977
+ /**
978
+ * Persist a chain resolution that does NOT change the audit verdict: the
979
+ * user chose re-run (status → 'todo'; the failed audit + counter stay so the
980
+ * next verdict counts as consecutive), chose Mark blocked, or a guarded stop
981
+ * needs its lastStop recorded. Returns the saved path.
982
+ */
983
+ export async function applyPhaseResolution(envelopePath, phaseId, opts) {
984
+ return savePlanPatch(envelopePath, phaseId, {
985
+ audit: opts.clearAudit ? 'clear' : 'keep',
986
+ status: opts.status,
987
+ lastStop: opts.lastStop,
988
+ summary: opts.summary,
989
+ }, opts.log);
308
990
  }