@cspeach/cli 0.9.0 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/README.md +1 -1
  2. package/dist/agent/intent-system-prompt.js +1 -1
  3. package/dist/agent/loop.js +209 -20
  4. package/dist/agent/providers/license-gate.js +44 -0
  5. package/dist/agent/skill-checkpoint.js +1 -1
  6. package/dist/agent/tool-dispatch.js +15 -0
  7. package/dist/approvals/canonical.js +91 -0
  8. package/dist/approvals/jwt.js +39 -2
  9. package/dist/auth/org-anthropic-key.js +25 -0
  10. package/dist/classifier/client.js +18 -3
  11. package/dist/commands/config-set.js +95 -0
  12. package/dist/commands/login.js +31 -14
  13. package/dist/commands/plan-model-tier.js +83 -0
  14. package/dist/commands/plan-resume.js +148 -21
  15. package/dist/config/loader.js +95 -1
  16. package/dist/doctor/checks/_http-probe.js +1 -0
  17. package/dist/doctor/checks/cert.js +14 -3
  18. package/dist/doctor/checks/sap.js +30 -8
  19. package/dist/doctor/checks/zcspeach.js +19 -4
  20. package/dist/one-shot.js +52 -4
  21. package/dist/projects/answer-blockers.js +137 -0
  22. package/dist/projects/extract-cca.js +108 -16
  23. package/dist/projects/extract-modernize.js +1 -1
  24. package/dist/projects/extract-plan.js +130 -37
  25. package/dist/projects/extract-spec-gap.js +34 -7
  26. package/dist/projects/extract-test-coverage.js +1 -1
  27. package/dist/projects/extract-upgrade.js +113 -22
  28. package/dist/projects/index.js +5 -2
  29. package/dist/projects/merge-cca.js +292 -0
  30. package/dist/projects/merge-upgrade.js +173 -0
  31. package/dist/projects/migration.js +103 -1
  32. package/dist/projects/output-paths.js +27 -0
  33. package/dist/projects/plan-run.js +159 -25
  34. package/dist/projects/plan-schema.js +63 -3
  35. package/dist/projects/promote-command.js +25 -2
  36. package/dist/projects/promote.js +128 -0
  37. package/dist/projects/save-command.js +247 -20
  38. package/dist/projects/status.js +3 -1
  39. package/dist/projects/validate.js +1 -1
  40. package/dist/projects/workspace.js +164 -20
  41. package/dist/renderer/notices.js +64 -0
  42. package/dist/renderer/progress-chatter.js +8 -0
  43. package/dist/renderer/tool-widget.js +18 -4
  44. package/dist/renderer/tty.js +43 -4
  45. package/dist/renderer/verify-chain.js +77 -0
  46. package/dist/repl/at-picker.js +60 -7
  47. package/dist/repl/builtin-commands.js +37 -0
  48. package/dist/repl/early-line-buffer.js +68 -0
  49. package/dist/repl/inquirer-guard.js +70 -5
  50. package/dist/repl/numbered-menu.js +131 -0
  51. package/dist/repl/post-turn-status.js +2 -2
  52. package/dist/repl/rule8-detector.js +17 -2
  53. package/dist/repl/safety-confirm.js +111 -2
  54. package/dist/repl/safety-mode-state.js +19 -3
  55. package/dist/repl/slash-picker.js +10 -15
  56. package/dist/repl.js +301 -35
  57. package/dist/router/classifier.js +150 -6
  58. package/dist/sap/capability-matrix.js +20 -0
  59. package/dist/sap/capability-matrix.json +11236 -0
  60. package/dist/sap/capability.js +146 -0
  61. package/dist/sap/connection-manager.js +19 -1
  62. package/dist/sap/onboarding.js +42 -4
  63. package/dist/session/pending.js +27 -0
  64. package/dist/skill-catalog.js +48 -43
  65. package/dist/skills/bundled-skills.js +279 -1
  66. package/dist/skills/promotion-dispatch.js +23 -0
  67. package/dist/tools/_command-shared.js +36 -12
  68. package/dist/tools/_filesystem-shared.js +139 -4
  69. package/dist/tools/_flag.js +25 -0
  70. package/dist/tools/approval.js +64 -21
  71. package/dist/tools/ask-question.js +96 -4
  72. package/dist/tools/capability/tool.js +74 -0
  73. package/dist/tools/dispatch-skill.js +22 -1
  74. package/dist/tools/extend-model/anchored-insert.js +810 -0
  75. package/dist/tools/extend-model/tool.js +188 -0
  76. package/dist/tools/filesystem/extract-document.js +57 -0
  77. package/dist/tools/filesystem/file-edit.js +12 -2
  78. package/dist/tools/filesystem/file-read.js +2 -2
  79. package/dist/tools/filesystem/file-write.js +11 -2
  80. package/dist/tools/filesystem/glob.js +11 -0
  81. package/dist/tools/filesystem/grep.js +10 -0
  82. package/dist/tools/filesystem/read-document.js +107 -0
  83. package/dist/tools/fiori/apply.js +50 -0
  84. package/dist/tools/fiori/bin.js +3 -0
  85. package/dist/tools/fiori/catalog/index.js +27 -0
  86. package/dist/tools/fiori/catalog/value-help.js +230 -0
  87. package/dist/tools/fiori/catalog/viz-chart.js +177 -0
  88. package/dist/tools/fiori/cli.js +71 -0
  89. package/dist/tools/fiori/deploy-config.js +73 -0
  90. package/dist/tools/fiori/fe-scaffold.js +45 -0
  91. package/dist/tools/fiori/i18n.js +39 -0
  92. package/dist/tools/fiori/manifest.js +70 -0
  93. package/dist/tools/fiori/render.js +77 -0
  94. package/dist/tools/fiori/scaffold.js +39 -0
  95. package/dist/tools/fiori/tools.js +356 -0
  96. package/dist/tools/fiori/types.js +1 -0
  97. package/dist/tools/local-build.js +76 -0
  98. package/dist/tools/local-files.js +31 -0
  99. package/dist/tools/project/_merge-shared.js +68 -0
  100. package/dist/tools/project/cca_merge.js +164 -0
  101. package/dist/tools/project/playbook_get.js +1 -1
  102. package/dist/tools/project/upgrade_merge_progress.js +206 -0
  103. package/dist/tools/sap-read.js +53 -9
  104. package/dist/tools/sap-write.js +530 -21
  105. package/dist/tools/shell/shell_exec.js +41 -6
  106. package/dist/tools/snapshot.js +37 -14
  107. package/dist/tools/subagent/background_run.js +17 -1
  108. package/dist/tools/transport-resolution.js +86 -0
  109. package/dist/tools/transport.js +224 -5
  110. package/dist/tools/write-mode.js +4 -0
  111. package/dist/ui/app.js +6 -2
  112. package/dist/ui/body.js +13 -0
  113. package/dist/ui/footer.js +20 -6
  114. package/dist/ui/line-resolution.js +17 -6
  115. package/dist/ui/session-timeline.js +1 -0
  116. package/dist/ui/text-input.js +150 -0
  117. package/dist/ui/widgets/ask-question-modal.js +4 -1
  118. package/package.json +19 -3
  119. package/bench/README.md +0 -78
  120. package/bench/prompts/abap-document-cds.md +0 -44
  121. package/bench/prompts/abap-explain-bdef-handler.md +0 -57
  122. package/bench/prompts/abap-test-method.md +0 -42
  123. package/bench/results/abap-document-cds/claude-haiku-4-5.md +0 -189
  124. package/bench/results/abap-document-cds/claude-opus-4-7.md +0 -120
  125. package/bench/results/abap-document-cds/claude-sonnet-4-6.md +0 -151
  126. package/bench/results/abap-explain-bdef-handler/claude-haiku-4-5.md +0 -112
  127. package/bench/results/abap-explain-bdef-handler/claude-opus-4-7.md +0 -101
  128. package/bench/results/abap-explain-bdef-handler/claude-sonnet-4-6.md +0 -101
  129. package/bench/results/abap-test-method/claude-haiku-4-5.md +0 -186
  130. package/bench/results/abap-test-method/claude-opus-4-7.md +0 -193
  131. package/bench/results/abap-test-method/claude-sonnet-4-6.md +0 -234
@@ -8,8 +8,10 @@ const RULES = [
8
8
  { skill: 'abap-document', patterns: [/developer\s+documentation|document\s+this\s+(class|program|object)/i] },
9
9
  // abap-incident MUST precede dump rules — pipeline framing beats keyword-only matches
10
10
  { skill: 'abap-incident', patterns: [/incident|resolve.*dump.*to.*fix|pipeline.*(short\s+dump|st22|dump)|(dump|st22).*to.*transport|formal\s+(incident|dump)\s+report|end.to.end.*dump/i] },
11
- { skill: 'abap-dump-triage', patterns: [/dump\s+triage|structured\s+investigation.*dump|dump\s+forensic/i] },
12
- { skill: 'abap-dump-analyze', patterns: [/short\s+dump|runtime\s+error|\bst22\b|dump\s+analys/i] },
11
+ // abap-dump (2026-06-12) — merger of abap-dump-analyze + abap-dump-triage.
12
+ // Triage-vs-fix is an intent decision INSIDE the skill now, so both
13
+ // pattern sets route here.
14
+ { skill: 'abap-dump', patterns: [/short\s+dump|runtime\s+error|\bst22\b|dump\s+(analys|triage|forensic)|structured\s+investigation.*dump/i] },
13
15
  { skill: 'abap-eml', patterns: [/\beml\b|entity\s+manipulation|behavior\s+handler/i] },
14
16
  { skill: 'abap-enhance', patterns: [/\bbadi\b|enhancement\s+spot|implicit\s+enhancement|customer\s+exit|user\s+exit|(hook\s+into|extend)\s+(sap|standard|me\d+|migo|miro|va\d+|vf\d+|vl\d+|fb\d+)/i] },
15
17
  { skill: 'abap-estimate', patterns: [/effort\s+estimat|ticket\s+estimat|size\s+this\s+(work|task|ticket|story)|how\s+(many|long).*(hours|days).*(build|develop|implement)/i] },
@@ -20,11 +22,24 @@ const RULES = [
20
22
  // abap-modernize MUST precede abap-rap so "rebuild as RAP" / "convert to RAP" stays
21
23
  // on modernize (rebuild intent) instead of falling to abap-rap's bare \brap\b match.
22
24
  { skill: 'abap-modernize', patterns: [/modernise|modernize|classic.*to.*modern|convert.*to\s+(rap|fiori|odata)|rebuild.*as\s+(rap|fiori)/i] },
25
+ // abap-fiori-build (D12, 2026-06-12) MUST follow abap-modernize ("convert
26
+ // this WRITE report to Fiori" = rebuild intent) and precede abap-rap so
27
+ // "fiori app for my RAP/OData service" lands on the UI skill, not the
28
+ // backend scaffolder.
29
+ { skill: 'abap-fiori-build', patterns: [/\bfiori\b|\b(sap)?ui5\b|freestyle\s+app|\bui\s+app\b|front.?end\s+(for|on)\s/i] },
23
30
  // abap-rap MUST precede abap-generate so RAP-stack prompts route to rap, not the
24
31
  // generic generator (cluster I, 2026-05-03). Also captures "scaffold a CDS view +
25
32
  // BDEF + service binding" type prompts which collide with abap-generate.
26
33
  { skill: 'abap-rap', patterns: [/\brap\b|business\s+object(?!s?\s+for)|bdef|service\s+binding|srvb|scaffold.*(cds|odata\s+v4|bdef|behavior)|odata\s+v4\s+service/i] },
27
- { skill: 'abap-generate', patterns: [/\bgenerate\b|create\s+(a\s+)?(new\s+)?(class|program|report|include)|build.*from\s+scratch/i] },
34
+ // Daily-edit phrasings (D12, 2026-06-12): "add a validation to ZCL_X",
35
+ // "change/fix ZFOO" are abap-generate work — the old rules only matched
36
+ // explicit "generate"/"create" verbs, so daily edits fell through and the
37
+ // coaching picker offered enhance/refactor/test instead.
38
+ { skill: 'abap-generate', patterns: [
39
+ /\bgenerate\b|create\s+(a\s+)?(new\s+)?(class|program|report|include)|build.*from\s+scratch/i,
40
+ /\badd\s+(a\s+|an\s+)?[\w\s-]{1,40}\bto\s+z[_a-z0-9]+/i,
41
+ /\b(change|fix|update|modify|adjust)\s+(the\s+)?z[_a-z0-9]+\b/i,
42
+ ] },
28
43
  { skill: 'abap-impact', patterns: [/impact\s+analysis|where.*used|what.*affected\s+by/i] },
29
44
  // abap-upgrade-* must precede abap-migrate
30
45
  { skill: 'abap-upgrade-verify', patterns: [/upgrade\s+verify|remediation\s+report|verify\s+upgrade/i] },
@@ -10,6 +10,8 @@ const MUTABLE_KEYS = new Set([
10
10
  'effort',
11
11
  'telemetry',
12
12
  'classifier.safe_mode',
13
+ 'local_build',
14
+ 'local_files', // deprecated alias for local_build (2026-06-13) — still accepted
13
15
  'write_mode',
14
16
  'llm.mode',
15
17
  'llm.ai_hub_base_url',
@@ -22,6 +24,25 @@ const MUTABLE_KEYS = new Set([
22
24
  ]);
23
25
  // SAP-scoped key pattern: sap.<alias>.auto_approve
24
26
  const SAP_AUTO_APPROVE_RE = /^sap\.([A-Z0-9_-]{2,32})\.auto_approve$/;
27
+ // SAP-scoped TLS keys: sap.<alias>.ca_cert_path | sap.<alias>.insecure_skip_tls_verify
28
+ const SAP_CA_CERT_PATH_RE = /^sap\.([A-Z0-9_-]{2,32})\.ca_cert_path$/;
29
+ const SAP_INSECURE_TLS_RE = /^sap\.([A-Z0-9_-]{2,32})\.insecure_skip_tls_verify$/;
30
+ /**
31
+ * Coerce a user-typed on/off-style string to a boolean. Accepts the common
32
+ * spellings a developer expects from a CLI toggle (`on`/`true`/`1` → true,
33
+ * `off`/`false`/`0` → false). Returns null for anything else so the caller
34
+ * can reject with a clear error. Case-insensitive.
35
+ */
36
+ const TRUE_TOKENS = new Set(['on', 'true', '1']);
37
+ const FALSE_TOKENS = new Set(['off', 'false', '0']);
38
+ function coerceBool(value) {
39
+ const v = value.trim().toLowerCase();
40
+ if (TRUE_TOKENS.has(v))
41
+ return true;
42
+ if (FALSE_TOKENS.has(v))
43
+ return false;
44
+ return null;
45
+ }
25
46
  const EFFORT_VALUES = ['low', 'medium', 'high', 'xhigh', 'max'];
26
47
  const TELEMETRY_VALUES = ['minimal', 'full'];
27
48
  const AUTO_APPROVE_VALUES = ['never', 'low', 'medium'];
@@ -73,6 +94,45 @@ export async function runConfigSet(args) {
73
94
  console.log(chalk.green(`Set ${key} = ${value}`));
74
95
  return;
75
96
  }
97
+ // SAP-scoped TLS: sap.<alias>.ca_cert_path — trust a CA, verification STAYS ON.
98
+ const caMatch = SAP_CA_CERT_PATH_RE.exec(key);
99
+ if (caMatch) {
100
+ const alias = caMatch[1];
101
+ const cfg = await loadConfig();
102
+ if (!cfg.sap[alias]) {
103
+ err(`SAP system "${alias}" not found in config. Add it first with: cspeach config add-sap`);
104
+ }
105
+ // Path is stored as-is (no existence check here — the connection fails loudly
106
+ // and fail-closed if the CA can't be read at connect time). Setting a CA also
107
+ // clears any insecure flag so the two never conflict.
108
+ cfg.sap[alias].ca_cert_path = value;
109
+ delete cfg.sap[alias].insecure_skip_tls_verify;
110
+ await saveConfig(cfg);
111
+ console.log(chalk.green(`Set ${key} = ${value}`));
112
+ console.log(chalk.dim(' → TLS verification stays ON; this CA is added to the trust store.'));
113
+ return;
114
+ }
115
+ // SAP-scoped TLS: sap.<alias>.insecure_skip_tls_verify — dev-only opt-out.
116
+ const insecMatch = SAP_INSECURE_TLS_RE.exec(key);
117
+ if (insecMatch) {
118
+ const alias = insecMatch[1];
119
+ const b = coerceBool(value);
120
+ if (b === null) {
121
+ err(`${key} must be on/off (also accepts true/false, 1/0).`);
122
+ }
123
+ const cfg = await loadConfig();
124
+ if (!cfg.sap[alias]) {
125
+ err(`SAP system "${alias}" not found in config. Add it first with: cspeach config add-sap`);
126
+ }
127
+ cfg.sap[alias].insecure_skip_tls_verify = b;
128
+ if (b) {
129
+ console.log(chalk.yellow(` ⚠ TLS verification will be DISABLED for ${alias} — dev only, NEVER use in production. ` +
130
+ `Prefer: cspeach config set sap.${alias}.ca_cert_path <path-to-corporate-CA>`));
131
+ }
132
+ await saveConfig(cfg);
133
+ console.log(chalk.green(`Set ${key} = ${b}`));
134
+ return;
135
+ }
76
136
  // Virtual keys that store secrets in the OS keychain instead of config.toml.
77
137
  if (key === 'llm.byok_key' || key === 'llm.ai_hub_token') {
78
138
  // Support stdin piping for secrets: `echo "key" | cspeach config set llm.byok_key -`
@@ -142,6 +202,31 @@ export async function runConfigSet(args) {
142
202
  }
143
203
  cfg.classifier.safe_mode = value === 'true';
144
204
  break;
205
+ case 'local_files':
206
+ // Deprecated alias (2026-06-13). Still applied — but it WRITES to the new
207
+ // `local_build` key so the user's config converges on the new name. The
208
+ // shared `local_build` case below also DELETES any stale legacy
209
+ // `local_files` key so it doesn't linger after migration. One-line
210
+ // deprecation note, then fall through.
211
+ console.log(chalk.yellow(' → "local_files" is deprecated; use "local_build". Applying to local_build.'));
212
+ // fallthrough
213
+ case 'local_build': {
214
+ const b = coerceBool(value);
215
+ if (b === null) {
216
+ err(`${key} must be on/off (also accepts true/false, 1/0).`);
217
+ }
218
+ cfg.local_build = b;
219
+ // Migration hygiene (2026-06-13): whether the user typed `local_build` or
220
+ // the deprecated `local_files` alias, we write the new key AND drop any
221
+ // stale legacy `local_files` so it doesn't linger in `config show` after
222
+ // migration. resolveLocalBuild prefers local_build anyway, but a leftover
223
+ // legacy key is confusing and could surprise a future reader.
224
+ delete cfg.local_files;
225
+ // Config is read at startup; a running session isn't retroactively
226
+ // changed (the tool override is applied once, before the first turn).
227
+ console.log(chalk.dim(' → takes effect on the next `cspeach` launch.'));
228
+ break;
229
+ }
145
230
  case 'write_mode':
146
231
  if (!isValidWriteMode(value)) {
147
232
  err(`Invalid value for write_mode. Must be one of: auto, approval-gated, advisory-only`);
@@ -201,6 +286,13 @@ export function validateValue(key, value) {
201
286
  if (SAP_AUTO_APPROVE_RE.test(key)) {
202
287
  return AUTO_APPROVE_VALUES.includes(value);
203
288
  }
289
+ // SAP-scoped TLS keys.
290
+ if (SAP_CA_CERT_PATH_RE.test(key)) {
291
+ return value.length > 0; // any non-empty path; existence checked at connect
292
+ }
293
+ if (SAP_INSECURE_TLS_RE.test(key)) {
294
+ return coerceBool(value) !== null;
295
+ }
204
296
  if (!MUTABLE_KEYS.has(key))
205
297
  return false;
206
298
  switch (key) {
@@ -220,6 +312,9 @@ export function validateValue(key, value) {
220
312
  return TELEMETRY_VALUES.includes(value);
221
313
  case 'classifier.safe_mode':
222
314
  return value === 'true' || value === 'false';
315
+ case 'local_build':
316
+ case 'local_files': // deprecated alias — same on/off coercion
317
+ return coerceBool(value) !== null;
223
318
  case 'write_mode':
224
319
  return isValidWriteMode(value);
225
320
  case 'llm.mode':
@@ -13,7 +13,11 @@
13
13
  * Dependencies (open, print, prompt) are injected so the command is
14
14
  * unit-testable without stubbing platform globals.
15
15
  */
16
+ import keytar from 'keytar';
16
17
  import { writeAuthFile } from '../auth/auth-file.js';
18
+ import { fetchOrgAnthropicKey } from '../auth/org-anthropic-key.js';
19
+ const ANTHROPIC_KEYCHAIN_SERVICE = 'cspeach.anthropic';
20
+ const ANTHROPIC_KEYCHAIN_ACCOUNT = 'api-key';
17
21
  export async function runLoginCommand(deps) {
18
22
  const proxy = deps.proxyUrl.replace(/\/$/, '');
19
23
  const pollInterval = deps.pollIntervalMs ?? 5000;
@@ -92,22 +96,35 @@ export async function runLoginCommand(deps) {
92
96
  const me = await r.json();
93
97
  deps.print(`✓ Signed in as ${me.email} (${me.plan})`);
94
98
  if (me.trial_mode === 'byok' || me.plan === 'byo-key') {
95
- deps.print('');
96
- deps.print('This account is BYOK — you provide your own Anthropic API key.');
97
- deps.print('Get one at https://console.anthropic.com/settings/keys');
98
- const ak = (await deps.prompt('Paste your Anthropic API key (sk-ant-...): ')).trim();
99
- if (ak.startsWith('sk-ant-') && ak.length > 20) {
100
- // The non-BYOK save above just persisted these exact values,
101
- // so reading them back from disk would be redundant.
102
- await writeAuthFile({
103
- apiKey: polled.api_key,
104
- customerId: polled.customer_id,
105
- byokAnthropicKey: ak,
106
- });
107
- deps.print('✓ Anthropic key saved');
99
+ // Try to fetch the org's Anthropic key first — non-fatal if unavailable.
100
+ const orgKey = await fetchOrgAnthropicKey(proxy, polled.api_key, {
101
+ fetchImpl: deps.fetchImpl,
102
+ });
103
+ const setKey = deps.setKeychainKey ??
104
+ ((svc, acct, k) => keytar.setPassword(svc, acct, k));
105
+ if (orgKey !== null) {
106
+ await setKey(ANTHROPIC_KEYCHAIN_SERVICE, ANTHROPIC_KEYCHAIN_ACCOUNT, orgKey);
107
+ deps.print('Using your organization\'s Anthropic key.');
108
108
  }
109
109
  else {
110
- deps.print('error: Anthropic key looks malformed (must start with sk-ant-). Skipped — run `cspeach login` again later.');
110
+ // Fall back to the manual paste prompt — behaviour unchanged.
111
+ deps.print('');
112
+ deps.print('This account is BYOK — you provide your own Anthropic API key.');
113
+ deps.print('Get one at https://console.anthropic.com/settings/keys');
114
+ const ak = (await deps.prompt('Paste your Anthropic API key (sk-ant-...): ')).trim();
115
+ if (ak.startsWith('sk-ant-') && ak.length > 20) {
116
+ // The non-BYOK save above just persisted these exact values,
117
+ // so reading them back from disk would be redundant.
118
+ await writeAuthFile({
119
+ apiKey: polled.api_key,
120
+ customerId: polled.customer_id,
121
+ byokAnthropicKey: ak,
122
+ });
123
+ deps.print('✓ Anthropic key saved');
124
+ }
125
+ else {
126
+ deps.print('error: Anthropic key looks malformed (must start with sk-ant-). Skipped — run `cspeach login` again later.');
127
+ }
111
128
  }
112
129
  }
113
130
  }
@@ -0,0 +1,83 @@
1
+ // A4 (2026-06-11) — per-phase model tiering for the plan runner.
2
+ //
3
+ // Cost-control Phase 3, pre-approved by the 2026-05-18 model bench
4
+ // (cspeach-cli/scripts/bench-models.ts): Sonnet matches Opus on
5
+ // explain/document/test/design-shaped ABAP work at ~85% lower cost, while
6
+ // Haiku is UNSAFE on ABAP content (hallucinates SAP terms — bench memo) and
7
+ // is therefore NEVER a tiering target here.
8
+ //
9
+ // FLAG-GATED, OFF BY DEFAULT. Two switches, env wins when set:
10
+ // - config: plan_model_tiering = true (~/.cspeach/config.toml)
11
+ // - env: CSPEACH_PLAN_MODEL_TIERING=on (also 1/true; off/0/false force-disable)
12
+ //
13
+ // Classification is STRUCTURAL, not id-string parsing: each plan phase
14
+ // carries a `delegateTo` skill (Zod enum PLAN_DELEGATE_SKILLS in
15
+ // projects/plan-schema.ts). Only the read/reason-shaped skills the bench
16
+ // cleared go to Sonnet; every write/codegen delegate — and anything
17
+ // ambiguous or unknown — stays on the session's default model.
18
+ //
19
+ // Extra guard: tiering only fires when the session default is Opus-family.
20
+ // A user who deliberately set default_model to Sonnet (or anything cheaper)
21
+ // must never be silently moved to a different model by this feature.
22
+ //
23
+ // Wiring (single read path): repl.tsx computes the choice right after
24
+ // preparePlanResume succeeds (both Ink and classic paths), prints the dim
25
+ // notice for cost auditability, and threads the model into runTurn via
26
+ // RunTurnParams.modelOverride — which also feeds the per-turn cost-log
27
+ // entry so ~/.cspeach/sessions/<id>-cost.jsonl records the ACTUAL model.
28
+ /** The only model this feature ever tiers down to. Never Haiku. */
29
+ export const PLAN_TIER_SONNET_MODEL = 'claude-sonnet-4-6';
30
+ /**
31
+ * Delegate skills whose phases are design/test/document-shaped — the shapes
32
+ * the 2026-05-18 bench cleared for Sonnet. `abap-document` is not in
33
+ * PLAN_DELEGATE_SKILLS today; included so an enum extension is covered
34
+ * without touching this file. Everything else (abap-data-model, abap-rap,
35
+ * abap-eml, abap-generate, abap-segw, abap-fiori-build) writes SAP objects
36
+ * or code and stays on the default model.
37
+ */
38
+ export const SONNET_TIER_DELEGATES = new Set([
39
+ 'abap-design',
40
+ 'abap-test',
41
+ 'abap-document',
42
+ ]);
43
+ const ENV_FLAG = 'CSPEACH_PLAN_MODEL_TIERING';
44
+ /**
45
+ * Resolve the feature flag. Env var (when set) wins over config so a single
46
+ * shell can A/B the feature without editing config.toml; the config flag is
47
+ * the persistent opt-in. Strict `=== true` on the config value — TOML is
48
+ * hand-edited, and a typo'd string must read as OFF, never ON.
49
+ */
50
+ export function isPlanModelTieringEnabled(cfg, env = process.env) {
51
+ const raw = env[ENV_FLAG];
52
+ if (raw !== undefined && raw !== '') {
53
+ const v = raw.trim().toLowerCase();
54
+ return v === 'on' || v === '1' || v === 'true';
55
+ }
56
+ return cfg.plan_model_tiering === true;
57
+ }
58
+ const NO_OVERRIDE = { model: null, notice: null };
59
+ /**
60
+ * Pure decision function: which model should this plan phase run on?
61
+ *
62
+ * Returns an override ONLY when all of:
63
+ * - the flag is on,
64
+ * - the phase's delegateTo is a bench-cleared design/test/document shape,
65
+ * - the session default is an Opus-family model (we only tier DOWN from
66
+ * Opus — never sideways/up from a deliberately cheaper default).
67
+ *
68
+ * Anything ambiguous resolves to "no override". The returned model is
69
+ * claude-sonnet-4-6 or nothing — Haiku is unreachable by construction
70
+ * (pinned in plan-model-tier.test.ts).
71
+ */
72
+ export function selectPlanPhaseModel(args) {
73
+ if (!args.enabled)
74
+ return NO_OVERRIDE;
75
+ if (!args.delegateTo || !SONNET_TIER_DELEGATES.has(args.delegateTo))
76
+ return NO_OVERRIDE;
77
+ if (!/opus/i.test(args.defaultModel))
78
+ return NO_OVERRIDE;
79
+ return {
80
+ model: PLAN_TIER_SONNET_MODEL,
81
+ notice: `phase ${args.phaseId} → ${PLAN_TIER_SONNET_MODEL} (plan model tiering)`,
82
+ };
83
+ }
@@ -16,7 +16,20 @@
16
16
  // precede a closing ask_question, see turn-assistant-text.ts) →
17
17
  // extract the LAST csforge:plan-manifest block → build the version
18
18
  // N+1 revision (same id, history appended) → save → print the
19
- // updated tracker + next-step hint.
19
+ // updated tracker + next-step hint → return the saved path + the
20
+ // HARNESS-built resume command for the next phase (A1, 2026-06-10).
21
+ //
22
+ // A1 (defect D23) — true bounded phases: phase end = turn end.
23
+ // The model's resume turn ends at the manifest block. It must NOT ask a
24
+ // continuation question, NOT call dispatch_skill, and NOT execute a
25
+ // second phase in-turn (the old contract allowed in-session "continue",
26
+ // which snowballed a 6-phase run into one 3.24M-token-context session).
27
+ // Continuation is harness-owned: repl.tsx calls offerNextPhaseAutoRun
28
+ // with the PlanResumeOutcome; on consent it queues the EXACT command
29
+ // buildResumeCommand produced from the just-saved path (the model once
30
+ // hallucinated a wrong base when it authored this text itself), and
31
+ // every plan-resume turn starts from reset session messages
32
+ // (resetContextForPlanResume).
20
33
  //
21
34
  // The save hook inside runTurn is suppressed for resume turns
22
35
  // (RunTurnParams.suppressSaveHook) — it would otherwise offer to save a
@@ -28,12 +41,22 @@ import { basename, dirname, join } from 'node:path';
28
41
  import { readProjectFile } from '../projects/status.js';
29
42
  import { resolveAtTokenAsync, formatProjectFileList, ensureWorkspace } from '../projects/workspace.js';
30
43
  import { parsePlanContent } from '../projects/plan-schema.js';
31
- import { statusesFromItems, computeNextPhase, renderPlanTracker, buildPlanRevision, } from '../projects/plan-run.js';
44
+ import { statusesFromItems, computeNextPhase, renderPlanTracker, buildPlanRevision, isPhaseSatisfied, planCompletionLines, } from '../projects/plan-run.js';
32
45
  import { extractPlan } from '../projects/extract-plan.js';
33
46
  import { saveProject } from '../projects/save.js';
34
47
  import { validateEnvelope } from '../projects/validate.js';
35
48
  import { collectTurnAssistantText } from '../agent/turn-assistant-text.js';
36
49
  import { getAuthorIdentity } from '../agent/loop.js';
50
+ /**
51
+ * A1 (defect D23) — the harness-owned continuation contract, stated to the
52
+ * model verbatim in every resume prompt: the manifest ends the turn; the
53
+ * model must not call dispatch_skill or ask a continuation question — the
54
+ * CLI dispatches the next phase itself. Exported as a named constant so
55
+ * tests pin the prompt to this exact block instead of brittle prose
56
+ * regexes (the old contract let the model keep executing phases in-turn —
57
+ * a 6-phase run once snowballed to a 3.24M-token context).
58
+ */
59
+ export const PLAN_RESUME_HARNESS_OVERRIDE = 'HARNESS OVERRIDE of the skill\'s Mode 2 step 8: do NOT ask a continuation question, do NOT call dispatch_skill, and NEVER start another phase in this turn. After your manifest is persisted, the CLI itself asks the user whether to run the next phase and dispatches it with the exact saved file in a fresh bounded context. Phase end = turn end.';
37
60
  export async function preparePlanResume(args) {
38
61
  const tokens = args.body
39
62
  .split(/\s+/)
@@ -116,10 +139,14 @@ export async function preparePlanResume(args) {
116
139
  }));
117
140
  args.log('');
118
141
  if (!next) {
119
- const allValidated = content.phases.every((p) => statuses[p.id] === 'validated');
142
+ // C1 (D30 waiver): 'validated-with-waiver' is satisfied — a plan whose
143
+ // last gate was explicitly waived is complete, not stuck.
144
+ const allValidated = content.phases.every((p) => isPhaseSatisfied(statuses[p.id]));
120
145
  if (allValidated) {
121
- args.log('Plan complete — every phase is validated.');
122
- args.log('Consider /abap-preflight on the produced transport(s) before release.');
146
+ // C2 (D24): a finished UI-less backend stack chains to /abap-fiori-build
147
+ // with the exact binding name — the marketed idea→app story must not
148
+ // silently end at the service binding.
149
+ args.log(...planCompletionLines(content.phases));
123
150
  }
124
151
  else {
125
152
  const blocked = content.phases.filter((p) => statuses[p.id] === 'blocked').map((p) => p.id);
@@ -143,10 +170,39 @@ export async function preparePlanResume(args) {
143
170
  }
144
171
  }
145
172
  const planState = JSON.stringify({ title: envelope.title, version: envelope.version, statuses, content }, null, 2);
173
+ // A2 (defects D23/D30) — compact write-back: the model re-emitting the
174
+ // full plan content every phase cost ~6–10k output tokens/phase at Opus
175
+ // pricing for data the envelope already holds. The resume turn emits
176
+ // statuses + a `changed` entry for the executed phase only; extractPlan
177
+ // merges it into the prior content. The example is built with the REAL
178
+ // phase ids and current statuses so the model copies, not reconstructs.
179
+ const compactExample = [
180
+ '<!-- csforge:plan-manifest',
181
+ JSON.stringify({
182
+ title: envelope.title,
183
+ statuses: { ...statuses, [next.id]: '<validated | validated-with-waiver | blocked>' },
184
+ changed: {
185
+ [next.id]: {
186
+ work: {
187
+ generated: ['<object names created/changed>'],
188
+ transport: '<transport number — omit the key if none>',
189
+ // C2 (D24): the service phase records the published SRVB name in
190
+ // work.binding — the plan-complete message hands it to
191
+ // /abap-fiori-build. Only shown when this turn IS the service phase.
192
+ ...(next.layer === 'service'
193
+ ? { binding: '<published service binding name, e.g. ZUI_MAINTREQ_O4>' }
194
+ : {}),
195
+ notes: '<decisions, substitutions, blocker details worth keeping — omit if none>',
196
+ },
197
+ },
198
+ },
199
+ }, null, 2),
200
+ '-->',
201
+ ].join('\n');
146
202
  const llmPrompt = [
147
203
  `Resume execution of the project plan "${envelope.title}" (envelope v${envelope.version}).`,
148
204
  '',
149
- `Execute Mode 2 of the abap-plan skill for phase "${next.id}" ONLY — it is the computed next eligible phase. Do not execute any other phase unless the user explicitly picks "continue" at the phase-end question.`,
205
+ `Execute Mode 2 of the abap-plan skill for phase "${next.id}" ONLY — it is the computed next eligible phase. Never execute any other phase in this turn; the CLI itself offers and dispatches the next phase (in a fresh bounded context) after this one is persisted.`,
150
206
  '',
151
207
  '<plan_state>',
152
208
  planState,
@@ -156,13 +212,26 @@ export async function preparePlanResume(args) {
156
212
  ...ruleBlocks,
157
213
  '</phase_rules>',
158
214
  '',
159
- // 2026-06-06 live-smoke lesson #3: the model asked the continuation
160
- // question FIRST, then ended the turn on the user's "exit" answer with
161
- // a one-line acknowledgment — and the whole phase result was lost. The
162
- // ordering must be explicit and the consequence named.
163
- 'CRITICAL — write-back ordering: emit the COMPLETE updated <!-- csforge:plan-manifest --> block (full phase list, statuses entry for every phase, this phase\'s work filled in — including work.notes with the compact decision register when the phase produced decisions rather than SAP objects) BEFORE the phase-end continuation question. The manifest block is how the result is persisted; a turn that ends without it LOSES the phase. After the user answers "exit", reply with at most one short line — the manifest must already be in the transcript by then.',
215
+ // A1 (2026-06-10, defect D23): the old contract let the model ask a
216
+ // continuation question and keep executing phases in-turn — a 6-phase
217
+ // run snowballed to a 3.24M-token context because the turn never
218
+ // ended. The manifest now ENDS the turn; the harness owns continuation
219
+ // (offerNextPhaseAutoRun) and dispatches the next phase itself with
220
+ // the exact saved path, in reset context.
221
+ 'CRITICAL — the write-back ENDS the turn: emit ONE COMPACT <!-- csforge:plan-manifest --> block as the LAST thing in your output, then END THE TURN. Compact shape = "title" + a "statuses" entry for EVERY phase + a "changed" map carrying ONLY the phase(s) you touched this turn (normally exactly this one). Each "changed" entry holds the phase\'s COMPLETE updated "work" (generated / transport / snapshot / notes — put the compact decision register in work.notes when the phase produced decisions rather than SAP objects); it replaces that phase\'s prior work wholesale. Do NOT re-emit "content" or the full phase list — the CLI already holds the full plan, merges your "changed" entries into it, and recomputes the summary. Exact shape for this turn:',
222
+ '',
223
+ compactExample,
224
+ '',
225
+ // C1 (D30 waiver) — the waiver status exists so an explicitly-waived exit
226
+ // gate is recorded as what it is, instead of being laundered into a
227
+ // "validated" the gate never earned (the c1.test AUnit-gap incident).
228
+ 'Status rules: "validated" ONLY when the exit gate actually held and was verified. If the gate could NOT be met but the user EXPLICITLY waived it this turn (e.g. "skip the test gate, continue anyway"), use "validated-with-waiver" and record what was waived and why in work.notes — never mark an unmet gate "validated", and never use the waiver status without an explicit user waiver. Otherwise the phase is "blocked".',
229
+ '',
230
+ 'Only if the plan itself must change structurally (a phase added, removed, or re-sequenced) fall back to the full shape with "content" — the CLI accepts both. The manifest block is how the result is persisted; a turn that ends without it LOSES the phase.',
231
+ '',
232
+ PLAN_RESUME_HARNESS_OVERRIDE,
164
233
  ].join('\n');
165
- return { path, envelope, statuses, nextPhaseId: next.id, llmPrompt };
234
+ return { path, envelope, statuses, nextPhaseId: next.id, nextDelegateTo: next.delegateTo, llmPrompt };
166
235
  }
167
236
  /**
168
237
  * Given a set of candidate file paths (e.g. the matches from an ambiguous
@@ -193,14 +262,57 @@ export function pickNewestPlanVersion(paths) {
193
262
  return items[0].path;
194
263
  }
195
264
  /**
196
- * True when a queued dispatch command is a `/abap-plan --resume …`. The REPL
197
- * uses this to clear session.messages before re-entering, so the auto-fired
198
- * next phase runs in fresh, bounded context (the cheap path).
265
+ * True when a queued dispatch command is a `/abap-plan --resume …`.
266
+ * Consumer: dispatch_skill REJECTS model-authored plan-resume dispatches
267
+ * (A1 — the harness owns that command; the model once hallucinated a wrong
268
+ * base token). The transcript clear for resume turns is owned solely by
269
+ * resetContextForPlanResume, called from the REPL's plan-resume prepare
270
+ * branch after preparePlanResume succeeds.
199
271
  */
200
272
  export function isPlanResumeCommand(cmd) {
201
273
  const c = cmd.trim();
202
274
  return /^\/abap-plan\b/.test(c) && /(^|\s)--resume(\s|$)/.test(c);
203
275
  }
276
+ /**
277
+ * A1 (defect D23) — the HARNESS builds the next-phase resume command from
278
+ * the exact path finishPlanResume just saved. Single construction site:
279
+ * the model never authors this text (it once invented `@…-plan-c1` when
280
+ * the real file was `…-plan-c9b3-v14.cspeach.json`). The basename keeps
281
+ * the version suffix — it resolves to that exact file, and the
282
+ * newest-version redirect in preparePlanResume still protects against a
283
+ * concurrent later save.
284
+ */
285
+ export function buildResumeCommand(savedPath) {
286
+ return `/abap-plan --resume @${basename(savedPath)}`;
287
+ }
288
+ /**
289
+ * A1 — every plan-resume turn starts from RESET session messages: the
290
+ * envelope + the bounded preparePlanResume prompt carry all needed state,
291
+ * so prior-phase (or prior-chat) transcript is pure cost. Called by the
292
+ * REPL right after preparePlanResume succeeds, which covers BOTH the
293
+ * harness-dispatched auto-run path and a manually typed `--resume`.
294
+ * Returns true when messages were actually cleared (caller logs a notice).
295
+ * Property reassignment (not splice) — session.messages consumers always
296
+ * re-read the property.
297
+ */
298
+ export function resetContextForPlanResume(session) {
299
+ if (session.messages.length === 0)
300
+ return false;
301
+ session.messages = [];
302
+ return true;
303
+ }
304
+ export async function offerNextPhaseAutoRun(args) {
305
+ if (!args.outcome.nextPhaseId)
306
+ return false;
307
+ const delegate = args.outcome.nextDelegateTo ? ` (${args.outcome.nextDelegateTo})` : '';
308
+ const answer = (await args.prompt(`Run next phase ${args.outcome.nextPhaseId}${delegate} now in a fresh context? [y/N]: `)).trim().toLowerCase();
309
+ if (answer !== 'y' && answer !== 'yes') {
310
+ args.log(`Exiting — resume later with: ${args.outcome.resumeCommand}`);
311
+ return false;
312
+ }
313
+ args.queueDispatch(args.outcome.resumeCommand);
314
+ return true;
315
+ }
204
316
  /**
205
317
  * Find the newest sibling version of the envelope at `path` — same
206
318
  * filename family (`<slug>-<shortid>-vN[...]`) AND same envelope id (the
@@ -250,11 +362,18 @@ export async function finishPlanResume(args) {
250
362
  args.log('');
251
363
  args.log('[plan] turn produced no assistant output — plan envelope UNCHANGED.');
252
364
  args.log(`[plan] re-run: /abap-plan --resume @${basename(args.prepared.path)}`);
253
- return;
365
+ return null;
254
366
  }
255
367
  let extract;
256
368
  try {
257
- extract = extractPlan(text);
369
+ // A2 — pass the prior content so a COMPACT manifest (statuses + changed
370
+ // map) can be merged into it. prepared.envelope.content already passed
371
+ // parsePlanContent in preparePlanResume; re-parsing here hands extractPlan
372
+ // the normalised PlanContent without widening PreparedPlanResume. If the
373
+ // parse somehow fails, prior stays undefined and a compact manifest fails
374
+ // loudly below (full manifests are unaffected).
375
+ const prior = parsePlanContent(args.prepared.envelope.content);
376
+ extract = extractPlan(text, prior.ok ? prior.content : undefined);
258
377
  }
259
378
  catch (e) {
260
379
  // Loud by design: the phase may have built real SAP objects, but the
@@ -265,14 +384,15 @@ export async function finishPlanResume(args) {
265
384
  args.log('[plan] The plan envelope is unchanged. Check what the phase actually built');
266
385
  args.log('[plan] (transport, activated objects), update the envelope statuses by hand or');
267
386
  args.log(`[plan] re-run: /abap-plan --resume @${basename(args.prepared.path)}`);
268
- return;
387
+ return null;
269
388
  }
270
389
  const revision = buildPlanRevision(args.prepared.envelope, extract, getAuthorIdentity(), new Date().toISOString());
390
+ // JSON round-trip simulates disk serialization for the validator.
271
391
  const check = validateEnvelope(JSON.parse(JSON.stringify(revision)));
272
392
  if (!check.ok) {
273
393
  args.log('');
274
394
  args.log(`[plan] PHASE RESULT NOT PERSISTED — revision failed validation: ${check.error.message}`);
275
- return;
395
+ return null;
276
396
  }
277
397
  let outDir;
278
398
  try {
@@ -298,11 +418,18 @@ export async function finishPlanResume(args) {
298
418
  if (next) {
299
419
  args.log(`Next: ${next.id} (${next.delegateTo}) — run /abap-plan --resume @${basename(savedPath)} in a fresh session.`);
300
420
  }
301
- else if (extract.content.phases.every((p) => statuses[p.id] === 'validated')) {
302
- args.log('Plan complete — every phase validated. Consider /abap-preflight before release.');
421
+ else if (extract.content.phases.every((p) => isPhaseSatisfied(statuses[p.id]))) {
422
+ // C2 (D24): same chain as preparePlanResume — single source in plan-run.ts.
423
+ args.log(...planCompletionLines(extract.content.phases));
303
424
  }
304
425
  else {
305
426
  const blocked = extract.content.phases.filter((p) => statuses[p.id] === 'blocked').map((p) => p.id);
306
427
  args.log(`No eligible next phase. Blocked: ${blocked.join(', ')}. Unblock, then resume again.`);
307
428
  }
429
+ return {
430
+ savedPath,
431
+ resumeCommand: buildResumeCommand(savedPath),
432
+ nextPhaseId: next?.id ?? null,
433
+ nextDelegateTo: next?.delegateTo ?? null,
434
+ };
308
435
  }