@cspeach/cli 1.0.0 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/dist/agent/loop.js +22 -9
  2. package/dist/approvals/op-labels.js +124 -0
  3. package/dist/approvals/render.js +42 -36
  4. package/dist/cli.js +15 -0
  5. package/dist/commands/compact.js +28 -2
  6. package/dist/commands/config-set.js +189 -0
  7. package/dist/commands/config-show.js +20 -0
  8. package/dist/commands/export-audit.js +43 -0
  9. package/dist/commands/help.js +5 -0
  10. package/dist/commands/plan-audit-evidence.js +266 -0
  11. package/dist/commands/plan-audit.js +692 -0
  12. package/dist/commands/plan-chain.js +671 -0
  13. package/dist/commands/plan-continue.js +179 -0
  14. package/dist/commands/plan-gate.js +154 -0
  15. package/dist/commands/plan-resume.js +588 -33
  16. package/dist/config/loader.js +128 -4
  17. package/dist/config/model-defaults.js +14 -0
  18. package/dist/cost/pricing.js +27 -1
  19. package/dist/doctor/checks/system-roles.js +41 -0
  20. package/dist/doctor/run.js +2 -0
  21. package/dist/models/resolve.js +61 -0
  22. package/dist/models/server-config.js +155 -0
  23. package/dist/one-shot.js +25 -3
  24. package/dist/projects/extract-cca.js +3 -1
  25. package/dist/projects/extract-modernize.js +3 -1
  26. package/dist/projects/extract-plan.js +60 -6
  27. package/dist/projects/extract-test-coverage.js +3 -1
  28. package/dist/projects/extract-upgrade.js +3 -1
  29. package/dist/projects/handover-md.js +195 -0
  30. package/dist/projects/index.js +1 -1
  31. package/dist/projects/plan-run.js +137 -13
  32. package/dist/projects/plan-schema.js +73 -0
  33. package/dist/projects/run-lease.js +157 -0
  34. package/dist/projects/save-command.js +26 -15
  35. package/dist/renderer/status-footer.js +22 -12
  36. package/dist/renderer/thinking-heartbeat.js +64 -8
  37. package/dist/renderer/todo-block.js +51 -0
  38. package/dist/renderer/tool-widget.js +37 -0
  39. package/dist/repl/bracketed-paste.js +28 -19
  40. package/dist/repl/builtin-commands.js +5 -0
  41. package/dist/repl/current-transport.js +10 -0
  42. package/dist/repl/history.js +86 -0
  43. package/dist/repl/ink-stdin-guard.js +64 -0
  44. package/dist/repl/mode-ceiling.js +16 -0
  45. package/dist/repl/mode-cycle.js +104 -0
  46. package/dist/repl/post-turn-status.js +24 -4
  47. package/dist/repl/slash-completer.js +5 -0
  48. package/dist/repl.js +954 -83
  49. package/dist/rewind/candidates.js +194 -0
  50. package/dist/rewind/cli.js +137 -0
  51. package/dist/rewind/format.js +27 -0
  52. package/dist/rewind/restore.js +245 -0
  53. package/dist/session/audit-export.js +459 -0
  54. package/dist/session/context-report.js +163 -0
  55. package/dist/session/recap.js +160 -0
  56. package/dist/skill-catalog.js +9 -3
  57. package/dist/skills/bundled-skills.js +59 -66
  58. package/dist/tools/approval.js +115 -7
  59. package/dist/tools/ask-question.js +304 -3
  60. package/dist/tools/extend-model/anchored-insert.js +604 -0
  61. package/dist/tools/extend-model/tool.js +162 -10
  62. package/dist/tools/fiori/fe-extend.js +76 -0
  63. package/dist/tools/fiori/fe-scaffold.js +29 -3
  64. package/dist/tools/fiori/floorplan-map.js +19 -0
  65. package/dist/tools/fiori/samples/data/index.json +13602 -0
  66. package/dist/tools/fiori/samples/data/sources.generated.js +808 -0
  67. package/dist/tools/fiori/samples/loader.js +248 -0
  68. package/dist/tools/fiori/samples/search.js +63 -0
  69. package/dist/tools/fiori/samples/types.js +2 -0
  70. package/dist/tools/fiori/smoke/assertions.js +74 -0
  71. package/dist/tools/fiori/smoke/browser.js +52 -0
  72. package/dist/tools/fiori/smoke/driver.js +89 -0
  73. package/dist/tools/fiori/smoke/freestyle-spec.js +317 -0
  74. package/dist/tools/fiori/smoke/run-smoke.js +149 -0
  75. package/dist/tools/fiori/tools.js +328 -3
  76. package/dist/tools/local-build.js +11 -1
  77. package/dist/tools/sap-read.js +79 -11
  78. package/dist/tools/sap-write.js +24 -4
  79. package/dist/tools/snapshot.js +27 -1
  80. package/dist/tools/subagent/agent_run.js +27 -3
  81. package/dist/tools/todo.js +144 -0
  82. package/dist/ui/app.js +372 -19
  83. package/dist/ui/approval-modal.js +49 -16
  84. package/dist/ui/ask-question-emitter.js +14 -0
  85. package/dist/ui/context-grid.js +108 -0
  86. package/dist/ui/footer.js +109 -30
  87. package/dist/ui/header.js +7 -0
  88. package/dist/ui/line-resolution.js +18 -2
  89. package/dist/ui/rewind-emitter.js +10 -0
  90. package/dist/ui/rewind-panel.js +81 -0
  91. package/dist/ui/sap-state-store.js +1 -0
  92. package/dist/ui/status-line.js +43 -0
  93. package/dist/ui/text-input.js +72 -8
  94. package/dist/ui/todo-emitter.js +25 -0
  95. package/dist/ui/todo-panel.js +64 -0
  96. package/dist/ui/turn-status-emitter.js +50 -4
  97. package/dist/ui/turn-status.js +18 -3
  98. package/dist/ui/widgets/ask-form.js +242 -0
  99. package/dist/ui/widgets/ask-question-modal.js +17 -7
  100. package/package.json +4 -1
@@ -7,12 +7,16 @@ import { loadConfig, saveConfig, isValidWriteMode } from '../config/loader.js';
7
7
  const MUTABLE_KEYS = new Set([
8
8
  'proxy_url',
9
9
  'default_model',
10
+ 'audit_model',
11
+ 'compact_model',
10
12
  'effort',
11
13
  'telemetry',
12
14
  'classifier.safe_mode',
13
15
  'local_build',
14
16
  'local_files', // deprecated alias for local_build (2026-06-13) — still accepted
15
17
  'write_mode',
18
+ 'plan_mode',
19
+ 'plan_audit',
16
20
  'llm.mode',
17
21
  'llm.ai_hub_base_url',
18
22
  'llm.ai_hub_model_alias',
@@ -24,6 +28,8 @@ const MUTABLE_KEYS = new Set([
24
28
  ]);
25
29
  // SAP-scoped key pattern: sap.<alias>.auto_approve
26
30
  const SAP_AUTO_APPROVE_RE = /^sap\.([A-Z0-9_-]{2,32})\.auto_approve$/;
31
+ // SAP-scoped key pattern: sap.<alias>.role (system role → write-mode ceiling)
32
+ const SAP_ROLE_RE = /^sap\.([A-Z0-9_-]{2,32})\.role$/;
27
33
  // SAP-scoped TLS keys: sap.<alias>.ca_cert_path | sap.<alias>.insecure_skip_tls_verify
28
34
  const SAP_CA_CERT_PATH_RE = /^sap\.([A-Z0-9_-]{2,32})\.ca_cert_path$/;
29
35
  const SAP_INSECURE_TLS_RE = /^sap\.([A-Z0-9_-]{2,32})\.insecure_skip_tls_verify$/;
@@ -46,7 +52,10 @@ function coerceBool(value) {
46
52
  const EFFORT_VALUES = ['low', 'medium', 'high', 'xhigh', 'max'];
47
53
  const TELEMETRY_VALUES = ['minimal', 'full'];
48
54
  const AUTO_APPROVE_VALUES = ['never', 'low', 'medium'];
55
+ const SYSTEM_ROLE_VALUES = ['dev', 'qas', 'prd'];
49
56
  const LLM_MODE_VALUES = ['managed', 'byok', 'ai-hub', 'local'];
57
+ const PLAN_MODE_VALUES = ['step', 'guarded'];
58
+ const PLAN_AUDIT_VALUES = ['on', 'off'];
50
59
  /**
51
60
  * Thrown on validation failure. The outer CLI dispatcher converts to
52
61
  * `console.error` + `process.exit(exitCode)`. Throwing instead of calling
@@ -65,6 +74,119 @@ export class ConfigSetError extends Error {
65
74
  function err(msg) {
66
75
  throw new ConfigSetError(`[config set] ${msg}`);
67
76
  }
77
+ function errUnset(msg) {
78
+ throw new ConfigSetError(`[config unset] ${msg}`);
79
+ }
80
+ /**
81
+ * Internal, loader-derived flags that are never serialised (stripped by
82
+ * saveConfig). They must never be targets of `config set`/`config unset`.
83
+ */
84
+ const INTERNAL_KEYS = new Set(['default_model_set', 'file_keys']);
85
+ /**
86
+ * Virtual keys whose values live in the OS keychain, not config.toml. `config
87
+ * unset` cannot remove them — that is a keychain operation — so they are
88
+ * rejected with a pointer rather than silently no-op'd.
89
+ */
90
+ const VIRTUAL_KEYCHAIN_KEYS = new Set(['llm.byok_key', 'llm.ai_hub_token']);
91
+ /**
92
+ * MUTABLE_KEYS that `config unset` can act on (excludes keychain virtuals and
93
+ * the deprecated `local_files` alias — the canonical name is `local_build`).
94
+ */
95
+ function unsettableKeyList() {
96
+ return [...MUTABLE_KEYS]
97
+ .filter((k) => !VIRTUAL_KEYCHAIN_KEYS.has(k) && k !== 'local_files')
98
+ .join(', ');
99
+ }
100
+ /**
101
+ * Mark a top-level config key as explicitly user-set so sparse serialisation
102
+ * (saveConfig) preserves it even when its value equals the built-in default.
103
+ * `file_keys` tracks TOP-LEVEL keys, so we push the first dotted segment
104
+ * (`default_model` → `default_model`, `classifier.safe_mode` → `classifier`).
105
+ * Dedup-safe. Closes the one edge the value-diff can't see: `config set
106
+ * default_model claude-opus-4-8` (equal to the built-in default) on a file that
107
+ * lacked the key would otherwise be dropped on save, permanently opting the
108
+ * user out of the model-governance server steering (default_model_set semantics).
109
+ */
110
+ export function markExplicit(cfg, key) {
111
+ const top = key.split('.')[0];
112
+ if (!Array.isArray(cfg.file_keys))
113
+ cfg.file_keys = [];
114
+ if (!cfg.file_keys.includes(top))
115
+ cfg.file_keys.push(top);
116
+ }
117
+ /**
118
+ * Delete a (possibly dotted, max 2-level) key from a loaded config in place.
119
+ * Returns true if the key was present and removed, false if it was already
120
+ * absent (the caller renders a friendly no-op). Also drops the top-level key
121
+ * from `file_keys` — but ONLY when the whole top-level key was removed. If a
122
+ * nested sibling survives (e.g. unsetting `llm.local_model` while `llm.mode`
123
+ * remains), keeping the top-level entry in `file_keys` is what stops sparse
124
+ * save from dropping the now-default-equal parent table.
125
+ */
126
+ function unsetConfigKey(cfg, key) {
127
+ const parts = key.split('.');
128
+ const top = parts[0];
129
+ const obj = cfg;
130
+ let removedTop = false;
131
+ if (parts.length === 1) {
132
+ if (!Object.prototype.hasOwnProperty.call(obj, top))
133
+ return false;
134
+ delete obj[top];
135
+ removedTop = true;
136
+ }
137
+ else {
138
+ const parent = obj[top];
139
+ if (!parent || typeof parent !== 'object')
140
+ return false;
141
+ const p = parent;
142
+ const leaf = parts.slice(1).join('.');
143
+ if (!Object.prototype.hasOwnProperty.call(p, leaf))
144
+ return false;
145
+ delete p[leaf];
146
+ if (Object.keys(p).length === 0) {
147
+ delete obj[top];
148
+ removedTop = true;
149
+ }
150
+ }
151
+ if (removedTop && Array.isArray(cfg.file_keys)) {
152
+ cfg.file_keys = cfg.file_keys.filter((k) => k !== top);
153
+ }
154
+ return true;
155
+ }
156
+ /**
157
+ * Executes `cspeach config unset <key>`.
158
+ * Removes a MUTABLE top-level key so it reverts to the built-in default (and,
159
+ * for the model keys, re-enables server model-steering). Rejects internal keys,
160
+ * keychain virtuals, and non-MUTABLE keys. Unsetting an absent key is a friendly
161
+ * no-op. Throws ConfigSetError on validation failure (the CLI dispatcher in
162
+ * cli.ts converts it to console.error + exit).
163
+ */
164
+ export async function runConfigUnset(args) {
165
+ const key = args[0];
166
+ if (!key) {
167
+ errUnset('Usage: cspeach config unset <key>\n Valid keys: ' + unsettableKeyList());
168
+ }
169
+ if (INTERNAL_KEYS.has(key)) {
170
+ errUnset(`"${key}" is an internal key (derived on load, never stored) and cannot be unset.`);
171
+ }
172
+ if (VIRTUAL_KEYCHAIN_KEYS.has(key)) {
173
+ errUnset(`"${key}" is stored in the OS keychain, not config.toml — it cannot be removed with config unset.`);
174
+ }
175
+ if (key.startsWith('sap.')) {
176
+ errUnset(`"${key}" is a SAP-scoped key — those are managed per system via \`cspeach config set ${key} <value>\`, not config unset.`);
177
+ }
178
+ if (!MUTABLE_KEYS.has(key)) {
179
+ errUnset(`Unknown key "${key}". Valid keys: ${unsettableKeyList()}`);
180
+ }
181
+ const cfg = await loadConfig();
182
+ const removed = unsetConfigKey(cfg, key);
183
+ if (!removed) {
184
+ console.log(chalk.dim(`${key} was not set — nothing to unset.`));
185
+ return;
186
+ }
187
+ await saveConfig(cfg);
188
+ console.log(chalk.green(`Unset ${key} — it now falls back to the built-in default (and any server steering).`));
189
+ }
68
190
  /**
69
191
  * Executes `cspeach config set <key> <value>`.
70
192
  * Exits process with code 1 on validation error.
@@ -94,6 +216,23 @@ export async function runConfigSet(args) {
94
216
  console.log(chalk.green(`Set ${key} = ${value}`));
95
217
  return;
96
218
  }
219
+ // SAP-scoped key: sap.<alias>.role — system role, feeds the write-mode ceiling
220
+ // (prd → advisory-only, qas → approval-gated, dev → no ceiling).
221
+ const roleMatch = SAP_ROLE_RE.exec(key);
222
+ if (roleMatch) {
223
+ const alias = roleMatch[1];
224
+ if (!SYSTEM_ROLE_VALUES.includes(value)) {
225
+ err(`Invalid value for ${key}. Allowed: ${SYSTEM_ROLE_VALUES.join(', ')}`);
226
+ }
227
+ const cfg = await loadConfig();
228
+ if (!cfg.sap[alias]) {
229
+ err(`SAP system "${alias}" not found in config. Add it first with: cspeach config add-sap`);
230
+ }
231
+ cfg.sap[alias].role = value;
232
+ await saveConfig(cfg);
233
+ console.log(chalk.green(`Set ${key} = ${value}`));
234
+ return;
235
+ }
97
236
  // SAP-scoped TLS: sap.<alias>.ca_cert_path — trust a CA, verification STAYS ON.
98
237
  const caMatch = SAP_CA_CERT_PATH_RE.exec(key);
99
238
  if (caMatch) {
@@ -184,6 +323,24 @@ export async function runConfigSet(args) {
184
323
  }
185
324
  cfg.default_model = value;
186
325
  break;
326
+ case 'audit_model':
327
+ // model-governance Step 1 — the per-phase auditor's model. Same claude-
328
+ // prefix guard as default_model; env CSPEACH_AUDIT_MODEL still overrides
329
+ // this at runtime (see resolveAuditModel).
330
+ if (!value.startsWith('claude-')) {
331
+ err(`model must start with "claude-", got "${value}".`);
332
+ }
333
+ cfg.audit_model = value;
334
+ break;
335
+ case 'compact_model':
336
+ // model-governance step 2d — the model `/compact` summarisation runs on.
337
+ // Same claude- prefix guard as default_model/audit_model; env
338
+ // CSPEACH_COMPACT_MODEL still overrides this at runtime (resolveModelRole).
339
+ if (!value.startsWith('claude-')) {
340
+ err(`model must start with "claude-", got "${value}".`);
341
+ }
342
+ cfg.compact_model = value;
343
+ break;
187
344
  case 'effort':
188
345
  if (!EFFORT_VALUES.includes(value)) {
189
346
  err(`Invalid effort value "${value}". Allowed: ${EFFORT_VALUES.join(', ')}`);
@@ -233,6 +390,24 @@ export async function runConfigSet(args) {
233
390
  }
234
391
  cfg.write_mode = value;
235
392
  break;
393
+ case 'plan_mode':
394
+ // 'step' = prompt before every phase (default when unset);
395
+ // 'guarded' = auto-chain non-write phases, stop on write phases /
396
+ // unresolved audits. Per-run --guarded/--step flags override this.
397
+ if (!PLAN_MODE_VALUES.includes(value)) {
398
+ err(`Invalid plan_mode "${value}". Allowed: ${PLAN_MODE_VALUES.join(', ')}`);
399
+ }
400
+ cfg.plan_mode = value;
401
+ break;
402
+ case 'plan_audit':
403
+ // 'on' = audit every phase (default when unset); 'off' = skip the
404
+ // independent auditor entirely (owner off-switch, 2026-07-06). Per-run
405
+ // --audit/--no-audit flags override this default.
406
+ if (!PLAN_AUDIT_VALUES.includes(value)) {
407
+ err(`Invalid plan_audit "${value}". Allowed: ${PLAN_AUDIT_VALUES.join(', ')}`);
408
+ }
409
+ cfg.plan_audit = value;
410
+ break;
236
411
  case 'llm.mode':
237
412
  if (!LLM_MODE_VALUES.includes(value)) {
238
413
  err(`Invalid llm.mode "${value}". Allowed: ${LLM_MODE_VALUES.join(', ')}`);
@@ -270,6 +445,10 @@ export async function runConfigSet(args) {
270
445
  default:
271
446
  err(`Unhandled key "${key}".`);
272
447
  }
448
+ // Mark this top-level key as explicitly user-set so sparse serialisation
449
+ // keeps it even when the value equals the built-in default (the edge the
450
+ // value-diff can't see — e.g. `config set default_model claude-opus-4-8`).
451
+ markExplicit(cfg, key);
273
452
  await saveConfig(cfg);
274
453
  console.log(chalk.green(`Set ${key} = ${value}`));
275
454
  }
@@ -286,6 +465,10 @@ export function validateValue(key, value) {
286
465
  if (SAP_AUTO_APPROVE_RE.test(key)) {
287
466
  return AUTO_APPROVE_VALUES.includes(value);
288
467
  }
468
+ // SAP-scoped key: sap.<alias>.role
469
+ if (SAP_ROLE_RE.test(key)) {
470
+ return SYSTEM_ROLE_VALUES.includes(value);
471
+ }
289
472
  // SAP-scoped TLS keys.
290
473
  if (SAP_CA_CERT_PATH_RE.test(key)) {
291
474
  return value.length > 0; // any non-empty path; existence checked at connect
@@ -305,6 +488,8 @@ export function validateValue(key, value) {
305
488
  return false;
306
489
  }
307
490
  case 'default_model':
491
+ case 'audit_model':
492
+ case 'compact_model':
308
493
  return value.startsWith('claude-');
309
494
  case 'effort':
310
495
  return EFFORT_VALUES.includes(value);
@@ -317,6 +502,10 @@ export function validateValue(key, value) {
317
502
  return coerceBool(value) !== null;
318
503
  case 'write_mode':
319
504
  return isValidWriteMode(value);
505
+ case 'plan_mode':
506
+ return PLAN_MODE_VALUES.includes(value);
507
+ case 'plan_audit':
508
+ return PLAN_AUDIT_VALUES.includes(value);
320
509
  case 'llm.mode':
321
510
  return LLM_MODE_VALUES.includes(value);
322
511
  case 'llm.ai_hub_base_url':
@@ -14,6 +14,16 @@ const REDACTED_KEYS = new Set([
14
14
  'llm.byok_key', // Anthropic API key — keychain "cspeach.anthropic"
15
15
  'llm.ai_hub_token', // SAP AI Hub bearer — keychain "cspeach.ai-hub"
16
16
  ]);
17
+ /**
18
+ * INTERNAL, loader-derived flags that are attached to the config object at load
19
+ * time but never serialised (stripped by saveConfig). They must not render in
20
+ * `config show` — `default_model_set` was a pre-existing leak, `file_keys` was
21
+ * added by sparse-config task 1. Both are top-level only.
22
+ */
23
+ const INTERNAL_KEYS = new Set([
24
+ 'default_model_set',
25
+ 'file_keys',
26
+ ]);
17
27
  function formatTomlValue(value) {
18
28
  if (typeof value === 'string')
19
29
  return `"${value}"`;
@@ -61,6 +71,9 @@ function printConfigAsToml(cfg, prefix = '') {
61
71
  const tables = entries.filter(([, v]) => typeof v === 'object' && v !== null && !Array.isArray(v));
62
72
  for (const [k, v] of scalars) {
63
73
  const fullKey = prefix ? `${prefix}.${k}` : k;
74
+ // Skip internal, never-serialised keys so they never leak into output.
75
+ if (INTERNAL_KEYS.has(fullKey))
76
+ continue;
64
77
  if (REDACTED_KEYS.has(fullKey)) {
65
78
  console.log(`${k} = [REDACTED]`);
66
79
  }
@@ -126,6 +139,13 @@ export async function runConfigShow(args) {
126
139
  return;
127
140
  }
128
141
  const keyPath = args[0];
142
+ // Reject internal, loader-derived keys in the single-key path too — the full
143
+ // dump already filters INTERNAL_KEYS, but resolveDottedKey would happily
144
+ // print e.g. the raw `file_keys` array. Mirror `config unset`'s wording.
145
+ if (INTERNAL_KEYS.has(keyPath)) {
146
+ console.error(chalk.red(`[config show] "${keyPath}" is an internal key (derived on load, never stored) and cannot be shown.`));
147
+ process.exit(1);
148
+ }
129
149
  const resolved = resolveDottedKey(cfg, keyPath);
130
150
  if (resolved === undefined) {
131
151
  console.error(chalk.red(`[config show] Key "${keyPath}" not found.`));
@@ -0,0 +1,43 @@
1
+ // cspeach-cli/src/commands/export-audit.ts
2
+ //
3
+ // `/export audit [filename]` — write the signable session audit record to a
4
+ // markdown file in the cwd and print the path.
5
+ //
6
+ // Thin I/O wrapper over the pure buildAuditReport builder (session/audit-export
7
+ // .ts): reads the session's cost JSONL, resolves the target filename (explicit
8
+ // arg or a deterministic default), writes the file, and emits a confirmation
9
+ // line through the caller-supplied `emit` (chunkEmitter in Ink mode, console
10
+ // .log in classic mode — same seam as printHelp / runCostCommand).
11
+ //
12
+ // The default filename's date comes from SESSION DATA (last_turn_at), never
13
+ // now() — a re-export of an old session names the file for when the work
14
+ // happened, matching the deterministic report body.
15
+ import fs from 'node:fs/promises';
16
+ import path from 'node:path';
17
+ import { buildAuditReport } from '../session/audit-export.js';
18
+ import { readCostLog } from '../cost/cost-log.js';
19
+ /** Parse the text after `/export` into a subcommand + optional filename. */
20
+ export function parseExportArgs(rest) {
21
+ const parts = rest.trim().split(/\s+/).filter(Boolean);
22
+ return { sub: (parts[0] ?? '').toLowerCase(), filename: parts[1] };
23
+ }
24
+ /** Deterministic default filename: date from session data, not now(). */
25
+ export function defaultAuditFilename(session) {
26
+ const date = (session.last_turn_at || session.started_at || '').slice(0, 10) || 'undated';
27
+ return `cspeach-audit-${session.id}-${date}.md`;
28
+ }
29
+ export async function runExportAuditCommand(opts) {
30
+ const { session, transport, emit } = opts;
31
+ const cwd = opts.cwd ?? process.cwd();
32
+ const costEntries = opts.costEntries ?? (await readCostLog(session.id));
33
+ const report = buildAuditReport({ session, costEntries, transport });
34
+ const name = opts.filename && opts.filename.trim().length > 0
35
+ ? opts.filename.trim()
36
+ : defaultAuditFilename(session);
37
+ const outPath = path.resolve(cwd, name);
38
+ await fs.writeFile(outPath, report, 'utf-8');
39
+ emit('');
40
+ emit(`Audit written to ${outPath}`);
41
+ emit('');
42
+ return { path: outPath };
43
+ }
@@ -93,7 +93,12 @@ export function printHelp(emit = (s) => console.log(s)) {
93
93
  emit(` ${chalk.bold('/ui [auto|ink|classic]'.padEnd(NAME_COL_PAD))} View or change UI rendering mode`);
94
94
  emit(` ${chalk.bold('/reroute <skill>'.padEnd(NAME_COL_PAD))} Re-run your last prompt with a different skill`);
95
95
  emit(` ${chalk.bold('/new'.padEnd(NAME_COL_PAD))} End the current Q&A chain — next prompt is classified fresh`);
96
+ emit(` ${chalk.bold('/mode [advisory|gated|auto]'.padEnd(NAME_COL_PAD))} Cycle or set the session write mode (Shift+Tab / Alt+M also cycle)`);
97
+ emit(` ${chalk.bold('/recap'.padEnd(NAME_COL_PAD))} Show the session recap (last session · open tasks · plan progress)`);
98
+ emit(` ${chalk.bold('/rewind'.padEnd(NAME_COL_PAD))} Undo a write from this session — restore an object to a pre-write snapshot (Esc Esc)`);
96
99
  emit(` ${chalk.bold('/cost'.padEnd(NAME_COL_PAD))} Show this session's API spend so far ($ + token breakdown)`);
100
+ emit(` ${chalk.bold('/context'.padEnd(NAME_COL_PAD))} Show what fills the model context window — a composition grid of bars`);
101
+ emit(` ${chalk.bold('/export audit [file]'.padEnd(NAME_COL_PAD))} Write the signable session audit record to a markdown file`);
97
102
  emit(` ${chalk.bold('/compact'.padEnd(NAME_COL_PAD))} Summarise older turns into a compact context block — cuts subsequent turn cost by 80-90%`);
98
103
  emit(` ${chalk.bold('/exit'.padEnd(NAME_COL_PAD))} Quit CSPeach`);
99
104
  emit('');
@@ -0,0 +1,266 @@
1
+ /**
2
+ * Ground-truth extractor for the phase auditor.
3
+ *
4
+ * The agent loop writes a JSONL audit log of every tool call to
5
+ * `~/.cspeach/checkpoints/<sessionId>/tool-calls.jsonl` (Layer 2 default
6
+ * handler in `agent/skill-checkpoint.ts`). The phase auditor must judge a
7
+ * phase's self-report against evidence the audited model did NOT author —
8
+ * this module produces that evidence: a bounded, compact excerpt of the
9
+ * write + verification tool calls (set_source, activate, syntax_check,
10
+ * transport, snapshot, ...), newest-last.
11
+ *
12
+ * Pure read module — no writes, no new dependencies.
13
+ */
14
+ import { promises as fs } from 'node:fs';
15
+ import * as path from 'node:path';
16
+ import { checkpointsRoot } from '../agent/skill-checkpoint.js';
17
+ /**
18
+ * PROTECTED evidence — write + verification calls that PROVE a phase's core
19
+ * claims and must NEVER be evicted by read rows (see the two-tier cap in
20
+ * extractAuditEvidence). Transport tools are matched by their EXACT
21
+ * write-relevant names (fix-wave): the registered family is
22
+ * sap_transport_create / _for_object / _list / _release (tools/transport.ts),
23
+ * and a bare `sap_transport` prefix let list-shaped read polling
24
+ * (sap_transport_list, sap_transport_for_object) evict real write rows from
25
+ * the 60-row cap.
26
+ *
27
+ * D-A (2026-07-05): sap_atc_run and sap_inactive_objects are VERIFICATION
28
+ * tools the audit contract + phase exit gates explicitly demand (ATC clean /
29
+ * no inactive versions remain), yet they were missing from this filter — so
30
+ * a phase that legitimately ran ATC or checked for inactive objects had that
31
+ * evidence stripped, and the auditor false-failed it with "no ATC call
32
+ * appears anywhere in the evidence". They are reads in the sense of not
33
+ * mutating source, but they ARE the proof of the verify steps, so they
34
+ * belong in the evidence excerpt. (Unlike the transport LIST polling, these
35
+ * fire once per verify, not in a loop, so they do not threaten the row cap.)
36
+ *
37
+ * Task 2 (audit-confidence-tiered-redesign, §6): request_approval is admitted
38
+ * here too. It is the PROOF-of-consent the §6 escalation exception keys on — a
39
+ * write to an infra base-object (DEVC / transport / number-range) under a
40
+ * writes:false phase is a sanctioned escalation ONLY IF the evidence carries an
41
+ * OK request_approval row for it. With the old sap_*-only filter that row was
42
+ * dropped, so the exception could never be satisfied. Like the existence-read
43
+ * proofs it is ground truth (the user's recorded consent), and the whole
44
+ * contract branch depends on it — so it must NEVER be evicted by the row cap:
45
+ * PROTECTED, not the fillable existence tier. Matched EXACTLY (^…$ anchored) so
46
+ * a near-miss tool name can never masquerade as consent.
47
+ */
48
+ const PROTECTED_TOOL_RE = /^sap_(set_source|update_method|create_object|delete_object|activate|syntax_check|atc_run|inactive_objects|transport_create|transport_release|service_binding_publish|snapshot)|^request_approval$/;
49
+ /**
50
+ * D-F (2026-07-05): EXISTENCE / VERIFY-READ evidence. A verify-only phase
51
+ * re-run (object already built in a prior attempt; this attempt proves it
52
+ * exists / is active / is clean and writes NOTHING) records only reads:
53
+ * sap_object_structure (version=active), sap_get_source (fields match), and
54
+ * sap_transport_for_object (object locked in a transport). With the old
55
+ * write-only filter those rows were stripped, the evidence went empty, and
56
+ * the auditor false-FAILED the re-run ("no write call in evidence"). These
57
+ * reads are legitimate ground truth: a read against the live harness log
58
+ * cannot be fabricated — a read of a nonexistent object ERRORS, so an `ok`
59
+ * existence read IS the system agreeing the object is there. They are
60
+ * admitted here so the auditor can satisfy a pre-existence claim on
61
+ * verification evidence alone (contract Rule 1, D-F branch).
62
+ *
63
+ * Cap safety (two-tier, see extractAuditEvidence): these reads are also used
64
+ * heavily in analysis phases and could, in a single-bucket cap, evict real
65
+ * write rows. They are therefore the FILLABLE tier — the newest of them fill
66
+ * whatever budget remains after every PROTECTED row is kept; a write row is
67
+ * never dropped to make room for one of these.
68
+ */
69
+ const EXISTENCE_READ_RE = /^sap_(object_structure|get_source|transport_for_object)/;
70
+ /** Any row the auditor may see is the union of the two tiers. */
71
+ const EVIDENCE_TOOL_RE = new RegExp(`${PROTECTED_TOOL_RE.source}|${EXISTENCE_READ_RE.source}`);
72
+ const NO_LOG_SENTINEL = '(no tool-call log found for this session)';
73
+ const NO_EVIDENCE_ROWS = '(tool-call log present, but no write/verify calls recorded for this session)';
74
+ const DEFAULT_MAX_ROWS = 60;
75
+ const RESULT_EXCERPT_CHARS = 200;
76
+ const ARGS_COMPACT_CHARS = 80;
77
+ /**
78
+ * audit-timeout (2026-07-06) — per-tool result caps.
79
+ *
80
+ * Live false-FAIL #4: a c1.behavior audit failed with "both sap_inactive_objects
81
+ * results are truncated (only ZI_FRG3_DC and ZI_FRG_V2DC visible, followed by
82
+ * ellipsis)". The default 200-char cap beheaded the inactive-objects LIST,
83
+ * destroying the absence-from-inactive-list proof the D-F activeness arm depends
84
+ * on — and the auditor was CORRECTLY refusing to certify activeness from clipped
85
+ * evidence. The fix: render the inactive-objects list generously (it is names
86
+ * only, bounded in practice) so absence stays judgeable; only if it is
87
+ * pathologically huge do we clip it AND mark the clip HONESTLY, so the contract
88
+ * can fall back to the object_structure "version": "active" proof rather than
89
+ * inferring absence from a beheaded list. ATC results get a modest bump because
90
+ * the auditor reads finding priorities/counts from them. Everything else keeps
91
+ * the 200-char cap (object_structure surfaces "version" early, so 200 is fine).
92
+ */
93
+ const INACTIVE_LIST_CAP = 4000;
94
+ const ATC_RESULT_CAP = 500;
95
+ /** Appended when the inactive-objects list itself is clipped — see the contract. */
96
+ const INACTIVE_CLIP_MARKER = ' … (list clipped — absence not verifiable from this row; use object_structure version)';
97
+ function resultCapFor(toolName) {
98
+ if (/^sap_inactive_objects/.test(toolName))
99
+ return INACTIVE_LIST_CAP;
100
+ if (/^sap_atc_run/.test(toolName))
101
+ return ATC_RESULT_CAP;
102
+ return RESULT_EXCERPT_CHARS;
103
+ }
104
+ function normalizeRow(raw) {
105
+ if (raw === null || typeof raw !== 'object')
106
+ return null;
107
+ const r = raw;
108
+ const toolName = typeof r.toolName === 'string' ? r.toolName
109
+ : typeof r.tool === 'string' ? r.tool
110
+ : null;
111
+ if (!toolName)
112
+ return null;
113
+ const isError = typeof r.isError === 'boolean' ? r.isError
114
+ : typeof r.ok === 'boolean' ? !r.ok
115
+ : false;
116
+ const at = typeof r.completedAt === 'string' ? r.completedAt
117
+ : typeof r.at === 'string' ? r.at
118
+ : null;
119
+ // PROTECTED is checked first: it is a strict-enough family that no
120
+ // existence-read name can collide with it, but ordering makes the tiering
121
+ // explicit for the reader.
122
+ const kind = PROTECTED_TOOL_RE.test(toolName) ? 'protected' : 'existence';
123
+ return { toolName, args: r.args, result: r.result, isError, at, kind };
124
+ }
125
+ /** Render the args as `name/type` when available, else a compact fallback. */
126
+ function renderArgs(args) {
127
+ if (args === null || args === undefined || typeof args !== 'object')
128
+ return '';
129
+ const a = args;
130
+ const name = a.name ?? a.object_name;
131
+ const type = a.type ?? a.object_type;
132
+ if (typeof name === 'string' && name.length > 0) {
133
+ return typeof type === 'string' && type.length > 0 ? `${name}/${type}` : name;
134
+ }
135
+ // No name/type — render what's available, compactly.
136
+ try {
137
+ const json = JSON.stringify(a);
138
+ if (!json || json === '{}')
139
+ return '';
140
+ return json.length <= ARGS_COMPACT_CHARS ? json : json.slice(0, ARGS_COMPACT_CHARS) + '…';
141
+ }
142
+ catch {
143
+ return '';
144
+ }
145
+ }
146
+ /**
147
+ * Result excerpt, flattened to a single line, capped per-tool (see
148
+ * resultCapFor). For sap_inactive_objects, a clip is marked honestly so a
149
+ * beheaded list can never masquerade as a complete absence proof.
150
+ */
151
+ function renderResult(result, toolName) {
152
+ let text;
153
+ if (typeof result === 'string') {
154
+ text = result;
155
+ }
156
+ else if (result === null || result === undefined) {
157
+ text = '';
158
+ }
159
+ else {
160
+ try {
161
+ text = JSON.stringify(result) ?? '';
162
+ }
163
+ catch {
164
+ text = String(result);
165
+ }
166
+ }
167
+ const flat = text.replace(/\s+/g, ' ').trim();
168
+ const cap = resultCapFor(toolName);
169
+ if (flat.length <= cap)
170
+ return flat;
171
+ if (/^sap_inactive_objects/.test(toolName)) {
172
+ // Absence-from-inactive-list is an activeness PROOF; a bare ellipsis would
173
+ // silently behead the list and make absence unverifiable. Mark it honestly.
174
+ return flat.slice(0, cap) + INACTIVE_CLIP_MARKER;
175
+ }
176
+ return flat.slice(0, cap);
177
+ }
178
+ function renderRow(row) {
179
+ const status = row.isError ? 'ERROR' : 'ok';
180
+ return `${row.toolName}(${renderArgs(row.args)}) → ${status} — ${renderResult(row.result, row.toolName)}`;
181
+ }
182
+ /**
183
+ * Extract a bounded, auditor-ready excerpt of the session's write +
184
+ * verification tool calls from the JSONL audit log.
185
+ *
186
+ * - Only rows whose toolName matches {@link EVIDENCE_TOOL_RE} are included.
187
+ * - D-B: when `sinceIso` is given, rows whose timestamp is strictly BEFORE it
188
+ * are excluded — so an audit judging attempt N of a phase never sees the
189
+ * rows of an earlier attempt that shares the same session JSONL (two
190
+ * attempts in one CLI process). Untimestamped rows are kept (fail-open).
191
+ * - D-F two-tier cap: chronological order is preserved (newest-last). When
192
+ * the evidence exceeds `maxRows` (default 60), the OLDEST existence-read
193
+ * rows are dropped first; a PROTECTED (write / verify-proof) row is dropped
194
+ * only if no existence-read rows remain to drop. So a write row is never
195
+ * evicted to make room for a read — the exact regression a single bucket +
196
+ * the new existence reads would have introduced. The 60 cap is kept (not
197
+ * raised): the tiering, not a bigger budget, is what guarantees write
198
+ * survival, and a fixed cap keeps the audit prompt's token cost bounded.
199
+ * - Missing file/dir returns the sentinel — never throws.
200
+ * - Malformed JSONL lines are skipped silently.
201
+ */
202
+ export async function extractAuditEvidence(sessionId, opts) {
203
+ const maxRows = opts?.maxRows ?? DEFAULT_MAX_ROWS;
204
+ const sinceIso = opts?.sinceIso;
205
+ const filePath = path.join(checkpointsRoot(), sessionId, 'tool-calls.jsonl');
206
+ let content;
207
+ try {
208
+ content = await fs.readFile(filePath, 'utf-8');
209
+ }
210
+ catch {
211
+ return NO_LOG_SENTINEL;
212
+ }
213
+ const rows = [];
214
+ for (const line of content.split('\n')) {
215
+ const trimmed = line.trim();
216
+ if (!trimmed)
217
+ continue;
218
+ let parsed;
219
+ try {
220
+ parsed = JSON.parse(trimmed);
221
+ }
222
+ catch {
223
+ continue; // malformed line — skip silently
224
+ }
225
+ const row = normalizeRow(parsed);
226
+ if (!row || !EVIDENCE_TOOL_RE.test(row.toolName))
227
+ continue;
228
+ // D-B attempt-window: drop rows recorded before this attempt started.
229
+ // ISO-8601 UTC strings compare correctly lexicographically. A row with no
230
+ // timestamp is never excluded (fail-open).
231
+ if (sinceIso && row.at !== null && row.at < sinceIso)
232
+ continue;
233
+ rows.push(row);
234
+ }
235
+ if (rows.length === 0)
236
+ return NO_EVIDENCE_ROWS;
237
+ return capRows(rows, maxRows).map(renderRow).join('\n');
238
+ }
239
+ /**
240
+ * D-F two-tier cap. Returns at most `maxRows` rows in chronological order.
241
+ * Eviction order when over budget: oldest EXISTENCE reads first, then (only
242
+ * if still over and no reads remain) oldest PROTECTED rows. Writes are thus
243
+ * never evicted by reads.
244
+ */
245
+ function capRows(rows, maxRows) {
246
+ if (rows.length <= maxRows)
247
+ return rows;
248
+ let over = rows.length - maxRows;
249
+ const dropped = new Set();
250
+ // Pass 1 — drop oldest existence reads.
251
+ for (let i = 0; i < rows.length && over > 0; i++) {
252
+ if (rows[i].kind === 'existence') {
253
+ dropped.add(i);
254
+ over--;
255
+ }
256
+ }
257
+ // Pass 2 — reads exhausted, still over: drop oldest protected rows (writes
258
+ // evicting older writes is the original 60-cap behavior, not a read winning).
259
+ for (let i = 0; i < rows.length && over > 0; i++) {
260
+ if (!dropped.has(i)) {
261
+ dropped.add(i);
262
+ over--;
263
+ }
264
+ }
265
+ return rows.filter((_, i) => !dropped.has(i));
266
+ }