@cspeach/cli 1.1.19 → 1.1.20

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/dist/agent/anthropic-provider.js +30 -10
  2. package/dist/agent/cache-keepalive.js +162 -0
  3. package/dist/agent/cold-prune.js +116 -0
  4. package/dist/agent/loop.js +675 -148
  5. package/dist/agent/provider-shape.js +263 -0
  6. package/dist/agent/providers/ai-hub-provider.js +17 -2
  7. package/dist/agent/providers/byok-provider.js +33 -3
  8. package/dist/agent/providers/local-provider.js +8 -1
  9. package/dist/agent/repair-partial.js +66 -5
  10. package/dist/agent/summarise-via-provider.js +6 -1
  11. package/dist/agent/system-prompt.js +38 -0
  12. package/dist/agent/tool-dispatch.js +8 -0
  13. package/dist/agent/tool-loading-pin.js +100 -0
  14. package/dist/cli.js +11 -0
  15. package/dist/commands/auto-compact.js +33 -16
  16. package/dist/commands/compact.js +37 -2
  17. package/dist/commands/config-set.js +10 -1
  18. package/dist/commands/config-show.js +11 -0
  19. package/dist/commands/cost.js +14 -2
  20. package/dist/commands/plan-audit.js +1 -0
  21. package/dist/config/loader.js +55 -2
  22. package/dist/cost/cost-log.js +62 -2
  23. package/dist/cost/pricing.js +6 -2
  24. package/dist/lib/spill-labels.js +13 -0
  25. package/dist/models/resolve.js +93 -2
  26. package/dist/models/server-config.js +158 -3
  27. package/dist/one-shot.js +15 -5
  28. package/dist/projects/image-attachments.js +15 -2
  29. package/dist/renderer/footer-line.js +6 -2
  30. package/dist/renderer/startup-lines.js +5 -3
  31. package/dist/renderer/tool-labels.js +33 -2
  32. package/dist/renderer/ui-width.js +13 -0
  33. package/dist/repl/current-transport.js +13 -0
  34. package/dist/repl/post-turn-status.js +8 -1
  35. package/dist/repl.js +71 -10
  36. package/dist/session/repin-model.js +18 -0
  37. package/dist/session/store.js +16 -2
  38. package/dist/skills/bundled-skills.js +1 -1
  39. package/dist/skills/preamble.js +75 -0
  40. package/dist/skills/source-manifest.js +11 -1
  41. package/dist/tools/filesystem/file-read.js +11 -1
  42. package/dist/tools/result-spill.js +238 -0
  43. package/dist/tools/sap-read.js +58 -14
  44. package/dist/tools/shell/shell_exec.js +9 -0
  45. package/dist/tools/subagent/adt-serial.js +33 -0
  46. package/dist/tools/subagent/agent_run.js +2 -0
  47. package/dist/tools/subagent/read_agent.js +178 -0
  48. package/dist/tools/subagent/reader-prompt.js +48 -0
  49. package/dist/tools/todo.js +3 -1
  50. package/dist/tools/tool-loading.js +255 -0
  51. package/dist/tools/tool-output-read.js +117 -0
  52. package/dist/tools/transport.js +6 -1
  53. package/dist/ui/footer.js +5 -5
  54. package/dist/ui/sap-state-store.js +1 -1
  55. package/dist/ui/turn-status-emitter.js +37 -0
  56. package/dist/ui/turn-status.js +1 -1
  57. package/package.json +2 -1
package/dist/cli.js CHANGED
@@ -119,6 +119,17 @@ export async function main(argv) {
119
119
  }
120
120
  if (args[0] === 'config' && args[1] === 'show') {
121
121
  const { runConfigShow } = await import('./commands/config-show.js');
122
+ if (args.length === 2) {
123
+ // Task 12 — the full dump ends with "# model set by your account admin"
124
+ // when /v1/me/model-config says the admin's model is enforced. Capped at
125
+ // 1.5 s and fail-open: offline / signed out ⇒ the output is unchanged.
126
+ const [{ loadConfig }, { getStore }, { loadCustomerModelConfig }] = await Promise.all([
127
+ import('./config/loader.js'),
128
+ import('./auth/api-key.js'),
129
+ import('./models/server-config.js'),
130
+ ]);
131
+ await loadCustomerModelConfig(await loadConfig(), async () => (await (await getStore()).get()) ?? '', { timeoutMs: 1500 }).catch(() => { });
132
+ }
122
133
  await runConfigShow(args.slice(2));
123
134
  return 0;
124
135
  }
@@ -1,9 +1,11 @@
1
1
  /**
2
2
  * Auto-compact orchestration — Phase 2 #9 (2026-05-16).
3
3
  *
4
- * Calls `runCompact` end-of-turn when the session has crossed
5
- * `compact.auto_threshold_tokens` non-cached input tokens AND enough turns
6
- * have passed since the last auto-compact run.
4
+ * Calls `runCompact` end-of-turn when the last model call's full context
5
+ * (input + cache read + cache write — Task 19, ruling F12) has reached
6
+ * `compact.auto_threshold_tokens` AND enough turns have passed since the last
7
+ * auto-compact run. The caller passes the RESOLVED threshold (file > served
8
+ * `auto_compact_tokens` > 150 000; see models/resolve.ts).
7
9
  *
8
10
  * Why end-of-turn (not start-of-turn): firing at the end means the cost
9
11
  * (one Haiku call to summarise) is amortised over the user's natural
@@ -21,7 +23,11 @@
21
23
  */
22
24
  import chalk from 'chalk';
23
25
  import { saveSession } from '../session/store.js';
24
- import { runCompact, formatCompactResult, } from './compact.js';
26
+ import { runCompact, planCompaction, canCompact } from './compact.js';
27
+ /** The number the gate compares: the context size, else the legacy counter. */
28
+ function gateTokens(input) {
29
+ return input.lastContextTokens ?? input.sessionInputTokens ?? 0;
30
+ }
25
31
  /**
26
32
  * Pure decision function — no IO. Decides whether auto-compact should
27
33
  * fire given the current session state and config. Extracted so the
@@ -29,11 +35,11 @@ import { runCompact, formatCompactResult, } from './compact.js';
29
35
  * Summariser.
30
36
  */
31
37
  export function shouldAutoCompact(input) {
32
- const { sessionInputTokens, turnNumber, lastAutoCompactTurn, config } = input;
38
+ const { turnNumber, lastAutoCompactTurn, config } = input;
33
39
  if (!config.auto_enabled) {
34
40
  return { fire: false, reason: 'disabled' };
35
41
  }
36
- if (sessionInputTokens < config.auto_threshold_tokens) {
42
+ if (gateTokens(input) < config.auto_threshold_tokens) {
37
43
  return { fire: false, reason: 'below-threshold' };
38
44
  }
39
45
  // Throttle: don't bounce. If we just ran auto-compact a few turns ago
@@ -43,7 +49,10 @@ export function shouldAutoCompact(input) {
43
49
  const turnsSince = lastAutoCompactTurn === undefined
44
50
  ? Number.POSITIVE_INFINITY
45
51
  : turnNumber - lastAutoCompactTurn;
46
- if (turnsSince < config.min_turns_between_auto) {
52
+ // Final review I4 — a marker ahead of the turn number (a legacy marker from
53
+ // the old user-message count, which shrinks after a compaction) never
54
+ // throttles; the loop now passes a monotonic per-session counter.
55
+ if (turnsSince >= 0 && turnsSince < config.min_turns_between_auto) {
47
56
  return { fire: false, reason: 'throttled' };
48
57
  }
49
58
  return { fire: true };
@@ -58,19 +67,28 @@ export function shouldAutoCompact(input) {
58
67
  export async function maybeAutoCompact(params) {
59
68
  const { session, config, turnNumber, summarise } = params;
60
69
  const emit = params.emit ?? ((line) => console.log(line));
61
- const decision = shouldAutoCompact({
70
+ const gate = {
71
+ lastContextTokens: session.last_context_tokens,
62
72
  sessionInputTokens: session.usage?.input_tokens ?? 0,
73
+ };
74
+ const decision = shouldAutoCompact({
75
+ ...gate,
63
76
  turnNumber,
64
77
  lastAutoCompactTurn: session.lastAutoCompactTurn,
65
78
  config,
66
79
  });
67
80
  if (!decision.fire)
68
81
  return;
69
- // Banner BEFORE the work so the user understands the pause they're about
70
- // to see. ~3s for a Haiku call on a long history.
71
- emit('');
72
- emit(chalk.yellow(`⚙ auto-compact: session crossed ${config.auto_threshold_tokens.toLocaleString()} input tokens`
73
- + ` — summarising older turns to keep the next turn cheap.`));
82
+ // Task 19 — at 150k a session can cross the gate with too few user turns
83
+ // for runCompact to cut anything (it would return "nothing to do"). Stay
84
+ // silent then, instead of printing the line on every turn (canCompact is
85
+ // the predicate runCompact's own bail-outs use).
86
+ if (!canCompact(planCompaction(session.messages, config.keep_recent_turns), config.keep_recent_turns))
87
+ return;
88
+ // Task 19 — ONE short line, printed when the work is done so it can carry
89
+ // the backup path (the one-step rollback). A failure prints its own line.
90
+ const head = `⚙ auto-compact: context ${gateTokens(gate).toLocaleString('en-US')} tokens ≥ `
91
+ + `${config.auto_threshold_tokens.toLocaleString('en-US')}`;
74
92
  try {
75
93
  const result = await runCompact({
76
94
  session,
@@ -82,13 +100,12 @@ export async function maybeAutoCompact(params) {
82
100
  // runCompact already saved the session inside its happy path, but we
83
101
  // mutate one more field (the throttle marker) so re-save to capture it.
84
102
  await saveSession(session);
103
+ emit(chalk.yellow(`${head} — older turns summarised · backup: ${result.backupPath}`));
85
104
  }
86
- emit(formatCompactResult(result, config.keep_recent_turns));
87
105
  }
88
106
  catch (err) {
89
107
  // Defensive: never let a summariser blow take down the turn.
90
108
  const msg = err instanceof Error ? err.message : String(err);
91
- emit(chalk.yellow(`⚙ auto-compact bailed: ${msg}`));
92
- emit(chalk.dim(' session left intact; /compact still available manually.'));
109
+ emit(chalk.yellow(`⚙ auto-compact bailed: ${msg} — session left intact; /compact still works.`));
93
110
  }
94
111
  }
@@ -216,6 +216,32 @@ export function buildSummaryMessage(summary, droppedTurns, todos) {
216
216
  ].join('\n'),
217
217
  };
218
218
  }
219
+ const THINKING_BLOCK_TYPES = new Set(['thinking', 'redacted_thinking']);
220
+ /** Text left in an assistant message that held only thinking (an interrupted turn): the API rejects empty content. */
221
+ export const THINKING_ONLY_PLACEHOLDER = '(thinking removed by /compact)';
222
+ /**
223
+ * Task 7 — drop `thinking` / `redacted_thinking` blocks from an assistant
224
+ * message. User messages and string content are returned unchanged (the same
225
+ * object); every other block is kept as-is and in order. Never mutates the
226
+ * input. Idempotent.
227
+ */
228
+ export function stripThinking(m) {
229
+ if (m.role !== 'assistant' || !Array.isArray(m.content))
230
+ return m;
231
+ const blocks = m.content;
232
+ if (!blocks.some((b) => THINKING_BLOCK_TYPES.has(b?.type ?? '')))
233
+ return m;
234
+ const rest = blocks.filter((b) => !THINKING_BLOCK_TYPES.has(b?.type ?? ''));
235
+ return { ...m, content: rest.length > 0 ? rest : [{ type: 'text', text: THINKING_ONLY_PLACEHOLDER }] };
236
+ }
237
+ /**
238
+ * Task 19 — true when `plan` leaves enough older history for runCompact to
239
+ * summarise. The one predicate runCompact's bail-outs and the auto-compact
240
+ * silent check share.
241
+ */
242
+ export function canCompact(plan, keepRecentTurns) {
243
+ return plan.userPromptCount >= keepRecentTurns && plan.toSummarise.length >= MIN_COMPACT_THRESHOLD;
244
+ }
219
245
  /**
220
246
  * Top-level compact orchestration. Pure async function — caller wires it
221
247
  * into the REPL slash dispatcher. Throws only on catastrophic IO failure
@@ -229,7 +255,8 @@ export async function runCompact(opts) {
229
255
  const { session, summarise } = opts;
230
256
  const keep = opts.keepRecentTurns ?? KEEP_RECENT_TURNS;
231
257
  const plan = planCompaction(session.messages, keep);
232
- // Bail if there's not enough older history to be worth compacting.
258
+ // Bail if there's not enough older history to be worth compacting
259
+ // (the two checks canCompact combines; kept apart for their messages).
233
260
  if (plan.userPromptCount < keep) {
234
261
  return {
235
262
  performed: false,
@@ -262,7 +289,15 @@ export async function runCompact(opts) {
262
289
  // we accept losing the original first prompt because the summary should
263
290
  // mention what the goal was.)
264
291
  const summaryMsg = buildSummaryMessage(summary, plan.toSummarise.length, session.todos);
265
- session.messages = [summaryMsg, ...plan.toKeep];
292
+ // Task 7 — the API's conversation check rejects a replayed thinking block
293
+ // whose earlier turns changed (Opus 5.5 binds each block to the exact prefix
294
+ // that produced it). The summary changed them, so the kept tail goes out
295
+ // without thinking; text and tool calls stay. The backup above keeps all.
296
+ const kept = plan.toKeep.map(stripThinking);
297
+ session.messages = [summaryMsg, ...kept];
298
+ // Task 19 — the recorded context size described the old history; the next
299
+ // call records the new one. Clearing it keeps the auto-compact gate honest.
300
+ delete session.last_context_tokens;
266
301
  await saveSession(session);
267
302
  return {
268
303
  performed: true,
@@ -96,7 +96,7 @@ function errUnset(msg) {
96
96
  * Internal, loader-derived flags that are never serialised (stripped by
97
97
  * saveConfig). They must never be targets of `config set`/`config unset`.
98
98
  */
99
- const INTERNAL_KEYS = new Set(['default_model_set', 'file_keys']);
99
+ const INTERNAL_KEYS = new Set(['default_model_set', 'effort_set', 'file_keys']);
100
100
  /**
101
101
  * Virtual keys whose values live in the OS keychain, not config.toml. `config
102
102
  * unset` cannot remove them — that is a keychain operation — so they are
@@ -266,6 +266,10 @@ export async function runConfigUnset(args) {
266
266
  }
267
267
  const cfg = await loadConfig();
268
268
  const removed = unsetConfigKey(cfg, key);
269
+ // Task 4 — `effort` and its explicit marker go together: removing the value
270
+ // without the marker would leave a stale marker for a later bare effort.
271
+ if (key === 'effort')
272
+ unsetConfigKey(cfg, 'effort_explicit');
269
273
  if (!removed) {
270
274
  console.log(chalk.dim(`${key} was not set — nothing to unset.`));
271
275
  return;
@@ -459,6 +463,11 @@ export async function runConfigSet(args) {
459
463
  err(`Invalid effort value "${value}". Allowed: ${EFFORT_VALUES.join(', ')}`);
460
464
  }
461
465
  cfg.effort = value;
466
+ // Task 4 (ruling F2) — the marker makes this an explicit act: only then
467
+ // does resolveEffort send output_config.effort. Kept on save even when
468
+ // the value equals the built-in default (xhigh).
469
+ cfg.effort_explicit = true;
470
+ markExplicit(cfg, 'effort_explicit');
462
471
  break;
463
472
  case 'telemetry':
464
473
  if (!TELEMETRY_VALUES.includes(value)) {
@@ -2,6 +2,7 @@ import chalk from 'chalk';
2
2
  import { loadConfig } from '../config/loader.js';
3
3
  import { paint } from '../renderer/theme.js';
4
4
  import { formatAttention } from '../renderer/notice-shape.js';
5
+ import { getCustomerModelConfig } from '../models/server-config.js';
5
6
  /**
6
7
  * Dotted paths of keys that must never appear in plaintext config output.
7
8
  * SAP passwords live in the keychain and should never land in config.toml,
@@ -25,6 +26,10 @@ const REDACTED_KEYS = new Set([
25
26
  const INTERNAL_KEYS = new Set([
26
27
  'default_model_set',
27
28
  'file_keys',
29
+ // Task 4 — effort_set is loader-derived; effort_explicit is the stored
30
+ // marker behind it (plumbing, not a user-facing setting).
31
+ 'effort_set',
32
+ 'effort_explicit',
28
33
  ]);
29
34
  function formatTomlValue(value) {
30
35
  if (typeof value === 'string')
@@ -138,6 +143,12 @@ export async function runConfigShow(args) {
138
143
  console.log(chalk.gray('# ~/.cspeach/config.toml'));
139
144
  printConfigAsToml(cfg);
140
145
  console.log('');
146
+ // Task 12 — one trailing line when the account admin's model is enforced
147
+ // by the proxy (/v1/me/model-config). No snapshot ⇒ output unchanged.
148
+ const customer = getCustomerModelConfig();
149
+ if (customer?.model && customer.enforced) {
150
+ console.log(chalk.gray(`# model set by your account admin: ${customer.model} (enforced)`));
151
+ }
141
152
  return;
142
153
  }
143
154
  const keyPath = args[0];
@@ -25,6 +25,9 @@ function summariseCredits(entries) {
25
25
  let total = 0;
26
26
  let unknown = 0;
27
27
  for (const e of entries) {
28
+ // Final review I6 — a zero-token reader rollup line is not a turn.
29
+ if (e.kind === 'reader_rollup')
30
+ continue;
28
31
  if (typeof e.credits_charged === 'number')
29
32
  total += e.credits_charged;
30
33
  else
@@ -73,6 +76,12 @@ export async function runCostCommand(sessionId, credits = globalCredits.snapshot
73
76
  emit(panelRow('output', s.totalTokens.output.toLocaleString().padStart(12)));
74
77
  emit(panelRow('cache read', s.totalTokens.cacheRead.toLocaleString().padStart(12)));
75
78
  emit(panelRow('cache create', s.totalTokens.cacheCreate.toLocaleString().padStart(12)));
79
+ // Task 18 fix M3 — reader subagent spend (already inside the total above).
80
+ // USD only: in credits mode the balance delta already carries it.
81
+ if (s.readerCalls > 0 && !inCredits) {
82
+ emit('');
83
+ emit(panelRow('Readers', `${s.readerCalls} run${s.readerCalls === 1 ? '' : 's'} ${formatCost(s.readerCost)}`));
84
+ }
76
85
  if (s.byModel.length > 1) {
77
86
  emit('');
78
87
  emit(chalk.bold('By model:'));
@@ -88,9 +97,12 @@ export async function runCostCommand(sessionId, credits = globalCredits.snapshot
88
97
  emit(panelRow('Model', modelText(s.byModel[0].model)));
89
98
  }
90
99
  // Recent turns — last 5, most recent first. Piece 2: one count format, "N of M" (P1's countLabel).
91
- const recent = entries.slice(-5).reverse();
100
+ // Final review I6 — reader rollup lines (interrupted turns) are not turns:
101
+ // they stay out of the list and the count (summarise() already skips them).
102
+ const turnEntries = entries.filter((e) => e.kind !== 'reader_rollup');
103
+ const recent = turnEntries.slice(-5).reverse();
92
104
  emit('');
93
- emit(chalk.bold(`Recent turns · ${countLabel(recent.length, entries.length)}`));
105
+ emit(chalk.bold(`Recent turns · ${countLabel(recent.length, turnEntries.length)}`));
94
106
  for (const e of recent) {
95
107
  const tok = (e.tokens.input + e.tokens.output).toLocaleString();
96
108
  // Credits mode: the per-turn figure is what the LEDGER charged (a balance
@@ -635,6 +635,7 @@ export async function runPhaseAudit(args) {
635
635
  // read-only filter (which still exposes file_read/grep/glob).
636
636
  toolFilter: () => false,
637
637
  modelOverride: auditModel,
638
+ modelRole: 'audit',
638
639
  suppressSaveHook: true,
639
640
  signal: controller.signal,
640
641
  });
@@ -12,6 +12,11 @@ import { sanitiseStandaloneProfile } from '../sap/standalone-profile.js';
12
12
  export function isValidWriteMode(v) {
13
13
  return v === 'auto' || v === 'approval-gated' || v === 'advisory-only';
14
14
  }
15
+ export const DEFAULT_KEEPALIVE = Object.freeze({
16
+ enabled: true,
17
+ in_turn_pings: 11,
18
+ idle_pings: 4,
19
+ });
15
20
  export const DEFAULT_CONFIG = {
16
21
  proxy_url: 'https://api.cspeach.dev',
17
22
  default_model: DEFAULT_SESSION_MODEL,
@@ -46,9 +51,10 @@ export const DEFAULT_CONFIG = {
46
51
  compact: {
47
52
  keep_recent_turns: 5,
48
53
  auto_enabled: true,
49
- auto_threshold_tokens: 1_000_000,
54
+ auto_threshold_tokens: 150_000,
50
55
  min_turns_between_auto: 5,
51
56
  },
57
+ keepalive: DEFAULT_KEEPALIVE,
52
58
  };
53
59
  /**
54
60
  * Coerce a possibly-bad `compact` block from disk into a safe CompactConfig.
@@ -78,6 +84,43 @@ function sanitiseCompact(raw) {
78
84
  min_turns_between_auto: intInRange(r.min_turns_between_auto, d.min_turns_between_auto, 1, 100),
79
85
  };
80
86
  }
87
+ /** The old DEFAULT the pre-sparse full save wrote into every config file. */
88
+ const LEGACY_SAVED_AUTO_THRESHOLD = 1_000_000;
89
+ /**
90
+ * Task 19 (ruling F12) — does the FILE's `[compact]` table set the threshold by
91
+ * an explicit act? A numeric value counts, except 1_000_000: that is the old
92
+ * default a full save wrote, so it reads as unset (same reasoning as effort,
93
+ * F2). The stored value itself is left as-is (so a later save keeps writing
94
+ * 1_000_000, which stays "unset", instead of baking in a value that would beat
95
+ * the served `auto_compact_tokens`).
96
+ */
97
+ function fileSetsAutoThreshold(raw) {
98
+ if (!raw || typeof raw !== 'object')
99
+ return false;
100
+ const v = raw.auto_threshold_tokens;
101
+ return typeof v === 'number' && Number.isFinite(v) && Math.floor(v) !== LEGACY_SAVED_AUTO_THRESHOLD;
102
+ }
103
+ /**
104
+ * Coerce a possibly-bad `keepalive` block from disk (same defensive pattern as
105
+ * sanitiseCompact): missing / wrong type → defaults; counts floored and
106
+ * clamped (in_turn_pings 0..20, idle_pings 0..10).
107
+ */
108
+ function sanitiseKeepalive(raw) {
109
+ const d = DEFAULT_KEEPALIVE;
110
+ if (!raw || typeof raw !== 'object')
111
+ return { ...d };
112
+ const r = raw;
113
+ const intInRange = (v, fallback, lo, hi) => {
114
+ if (typeof v !== 'number' || !Number.isFinite(v))
115
+ return fallback;
116
+ return Math.min(hi, Math.max(lo, Math.floor(v)));
117
+ };
118
+ return {
119
+ enabled: typeof r.enabled === 'boolean' ? r.enabled : d.enabled,
120
+ in_turn_pings: intInRange(r.in_turn_pings, d.in_turn_pings, 0, 20),
121
+ idle_pings: intInRange(r.idle_pings, d.idle_pings, 0, 10),
122
+ };
123
+ }
81
124
  /**
82
125
  * Resolve the effective TLS options for a SAP system, applying the legacy
83
126
  * `sslVerify: false` → insecure mapping. A configured `ca_cert_path` always
@@ -195,6 +238,13 @@ export async function loadConfig() {
195
238
  // `default_model` key (file presence, not the value). Lets an explicit user
196
239
  // choice beat a served roles.session_default. Stripped by saveConfig.
197
240
  default_model_set: Object.prototype.hasOwnProperty.call(parsed, 'default_model'),
241
+ // effort_set: INTERNAL flag (ruling F2) — true only when the FILE carries
242
+ // a string `effort` AND the `effort_explicit = true` marker that
243
+ // `config set effort` writes. Stripped by saveConfig.
244
+ effort_set: parsed.effort_explicit === true && typeof parsed.effort === 'string',
245
+ // compact_threshold_set: INTERNAL flag (Task 19) — only present when true,
246
+ // so a loaded config without it keeps its exact old shape.
247
+ ...(fileSetsAutoThreshold(parsed.compact) ? { compact_threshold_set: true } : {}),
198
248
  // sap: ALWAYS a fresh top-level object, never the shared DEFAULT_CONFIG.sap
199
249
  // reference. Without this, a file lacking `[sap.*]` would leave
200
250
  // `merged.sap === DEFAULT_CONFIG.sap`; the onboarding wizard then mutating
@@ -216,6 +266,7 @@ export async function loadConfig() {
216
266
  // field (e.g. just `keep_recent_turns = 10`) still gets the four
217
267
  // unspecified defaults — and a typo'd field falls through harmlessly.
218
268
  compact: sanitiseCompact(parsed.compact),
269
+ keepalive: sanitiseKeepalive(parsed.keepalive),
219
270
  shell_exec: parsed.shell_exec === undefined ? undefined : { allow },
220
271
  };
221
272
  // Drop the legacy `local_files` key entirely when it isn't a genuine on
@@ -369,7 +420,9 @@ export async function saveConfig(config) {
369
420
  const fileKeys = config.file_keys ?? [];
370
421
  // Strip the INTERNAL flags — both are derived on load (file presence /
371
422
  // file-key snapshot) and must never be written back into config.toml.
372
- const { default_model_set: _dropSet, file_keys: _dropKeys, ...toWrite } = config;
423
+ // effort_set is internal too (Task 4); the stored marker effort_explicit is kept.
424
+ // compact_threshold_set is internal too (Task 19).
425
+ const { default_model_set: _dropSet, effort_set: _dropEffortSet, compact_threshold_set: _dropThresholdSet, file_keys: _dropKeys, ...toWrite } = config;
373
426
  // Sparse serialisation: drop any key that the file did NOT carry AND whose
374
427
  // value is deep-equal to the built-in default. Keys absent from DEFAULT_CONFIG
375
428
  // (optional keys like audit_model) never deep-equal a defined value, so any
@@ -15,6 +15,29 @@ import fs from 'node:fs/promises';
15
15
  import path from 'node:path';
16
16
  import { sessionsDir } from '../config/paths.js';
17
17
  import { computeCost } from './pricing.js';
18
+ /**
19
+ * Task 7 — summarise `message_start.message.input_transformations` (probe P4
20
+ * shape: `[{ type: 'thinking_dropped', path, reason }]`). Counts the
21
+ * `thinking_dropped` entries and lists each distinct reason once ('unknown'
22
+ * when an entry has none). Returns undefined for absent, empty or malformed
23
+ * input, and when nothing was dropped — so the cost-line key stays absent.
24
+ */
25
+ export function summariseInputTransformations(raw) {
26
+ if (!Array.isArray(raw))
27
+ return undefined;
28
+ let dropped = 0;
29
+ const reasons = [];
30
+ for (const t of raw) {
31
+ if (!t || typeof t !== 'object' || t.type !== 'thinking_dropped')
32
+ continue;
33
+ dropped += 1;
34
+ const r = t.reason;
35
+ const reason = typeof r === 'string' && r ? r : 'unknown';
36
+ if (!reasons.includes(reason))
37
+ reasons.push(reason);
38
+ }
39
+ return dropped > 0 ? { dropped, reasons } : undefined;
40
+ }
18
41
  /**
19
42
  * Path to a session's cost log file. Co-located with the session JSON so
20
43
  * deletion cleans up both atomically.
@@ -39,6 +62,20 @@ export function buildEntry(opts) {
39
62
  entry.credits_charged = opts.credits.charged;
40
63
  entry.balance_credits = opts.credits.balance;
41
64
  }
65
+ if (opts.served_model !== undefined)
66
+ entry.served_model = opts.served_model;
67
+ if (opts.call_index !== undefined)
68
+ entry.call_index = opts.call_index;
69
+ if (opts.kind !== undefined)
70
+ entry.kind = opts.kind;
71
+ if (opts.refusal_category !== undefined)
72
+ entry.refusal_category = opts.refusal_category;
73
+ if (opts.input_transformations !== undefined)
74
+ entry.input_transformations = opts.input_transformations;
75
+ if (opts.reader && opts.reader.calls > 0) {
76
+ entry.reader_cost_usd = Math.round(opts.reader.cost * 10000) / 10000;
77
+ entry.reader_calls = opts.reader.calls;
78
+ }
42
79
  return entry;
43
80
  }
44
81
  /**
@@ -90,7 +127,22 @@ export function summarise(entries) {
90
127
  const totals = { input: 0, output: 0, cacheRead: 0, cacheCreate: 0 };
91
128
  let total = 0;
92
129
  const perModel = new Map();
130
+ const readerByTurn = new Map();
131
+ let lines = 0;
93
132
  for (const e of entries) {
133
+ if (typeof e.reader_calls === 'number' && e.reader_calls > 0) {
134
+ const prev = readerByTurn.get(e.turn);
135
+ if (!prev || e.reader_calls > prev.calls || (e.reader_cost_usd ?? 0) > prev.cost) {
136
+ readerByTurn.set(e.turn, {
137
+ cost: Math.max(prev?.cost ?? 0, e.reader_cost_usd ?? 0),
138
+ calls: Math.max(prev?.calls ?? 0, e.reader_calls),
139
+ });
140
+ }
141
+ }
142
+ // Fix M3 — a zero-token reader rollup line (interrupted turn) is not a model call.
143
+ if (e.kind === 'reader_rollup')
144
+ continue;
145
+ lines += 1;
94
146
  total += e.cost;
95
147
  totals.input += e.tokens.input;
96
148
  totals.output += e.tokens.output;
@@ -109,10 +161,18 @@ export function summarise(entries) {
109
161
  slot.tokens.cacheCreate += e.tokens.cacheCreate;
110
162
  perModel.set(e.model, slot);
111
163
  }
164
+ let readerCost = 0;
165
+ let readerCalls = 0;
166
+ for (const r of readerByTurn.values()) {
167
+ readerCost += r.cost;
168
+ readerCalls += r.calls;
169
+ }
112
170
  return {
113
- totalCost: Math.round(total * 10000) / 10000,
171
+ totalCost: Math.round((total + readerCost) * 10000) / 10000,
114
172
  totalTokens: totals,
115
- turns: entries.length,
173
+ turns: lines,
116
174
  byModel: [...perModel.entries()].map(([model, v]) => ({ model, ...v })),
175
+ readerCost: Math.round(readerCost * 10000) / 10000,
176
+ readerCalls,
117
177
  };
118
178
  }
@@ -11,7 +11,9 @@
11
11
  * BOTH this file AND the proxy's version. Verify on the public pricing page:
12
12
  * https://www.anthropic.com/pricing
13
13
  *
14
- * Last verified 2026-06-07. Same as proxy.
14
+ * Last verified 2026-09-27. Same as proxy. The claude-opus-5-5 row comes from
15
+ * the Opus 5.5 section of the model migration guide. The two tables are pinned
16
+ * in sync by src/cost/__tests__/pricing-drift-pin.test.ts.
15
17
  *
16
18
  * 2026-06-07 SYNC: e294cff switched the default model to claude-opus-4-8 AND
17
19
  * applied the 3x Opus correction, but only to the proxy copy — this CLI copy
@@ -25,9 +27,11 @@
25
27
  import { recordFullNotice } from '../renderer/notice-log.js';
26
28
  import { getServerModelConfig } from '../models/server-config.js';
27
29
  const PRICING = {
30
+ // Claude Opus 5.5 (2026-09-27, migration guide) — $4/$20, cache read $0.20
31
+ // (0.05x input, not 10%), 5-minute cache write $5 (1.25x input).
32
+ 'claude-opus-5-5': { inputPer1M: 4.00, outputPer1M: 20.00, cacheReadPer1M: 0.20, cacheCreatePer1M: 5.00 },
28
33
  // Claude 5 family (2026-09-15) — Opus 5 keeps the Opus 4.5+ rate ($5/$25);
29
34
  // Sonnet 5 is $2/$10. Cache read = 10% of input, cache write = 1.25x input.
30
- 'claude-opus-5-5': { inputPer1M: 4.00, outputPer1M: 20.00, cacheReadPer1M: 0.20, cacheCreatePer1M: 5.00 },
31
35
  'claude-opus-5': { inputPer1M: 5.00, outputPer1M: 25.00, cacheReadPer1M: 0.50, cacheCreatePer1M: 6.25 },
32
36
  'claude-sonnet-5': { inputPer1M: 2.00, outputPer1M: 10.00, cacheReadPer1M: 0.20, cacheCreatePer1M: 2.50 },
33
37
  // Claude Opus 4.5+ — input 5, output 25, cache_read 0.50, cache_write 6.25 (1.25x input).
@@ -0,0 +1,13 @@
1
+ // What a saved (spilled) tool output came from, so the tool_output_read row
2
+ // names the object ("ZTTT_GAME") instead of the internal spill file path.
3
+ // In-memory for this process only: after a resume the row reads "saved output".
4
+ const labels = new Map();
5
+ /** File name only, so '/' and '\' spellings of the same path match. */
6
+ const keyOf = (p) => p.split(/[\\/]/).pop() ?? p;
7
+ export function rememberSpillLabel(filePath, label) {
8
+ if (label.trim().length > 0)
9
+ labels.set(keyOf(filePath), label.trim());
10
+ }
11
+ export function spillLabelFor(filePath) {
12
+ return labels.get(keyOf(filePath));
13
+ }
@@ -8,6 +8,13 @@
8
8
  //
9
9
  // env > local config > server > built-in constant
10
10
  //
11
+ // Task 12 (2026-09-27) adds the CUSTOMER layer (GET /v1/me/model-config, the
12
+ // admin-set per-customer override) for session_default and effort, between
13
+ // local and server: env > local (set) > customer > server > built-in.
14
+ // For managed (enforced) customers this is only a prediction — the proxy
15
+ // enforces and the loop adopts the served model; for BYOK it is a default the
16
+ // developer's local config overrides.
17
+ //
11
18
  // Locked invariant: with NOTHING set (no env, default_model_set false, no server
12
19
  // snapshot) every role resolves to today's constant — byte-identical to before
13
20
  // this feature.
@@ -16,10 +23,11 @@
16
23
  // read is getServerModelConfig(), a synchronous in-memory snapshot the caller
17
24
  // (or the startup fetch) has already populated. `router` is proxy-side only and
18
25
  // is deliberately NOT a CLI role.
26
+ import { DEFAULT_CONFIG } from '../config/loader.js';
19
27
  import { DEFAULT_SESSION_MODEL } from '../config/model-defaults.js';
20
28
  import { PLAN_TIER_SONNET_MODEL } from '../commands/plan-model-tier.js';
21
29
  import { COMPACTION_MODEL } from '../commands/compact.js';
22
- import { getServerModelConfig } from './server-config.js';
30
+ import { getCustomerModelConfig, getServerModelConfig, isEffort } from './server-config.js';
23
31
  /** First non-empty (trimmed) candidate in precedence order, else the built-in. */
24
32
  function firstSet(candidates, builtin) {
25
33
  for (const c of candidates) {
@@ -31,7 +39,8 @@ function firstSet(candidates, builtin) {
31
39
  }
32
40
  /**
33
41
  * Resolve the model for a pipeline role. Precedence: env > local config >
34
- * server > built-in constant.
42
+ * server > built-in constant; `session_default` also reads the customer layer
43
+ * (Task 12) between local and server.
35
44
  *
36
45
  * `session_default` reads the local `default_model` ONLY when `default_model_set`
37
46
  * is true — i.e. the user's config file explicitly carried the key. The value
@@ -46,6 +55,7 @@ export function resolveModelRole(role, cfg, env = process.env) {
46
55
  return firstSet([
47
56
  env.CSPEACH_DEFAULT_MODEL,
48
57
  cfg.default_model_set ? cfg.default_model : undefined,
58
+ getCustomerModelConfig()?.model ?? undefined,
49
59
  server?.roles.session_default,
50
60
  ], DEFAULT_SESSION_MODEL);
51
61
  case 'audit':
@@ -59,3 +69,84 @@ export function resolveModelRole(role, cfg, env = process.env) {
59
69
  }
60
70
  }
61
71
  }
72
+ /**
73
+ * Integration finding N1 (2026-09-28) — the model a call is SENT with. For a
74
+ * managed session-role call of a customer whose model is ENFORCED
75
+ * (/v1/me/model-config `enforced: true`), the proxy serves the customer's
76
+ * model (or, when the row names none, the global session default) whatever
77
+ * the body asks. The CLI sends that model itself, so everything decided from
78
+ * the call's model — deferred vs static tools, max_tokens, block_binding,
79
+ * effort acceptance — is right on call 0, not only after the served id is
80
+ * adopted. Mirrors the proxy's shouldEnforce: session role only (audit,
81
+ * compact and reader are never enforced), managed only (BYOK has no proxy to
82
+ * enforce). Anything else ⇒ `requested` unchanged.
83
+ */
84
+ export function callModelFor(requested, i) {
85
+ if (i.providerMode !== 'managed')
86
+ return requested;
87
+ if ((i.role ?? 'session_default') !== 'session_default')
88
+ return requested;
89
+ const customer = getCustomerModelConfig();
90
+ if (customer?.enforced !== true)
91
+ return requested;
92
+ const enforced = customer.model?.trim() || getServerModelConfig()?.roles.session_default?.trim();
93
+ return enforced || requested;
94
+ }
95
+ // ── Effort (Task 4, 2026-09-27) ─────────────────────────────────────────────
96
+ //
97
+ // Ruling F2: `output_config.effort` is sent only when effort was set by an
98
+ // explicit act. Precedence:
99
+ //
100
+ // env CSPEACH_EFFORT > `config set effort` (effort_set) > served
101
+ // effort.session_default > undefined (⇒ output_config omitted)
102
+ //
103
+ // A bare stored `effort` (the pre-sparse full save wrote effort = "xhigh" into
104
+ // every config.toml) is ignored: loadConfig sets effort_set only when the file
105
+ // also carries the `effort_explicit = true` marker `config set effort` writes.
106
+ // Invalid values at any layer fall through. Task 12 adds the customer layer
107
+ // (/v1/me/model-config) between local and served.
108
+ export function resolveEffort(cfg, env = process.env) {
109
+ const fromEnv = env.CSPEACH_EFFORT?.trim();
110
+ if (isEffort(fromEnv))
111
+ return fromEnv;
112
+ if (cfg.effort_set === true && isEffort(cfg.effort))
113
+ return cfg.effort;
114
+ const customer = getCustomerModelConfig()?.effort;
115
+ if (isEffort(customer))
116
+ return customer;
117
+ const served = getServerModelConfig()?.effort?.session_default;
118
+ if (isEffort(served))
119
+ return served;
120
+ return undefined;
121
+ }
122
+ /**
123
+ * Probe P1 (2026-09-27): claude-haiku-4-5 rejects `output_config.effort` with
124
+ * a 400 ("This model does not support the effort parameter."). Every
125
+ * claude-haiku-4-5* id (incl. dated ids such as claude-haiku-4-5-20251001) is
126
+ * treated as not accepting effort, so the caller drops it — this covers the
127
+ * compact role, which runs on Haiku by default.
128
+ */
129
+ export function modelAcceptsEffort(model) {
130
+ return !model.trim().toLowerCase().startsWith('claude-haiku-4-5');
131
+ }
132
+ /**
133
+ * Task 19 (model control, 2026-09-27, ruling F12) — the effective auto-compact
134
+ * threshold, in context tokens (the last call's input + cache read + cache
135
+ * write):
136
+ *
137
+ * explicit file `[compact] auto_threshold_tokens`
138
+ * > served `auto_compact_tokens` (admin; managed mode)
139
+ * > DEFAULT_CONFIG.compact.auto_threshold_tokens (150 000)
140
+ *
141
+ * "Explicit" is `cfg.compact_threshold_set` (loadConfig): a stored 1 000 000
142
+ * is the old full-save default and counts as unset.
143
+ */
144
+ export function resolveAutoCompactThreshold(cfg) {
145
+ if (cfg.compact_threshold_set === true && Number.isFinite(cfg.compact?.auto_threshold_tokens)) {
146
+ return cfg.compact.auto_threshold_tokens;
147
+ }
148
+ const served = getServerModelConfig()?.auto_compact_tokens;
149
+ if (typeof served === 'number' && Number.isFinite(served))
150
+ return served;
151
+ return DEFAULT_CONFIG.compact.auto_threshold_tokens;
152
+ }