@cspeach/cli 0.6.5 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/intent-system-prompt.js +46 -0
- package/dist/agent/loop.js +180 -18
- package/dist/agent/parallel-write-guard.js +71 -0
- package/dist/agent/repair-partial.js +88 -1
- package/dist/agent/retry-cap.js +121 -0
- package/dist/agent/summarise-via-provider.js +51 -0
- package/dist/approvals/jwt.js +33 -12
- package/dist/commands/auto-compact.js +94 -0
- package/dist/commands/compact.js +265 -0
- package/dist/commands/cost.js +56 -0
- package/dist/commands/help.js +21 -14
- package/dist/config/loader.js +39 -0
- package/dist/cost/cost-log.js +113 -0
- package/dist/cost/pricing.js +91 -0
- package/dist/one-shot.js +1 -1
- package/dist/renderer/markdown.js +7 -1
- package/dist/renderer/status-footer.js +37 -14
- package/dist/renderer/thinking-heartbeat.js +15 -3
- package/dist/renderer/tool-widget.js +6 -1
- package/dist/renderer/tty.js +27 -0
- package/dist/repl/cspeach-shell-detect.js +33 -0
- package/dist/repl/post-turn-status.js +68 -0
- package/dist/repl/slash-completer.js +2 -0
- package/dist/repl/slash-picker.js +2 -0
- package/dist/repl.js +371 -46
- package/dist/sap-errors/parse-adt-exception.js +149 -0
- package/dist/session/schema.js +2 -1
- package/dist/session/store.js +19 -0
- package/dist/tools/_filesystem-shared.js +9 -1
- package/dist/tools/ask-question.js +19 -0
- package/dist/tools/sap-read.js +56 -4
- package/dist/tools/sap-write.js +19 -0
- package/dist/tools/subagent/schedule_draft_create.js +137 -0
- package/dist/ui/alt-screen.js +120 -0
- package/dist/ui/app.js +39 -15
- package/dist/ui/ask-question-emitter.js +13 -0
- package/dist/ui/body.js +39 -23
- package/dist/ui/coaching-picker-classic.js +8 -1
- package/dist/ui/footer.js +6 -1
- package/dist/ui/header.js +19 -4
- package/dist/ui/sap-state-store.js +13 -1
- package/dist/ui/sidebar.js +5 -3
- package/dist/ui/status-emitter.js +19 -0
- package/dist/ui/status-row.js +30 -4
- package/dist/ui/widgets/ask-question-modal.js +132 -0
- package/dist/ui/widgets/coaching-picker.js +55 -8
- package/package.json +1 -1
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* One-shot non-streaming text summarisation via an LLMProvider.
|
|
3
|
+
*
|
|
4
|
+
* The agent loop's createStream is designed for full agentic turns with
|
|
5
|
+
* tools, thinking, and multi-round interaction. /compact only needs the
|
|
6
|
+
* simplest case: send a system + user prompt to a small model, collect the
|
|
7
|
+
* text response, return it.
|
|
8
|
+
*
|
|
9
|
+
* This helper wraps provider.createStream into that simple shape so the
|
|
10
|
+
* compact command (and any future cheap-model utility) doesn't need to
|
|
11
|
+
* recreate the streaming event-loop boilerplate.
|
|
12
|
+
*
|
|
13
|
+
* NOT for tool-using calls. NOT for thinking-enabled calls. Pure text-in /
|
|
14
|
+
* text-out.
|
|
15
|
+
*/
|
|
16
|
+
/**
|
|
17
|
+
* Run a single non-streaming summarisation call.
|
|
18
|
+
*
|
|
19
|
+
* `system` is the instruction (what kind of summary you want).
|
|
20
|
+
* `user` is the content to summarise.
|
|
21
|
+
*
|
|
22
|
+
* Returns the accumulated text from the model's response. Throws on
|
|
23
|
+
* provider error or empty output.
|
|
24
|
+
*/
|
|
25
|
+
export async function summariseViaProvider(provider, system, user, opts) {
|
|
26
|
+
const stream = await provider.createStream({
|
|
27
|
+
model: opts.model,
|
|
28
|
+
max_tokens: opts.maxTokens ?? 4000,
|
|
29
|
+
system,
|
|
30
|
+
messages: [{ role: 'user', content: user }],
|
|
31
|
+
}, {
|
|
32
|
+
headers: opts.headers,
|
|
33
|
+
signal: opts.signal,
|
|
34
|
+
});
|
|
35
|
+
let accumulated = '';
|
|
36
|
+
for await (const event of stream) {
|
|
37
|
+
const type = event?.type;
|
|
38
|
+
if (type === 'content_block_delta') {
|
|
39
|
+
const delta = event.delta;
|
|
40
|
+
if (delta?.type === 'text_delta' && typeof delta.text === 'string') {
|
|
41
|
+
accumulated += delta.text;
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
// Ignore everything else — message_start/stop, usage events, etc.
|
|
45
|
+
// For one-shot text summarisation only content_block_delta matters.
|
|
46
|
+
}
|
|
47
|
+
if (!accumulated || accumulated.trim().length === 0) {
|
|
48
|
+
throw new Error('Summariser returned empty output.');
|
|
49
|
+
}
|
|
50
|
+
return accumulated;
|
|
51
|
+
}
|
package/dist/approvals/jwt.js
CHANGED
|
@@ -12,22 +12,43 @@ export async function mintApprovalId(payload) {
|
|
|
12
12
|
.sign(sessionKey);
|
|
13
13
|
}
|
|
14
14
|
export async function verifyAndSpendApprovalId(jwt, expectedObject, expectedOp) {
|
|
15
|
+
// 2026-05-15 (bug 7): the previous catch lumped EVERY thrown failure
|
|
16
|
+
// into `'invalid_signature'`, including expirations (which used to be
|
|
17
|
+
// sniffed via a fragile substring match on err.message). During the
|
|
18
|
+
// live session this surfaced `op_mismatch` errors as `invalid_signature`
|
|
19
|
+
// — extremely disorienting for whoever was debugging since it implied
|
|
20
|
+
// cryptographic corruption when the JWT was actually fine.
|
|
21
|
+
//
|
|
22
|
+
// The expiry path now uses jose's documented error code
|
|
23
|
+
// (`ERR_JWT_EXPIRED`); the signature path is reserved for the actual
|
|
24
|
+
// signature-verification failure code. Mismatch reasons (nonce, object,
|
|
25
|
+
// op) fall out of the post-verify checks below and are returned
|
|
26
|
+
// verbatim — no catch involved.
|
|
27
|
+
let payload;
|
|
15
28
|
try {
|
|
16
|
-
const
|
|
17
|
-
|
|
18
|
-
if (usedNonces.has(p.nonce))
|
|
19
|
-
return { ok: false, reason: 'nonce_spent' };
|
|
20
|
-
if (p.object !== expectedObject)
|
|
21
|
-
return { ok: false, reason: 'object_mismatch' };
|
|
22
|
-
if (p.op !== expectedOp)
|
|
23
|
-
return { ok: false, reason: 'op_mismatch' };
|
|
24
|
-
usedNonces.add(p.nonce);
|
|
25
|
-
return { ok: true, payload: p };
|
|
29
|
+
const verified = await jwtVerify(jwt, sessionKey);
|
|
30
|
+
payload = verified.payload;
|
|
26
31
|
}
|
|
27
32
|
catch (err) {
|
|
28
|
-
|
|
33
|
+
const code = typeof err?.code === 'string'
|
|
34
|
+
? err.code
|
|
35
|
+
: '';
|
|
36
|
+
if (code === 'ERR_JWT_EXPIRED')
|
|
29
37
|
return { ok: false, reason: 'expired' };
|
|
30
|
-
|
|
38
|
+
// Everything else from jose at this point is a structural / signature
|
|
39
|
+
// problem with the token itself (ERR_JWS_SIGNATURE_VERIFICATION_FAILED,
|
|
40
|
+
// ERR_JWS_INVALID, ERR_JWT_INVALID, …). All five surface to the user
|
|
41
|
+
// as the same actionable: "this token is not trustworthy — get a new
|
|
42
|
+
// one". Mapping them all to invalid_signature keeps the user-facing
|
|
43
|
+
// taxonomy short. Mismatch / spent-nonce paths NEVER reach this catch.
|
|
31
44
|
return { ok: false, reason: 'invalid_signature' };
|
|
32
45
|
}
|
|
46
|
+
if (usedNonces.has(payload.nonce))
|
|
47
|
+
return { ok: false, reason: 'nonce_spent' };
|
|
48
|
+
if (payload.object !== expectedObject)
|
|
49
|
+
return { ok: false, reason: 'object_mismatch' };
|
|
50
|
+
if (payload.op !== expectedOp)
|
|
51
|
+
return { ok: false, reason: 'op_mismatch' };
|
|
52
|
+
usedNonces.add(payload.nonce);
|
|
53
|
+
return { ok: true, payload };
|
|
33
54
|
}
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Auto-compact orchestration — Phase 2 #9 (2026-05-16).
|
|
3
|
+
*
|
|
4
|
+
* Calls `runCompact` end-of-turn when the session has crossed
|
|
5
|
+
* `compact.auto_threshold_tokens` non-cached input tokens AND enough turns
|
|
6
|
+
* have passed since the last auto-compact run.
|
|
7
|
+
*
|
|
8
|
+
* Why end-of-turn (not start-of-turn): firing at the end means the cost
|
|
9
|
+
* (one Haiku call to summarise) is amortised over the user's natural
|
|
10
|
+
* "think before typing" pause. Firing at the start would delay the user's
|
|
11
|
+
* next response by ~3s. Same eventual savings; better felt latency.
|
|
12
|
+
*
|
|
13
|
+
* Why a separate file: the gating logic (threshold + throttle + opt-out)
|
|
14
|
+
* has enough branches to deserve isolated unit tests, and dispatch from
|
|
15
|
+
* `loop.ts` should stay one short call site.
|
|
16
|
+
*
|
|
17
|
+
* Safety: if the summariser throws, we LOG and CONTINUE — auto-compact
|
|
18
|
+
* failure must not take down the turn that just completed successfully.
|
|
19
|
+
* Worst case: the user sees a yellow note that auto-compact bailed, and
|
|
20
|
+
* the next /compact-eligible threshold check fires again.
|
|
21
|
+
*/
|
|
22
|
+
import chalk from 'chalk';
|
|
23
|
+
import { saveSession } from '../session/store.js';
|
|
24
|
+
import { runCompact, formatCompactResult, } from './compact.js';
|
|
25
|
+
/**
|
|
26
|
+
* Pure decision function — no IO. Decides whether auto-compact should
|
|
27
|
+
* fire given the current session state and config. Extracted so the
|
|
28
|
+
* branching logic can be unit-tested without standing up a fake
|
|
29
|
+
* Summariser.
|
|
30
|
+
*/
|
|
31
|
+
export function shouldAutoCompact(input) {
|
|
32
|
+
const { sessionInputTokens, turnNumber, lastAutoCompactTurn, config } = input;
|
|
33
|
+
if (!config.auto_enabled) {
|
|
34
|
+
return { fire: false, reason: 'disabled' };
|
|
35
|
+
}
|
|
36
|
+
if (sessionInputTokens < config.auto_threshold_tokens) {
|
|
37
|
+
return { fire: false, reason: 'below-threshold' };
|
|
38
|
+
}
|
|
39
|
+
// Throttle: don't bounce. If we just ran auto-compact a few turns ago
|
|
40
|
+
// and the counter happens to be over again (it should fall sharply
|
|
41
|
+
// after a successful compact, but tool-heavy turns can still grow
|
|
42
|
+
// it), wait `min_turns_between_auto` before firing again.
|
|
43
|
+
const turnsSince = lastAutoCompactTurn === undefined
|
|
44
|
+
? Number.POSITIVE_INFINITY
|
|
45
|
+
: turnNumber - lastAutoCompactTurn;
|
|
46
|
+
if (turnsSince < config.min_turns_between_auto) {
|
|
47
|
+
return { fire: false, reason: 'throttled' };
|
|
48
|
+
}
|
|
49
|
+
return { fire: true };
|
|
50
|
+
}
|
|
51
|
+
/**
|
|
52
|
+
* End-of-turn hook called by runTurn. Reads the gate, runs runCompact if
|
|
53
|
+
* the gate passes, surfaces a banner + the standard /compact result.
|
|
54
|
+
*
|
|
55
|
+
* Always swallows non-fatal errors — auto-compact must not regress the
|
|
56
|
+
* turn that just succeeded.
|
|
57
|
+
*/
|
|
58
|
+
export async function maybeAutoCompact(params) {
|
|
59
|
+
const { session, config, turnNumber, summarise } = params;
|
|
60
|
+
const emit = params.emit ?? ((line) => console.log(line));
|
|
61
|
+
const decision = shouldAutoCompact({
|
|
62
|
+
sessionInputTokens: session.usage?.input_tokens ?? 0,
|
|
63
|
+
turnNumber,
|
|
64
|
+
lastAutoCompactTurn: session.lastAutoCompactTurn,
|
|
65
|
+
config,
|
|
66
|
+
});
|
|
67
|
+
if (!decision.fire)
|
|
68
|
+
return;
|
|
69
|
+
// Banner BEFORE the work so the user understands the pause they're about
|
|
70
|
+
// to see. ~3s for a Haiku call on a long history.
|
|
71
|
+
emit('');
|
|
72
|
+
emit(chalk.yellow(`⚙ auto-compact: session crossed ${config.auto_threshold_tokens.toLocaleString()} input tokens`
|
|
73
|
+
+ ` — summarising older turns to keep the next turn cheap.`));
|
|
74
|
+
try {
|
|
75
|
+
const result = await runCompact({
|
|
76
|
+
session,
|
|
77
|
+
summarise,
|
|
78
|
+
keepRecentTurns: config.keep_recent_turns,
|
|
79
|
+
});
|
|
80
|
+
if (result.performed) {
|
|
81
|
+
session.lastAutoCompactTurn = turnNumber;
|
|
82
|
+
// runCompact already saved the session inside its happy path, but we
|
|
83
|
+
// mutate one more field (the throttle marker) so re-save to capture it.
|
|
84
|
+
await saveSession(session);
|
|
85
|
+
}
|
|
86
|
+
emit(formatCompactResult(result, config.keep_recent_turns));
|
|
87
|
+
}
|
|
88
|
+
catch (err) {
|
|
89
|
+
// Defensive: never let a summariser blow take down the turn.
|
|
90
|
+
const msg = err instanceof Error ? err.message : String(err);
|
|
91
|
+
emit(chalk.yellow(`⚙ auto-compact bailed: ${msg}`));
|
|
92
|
+
emit(chalk.dim(' session left intact; /compact still available manually.'));
|
|
93
|
+
}
|
|
94
|
+
}
|
|
@@ -0,0 +1,265 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `/compact` slash command — Phase 2 conversation hygiene (2026-05-16).
|
|
3
|
+
*
|
|
4
|
+
* Compacts older turns of the active session into a single Haiku-generated
|
|
5
|
+
* summary block so subsequent turns ingest ~5k tokens of context instead of
|
|
6
|
+
* 500k+. Built in response to live evidence from session 0618a42d where
|
|
7
|
+
* simple Q&A turns burned $3-8 each because of accumulated message history.
|
|
8
|
+
*
|
|
9
|
+
* Cost model: one Haiku call (typically ~0.01-0.05 USD) to save 80-90% on
|
|
10
|
+
* every subsequent turn until the session grows large again. Pays for
|
|
11
|
+
* itself after one follow-up turn.
|
|
12
|
+
*
|
|
13
|
+
* Quality boundary (set 2026-05-16 by Laeeq): summarisation is a mechanical
|
|
14
|
+
* read-and-condense task — Haiku 4.5 does this well. ABAP planning + code
|
|
15
|
+
* gen stay on Opus; only this one operation drops to Haiku.
|
|
16
|
+
*
|
|
17
|
+
* Safety:
|
|
18
|
+
* - Before mutating session.messages, the prior shape is backed up to
|
|
19
|
+
* `~/.cspeach/sessions/<id>.pre-compact-<timestamp>.json` so a rollback
|
|
20
|
+
* is one file-rename away.
|
|
21
|
+
* - The recent N user-prompt-anchored turns are kept verbatim — we never
|
|
22
|
+
* summarise the most recent context.
|
|
23
|
+
* - Tool-use / tool-result pairing is preserved by anchoring the cut on
|
|
24
|
+
* user-prompt boundaries (where content is a string, not a tool_result
|
|
25
|
+
* array).
|
|
26
|
+
*/
|
|
27
|
+
import fs from 'node:fs/promises';
|
|
28
|
+
import path from 'node:path';
|
|
29
|
+
import chalk from 'chalk';
|
|
30
|
+
import { sessionsDir } from '../config/paths.js';
|
|
31
|
+
import { saveSession } from '../session/store.js';
|
|
32
|
+
/** Keep this many recent user-prompt anchored turns verbatim. */
|
|
33
|
+
export const KEEP_RECENT_TURNS = 5;
|
|
34
|
+
/** Don't bother compacting if there's only a handful of older turns. */
|
|
35
|
+
export const MIN_COMPACT_THRESHOLD = 3;
|
|
36
|
+
/** Model used for summarisation — see quality boundary in the module doc. */
|
|
37
|
+
export const COMPACTION_MODEL = 'claude-haiku-4-5';
|
|
38
|
+
/**
|
|
39
|
+
* Decide which message indices to summarise vs keep verbatim.
|
|
40
|
+
*
|
|
41
|
+
* Strategy:
|
|
42
|
+
* - Walk messages backwards looking for user prompts (content is a string,
|
|
43
|
+
* not a tool_result array — that's the signal that a NEW user turn began)
|
|
44
|
+
* - Mark the index of the Nth-most-recent user prompt as the "keep from
|
|
45
|
+
* here on" boundary
|
|
46
|
+
* - Everything before that boundary becomes the summarisation slice
|
|
47
|
+
*
|
|
48
|
+
* Returned `cutAt` is an exclusive end index for the summarisation slice —
|
|
49
|
+
* i.e. messages[0..cutAt-1] get summarised, messages[cutAt..] stay.
|
|
50
|
+
*
|
|
51
|
+
* If the session has fewer than KEEP_RECENT_TURNS user prompts, `cutAt` is
|
|
52
|
+
* 0 (nothing to compact). The caller bails in that case.
|
|
53
|
+
*/
|
|
54
|
+
export function planCompaction(messages,
|
|
55
|
+
/** Optional override — falls back to KEEP_RECENT_TURNS constant for back-compat. */
|
|
56
|
+
keepRecentTurns = KEEP_RECENT_TURNS) {
|
|
57
|
+
let userPromptsSeen = 0;
|
|
58
|
+
let cutAt = 0;
|
|
59
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
60
|
+
const msg = messages[i];
|
|
61
|
+
// A user PROMPT (not a tool_result block) — content is a string.
|
|
62
|
+
if (msg.role === 'user' && typeof msg.content === 'string') {
|
|
63
|
+
userPromptsSeen++;
|
|
64
|
+
if (userPromptsSeen === keepRecentTurns) {
|
|
65
|
+
cutAt = i;
|
|
66
|
+
break;
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
return {
|
|
71
|
+
cutAt,
|
|
72
|
+
toSummarise: messages.slice(0, cutAt),
|
|
73
|
+
toKeep: messages.slice(cutAt),
|
|
74
|
+
userPromptCount: userPromptsSeen,
|
|
75
|
+
};
|
|
76
|
+
}
|
|
77
|
+
/**
|
|
78
|
+
* Build the Haiku prompt for summarising a session slice. Kept as a separate
|
|
79
|
+
* function so tests can assert on its shape and we can tune the prompt
|
|
80
|
+
* without touching the orchestration code.
|
|
81
|
+
*
|
|
82
|
+
* The output style we ask for is structured: decisions, objects, current
|
|
83
|
+
* state, open issues. That preserves the kind of context a follow-up turn
|
|
84
|
+
* actually needs while dropping the verbose tool-call chatter that drives
|
|
85
|
+
* the cost.
|
|
86
|
+
*/
|
|
87
|
+
export function buildSummarisationPrompt() {
|
|
88
|
+
return [
|
|
89
|
+
'You are compacting the older turns of a CSPeach (ABAP development CLI) conversation history.',
|
|
90
|
+
'',
|
|
91
|
+
'Your output replaces these turns in the session\'s context — the model will read your summary on the NEXT turn and continue working from it. So you must preserve everything the next turn needs to be useful.',
|
|
92
|
+
'',
|
|
93
|
+
'PRESERVE in your summary:',
|
|
94
|
+
'- Decisions made (architectural, naming, scope) with brief rationale',
|
|
95
|
+
'- SAP objects created or modified, with their types and active/inactive state',
|
|
96
|
+
'- Current state of any in-progress build (which phases done, which still open)',
|
|
97
|
+
'- Open issues, errors hit, deferred items',
|
|
98
|
+
'- Critical preferences/constraints the user set (model boundaries, naming conventions, etc.)',
|
|
99
|
+
'',
|
|
100
|
+
'DROP from your summary:',
|
|
101
|
+
'- Verbose tool-call args / results (just say "probed PACKAGE for existence" etc.)',
|
|
102
|
+
'- Repetitive content (the same class written 3 times — just the final shape)',
|
|
103
|
+
'- Intermediate failures already resolved',
|
|
104
|
+
'- TL;DR / Next sections from prior turns',
|
|
105
|
+
'- ASCII art separators, spinner output, footer rows',
|
|
106
|
+
'',
|
|
107
|
+
'OUTPUT FORMAT: a single markdown block with sections in this order:',
|
|
108
|
+
'1. ## Build state (one paragraph)',
|
|
109
|
+
'2. ## Decisions',
|
|
110
|
+
'3. ## Objects on SAP (table or bullet list)',
|
|
111
|
+
'4. ## Open / Deferred',
|
|
112
|
+
'5. ## Constraints set by user',
|
|
113
|
+
'',
|
|
114
|
+
'Aim for 1500-3000 tokens. Better to drop a detail than to bloat past 3000.',
|
|
115
|
+
'',
|
|
116
|
+
'The conversation history to compact follows below as JSON.',
|
|
117
|
+
].join('\n');
|
|
118
|
+
}
|
|
119
|
+
/**
|
|
120
|
+
* Serialise a message array into a compact JSON string for the summariser
|
|
121
|
+
* to ingest. We send role + a flattened representation of content because
|
|
122
|
+
* the raw structure (tool_use blocks with partial_json mid-stream, etc.) is
|
|
123
|
+
* noisy and doesn't add summarisation signal.
|
|
124
|
+
*/
|
|
125
|
+
export function serialiseForSummariser(messages) {
|
|
126
|
+
const out = [];
|
|
127
|
+
for (const msg of messages) {
|
|
128
|
+
if (typeof msg.content === 'string') {
|
|
129
|
+
out.push({ role: msg.role, content: msg.content });
|
|
130
|
+
}
|
|
131
|
+
else if (Array.isArray(msg.content)) {
|
|
132
|
+
const flat = msg.content.map((b) => {
|
|
133
|
+
if (b == null)
|
|
134
|
+
return '';
|
|
135
|
+
if (b.type === 'text')
|
|
136
|
+
return b.text ?? '';
|
|
137
|
+
if (b.type === 'thinking')
|
|
138
|
+
return `[thinking: ${(b.thinking ?? '').slice(0, 200)}]`;
|
|
139
|
+
if (b.type === 'tool_use')
|
|
140
|
+
return `[tool_use: ${b.name}(${JSON.stringify(b.input ?? {}).slice(0, 300)})]`;
|
|
141
|
+
if (b.type === 'tool_result') {
|
|
142
|
+
const content = typeof b.content === 'string' ? b.content : JSON.stringify(b.content);
|
|
143
|
+
return `[tool_result: ${content.slice(0, 500)}${content.length > 500 ? '...' : ''}]`;
|
|
144
|
+
}
|
|
145
|
+
return '';
|
|
146
|
+
}).filter(Boolean).join('\n');
|
|
147
|
+
out.push({ role: msg.role, content: flat });
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
return JSON.stringify(out, null, 2);
|
|
151
|
+
}
|
|
152
|
+
/**
|
|
153
|
+
* Back up a session's pre-compact state to a sibling file so /compact is
|
|
154
|
+
* recoverable. Returns the backup file path.
|
|
155
|
+
*/
|
|
156
|
+
export async function backupBeforeCompact(session) {
|
|
157
|
+
const ts = new Date().toISOString().replace(/[:.]/g, '-');
|
|
158
|
+
const dir = sessionsDir();
|
|
159
|
+
await fs.mkdir(dir, { recursive: true });
|
|
160
|
+
const filePath = path.join(dir, `${session.id}.pre-compact-${ts}.json`);
|
|
161
|
+
await fs.writeFile(filePath, JSON.stringify(session, null, 2), 'utf-8');
|
|
162
|
+
return filePath;
|
|
163
|
+
}
|
|
164
|
+
/**
|
|
165
|
+
* The compacted "synthesised earlier history" message that replaces the
|
|
166
|
+
* summarised slice in session.messages. Role=user because (a) Anthropic
|
|
167
|
+
* accepts multiple consecutive user messages, (b) putting it in user role
|
|
168
|
+
* makes it visually distinct from the model's prior assistant outputs, and
|
|
169
|
+
* (c) the explicit "context compacted by /compact" framing tells the next
|
|
170
|
+
* turn's model exactly what it's looking at.
|
|
171
|
+
*/
|
|
172
|
+
export function buildSummaryMessage(summary, droppedTurns) {
|
|
173
|
+
const ts = new Date().toISOString();
|
|
174
|
+
return {
|
|
175
|
+
role: 'user',
|
|
176
|
+
content: [
|
|
177
|
+
`[Context summary auto-generated by /compact at ${ts}.`,
|
|
178
|
+
`Original turns (${droppedTurns}) were summarised by ${COMPACTION_MODEL} to`,
|
|
179
|
+
`reduce cache cost on subsequent turns. The original session is backed up`,
|
|
180
|
+
`to ~/.cspeach/sessions/${'<session-id>'}.pre-compact-<timestamp>.json]`,
|
|
181
|
+
'',
|
|
182
|
+
summary,
|
|
183
|
+
].join('\n'),
|
|
184
|
+
};
|
|
185
|
+
}
|
|
186
|
+
/**
|
|
187
|
+
* Top-level compact orchestration. Pure async function — caller wires it
|
|
188
|
+
* into the REPL slash dispatcher. Throws only on catastrophic IO failure
|
|
189
|
+
* (the summariser swallowing its own errors is fine because we never mutate
|
|
190
|
+
* session.messages until we have a valid summary in hand).
|
|
191
|
+
*
|
|
192
|
+
* Returns a result struct describing what happened. Caller renders the
|
|
193
|
+
* user-facing output.
|
|
194
|
+
*/
|
|
195
|
+
export async function runCompact(opts) {
|
|
196
|
+
const { session, summarise } = opts;
|
|
197
|
+
const keep = opts.keepRecentTurns ?? KEEP_RECENT_TURNS;
|
|
198
|
+
const plan = planCompaction(session.messages, keep);
|
|
199
|
+
// Bail if there's not enough older history to be worth compacting.
|
|
200
|
+
if (plan.userPromptCount < keep) {
|
|
201
|
+
return {
|
|
202
|
+
performed: false,
|
|
203
|
+
reason: `Session has ${plan.userPromptCount} user prompts; need at least ${keep + 1} for /compact to find anything to summarise.`,
|
|
204
|
+
};
|
|
205
|
+
}
|
|
206
|
+
if (plan.toSummarise.length < MIN_COMPACT_THRESHOLD) {
|
|
207
|
+
return {
|
|
208
|
+
performed: false,
|
|
209
|
+
reason: `Only ${plan.toSummarise.length} messages before the recent window — nothing meaningful to compact yet.`,
|
|
210
|
+
};
|
|
211
|
+
}
|
|
212
|
+
// Run the Haiku call FIRST. If it fails we don't touch the session.
|
|
213
|
+
let summary;
|
|
214
|
+
try {
|
|
215
|
+
summary = await summarise(plan.toSummarise);
|
|
216
|
+
}
|
|
217
|
+
catch (err) {
|
|
218
|
+
throw new Error(`Compaction summariser failed: ${err instanceof Error ? err.message : String(err)}. Session unchanged.`);
|
|
219
|
+
}
|
|
220
|
+
if (!summary || summary.trim().length === 0) {
|
|
221
|
+
throw new Error('Compaction summariser returned empty output. Session unchanged.');
|
|
222
|
+
}
|
|
223
|
+
// Back up the prior session state to a sibling file (one-shot rollback).
|
|
224
|
+
const backupPath = await backupBeforeCompact(session);
|
|
225
|
+
// Mutate session.messages: drop the summarised slice, prepend the synthesised
|
|
226
|
+
// summary message immediately before the kept tail. The session retains its
|
|
227
|
+
// FIRST user message intact only if it was already in the toKeep slice —
|
|
228
|
+
// which it will be unless the session is very long. (For very long sessions
|
|
229
|
+
// we accept losing the original first prompt because the summary should
|
|
230
|
+
// mention what the goal was.)
|
|
231
|
+
const summaryMsg = buildSummaryMessage(summary, plan.toSummarise.length);
|
|
232
|
+
session.messages = [summaryMsg, ...plan.toKeep];
|
|
233
|
+
await saveSession(session);
|
|
234
|
+
return {
|
|
235
|
+
performed: true,
|
|
236
|
+
droppedTurns: plan.toSummarise.length,
|
|
237
|
+
keptTurns: plan.toKeep.length,
|
|
238
|
+
summaryTokens: Math.ceil(summary.length / 4), // rough — 4 chars per token
|
|
239
|
+
backupPath,
|
|
240
|
+
};
|
|
241
|
+
}
|
|
242
|
+
/**
|
|
243
|
+
* Format the result of a compact run for terminal display. Separate so
|
|
244
|
+
* tests can assert on shape and the REPL just renders.
|
|
245
|
+
*/
|
|
246
|
+
export function formatCompactResult(result,
|
|
247
|
+
/** Optional — falls back to the legacy constant for back-compat. */
|
|
248
|
+
keepRecentTurns = KEEP_RECENT_TURNS) {
|
|
249
|
+
if (!result.performed) {
|
|
250
|
+
return chalk.dim(`/compact: ${result.reason ?? 'nothing to do'}`);
|
|
251
|
+
}
|
|
252
|
+
const lines = [
|
|
253
|
+
'',
|
|
254
|
+
chalk.bold('/compact complete'),
|
|
255
|
+
chalk.dim(` dropped: ${result.droppedTurns} messages`),
|
|
256
|
+
chalk.dim(` kept: ${result.keptTurns} messages (most recent ${keepRecentTurns} user turns intact)`),
|
|
257
|
+
chalk.dim(` summary: ~${result.summaryTokens?.toLocaleString()} tokens`),
|
|
258
|
+
chalk.dim(` backup: ${result.backupPath}`),
|
|
259
|
+
'',
|
|
260
|
+
chalk.green('Next turn\'s context is now roughly summary-size instead of full history.'),
|
|
261
|
+
chalk.dim('Heads-up: the very next turn rebuilds the prompt cache (one-off cost), then turns after that are cheap again.'),
|
|
262
|
+
'',
|
|
263
|
+
];
|
|
264
|
+
return lines.join('\n');
|
|
265
|
+
}
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `/cost` slash command — Phase 1 cost tracking (2026-05-16).
|
|
3
|
+
*
|
|
4
|
+
* Reads the current session's per-turn cost log and prints a human-readable
|
|
5
|
+
* breakdown to the REPL: total spend, per-category tokens, and per-model
|
|
6
|
+
* split (will matter once Phase 2 lands and a single session crosses
|
|
7
|
+
* Haiku + Sonnet + Opus).
|
|
8
|
+
*
|
|
9
|
+
* No mutation, no network. Safe to run anytime.
|
|
10
|
+
*/
|
|
11
|
+
import chalk from 'chalk';
|
|
12
|
+
import { readCostLog, summarise } from '../cost/cost-log.js';
|
|
13
|
+
import { formatCost } from '../cost/pricing.js';
|
|
14
|
+
export async function runCostCommand(sessionId) {
|
|
15
|
+
const entries = await readCostLog(sessionId);
|
|
16
|
+
if (entries.length === 0) {
|
|
17
|
+
console.log('');
|
|
18
|
+
console.log(chalk.dim('No cost data yet for this session. Run a turn first.'));
|
|
19
|
+
console.log('');
|
|
20
|
+
return;
|
|
21
|
+
}
|
|
22
|
+
const s = summarise(entries);
|
|
23
|
+
// Header
|
|
24
|
+
console.log('');
|
|
25
|
+
console.log(chalk.bold(`Session cost so far: ${formatCost(s.totalCost)}`) + chalk.dim(` (${s.turns} turn${s.turns === 1 ? '' : 's'})`));
|
|
26
|
+
// Token breakdown
|
|
27
|
+
console.log('');
|
|
28
|
+
console.log(chalk.dim('Tokens:'));
|
|
29
|
+
console.log(` input ${s.totalTokens.input.toLocaleString().padStart(12)}`);
|
|
30
|
+
console.log(` output ${s.totalTokens.output.toLocaleString().padStart(12)}`);
|
|
31
|
+
console.log(` cache read ${s.totalTokens.cacheRead.toLocaleString().padStart(12)}`);
|
|
32
|
+
console.log(` cache create ${s.totalTokens.cacheCreate.toLocaleString().padStart(12)}`);
|
|
33
|
+
// Per-model split (Phase 2 will make this interesting)
|
|
34
|
+
if (s.byModel.length > 1) {
|
|
35
|
+
console.log('');
|
|
36
|
+
console.log(chalk.dim('By model:'));
|
|
37
|
+
for (const m of s.byModel) {
|
|
38
|
+
console.log(` ${m.model.padEnd(24)} ${m.turns} turn${m.turns === 1 ? '' : 's'} ${formatCost(m.cost)}`);
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
else if (s.byModel[0]) {
|
|
42
|
+
console.log('');
|
|
43
|
+
console.log(chalk.dim(`Model: ${s.byModel[0].model}`));
|
|
44
|
+
}
|
|
45
|
+
// Recent turns — last 5 lines, most recent first
|
|
46
|
+
const recent = entries.slice(-5).reverse();
|
|
47
|
+
console.log('');
|
|
48
|
+
console.log(chalk.dim(`Recent turns (last ${recent.length}):`));
|
|
49
|
+
for (const e of recent) {
|
|
50
|
+
const tok = (e.tokens.input + e.tokens.output).toLocaleString();
|
|
51
|
+
console.log(` turn ${String(e.turn).padStart(3)} ${tok.padStart(10)} tok ${formatCost(e.cost)}`);
|
|
52
|
+
}
|
|
53
|
+
console.log('');
|
|
54
|
+
console.log(chalk.dim(`Detailed JSONL: ~/.cspeach/sessions/${sessionId}-cost.jsonl`));
|
|
55
|
+
console.log('');
|
|
56
|
+
}
|
package/dist/commands/help.js
CHANGED
|
@@ -61,33 +61,40 @@ function wrapDescription(desc, width, maxLines) {
|
|
|
61
61
|
* Cyan category headings, bold skill names, default-color descriptions.
|
|
62
62
|
* Descriptions wrap to MAX_DESC_LINES lines max, aligned to the name column.
|
|
63
63
|
*/
|
|
64
|
-
|
|
64
|
+
/**
|
|
65
|
+
* Phase D5 (2026-05-17) — accepts an optional emit callback so Ink mode
|
|
66
|
+
* can route output through chunkEmitter instead of console.log (which
|
|
67
|
+
* corrupts the React frame). Default keeps the classic-mode behaviour.
|
|
68
|
+
*/
|
|
69
|
+
export function printHelp(emit = (s) => console.log(s)) {
|
|
65
70
|
const termWidth = process.stdout.columns ?? DEFAULT_TERM_WIDTH;
|
|
66
71
|
const descWidth = Math.max(MIN_DESC_WIDTH, termWidth - NAME_COL_WIDTH);
|
|
67
72
|
const continuationIndent = ' '.repeat(NAME_COL_WIDTH);
|
|
68
|
-
|
|
73
|
+
emit('');
|
|
69
74
|
for (const category of CATEGORY_ORDER) {
|
|
70
75
|
const skills = SKILL_CATALOG.filter((s) => s.category === category);
|
|
71
76
|
if (skills.length === 0)
|
|
72
77
|
continue;
|
|
73
|
-
|
|
78
|
+
emit(chalk.cyan.bold(category));
|
|
74
79
|
for (const skill of skills) {
|
|
75
80
|
const nameCol = ('/' + skill.name).padEnd(NAME_COL_PAD);
|
|
76
81
|
const descLines = wrapDescription(skill.description, descWidth, MAX_DESC_LINES);
|
|
77
82
|
const firstLine = descLines[0] ?? '';
|
|
78
|
-
|
|
83
|
+
emit(` ${chalk.bold(nameCol)} ${firstLine}`);
|
|
79
84
|
for (let j = 1; j < descLines.length; j++) {
|
|
80
|
-
|
|
85
|
+
emit(`${continuationIndent}${descLines[j]}`);
|
|
81
86
|
}
|
|
82
87
|
}
|
|
83
|
-
|
|
88
|
+
emit('');
|
|
84
89
|
}
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
90
|
+
emit(chalk.cyan.bold('Shortcuts'));
|
|
91
|
+
emit(` ${chalk.bold('/help'.padEnd(NAME_COL_PAD))} This screen`);
|
|
92
|
+
emit(` ${chalk.bold('/skills'.padEnd(NAME_COL_PAD))} Same as /help`);
|
|
93
|
+
emit(` ${chalk.bold('/ui [auto|ink|classic]'.padEnd(NAME_COL_PAD))} View or change UI rendering mode`);
|
|
94
|
+
emit(` ${chalk.bold('/reroute <skill>'.padEnd(NAME_COL_PAD))} Re-run your last prompt with a different skill`);
|
|
95
|
+
emit(` ${chalk.bold('/new'.padEnd(NAME_COL_PAD))} End the current Q&A chain — next prompt is classified fresh`);
|
|
96
|
+
emit(` ${chalk.bold('/cost'.padEnd(NAME_COL_PAD))} Show this session's API spend so far ($ + token breakdown)`);
|
|
97
|
+
emit(` ${chalk.bold('/compact'.padEnd(NAME_COL_PAD))} Summarise older turns into a compact context block — cuts subsequent turn cost by 80-90%`);
|
|
98
|
+
emit(` ${chalk.bold('/exit'.padEnd(NAME_COL_PAD))} Quit CSPeach`);
|
|
99
|
+
emit('');
|
|
93
100
|
}
|
package/dist/config/loader.js
CHANGED
|
@@ -24,7 +24,42 @@ const DEFAULT_CONFIG = {
|
|
|
24
24
|
write_mode: 'approval-gated',
|
|
25
25
|
// Default `managed` — lowest-friction entry point; uses the cspeach.dev proxy.
|
|
26
26
|
llm: { mode: 'managed' },
|
|
27
|
+
// Defaults tuned 2026-05-16 from session 0618a42d evidence — see CompactConfig doc.
|
|
28
|
+
compact: {
|
|
29
|
+
keep_recent_turns: 5,
|
|
30
|
+
auto_enabled: true,
|
|
31
|
+
auto_threshold_tokens: 1_000_000,
|
|
32
|
+
min_turns_between_auto: 5,
|
|
33
|
+
},
|
|
27
34
|
};
|
|
35
|
+
/**
|
|
36
|
+
* Coerce a possibly-bad `compact` block from disk into a safe CompactConfig.
|
|
37
|
+
* - missing/undefined → defaults
|
|
38
|
+
* - wrong type → defaults
|
|
39
|
+
* - out-of-range numbers → clamped to sane bounds (keep_recent_turns 1..50,
|
|
40
|
+
* auto_threshold_tokens 10_000..10_000_000, min_turns_between_auto 1..100)
|
|
41
|
+
*
|
|
42
|
+
* Defensive because TOML hand-editing makes typos likely and we never want
|
|
43
|
+
* a config typo to lock the user out of the REPL.
|
|
44
|
+
*/
|
|
45
|
+
function sanitiseCompact(raw) {
|
|
46
|
+
const d = DEFAULT_CONFIG.compact;
|
|
47
|
+
if (!raw || typeof raw !== 'object')
|
|
48
|
+
return { ...d };
|
|
49
|
+
const r = raw;
|
|
50
|
+
const intInRange = (v, fallback, lo, hi) => {
|
|
51
|
+
if (typeof v !== 'number' || !Number.isFinite(v))
|
|
52
|
+
return fallback;
|
|
53
|
+
const i = Math.floor(v);
|
|
54
|
+
return Math.min(hi, Math.max(lo, i));
|
|
55
|
+
};
|
|
56
|
+
return {
|
|
57
|
+
keep_recent_turns: intInRange(r.keep_recent_turns, d.keep_recent_turns, 1, 50),
|
|
58
|
+
auto_enabled: typeof r.auto_enabled === 'boolean' ? r.auto_enabled : d.auto_enabled,
|
|
59
|
+
auto_threshold_tokens: intInRange(r.auto_threshold_tokens, d.auto_threshold_tokens, 10_000, 10_000_000),
|
|
60
|
+
min_turns_between_auto: intInRange(r.min_turns_between_auto, d.min_turns_between_auto, 1, 100),
|
|
61
|
+
};
|
|
62
|
+
}
|
|
28
63
|
export async function loadConfig() {
|
|
29
64
|
try {
|
|
30
65
|
const raw = await fs.readFile(configFile(), 'utf-8');
|
|
@@ -52,6 +87,10 @@ export async function loadConfig() {
|
|
|
52
87
|
llm: { ...DEFAULT_CONFIG.llm, ...(parsed.llm ?? {}) },
|
|
53
88
|
classifier: { ...DEFAULT_CONFIG.classifier, ...(parsed.classifier ?? {}) },
|
|
54
89
|
ui: { ...DEFAULT_CONFIG.ui, ...(parsed.ui ?? {}) },
|
|
90
|
+
// Deep-merge `compact` so a user config that only overrides one
|
|
91
|
+
// field (e.g. just `keep_recent_turns = 10`) still gets the four
|
|
92
|
+
// unspecified defaults — and a typo'd field falls through harmlessly.
|
|
93
|
+
compact: sanitiseCompact(parsed.compact),
|
|
55
94
|
shell_exec: parsed.shell_exec === undefined ? undefined : { allow },
|
|
56
95
|
};
|
|
57
96
|
}
|