@knightcodeai/cli-linux-x64 0.9.0 → 0.9.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/CHANGELOG.md +38 -0
- package/bin/docs/custom-provider.md +3 -0
- package/bin/docs/environment-variables.md +1 -0
- package/bin/docs/extensions.md +19 -0
- package/bin/docs/models.md +32 -1
- package/bin/docs/providers.md +9 -1
- package/bin/docs/session-format.md +15 -1
- package/bin/docs/sessions.md +27 -0
- package/bin/docs/settings.md +24 -1
- package/bin/docs/usage.md +1 -0
- package/bin/knightcode +2 -2
- package/bin/package.json +6 -6
- package/package.json +1 -1
package/bin/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,43 @@
|
|
|
1
1
|
# @knightcodeai/cli
|
|
2
2
|
|
|
3
|
+
## 0.9.1
|
|
4
|
+
|
|
5
|
+
### Added
|
|
6
|
+
|
|
7
|
+
- Added the shipped Radius model catalog, so Radius models are listed in `/model` before the first gateway refresh and while offline; the fetched gateway catalog still overrides it.
|
|
8
|
+
|
|
9
|
+
- Added the Meta provider: `/login meta` signs in with a Muse subscription through Meta's device authorization flow and mints a Model API key that is refreshed automatically, and `META_API_KEY` works as an ordinary API key.
|
|
10
|
+
|
|
11
|
+
- Added prompt cache warming, which keeps a provider's prompt cache alive between turns where the model's cache lifetime is known, with `/settings` controls and a footer indicator. Extensions can observe or override each refresh.
|
|
12
|
+
|
|
13
|
+
- Added `/bug`, which collects a redacted report — version, runtime, model and provider configuration, extensions, settings and this session's error diagnostics, never API keys — and either uploads it to the KnightCode maintainers or writes it as a zip you can attach to an issue yourself. Including the transcript is optional, and declining it offers a model-written summary instead. Crashes are recorded and attached to the next report.
|
|
14
|
+
|
|
15
|
+
### Changed
|
|
16
|
+
|
|
17
|
+
- Changed extension loading to pull in its transform dependencies only when an extension actually needs transforming, which shortens startup for everyone who has no extensions installed.
|
|
18
|
+
|
|
19
|
+
- Changed the session picker to load progressively, so it opens immediately on a large session directory instead of waiting for every file to be read.
|
|
20
|
+
|
|
21
|
+
### Fixed
|
|
22
|
+
|
|
23
|
+
- Fixed sessions on z.ai stopping instead of compacting when the provider answers a too-long prompt with its `1261` error body rather than the usual wording.
|
|
24
|
+
|
|
25
|
+
- Fixed Cerebras requests failing with a 400 when extensions declare both strict and non-strict tools; strict tool schemas are no longer sent to Cerebras.
|
|
26
|
+
|
|
27
|
+
- Fixed compaction cancellation: aborting during auto-compaction could leave the turn retrying, run an extension handler after the abort, or report a cancelled compaction as a failure.
|
|
28
|
+
|
|
29
|
+
- Fixed clipboard copying in headless and remote sessions by restoring the OSC 52 fallback when no native clipboard is reachable, including under WSL.
|
|
30
|
+
|
|
31
|
+
- Fixed the terminal waiting on a remote prompt event that was never awaited, which could drop a prompt sent from the phone.
|
|
32
|
+
|
|
33
|
+
- Fixed display-math rendering of stacked sub/superscripts, the `\bf`-style font switches, and `cases` alignment and brace placement.
|
|
34
|
+
|
|
35
|
+
- Fixed slash-command ranking so a `skill:` command is matched on its bare name, putting `/idea` on `skill:research-idea` rather than `skill:deep-research`.
|
|
36
|
+
|
|
37
|
+
- Fixed fullscreen images disappearing in WezTerm, which erased a Kitty image when a later row on top of it was cleared.
|
|
38
|
+
|
|
39
|
+
- Fixed file-path autocomplete after CJK punctuation: a path typed after a full-width comma or colon now completes, and a completed path containing one is quoted.
|
|
40
|
+
|
|
3
41
|
## 0.9.0
|
|
4
42
|
|
|
5
43
|
### Added
|
|
@@ -733,6 +733,9 @@ interface ProviderModelConfig {
|
|
|
733
733
|
cacheWrite: number;
|
|
734
734
|
};
|
|
735
735
|
|
|
736
|
+
/** Best-effort prompt cache lifetime in seconds per retention tier. Unset disables cache warming. */
|
|
737
|
+
promptCache?: { short?: number; long?: number };
|
|
738
|
+
|
|
736
739
|
/** Maximum context window size in tokens. */
|
|
737
740
|
contextWindow: number;
|
|
738
741
|
|
|
@@ -89,6 +89,7 @@ These variables are read by KnightCode itself:
|
|
|
89
89
|
| `KNIGHTCODE_TELEMETRY` | Override install/update telemetry and provider attribution headers: `1`/`true`/`yes` or `0`/`false`/`no` |
|
|
90
90
|
| `KNIGHTCODE_CACHE_RETENTION` | Set to `long` for extended provider prompt caching where supported |
|
|
91
91
|
| `KNIGHTCODE_SHARE_VIEWER_URL` | Override the base URL used by `/share` |
|
|
92
|
+
| `KNIGHTCODE_RADIUS_GATEWAY` | Override the Radius gateway origin used by `/bug` uploads and Radius relay connections |
|
|
92
93
|
| `KNIGHTCODE_HARDWARE_CURSOR` | Set to `1` to show the hardware cursor; see [Terminal setup](terminal-setup.md) |
|
|
93
94
|
| `KNIGHTCODE_HYPERLINKS` | Override OSC 8 hyperlink detection with `1`, `0`, or `auto` |
|
|
94
95
|
| `KNIGHTCODE_IMAGE_PROTOCOL` | Override inline image detection with `kitty`, `iterm2`, `none`, or `auto` |
|
package/bin/docs/extensions.md
CHANGED
|
@@ -738,6 +738,25 @@ knightcode.on("after_provider_response", (event, ctx) => {
|
|
|
738
738
|
|
|
739
739
|
Header availability depends on provider and transport. Providers that abstract HTTP responses may not expose headers.
|
|
740
740
|
|
|
741
|
+
#### cache_warming_decision
|
|
742
|
+
|
|
743
|
+
Fired before each prompt-cache refresh with KnightCode's decision filled in. The event carries only KnightCode's cost estimates; use `ctx.model`, `ctx.isIdle()`, and `ctx.getContextUsage()` for everything else.
|
|
744
|
+
|
|
745
|
+
```typescript
|
|
746
|
+
knightcode.on("cache_warming_decision", (event, ctx) => {
|
|
747
|
+
// event.warmCost: price of this refresh
|
|
748
|
+
// event.missCost: extra price of the next request if the entry is lost
|
|
749
|
+
// event.continuationProbability: KnightCode's estimate that a request arrives in time
|
|
750
|
+
// event.action: "warm" | "stop", KnightCode's decision
|
|
751
|
+
|
|
752
|
+
if (ctx.model?.provider === "my-provider") {
|
|
753
|
+
return { action: "stop" };
|
|
754
|
+
}
|
|
755
|
+
});
|
|
756
|
+
```
|
|
757
|
+
|
|
758
|
+
Return `{ action: "warm" }` or `{ action: "stop" }` to override; the last handler that returns an action wins. `"stop"` ends warming until the next real request.
|
|
759
|
+
|
|
741
760
|
### Model Events
|
|
742
761
|
|
|
743
762
|
#### model_select
|
package/bin/docs/models.md
CHANGED
|
@@ -9,6 +9,7 @@ Add custom providers and models (Ollama, vLLM, LM Studio, proxies) via `~/.knigh
|
|
|
9
9
|
- [Supported APIs](#supported-apis)
|
|
10
10
|
- [Provider Configuration](#provider-configuration)
|
|
11
11
|
- [Model Configuration](#model-configuration)
|
|
12
|
+
- [Prompt Cache Lifetimes](#prompt-cache-lifetimes)
|
|
12
13
|
- [Overriding Built-in Providers](#overriding-built-in-providers)
|
|
13
14
|
- [Per-model Overrides](#per-model-overrides)
|
|
14
15
|
- [Anthropic Messages Compatibility](#anthropic-messages-compatibility)
|
|
@@ -208,6 +209,7 @@ If your command is slow, expensive, rate-limited, or should keep using a previou
|
|
|
208
209
|
| `maxTokens` | No | `16384` | Maximum output tokens |
|
|
209
210
|
| `samplingParams` | No | omitted | Sampling parameters merged verbatim into every request body (see below) |
|
|
210
211
|
| `cost` | No | all zeros | Per-million-token rates with optional request-wide input pricing tiers |
|
|
212
|
+
| `promptCache` | No | omitted | Best-effort prompt cache lifetime in seconds per retention tier (see below) |
|
|
211
213
|
| `compat` | No | provider `compat` | Provider compatibility overrides. Merged with provider-level `compat` when both are set. |
|
|
212
214
|
|
|
213
215
|
A cost tier supplies a complete alternate rate set and applies to the full request when total input usage (`input + cacheRead + cacheWrite`) exceeds `inputTokensAbove`. When multiple tiers match, the highest threshold wins.
|
|
@@ -236,6 +238,19 @@ Current behavior:
|
|
|
236
238
|
- `/model`, `--list-models`, and the interactive footer display entries by model `id`.
|
|
237
239
|
- The configured `name` is used for model matching and secondary model detail text. It does not replace the footer/status-bar model id.
|
|
238
240
|
|
|
241
|
+
### Prompt Cache Lifetimes
|
|
242
|
+
|
|
243
|
+
`promptCache` states how long the provider keeps a prompt cache entry alive for each retention tier KnightCode can request (`short` is the default tier; `long` is used when `KNIGHTCODE_CACHE_RETENTION=long`). Values are seconds and are estimates: providers publish ranges, so pick the conservative end.
|
|
244
|
+
|
|
245
|
+
```json
|
|
246
|
+
{
|
|
247
|
+
"id": "claude-sonnet-5",
|
|
248
|
+
"promptCache": { "short": 300, "long": 3600 }
|
|
249
|
+
}
|
|
250
|
+
```
|
|
251
|
+
|
|
252
|
+
The built-in catalog fills this in for direct Anthropic (5 min / 1 h). Other providers, including direct OpenAI, have no built-in lifetime until their cache-expiry and replay behavior has been validated for warming. A model without a value for the tier a request used is never warmed; custom models and provider overrides can opt in when the backing cache behavior is known. See [Cache Warming](settings.md#cache-warming).
|
|
253
|
+
|
|
239
254
|
### Sampling Parameters
|
|
240
255
|
|
|
241
256
|
`samplingParams` is a free-form object merged verbatim into every request body for the model, after the fields knightcode sets itself, so its keys win. Use it to send sampling parameters knightcode does not model — including server-specific ones like llama.cpp's `min_p` or vLLM's `top_k`:
|
|
@@ -359,7 +374,23 @@ Use `modelOverrides` to customize built-in models and matching extension-registe
|
|
|
359
374
|
}
|
|
360
375
|
```
|
|
361
376
|
|
|
362
|
-
`modelOverrides` supports these fields per model: `name`, `reasoning`, `thinkingLevelMap`, `input`, `cost` (partial), `contextWindow`, `maxTokens`, `samplingParams` (merged per key), `headers`, `compat`.
|
|
377
|
+
`modelOverrides` supports these fields per model: `name`, `reasoning`, `thinkingLevelMap`, `input`, `cost` (partial), `promptCache` (merged per tier), `contextWindow`, `maxTokens`, `samplingParams` (merged per key), `headers`, `compat`.
|
|
378
|
+
|
|
379
|
+
Use a `promptCache` override to enable cache warming through a proxy whose backing cache you know, for example OpenRouter routed to Anthropic:
|
|
380
|
+
|
|
381
|
+
```json
|
|
382
|
+
{
|
|
383
|
+
"providers": {
|
|
384
|
+
"openrouter": {
|
|
385
|
+
"modelOverrides": {
|
|
386
|
+
"anthropic/claude-sonnet-4": {
|
|
387
|
+
"promptCache": { "short": 300 }
|
|
388
|
+
}
|
|
389
|
+
}
|
|
390
|
+
}
|
|
391
|
+
}
|
|
392
|
+
}
|
|
393
|
+
```
|
|
363
394
|
|
|
364
395
|
Direct OpenAI GPT-5.6 Sol, Terra, and Luna default to a `272000` context window so requests remain within OpenAI's short-context pricing tier. To opt into OpenAI's 1.05M context window, increase it for each model you use:
|
|
365
396
|
|
package/bin/docs/providers.md
CHANGED
|
@@ -20,6 +20,7 @@ Use `/login` in interactive mode, then select a provider:
|
|
|
20
20
|
- Claude Pro/Max
|
|
21
21
|
- GitHub Copilot
|
|
22
22
|
- xAI (Grok/X subscription)
|
|
23
|
+
- Meta (Muse subscription)
|
|
23
24
|
- OpenRouter (OAuth-minted API key billed from OpenRouter credits)
|
|
24
25
|
- Radius
|
|
25
26
|
|
|
@@ -44,6 +45,12 @@ Anthropic subscription auth is active for Claude Pro/Max accounts. Third-party h
|
|
|
44
45
|
- Run `/login xai`, then select **Use a subscription**
|
|
45
46
|
- `XAI_API_KEY` remains available through **Use an API key**
|
|
46
47
|
|
|
48
|
+
### Meta (Muse subscription)
|
|
49
|
+
|
|
50
|
+
- Run `/login meta`, then select **Sign in with Meta** to open the device authorization flow
|
|
51
|
+
- The login mints a Model API key that is re-minted automatically about once a day
|
|
52
|
+
- `META_API_KEY` remains available through **Use an API key**
|
|
53
|
+
|
|
47
54
|
### OpenRouter
|
|
48
55
|
|
|
49
56
|
- Run `/login openrouter`, then select **Sign in with OpenRouter** to open the OpenRouter PKCE authorization flow
|
|
@@ -53,7 +60,7 @@ Anthropic subscription auth is active for Claude Pro/Max accounts. Third-party h
|
|
|
53
60
|
|
|
54
61
|
### Radius
|
|
55
62
|
|
|
56
|
-
Radius is a
|
|
63
|
+
Radius is a `knightcode-messages` gateway. KnightCode ships the public Radius model catalog for immediate and offline model lookup, then overlays it with the effective gateway catalog after authentication. `/login radius` stores OAuth tokens in `auth.json`; refreshed catalogs are cached in `models-store.json`. Custom Radius gateways can be declared in `models.json` with `"oauth": "radius"` and a gateway `baseUrl`; they do not inherit the public `radius.pi.dev` catalog.
|
|
57
64
|
|
|
58
65
|
## API Keys
|
|
59
66
|
|
|
@@ -95,6 +102,7 @@ knightcode
|
|
|
95
102
|
| Together AI | `TOGETHER_API_KEY` | `together` |
|
|
96
103
|
| Baseten | `BASETEN_API_KEY` | `baseten` |
|
|
97
104
|
| Kimi For Coding | `KIMI_API_KEY` | `kimi-coding` |
|
|
105
|
+
| Meta | `META_API_KEY` | `meta` |
|
|
98
106
|
| MiniMax | `MINIMAX_API_KEY` | `minimax` |
|
|
99
107
|
| MiniMax (China) | `MINIMAX_CN_API_KEY` | `minimax-cn` |
|
|
100
108
|
| Qwen Token Plan (existing catalog) | `QWEN_TOKEN_PLAN_API_KEY` | `qwen-token-plan` |
|
|
@@ -256,6 +256,16 @@ Emitted when the user changes the thinking/reasoning level.
|
|
|
256
256
|
{"type":"thinking_level_change","id":"e5f6g7h8","parentId":"d4e5f6g7","timestamp":"2024-12-03T14:06:00.000Z","thinkingLevel":"high"}
|
|
257
257
|
```
|
|
258
258
|
|
|
259
|
+
### UsageEntry
|
|
260
|
+
|
|
261
|
+
Records model-attributed usage that is not an assistant message and does not participate in LLM context. `kind` is an arbitrary string identifying the operation; for example, cache warming uses `"cache_warm"`.
|
|
262
|
+
|
|
263
|
+
```json
|
|
264
|
+
{"type":"usage","id":"f6g7h8i9","parentId":"e5f6g7h8","timestamp":"2024-12-03T14:08:00.000Z","kind":"cache_warm","provider":"anthropic","model":"claude-sonnet-4-5","usage":{"input":0,"output":0,"cacheRead":50000,"cacheWrite":0,"totalTokens":50000,"cost":{"input":0,"output":0,"cacheRead":0.015,"cacheWrite":0,"total":0.015}}}
|
|
265
|
+
```
|
|
266
|
+
|
|
267
|
+
Usage entries contribute to session token and cost totals. KnightCode hides them from the conversation tree. Consumers should treat unknown `kind` values as normal usage rather than rejecting them.
|
|
268
|
+
|
|
259
269
|
### CompactionEntry
|
|
260
270
|
|
|
261
271
|
Created when context is compacted. Stores a summary of earlier messages and a complete system prompt/tool checkpoint.
|
|
@@ -364,7 +374,7 @@ Entries normally form one tree, but navigation APIs can create multiple roots:
|
|
|
364
374
|
- `compaction` -> complete system checkpoint followed by `compactionSummary`
|
|
365
375
|
- `branch_summary` -> `branchSummary`
|
|
366
376
|
- `custom_message` -> `CustomMessage`
|
|
367
|
-
- `custom` -> no context message
|
|
377
|
+
- `usage` and `custom` -> no context message
|
|
368
378
|
|
|
369
379
|
The compaction summary replaces entries before `firstKeptEntryId`. Pre-compaction system messages are folded into the complete checkpoint rather than replayed from the retained range. Retained non-system entries and all entries after the compaction remain available to the LLM.
|
|
370
380
|
|
|
@@ -391,6 +401,9 @@ for (const line of lines) {
|
|
|
391
401
|
case "branch_summary":
|
|
392
402
|
console.log(`[${entry.id}] Branch from ${entry.fromId}`);
|
|
393
403
|
break;
|
|
404
|
+
case "usage":
|
|
405
|
+
console.log(`[${entry.id}] Usage (${entry.kind}): ${entry.usage.totalTokens} tokens`);
|
|
406
|
+
break;
|
|
394
407
|
case "custom":
|
|
395
408
|
console.log(`[${entry.id}] Custom (${entry.customType}): ${JSON.stringify(entry.data)}`);
|
|
396
409
|
break;
|
|
@@ -435,6 +448,7 @@ Key methods for working with sessions programmatically.
|
|
|
435
448
|
- `appendMessage(message)` - Add message
|
|
436
449
|
- `appendThinkingLevelChange(level)` - Record thinking change
|
|
437
450
|
- `appendModelChange(provider, modelId)` - Record model change
|
|
451
|
+
- `appendUsage(kind, provider, model, usage)` - Record model-attributed usage outside the conversation
|
|
438
452
|
- `appendCompaction(summary, firstKeptEntryId, tokensBefore, details?, fromHook?, usage?)` - Add compaction
|
|
439
453
|
- `appendCustomEntry(customType, data?)` - Extension state (not in context)
|
|
440
454
|
- `appendSessionInfo(name)` - Set session display name
|
package/bin/docs/sessions.md
CHANGED
|
@@ -34,6 +34,7 @@ For the JSONL file format and SessionManager API, see [Session Format](session-f
|
|
|
34
34
|
| `/compact [prompt]` | Summarize older context; see [Compaction](compaction.md) |
|
|
35
35
|
| `/export [file]` | Export session to HTML |
|
|
36
36
|
| `/share` | Upload as private GitHub gist with shareable HTML link |
|
|
37
|
+
| `/bug [description]` | Report a bug to the KnightCode developers; see [Reporting Bugs](#reporting-bugs) |
|
|
37
38
|
|
|
38
39
|
## Resuming and Deleting Sessions
|
|
39
40
|
|
|
@@ -175,6 +176,32 @@ When prompted, choose one of:
|
|
|
175
176
|
|
|
176
177
|
See [Compaction](compaction.md) for branch summarization internals and extension hooks.
|
|
177
178
|
|
|
179
|
+
## Reporting Bugs
|
|
180
|
+
|
|
181
|
+
`/bug [description]` collects a bug report for the KnightCode maintainers. The report is not shared publicly. The dialog asks for an optional description and whether to include the session transcript. If you decline the transcript, KnightCode offers to have the current model write a summary of what went wrong instead; the transcript is sent to your provider with your credentials, and only the summary is attached.
|
|
182
|
+
|
|
183
|
+
The last step chooses where the report goes:
|
|
184
|
+
|
|
185
|
+
- **Upload Report** sends it to the KnightCode maintainers through `remote.knightcode.dev`. No login is required and reports are always anonymous. If the upload fails, KnightCode offers to export the zip instead.
|
|
186
|
+
- **Export as Zip** writes a zip archive to the current directory. Attach it to an issue or send it to the developers yourself.
|
|
187
|
+
|
|
188
|
+
Both contain the same files:
|
|
189
|
+
|
|
190
|
+
| File | Content |
|
|
191
|
+
|------|---------|
|
|
192
|
+
| `report.json` | KnightCode version, runtime, OS, terminal, current model and provider configuration, loaded extensions, and settings. API keys, header values, URL credentials, and the analytics tracking id are never included. |
|
|
193
|
+
| `diagnostics.json` | Provider and runtime error diagnostics attached to assistant messages across the whole session (failed or aborted turns, retries, error messages), plus any recorded crashes. Always included; message content is not. |
|
|
194
|
+
| `session.jsonl` | The current branch of the session, only when you chose to include it. It contains file contents and command output read during the session. |
|
|
195
|
+
| `summary.md` | The model-written summary, only when you chose to generate one. |
|
|
196
|
+
|
|
197
|
+
Each report has a UUID. KnightCode shows it after upload or export and records it in the session as a `knightcode.bug-report` entry so you can refer to it later.
|
|
198
|
+
|
|
199
|
+
Set `KNIGHTCODE_RADIUS_GATEWAY` to upload to a different Radius deployment.
|
|
200
|
+
|
|
201
|
+
### Crashes
|
|
202
|
+
|
|
203
|
+
When KnightCode exits because of an uncaught exception or a fatal runtime error, it stores the error message and stack trace in `~/.knightcode/agent/crashes.json` (the newest five). The next interactive start shows a warning once; running `/bug` attaches the stored crashes to `diagnostics.json` and removes the file after the report is uploaded or exported. Resume the crashed session with `knightcode -r` first if you want the transcript in the report.
|
|
204
|
+
|
|
178
205
|
## Session Format
|
|
179
206
|
|
|
180
207
|
Session files are JSONL and contain message entries, model changes, thinking-level changes, labels, compactions, branch summaries, and extension entries.
|
package/bin/docs/settings.md
CHANGED
|
@@ -32,8 +32,31 @@ Use `/trust` in interactive mode to save a project trust decision for future ses
|
|
|
32
32
|
| `defaultThinkingLevel` | string | - | Startup thinking level (saved with Ctrl+S in `/thinking`, or edited manually): `"off"`, `"minimal"`, `"low"`, `"medium"`, `"high"`, `"xhigh"`, `"max"` |
|
|
33
33
|
| `modelThinkingLevels` | object | - | Per-model startup thinking levels keyed by `"provider/modelId"`; configure from `/settings` → Default thinking level per model or edit manually |
|
|
34
34
|
| `hideThinkingBlock` | boolean | `false` | Hide thinking blocks in output |
|
|
35
|
-
| `showCacheMissNotices` | boolean | `false` | Show transcript notices for significant prompt-cache misses, compaction or branch-summary usage, and provider recovery diagnostics such as dropped Anthropic thinking blocks |
|
|
35
|
+
| `showCacheMissNotices` | boolean | `false` | Show transcript notices for significant prompt-cache misses, successful cache-warming usage, compaction or branch-summary usage, and provider recovery diagnostics such as dropped Anthropic thinking blocks |
|
|
36
36
|
| `thinkingBudgets` | object | - | Custom token budgets per thinking level. Anthropic, Google, and Bedrock use these natively. OpenAI-compatible models use them when `compat.thinkingTokenBudgetField` (or `supportsThinkingTokenBudget`) is set. |
|
|
37
|
+
| `cacheWarming` | string | `"streaming"` | Prompt cache-warming mode: `"off"`, `"streaming"`, or `"idle"`. Global setting only. |
|
|
38
|
+
|
|
39
|
+
#### Cache Warming
|
|
40
|
+
|
|
41
|
+
Providers drop a prompt cache entry after a period of inactivity, so the first request after a pause pays full input price again. Cache warming re-sends the last request with a one-token output budget shortly before expiry:
|
|
42
|
+
|
|
43
|
+
- `"off"` disables warming.
|
|
44
|
+
- `"streaming"` protects expensive prefixes during long tool executions and stops as soon as the agent settles.
|
|
45
|
+
- `"idle"` also considers refreshes while waiting for your next prompt, using a fixed 15% continuation probability measured from real usage.
|
|
46
|
+
|
|
47
|
+
```json
|
|
48
|
+
{
|
|
49
|
+
"cacheWarming": "idle"
|
|
50
|
+
}
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
A refresh is sent only when the expected avoided cache-miss cost, minus the cost of the refresh, leaves at least $0.05 of expected savings. Active agent runs use 100% continuation probability. `/session` shows the next decision, continuation probability, expected savings, threshold, and estimated costs. When cache miss notices are enabled, each successful refresh appears in the transcript with its cost; notices identify extension overrides.
|
|
54
|
+
|
|
55
|
+
Warming stops when the context changes (model switch, compaction, branch navigation). Idle warming stops no later than 30 minutes after the last real provider request; warming during an active agent run stops after 60 minutes. Extensions can override each decision through the [`cache_warming_decision`](extensions.md#cache_warming_decision) event.
|
|
56
|
+
|
|
57
|
+
Each refresh is billed as a cache read of the full context plus one output token. Usage and cost show up in session totals but never enter model context. KnightCode schedules candidates at 90% of the cache lifetime while leaving at least ten seconds before expiry.
|
|
58
|
+
|
|
59
|
+
Warming needs a known cache lifetime for the model and the retention tier the request used (`short`, or `long` with `KNIGHTCODE_CACHE_RETENTION=long`). The built-in catalog carries lifetimes for direct Anthropic; custom models and other providers can declare theirs with `promptCache` in `models.json` (see [Prompt Cache Lifetimes](models.md#prompt-cache-lifetimes)). Claude models that use budget-based rather than adaptive thinking are skipped while thinking is on, because Anthropic derives the thinking budget from `max_tokens` and keys the message cache on it, so a one-token request cannot reproduce the entry.
|
|
37
60
|
|
|
38
61
|
#### thinkingBudgets
|
|
39
62
|
|
package/bin/docs/usage.md
CHANGED
|
@@ -57,6 +57,7 @@ Type `/` in the editor to open command completion. Extensions can register custo
|
|
|
57
57
|
| `/export [file]` | Export session to HTML or JSONL |
|
|
58
58
|
| `/import <file>` | Import and resume a session from a JSONL file |
|
|
59
59
|
| `/share` | Upload as private GitHub gist with shareable HTML link |
|
|
60
|
+
| `/bug [description]` | Report a bug to the KnightCode developers; see [Sessions](sessions.md#reporting-bugs) |
|
|
60
61
|
| `/reload` | Reload keybindings, extensions, skills, prompts, themes, and context files |
|
|
61
62
|
| `/hotkeys` | Show all keyboard shortcuts |
|
|
62
63
|
| `/changelog` | Display version history |
|
package/bin/knightcode
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
1
|
[diffend] Oversized file quarantined before diffing.
|
|
2
2
|
name: package/bin/knightcode
|
|
3
|
-
size:
|
|
4
|
-
sha256:
|
|
3
|
+
size: 117581290 bytes
|
|
4
|
+
sha256: 55b9ab58d9aa29d2bc21f39a279a44e92cb0f78d4b37bdd91f7907734d5a6d4b
|
package/bin/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@knightcodeai/cli",
|
|
3
|
-
"version": "0.9.
|
|
3
|
+
"version": "0.9.1",
|
|
4
4
|
"description": "KnightCode — a local, BYOK terminal coding agent powered by OpenRouter.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"repository": {
|
|
@@ -37,11 +37,11 @@
|
|
|
37
37
|
"test": "vitest --run"
|
|
38
38
|
},
|
|
39
39
|
"optionalDependencies": {
|
|
40
|
-
"@knightcodeai/cli-linux-x64": "0.9.
|
|
41
|
-
"@knightcodeai/cli-linux-arm64": "0.9.
|
|
42
|
-
"@knightcodeai/cli-darwin-x64": "0.9.
|
|
43
|
-
"@knightcodeai/cli-darwin-arm64": "0.9.
|
|
44
|
-
"@knightcodeai/cli-win32-x64": "0.9.
|
|
40
|
+
"@knightcodeai/cli-linux-x64": "0.9.1",
|
|
41
|
+
"@knightcodeai/cli-linux-arm64": "0.9.1",
|
|
42
|
+
"@knightcodeai/cli-darwin-x64": "0.9.1",
|
|
43
|
+
"@knightcodeai/cli-darwin-arm64": "0.9.1",
|
|
44
|
+
"@knightcodeai/cli-win32-x64": "0.9.1"
|
|
45
45
|
},
|
|
46
46
|
"devDependencies": {
|
|
47
47
|
"@agentclientprotocol/sdk": "1.4.0",
|