@knightcodeai/cli-linux-x64 0.9.1 → 0.9.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/CHANGELOG.md +50 -0
- package/bin/README.md +52 -19
- package/bin/docs/cli-integration.md +106 -0
- package/bin/docs/cli.md +270 -0
- package/bin/docs/compaction.md +56 -37
- package/bin/docs/configuration.md +46 -0
- package/bin/docs/containerization.md +86 -54
- package/bin/docs/custom-provider.md +132 -785
- package/bin/docs/docs.json +143 -103
- package/bin/docs/environment-variables.md +5 -4
- package/bin/docs/extensions.md +134 -2956
- package/bin/docs/how-knightcode-works.md +49 -0
- package/bin/docs/index.md +24 -69
- package/bin/docs/json.md +193 -65
- package/bin/docs/keybindings.md +56 -101
- package/bin/docs/llama-cpp.md +3 -3
- package/bin/docs/message-types.md +261 -0
- package/bin/docs/models.md +64 -547
- package/bin/docs/packages.md +66 -167
- package/bin/docs/prompt-templates.md +31 -68
- package/bin/docs/providers.md +103 -241
- package/bin/docs/quickstart.md +61 -106
- package/bin/docs/rpc-commands.md +854 -0
- package/bin/docs/rpc-extension-ui.md +200 -0
- package/bin/docs/rpc.md +129 -1556
- package/bin/docs/sdk.md +76 -1160
- package/bin/docs/security.md +70 -32
- package/bin/docs/session-format.md +25 -216
- package/bin/docs/sessions.md +38 -143
- package/bin/docs/settings.md +111 -389
- package/bin/docs/shell-aliases.md +85 -5
- package/bin/docs/skills.md +51 -189
- package/bin/docs/slash-commands.md +63 -0
- package/bin/docs/terminal-setup.md +107 -79
- package/bin/docs/termux.md +74 -83
- package/bin/docs/themes.md +68 -280
- package/bin/docs/tmux.md +31 -39
- package/bin/docs/tui.md +69 -923
- package/bin/docs/usage.md +79 -286
- package/bin/docs/windows.md +43 -17
- package/bin/export-html/template.js +6 -1
- package/bin/knightcode +2 -2
- package/bin/package.json +6 -6
- package/package.json +1 -1
- package/bin/docs/development.md +0 -71
package/bin/docs/models.md
CHANGED
|
@@ -1,606 +1,123 @@
|
|
|
1
|
-
#
|
|
1
|
+
# Choose a Model
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
For a built-in provider, start with `/login`, then choose a model with `/model`. Use custom model configuration only when KnightCode does not already include the provider or endpoint you need.
|
|
4
4
|
|
|
5
|
-
##
|
|
5
|
+
## Choose a connection
|
|
6
6
|
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
-
|
|
13
|
-
|
|
14
|
-
- [Per-model Overrides](#per-model-overrides)
|
|
15
|
-
- [Anthropic Messages Compatibility](#anthropic-messages-compatibility)
|
|
16
|
-
- [OpenAI Compatibility](#openai-compatibility)
|
|
7
|
+
| What you have | Recommended setup |
|
|
8
|
+
|---|---|
|
|
9
|
+
| A supported subscription | Sign in through `/login` |
|
|
10
|
+
| A provider API key | Store it through `/login` or set its environment variable |
|
|
11
|
+
| A local GGUF model | Connect KnightCode to the llama.cpp router |
|
|
12
|
+
| An OpenAI-, Anthropic-, or Google-compatible endpoint | Add it to `models.json` |
|
|
13
|
+
| A provider with a custom protocol or authentication flow | Build or install a provider extension |
|
|
17
14
|
|
|
18
|
-
|
|
15
|
+
Browse the [model catalog](https://knightcode.dev/models) for current providers, model IDs, capabilities, context limits, and pricing. KnightCode starts with its bundled catalog and can overlay newer catalog data from knightcode.dev. Cached catalog data remains available offline; run `knightcode update --models` to force a refresh.
|
|
19
16
|
|
|
20
|
-
|
|
17
|
+
## Authenticate
|
|
21
18
|
|
|
22
|
-
|
|
23
|
-
{
|
|
24
|
-
"providers": {
|
|
25
|
-
"ollama": {
|
|
26
|
-
"baseUrl": "http://localhost:11434/v1",
|
|
27
|
-
"api": "openai-completions",
|
|
28
|
-
"apiKey": "ollama",
|
|
29
|
-
"models": [
|
|
30
|
-
{ "id": "llama3.1:8b" },
|
|
31
|
-
{ "id": "qwen2.5-coder:7b" }
|
|
32
|
-
]
|
|
33
|
-
}
|
|
34
|
-
}
|
|
35
|
-
}
|
|
36
|
-
```
|
|
37
|
-
|
|
38
|
-
The `apiKey` value is a placeholder because Ollama ignores it. knightcode still treats models as requiring auth before they appear in `/model`, so keyless local servers should keep a dummy value, save a key for that provider with `/login`, or pass `--api-key` when selecting the model.
|
|
39
|
-
|
|
40
|
-
Some OpenAI-compatible servers do not understand the `developer` role used for reasoning-capable models. For those providers, set `compat.supportsDeveloperRole` to `false` so knightcode sends the system prompt as a `system` message instead. If the server also does not support `reasoning_effort`, set `compat.supportsReasoningEffort` to `false` too.
|
|
41
|
-
|
|
42
|
-
You can set `compat` at the provider level to apply to all models, or at the model level to override a specific model. This commonly applies to Ollama, vLLM, SGLang, and similar OpenAI-compatible servers.
|
|
43
|
-
|
|
44
|
-
```json
|
|
45
|
-
{
|
|
46
|
-
"providers": {
|
|
47
|
-
"ollama": {
|
|
48
|
-
"baseUrl": "http://localhost:11434/v1",
|
|
49
|
-
"api": "openai-completions",
|
|
50
|
-
"apiKey": "ollama",
|
|
51
|
-
"compat": {
|
|
52
|
-
"supportsDeveloperRole": false,
|
|
53
|
-
"supportsReasoningEffort": false
|
|
54
|
-
},
|
|
55
|
-
"models": [
|
|
56
|
-
{
|
|
57
|
-
"id": "gpt-oss:20b",
|
|
58
|
-
"reasoning": true
|
|
59
|
-
}
|
|
60
|
-
]
|
|
61
|
-
}
|
|
62
|
-
}
|
|
63
|
-
}
|
|
64
|
-
```
|
|
65
|
-
|
|
66
|
-
## Full Example
|
|
67
|
-
|
|
68
|
-
Override defaults when you need specific values:
|
|
69
|
-
|
|
70
|
-
```json
|
|
71
|
-
{
|
|
72
|
-
"providers": {
|
|
73
|
-
"ollama": {
|
|
74
|
-
"baseUrl": "http://localhost:11434/v1",
|
|
75
|
-
"api": "openai-completions",
|
|
76
|
-
"apiKey": "ollama",
|
|
77
|
-
"models": [
|
|
78
|
-
{
|
|
79
|
-
"id": "llama3.1:8b",
|
|
80
|
-
"name": "Llama 3.1 8B (Local)",
|
|
81
|
-
"reasoning": false,
|
|
82
|
-
"input": ["text"],
|
|
83
|
-
"contextWindow": 128000,
|
|
84
|
-
"maxTokens": 32000,
|
|
85
|
-
"cost": { "input": 0, "output": 0, "cacheRead": 0, "cacheWrite": 0 }
|
|
86
|
-
}
|
|
87
|
-
]
|
|
88
|
-
}
|
|
89
|
-
}
|
|
90
|
-
}
|
|
91
|
-
```
|
|
92
|
-
|
|
93
|
-
The file reloads each time you open `/model`. Edit during session; no restart needed.
|
|
19
|
+
Run `/login` and select a provider. KnightCode stores credentials in [`auth.json`](configuration.md#agent-directory). Run `/logout` to remove stored credentials for a provider.
|
|
94
20
|
|
|
95
|
-
|
|
21
|
+
You can instead provide an API key through the provider's environment variable. This is useful in CI and other environments where KnightCode should not write credentials. [Provider Authentication](providers.md) lists the variables and cloud-provider setup.
|
|
96
22
|
|
|
97
|
-
|
|
23
|
+
When several credential sources are configured, KnightCode uses a runtime `--api-key` first, then a stored `auth.json` credential, an `apiKey` from `models.json`, and finally the provider's environment variables or ambient cloud credentials. Provider extensions can define their own authentication behavior.
|
|
98
24
|
|
|
99
|
-
|
|
100
|
-
{
|
|
101
|
-
"providers": {
|
|
102
|
-
"my-google": {
|
|
103
|
-
"baseUrl": "https://generativelanguage.googleapis.com/v1beta",
|
|
104
|
-
"api": "google-generative-ai",
|
|
105
|
-
"apiKey": "$GEMINI_API_KEY",
|
|
106
|
-
"models": [
|
|
107
|
-
{
|
|
108
|
-
"id": "gemma-4-31b-it",
|
|
109
|
-
"name": "Gemma 4 31B",
|
|
110
|
-
"input": ["text", "image"],
|
|
111
|
-
"contextWindow": 262144,
|
|
112
|
-
"reasoning": true
|
|
113
|
-
}
|
|
114
|
-
]
|
|
115
|
-
}
|
|
116
|
-
}
|
|
117
|
-
}
|
|
118
|
-
```
|
|
119
|
-
|
|
120
|
-
The `baseUrl` is required when adding custom models to the `google-generative-ai` API type.
|
|
121
|
-
|
|
122
|
-
## Supported APIs
|
|
123
|
-
|
|
124
|
-
| API | Description |
|
|
125
|
-
|-----|-------------|
|
|
126
|
-
| `openai-completions` | OpenAI Chat Completions (most compatible) |
|
|
127
|
-
| `openai-responses` | OpenAI Responses API |
|
|
128
|
-
| `anthropic-messages` | Anthropic Messages API |
|
|
129
|
-
| `google-generative-ai` | Google Generative AI |
|
|
25
|
+
Keep `auth.json` and any credential commands private. Project settings and extensions can execute inside the KnightCode process after you trust a project. Review [Security](security.md) before loading configuration from an untrusted directory.
|
|
130
26
|
|
|
131
|
-
|
|
27
|
+
## Select a model
|
|
132
28
|
|
|
133
|
-
|
|
29
|
+
Run `/model` to search available models. The picker shows models whose providers have usable authentication. Press `Ctrl+S` (or `Ctrl+D`, if your terminal swallows `Ctrl+S`) on a model to save it as the default for new sessions.
|
|
134
30
|
|
|
135
|
-
|
|
136
|
-
|-------|-------------|
|
|
137
|
-
| `baseUrl` | API endpoint URL |
|
|
138
|
-
| `api` | API type (see above) |
|
|
139
|
-
| `apiKey` | Optional API key config (see value resolution below). Omit it when auth is provided by `/login`/`auth.json` or CLI `--api-key`. |
|
|
140
|
-
| `oauth` | Dynamic OAuth provider type. Currently supports `"radius"`; requires the gateway `baseUrl`. |
|
|
141
|
-
| `headers` | Custom headers (see value resolution below) |
|
|
142
|
-
| `authHeader` | Set `true` to add `Authorization: Bearer <apiKey>` automatically |
|
|
143
|
-
| `models` | Array of model configurations |
|
|
144
|
-
| `modelOverrides` | Per-model overrides for built-in or extension-registered models on this provider |
|
|
31
|
+
Run `/thinking` to select the thinking level for the current model. Press `Ctrl+S` or `Ctrl+D` there to save the startup level. KnightCode limits the choices to levels supported by the selected model.
|
|
145
32
|
|
|
146
|
-
|
|
33
|
+
`Ctrl+P` cycles through available models. Use `/scoped-models` to control that cycle and save the selection, or configure model patterns through [Settings](settings.md#model-cycling).
|
|
147
34
|
|
|
148
|
-
|
|
35
|
+
A session records model and thinking-level changes. Resuming the session restores them without changing defaults for new sessions.
|
|
149
36
|
|
|
150
|
-
|
|
37
|
+
## Connect local models
|
|
151
38
|
|
|
152
|
-
|
|
153
|
-
```json
|
|
154
|
-
"apiKey": "!security find-generic-password -ws 'anthropic'"
|
|
155
|
-
"apiKey": "!op read 'op://vault/item/credential'"
|
|
156
|
-
```
|
|
157
|
-
- **Environment interpolation:** `"$ENV_VAR"` or `"${ENV_VAR}"` uses the value of the named variable. Interpolation works inside larger literals.
|
|
158
|
-
```json
|
|
159
|
-
"apiKey": "$MY_API_KEY"
|
|
160
|
-
"apiKey": "${KEY_PREFIX}_${KEY_SUFFIX}"
|
|
161
|
-
```
|
|
162
|
-
`$FOO_BAR` is the variable `FOO_BAR`; use `${FOO}_BAR` when `BAR` is literal text. Missing environment variables make the value unresolved.
|
|
163
|
-
- **Escapes:** `"$$"` emits a literal `"$"`; `"$!"` emits a literal `"!"` without triggering command execution.
|
|
164
|
-
```json
|
|
165
|
-
"apiKey": "$$literal-dollar-prefix"
|
|
166
|
-
"apiKey": "$!literal-bang-prefix"
|
|
167
|
-
```
|
|
168
|
-
- **Literal value:** Used directly. Plain uppercase strings such as `MY_API_KEY` are literals; use `$MY_API_KEY` for environment variables.
|
|
169
|
-
```json
|
|
170
|
-
"apiKey": "sk-..."
|
|
171
|
-
```
|
|
39
|
+
KnightCode integrates directly with the llama.cpp router. The router discovers GGUF files and loads models on demand. KnightCode's `/llama` command manages the router, while `/model` selects one of its loaded models.
|
|
172
40
|
|
|
173
|
-
|
|
41
|
+
Follow [Local Models with llama.cpp](llama-cpp.md) for server startup, model layout, downloads, and connection troubleshooting.
|
|
174
42
|
|
|
175
|
-
|
|
43
|
+
For Ollama, LM Studio, vLLM, SGLang, and other compatible servers, [configure a compatible endpoint](#configure-a-compatible-endpoint) in `models.json`.
|
|
176
44
|
|
|
177
|
-
|
|
45
|
+
## Configure a compatible endpoint
|
|
178
46
|
|
|
179
|
-
|
|
47
|
+
Use [`models.json`](configuration.md#agent-directory) when an endpoint speaks an API KnightCode already supports. This includes most Ollama, LM Studio, vLLM, SGLang, and proxy deployments.
|
|
180
48
|
|
|
181
49
|
```json
|
|
182
50
|
{
|
|
183
51
|
"providers": {
|
|
184
|
-
"
|
|
185
|
-
"baseUrl": "
|
|
186
|
-
"
|
|
187
|
-
"
|
|
188
|
-
"
|
|
189
|
-
"
|
|
190
|
-
|
|
191
|
-
},
|
|
192
|
-
"models": [...]
|
|
193
|
-
}
|
|
194
|
-
}
|
|
195
|
-
}
|
|
196
|
-
```
|
|
197
|
-
|
|
198
|
-
## Model Configuration
|
|
199
|
-
|
|
200
|
-
| Field | Required | Default | Description |
|
|
201
|
-
|-------|----------|---------|-------------|
|
|
202
|
-
| `id` | Yes | — | Model identifier (passed to the API) |
|
|
203
|
-
| `name` | No | `id` | Human-readable model label. Used for matching (`--model` patterns) and shown as secondary model detail text. |
|
|
204
|
-
| `api` | No | provider's `api` | Override provider's API for this model |
|
|
205
|
-
| `reasoning` | No | `false` | Supports extended thinking |
|
|
206
|
-
| `thinkingLevelMap` | No | omitted | Maps knightcode thinking levels to provider values and marks unsupported levels (see below) |
|
|
207
|
-
| `input` | No | `["text"]` | Input types: `["text"]` or `["text", "image"]` |
|
|
208
|
-
| `contextWindow` | No | `128000` | Context window size in tokens |
|
|
209
|
-
| `maxTokens` | No | `16384` | Maximum output tokens |
|
|
210
|
-
| `samplingParams` | No | omitted | Sampling parameters merged verbatim into every request body (see below) |
|
|
211
|
-
| `cost` | No | all zeros | Per-million-token rates with optional request-wide input pricing tiers |
|
|
212
|
-
| `promptCache` | No | omitted | Best-effort prompt cache lifetime in seconds per retention tier (see below) |
|
|
213
|
-
| `compat` | No | provider `compat` | Provider compatibility overrides. Merged with provider-level `compat` when both are set. |
|
|
214
|
-
|
|
215
|
-
A cost tier supplies a complete alternate rate set and applies to the full request when total input usage (`input + cacheRead + cacheWrite`) exceeds `inputTokensAbove`. When multiple tiers match, the highest threshold wins.
|
|
216
|
-
|
|
217
|
-
```json
|
|
218
|
-
{
|
|
219
|
-
"cost": {
|
|
220
|
-
"input": 5,
|
|
221
|
-
"output": 30,
|
|
222
|
-
"cacheRead": 0.5,
|
|
223
|
-
"cacheWrite": 6.25,
|
|
224
|
-
"tiers": [
|
|
225
|
-
{
|
|
226
|
-
"inputTokensAbove": 272000,
|
|
227
|
-
"input": 10,
|
|
228
|
-
"output": 45,
|
|
229
|
-
"cacheRead": 1,
|
|
230
|
-
"cacheWrite": 12.5
|
|
231
|
-
}
|
|
232
|
-
]
|
|
233
|
-
}
|
|
234
|
-
}
|
|
235
|
-
```
|
|
236
|
-
|
|
237
|
-
Current behavior:
|
|
238
|
-
- `/model`, `--list-models`, and the interactive footer display entries by model `id`.
|
|
239
|
-
- The configured `name` is used for model matching and secondary model detail text. It does not replace the footer/status-bar model id.
|
|
240
|
-
|
|
241
|
-
### Prompt Cache Lifetimes
|
|
242
|
-
|
|
243
|
-
`promptCache` states how long the provider keeps a prompt cache entry alive for each retention tier KnightCode can request (`short` is the default tier; `long` is used when `KNIGHTCODE_CACHE_RETENTION=long`). Values are seconds and are estimates: providers publish ranges, so pick the conservative end.
|
|
244
|
-
|
|
245
|
-
```json
|
|
246
|
-
{
|
|
247
|
-
"id": "claude-sonnet-5",
|
|
248
|
-
"promptCache": { "short": 300, "long": 3600 }
|
|
249
|
-
}
|
|
250
|
-
```
|
|
251
|
-
|
|
252
|
-
The built-in catalog fills this in for direct Anthropic (5 min / 1 h). Other providers, including direct OpenAI, have no built-in lifetime until their cache-expiry and replay behavior has been validated for warming. A model without a value for the tier a request used is never warmed; custom models and provider overrides can opt in when the backing cache behavior is known. See [Cache Warming](settings.md#cache-warming).
|
|
253
|
-
|
|
254
|
-
### Sampling Parameters
|
|
255
|
-
|
|
256
|
-
`samplingParams` is a free-form object merged verbatim into every request body for the model, after the fields knightcode sets itself, so its keys win. Use it to send sampling parameters knightcode does not model — including server-specific ones like llama.cpp's `min_p` or vLLM's `top_k`:
|
|
257
|
-
|
|
258
|
-
```json
|
|
259
|
-
{
|
|
260
|
-
"id": "deepseek-v4-flash",
|
|
261
|
-
"samplingParams": {
|
|
262
|
-
"temperature": 1.0,
|
|
263
|
-
"top_p": 0.95,
|
|
264
|
-
"top_k": 0,
|
|
265
|
-
"min_p": 0.0
|
|
266
|
-
}
|
|
267
|
-
}
|
|
268
|
-
```
|
|
269
|
-
|
|
270
|
-
Only OpenAI-compatible APIs apply it (`openai-completions`, `openai-responses`, `azure-openai-responses`); other APIs ignore it. Keys override knightcode's named request fields (for example a `temperature` key here beats the request-level temperature), so prefer it as the single source of sampling truth for a model. In `modelOverrides`, `samplingParams` merges per key with the base model's value.
|
|
271
|
-
|
|
272
|
-
A constant thinking-token cap can go here too, but it will not follow `thinkingBudgets` or leave room for the answer. Prefer `compat.thinkingTokenBudgetField` (or the `supportsThinkingTokenBudget` alias) for that.
|
|
273
|
-
|
|
274
|
-
### Thinking Level Map
|
|
275
|
-
|
|
276
|
-
Use `thinkingLevelMap` on a model to describe model-specific thinking controls. Keys are knightcode thinking levels: `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, `max`. Maps may contain holes; for example, a model can expose `high` and `max` without exposing `xhigh`.
|
|
277
|
-
|
|
278
|
-
Values are tristate:
|
|
279
|
-
|
|
280
|
-
| Value | Meaning |
|
|
281
|
-
|-------|---------|
|
|
282
|
-
| omitted | Standard levels through `high` use the provider's default mapping; extended `xhigh` and `max` levels are unsupported |
|
|
283
|
-
| string | Level is supported and this value is sent to the provider |
|
|
284
|
-
| `null` | Level is unsupported and hidden/skipped/clamped away |
|
|
285
|
-
|
|
286
|
-
Example for a model that only supports off, high, and max reasoning:
|
|
287
|
-
|
|
288
|
-
```json
|
|
289
|
-
{
|
|
290
|
-
"id": "deepseek-v4-pro",
|
|
291
|
-
"reasoning": true,
|
|
292
|
-
"thinkingLevelMap": {
|
|
293
|
-
"minimal": null,
|
|
294
|
-
"low": null,
|
|
295
|
-
"medium": null,
|
|
296
|
-
"high": "high",
|
|
297
|
-
"xhigh": null,
|
|
298
|
-
"max": "max"
|
|
299
|
-
}
|
|
300
|
-
}
|
|
301
|
-
```
|
|
302
|
-
|
|
303
|
-
Example for a model where thinking cannot be disabled:
|
|
304
|
-
|
|
305
|
-
```json
|
|
306
|
-
{
|
|
307
|
-
"id": "always-thinking-model",
|
|
308
|
-
"reasoning": true,
|
|
309
|
-
"thinkingLevelMap": {
|
|
310
|
-
"off": null
|
|
311
|
-
}
|
|
312
|
-
}
|
|
313
|
-
```
|
|
314
|
-
|
|
315
|
-
Migration: older configs that used `compat.reasoningEffortMap` should move that mapping to model-level `thinkingLevelMap`. Use `null` for levels that should not appear in the UI.
|
|
316
|
-
|
|
317
|
-
## Overriding Built-in Providers
|
|
318
|
-
|
|
319
|
-
Route a built-in provider through a proxy without redefining models:
|
|
320
|
-
|
|
321
|
-
```json
|
|
322
|
-
{
|
|
323
|
-
"providers": {
|
|
324
|
-
"anthropic": {
|
|
325
|
-
"baseUrl": "https://my-proxy.example.com/v1"
|
|
52
|
+
"ollama": {
|
|
53
|
+
"baseUrl": "http://localhost:11434/v1",
|
|
54
|
+
"api": "openai-completions",
|
|
55
|
+
"apiKey": "ollama",
|
|
56
|
+
"models": [
|
|
57
|
+
{ "id": "qwen2.5-coder:7b" }
|
|
58
|
+
]
|
|
326
59
|
}
|
|
327
60
|
}
|
|
328
61
|
}
|
|
329
62
|
```
|
|
330
63
|
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
To merge custom models into a built-in provider, include the `models` array:
|
|
334
|
-
|
|
335
|
-
```json
|
|
336
|
-
{
|
|
337
|
-
"providers": {
|
|
338
|
-
"anthropic": {
|
|
339
|
-
"baseUrl": "https://my-proxy.example.com/v1",
|
|
340
|
-
"apiKey": "$ANTHROPIC_API_KEY",
|
|
341
|
-
"api": "anthropic-messages",
|
|
342
|
-
"models": [...]
|
|
343
|
-
}
|
|
344
|
-
}
|
|
345
|
-
}
|
|
346
|
-
```
|
|
64
|
+
The dummy key makes the model available to KnightCode; Ollama ignores it. For an authenticated endpoint, `apiKey` and header values can use `$NAME` or `${NAME}` environment interpolation, a literal value, or a leading `!command`. Commands in `models.json` run at request time and are not cached by KnightCode.
|
|
347
65
|
|
|
348
|
-
|
|
349
|
-
- Built-in models are kept.
|
|
350
|
-
- Custom models are upserted by `id` within the provider.
|
|
351
|
-
- If a custom model `id` matches a built-in model `id`, the custom model replaces that built-in model.
|
|
352
|
-
- If a custom model `id` is new, it is added alongside built-in models.
|
|
66
|
+
Opening `/model` reloads the file. A `models` entry adds or replaces a model with the same ID on that provider. Use `modelOverrides` to change metadata for an existing built-in or extension-provided model without replacing the provider's model list. Unknown override IDs are ignored.
|
|
353
67
|
|
|
354
|
-
|
|
68
|
+
### Describe model input and caching
|
|
355
69
|
|
|
356
|
-
Use `
|
|
70
|
+
Use `inputLimits.images.resize` to control how KnightCode encodes new image attachments, `read` results, and tool-result images before storing them in conversation history:
|
|
357
71
|
|
|
358
72
|
```json
|
|
359
73
|
{
|
|
360
|
-
"
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
}
|
|
370
|
-
}
|
|
74
|
+
"id": "vision-model",
|
|
75
|
+
"input": ["text", "image"],
|
|
76
|
+
"inputLimits": {
|
|
77
|
+
"images": {
|
|
78
|
+
"resize": {
|
|
79
|
+
"maxWidth": 1568,
|
|
80
|
+
"maxHeight": 1568,
|
|
81
|
+
"maxBytes": 524288,
|
|
82
|
+
"jpegQuality": 75
|
|
371
83
|
}
|
|
372
84
|
}
|
|
373
85
|
}
|
|
374
86
|
}
|
|
375
87
|
```
|
|
376
88
|
|
|
377
|
-
`
|
|
378
|
-
|
|
379
|
-
Use a `promptCache` override to enable cache warming through a proxy whose backing cache you know, for example OpenRouter routed to Anthropic:
|
|
89
|
+
`maxBytes` limits the base64-encoded payload. Omitted resize fields use conservative defaults of 2000 by 2000 pixels, 4.5 MiB encoded, and JPEG quality 80. Images are encoded once; changing models does not rewrite historical images. The catalog can also describe hard request limits with `inputLimits.maxRequestBytes`, `images.maxPerMessage`, and `images.maxPerRequest`, but KnightCode does not yet rewrite or reject history based on them.
|
|
380
90
|
|
|
381
|
-
|
|
382
|
-
{
|
|
383
|
-
"providers": {
|
|
384
|
-
"openrouter": {
|
|
385
|
-
"modelOverrides": {
|
|
386
|
-
"anthropic/claude-sonnet-4": {
|
|
387
|
-
"promptCache": { "short": 300 }
|
|
388
|
-
}
|
|
389
|
-
}
|
|
390
|
-
}
|
|
391
|
-
}
|
|
392
|
-
}
|
|
393
|
-
```
|
|
91
|
+
<a id="prompt-cache-lifetimes"></a>
|
|
394
92
|
|
|
395
|
-
|
|
93
|
+
Use `promptCache` to declare the provider's best-effort cache lifetime in seconds for the `short` or `long` retention tier:
|
|
396
94
|
|
|
397
95
|
```json
|
|
398
|
-
{
|
|
399
|
-
"providers": {
|
|
400
|
-
"openai": {
|
|
401
|
-
"modelOverrides": {
|
|
402
|
-
"gpt-5.6-sol": {
|
|
403
|
-
"contextWindow": 1050000
|
|
404
|
-
}
|
|
405
|
-
}
|
|
406
|
-
}
|
|
407
|
-
}
|
|
408
|
-
}
|
|
96
|
+
{ "id": "claude-sonnet-5", "promptCache": { "short": 300, "long": 3600 } }
|
|
409
97
|
```
|
|
410
98
|
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
Behavior notes:
|
|
414
|
-
- `modelOverrides` are applied to built-in provider models and matching extension-registered provider models.
|
|
415
|
-
- Unknown model IDs are ignored.
|
|
416
|
-
- You can combine provider-level `baseUrl`/`headers` with `modelOverrides`.
|
|
417
|
-
- Overriding `name` changes model matching and secondary detail text only; the footer and primary model lists continue to show the model `id`.
|
|
418
|
-
- If `models` is also defined for a provider, custom models are merged after built-in overrides. A custom model with the same `id` replaces the overridden built-in model entry.
|
|
419
|
-
|
|
420
|
-
## Anthropic Messages Compatibility
|
|
99
|
+
Choose the conservative end of any published range. A model without a lifetime for the active tier is not eligible for cache warming. A `modelOverrides` entry can set `inputLimits` or `promptCache` for a built-in or extension model, including a model accessed through a validated proxy. See [`cacheWarming`](settings.md#model-and-thinking).
|
|
421
100
|
|
|
422
|
-
|
|
101
|
+
Compatibility settings should describe verified differences in the endpoint's request or response behavior. Do not enable them based only on an endpoint advertising OpenAI or Anthropic compatibility.
|
|
423
102
|
|
|
424
|
-
|
|
103
|
+
## Add a custom provider
|
|
425
104
|
|
|
426
|
-
|
|
105
|
+
Use an extension when the provider needs custom streaming, model discovery, or authentication behavior. See [Custom Providers](custom-provider.md) for the extension workflow.
|
|
427
106
|
|
|
428
|
-
|
|
107
|
+
## Troubleshooting
|
|
429
108
|
|
|
430
|
-
|
|
109
|
+
### A model does not appear
|
|
431
110
|
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
```json
|
|
435
|
-
{
|
|
436
|
-
"providers": {
|
|
437
|
-
"anthropic-proxy": {
|
|
438
|
-
"baseUrl": "https://proxy.example.com",
|
|
439
|
-
"api": "anthropic-messages",
|
|
440
|
-
"apiKey": "$ANTHROPIC_PROXY_KEY",
|
|
441
|
-
"compat": {
|
|
442
|
-
"supportsEagerToolInputStreaming": false,
|
|
443
|
-
"supportsLongCacheRetention": true,
|
|
444
|
-
"forceAdaptiveThinking": true,
|
|
445
|
-
"allowEmptySignature": true
|
|
446
|
-
},
|
|
447
|
-
"models": [
|
|
448
|
-
{
|
|
449
|
-
"id": "claude-opus-4-7",
|
|
450
|
-
"reasoning": true,
|
|
451
|
-
"input": ["text", "image"]
|
|
452
|
-
}
|
|
453
|
-
]
|
|
454
|
-
}
|
|
455
|
-
}
|
|
456
|
-
}
|
|
457
|
-
```
|
|
111
|
+
Confirm that its provider has usable authentication. Custom models can load from `models.json` but remain unavailable in `/model` until KnightCode can resolve credentials. For llama.cpp, only models currently loaded by the router appear.
|
|
458
112
|
|
|
459
|
-
|
|
460
|
-
|-------|-------------|
|
|
461
|
-
| `supportsEagerToolInputStreaming` | Whether the provider accepts per-tool `eager_input_streaming`. Default: `true`. Set to `false` to omit that field and use the legacy fine-grained tool streaming beta header on tool-enabled requests. |
|
|
462
|
-
| `supportsLongCacheRetention` | Whether the provider accepts Anthropic long cache retention (`cache_control.ttl: "1h"`) when cache retention is `long`. Default: `true`. |
|
|
463
|
-
| `sendSessionAffinityHeaders` | Whether to send a session-affinity header from the session id when caching is enabled. Default: `true` for OpenRouter endpoints, auto-detected for other known providers. |
|
|
464
|
-
| `sessionAffinityFormat` | Session-affinity header name: `openrouter` sends `x-session-id`. When unset, `x-session-affinity` is sent. Default: `openrouter` for OpenRouter endpoints. |
|
|
465
|
-
| `supportsCacheControlOnTools` | Whether the provider accepts Anthropic-style `cache_control` markers on tool definitions. Default: `true`. |
|
|
466
|
-
| `forceAdaptiveThinking` | Whether to send adaptive thinking (`thinking.type: "adaptive"` plus `output_config.effort`) for this model. Built-in adaptive models set this automatically. Default: `false`. |
|
|
467
|
-
| `supportsMidConvoEffort` | Whether the exact Claude model transport supports per-turn effort system messages and thinking binding controls. KnightCode persists native effort levels and always sends `drop_block` when enabled. Default: `false`. |
|
|
468
|
-
| `allowEmptySignature` | Whether to replay empty thinking signatures as `signature: ""` instead of converting thinking to text. Default: `false`. |
|
|
469
|
-
| `supportsStrictTools` | Whether the provider accepts strict JSON-schema tool definitions. Default: `false`; built-in Anthropic models enable it in generated metadata. |
|
|
470
|
-
| `allowedFallbackModels` | Up to three server-side fallback models, each with `provider`, `model`, and complete `cost` metadata. An empty array disables fallback. |
|
|
113
|
+
### Authentication works in one shell only
|
|
471
114
|
|
|
472
|
-
|
|
115
|
+
Check whether the key came from an environment variable rather than `auth.json`. Environment variables must be present in the process that starts KnightCode.
|
|
473
116
|
|
|
474
|
-
|
|
117
|
+
### Sign-in opens a browser on a remote machine
|
|
475
118
|
|
|
476
|
-
|
|
477
|
-
- Model-level `compat` overrides provider-level values for that model.
|
|
119
|
+
Complete the provider's headless authentication flow when available. Some providers let you paste the final redirect URL or authorization code back into KnightCode. See [Authenticate interactively](providers.md#authenticate-interactively).
|
|
478
120
|
|
|
479
|
-
|
|
480
|
-
{
|
|
481
|
-
"providers": {
|
|
482
|
-
"local-llm": {
|
|
483
|
-
"baseUrl": "http://localhost:8080/v1",
|
|
484
|
-
"api": "openai-completions",
|
|
485
|
-
"compat": {
|
|
486
|
-
"supportsUsageInStreaming": false,
|
|
487
|
-
"maxTokensField": "max_tokens"
|
|
488
|
-
},
|
|
489
|
-
"models": [...]
|
|
490
|
-
}
|
|
491
|
-
}
|
|
492
|
-
}
|
|
493
|
-
```
|
|
121
|
+
### A compatible endpoint rejects requests
|
|
494
122
|
|
|
495
|
-
|
|
496
|
-
|-------|-------------|
|
|
497
|
-
| `supportsStore` | Provider supports `store` field |
|
|
498
|
-
| `supportsDeveloperRole` | Use `developer` vs `system` role |
|
|
499
|
-
| `supportsReasoningEffort` | Support for `reasoning_effort` parameter |
|
|
500
|
-
| `supportsUsageInStreaming` | Supports `stream_options: { include_usage: true }` (default: `true`) |
|
|
501
|
-
| `supportsFinishReason` | Whether streamed responses include `finish_reason`. When `false`, knightcode infers `stop` or `toolUse` when the stream ends. Default: `true`. |
|
|
502
|
-
| `maxTokensField` | Use `max_completion_tokens` or `max_tokens` |
|
|
503
|
-
| `requiresToolResultName` | Include `name` on tool result messages |
|
|
504
|
-
| `requiresAssistantAfterToolResult` | Insert an assistant message before a user message after tool results |
|
|
505
|
-
| `requiresThinkingAsText` | Convert thinking blocks to plain text |
|
|
506
|
-
| `requiresReasoningContentOnAssistantMessages` | Include empty `reasoning_content` on all replayed assistant messages when reasoning is enabled |
|
|
507
|
-
| `thinkingFormat` | Use `reasoning_effort`, `openrouter`, `deepseek`, `together`, `baseten`, `zai`, `qwen`, `chat-template`, or `qwen-chat-template` thinking parameters |
|
|
508
|
-
| `chatTemplateKwargs` | `chat_template_kwargs` values for `thinkingFormat: "chat-template"`; use `{ "$var": "thinking.enabled" }`, `{ "$var": "thinking.effort" }`, or `{ "$var": "thinking.budget" }` for knightcode-controlled thinking values |
|
|
509
|
-
| `chatTemplateArgs` | `chat_template_args` values for `thinkingFormat: "baseten"`; use `{ "$var": "thinking.enabled" }`, `{ "$var": "thinking.effort" }`, or `{ "$var": "thinking.budget" }` for knightcode-controlled thinking values |
|
|
510
|
-
| `thinkingTokenBudgetField` | Top-level request field used to cap reasoning tokens from `thinkingBudgets`, clamped so at least 1024 tokens remain for the answer. `"thinking_token_budget"` (vLLM), `"thinking_budget"` (Qwen/DashScope/SGLang), `"thinking_budget_tokens"` (llama.cpp). Off by default; not set on the generated catalog. |
|
|
511
|
-
| `supportsThinkingTokenBudget` | Alias for `thinkingTokenBudgetField: "thinking_token_budget"` (vLLM). Prefer `thinkingTokenBudgetField`. Default: `false`. |
|
|
512
|
-
| `cacheControlFormat` | Use Anthropic-style `cache_control` markers on the system prompt, last tool definition, and last user, assistant, or tool-result text content. Currently only `anthropic` is supported. |
|
|
513
|
-
| `sendSessionAffinityHeaders` | For `openai-completions`, send session-affinity headers from the session id when caching is enabled. Default: `true` for OpenRouter endpoints, `false` otherwise. |
|
|
514
|
-
| `sessionAffinityFormat` | For `openai-completions` and `openai-responses`, the session-affinity header format: `openai` sends `session_id`/`x-client-request-id` (completions also `x-session-affinity`), `openai-nosession` omits the underscore-containing `session_id` header, `openrouter` sends `x-session-id`. Does not affect the `prompt_cache_key` body param. Default: auto-detected. |
|
|
515
|
-
| `supportsStrictMode` | Whether the provider accepts strict JSON-schema function tool definitions. Defaults depend on the API; built-in OpenAI models carry explicit capability metadata. |
|
|
516
|
-
| `supportsOpenAIGrammarTools` | Whether OpenAI-compatible APIs emit custom Lark/regex grammar tools. When `false`, grammar-constrained tools fall back to normal function tools. Default: `false`; the built-in model catalog enables it for GPT-5+ models on OpenAI, OpenAI Codex, Azure OpenAI, GitHub Copilot, opencode, and Cloudflare AI Gateway. |
|
|
517
|
-
| `supportsLongCacheRetention` | Whether the provider accepts long cache retention when cache retention is `long`: `prompt_cache_options.ttl: "30m"` for GPT-5.6+ Responses models, `prompt_cache_retention: "24h"` for earlier OpenAI models, or `cache_control.ttl: "1h"` when `cacheControlFormat` is `anthropic`. Default: `true`. |
|
|
518
|
-
| `openRouterRouting` | OpenRouter provider routing preferences. This object is sent as-is in the `provider` field of the [OpenRouter API request](https://openrouter.ai/docs/guides/routing/provider-selection). |
|
|
519
|
-
| `vercelGatewayRouting` | Vercel AI Gateway routing config for provider selection (`only`, `order`) |
|
|
520
|
-
|
|
521
|
-
`openrouter` uses `reasoning: { effort }`. `together` uses `reasoning: { enabled }` and also `reasoning_effort` when `supportsReasoningEffort` is enabled. `qwen` uses top-level `enable_thinking`. Use `qwen-chat-template` for local Qwen-compatible servers that require `chat_template_kwargs.enable_thinking` and `preserve_thinking`. Use `chat-template` for vLLM/Hugging Face chat templates that need configurable `chat_template_kwargs`, such as `chatTemplateKwargs: { "thinking": { "$var": "thinking.enabled" } }` for DeepSeek V3.x templates. Use `thinkingFormat: "baseten"` with `chatTemplateArgs` for providers that expose toggle controls through `chat_template_args` and optionally support top-level `reasoning_effort`.
|
|
522
|
-
|
|
523
|
-
`thinkingTokenBudgetField` is independent of `thinkingFormat`. Do not enable it on the generated Qwen catalog: those models already send `reasoning_effort`, and DashScope rejects `thinking_budget` together with `reasoning_effort`.
|
|
524
|
-
|
|
525
|
-
`cacheControlFormat: "anthropic"` is for OpenAI-compatible providers that expose Anthropic-style prompt caching through `cache_control` markers on text content and tool definitions.
|
|
526
|
-
|
|
527
|
-
Example:
|
|
528
|
-
|
|
529
|
-
```json
|
|
530
|
-
{
|
|
531
|
-
"providers": {
|
|
532
|
-
"openrouter": {
|
|
533
|
-
"baseUrl": "https://openrouter.ai/api/v1",
|
|
534
|
-
"apiKey": "$OPENROUTER_API_KEY",
|
|
535
|
-
"api": "openai-completions",
|
|
536
|
-
"models": [
|
|
537
|
-
{
|
|
538
|
-
"id": "openrouter/anthropic/claude-3.5-sonnet",
|
|
539
|
-
"name": "OpenRouter Claude 3.5 Sonnet",
|
|
540
|
-
"compat": {
|
|
541
|
-
"openRouterRouting": {
|
|
542
|
-
"allow_fallbacks": true,
|
|
543
|
-
"require_parameters": false,
|
|
544
|
-
"data_collection": "deny",
|
|
545
|
-
"zdr": true,
|
|
546
|
-
"enforce_distillable_text": false,
|
|
547
|
-
"order": ["anthropic", "amazon-bedrock", "google-vertex"],
|
|
548
|
-
"only": ["anthropic", "amazon-bedrock"],
|
|
549
|
-
"ignore": ["gmicloud", "friendli"],
|
|
550
|
-
"quantizations": ["fp16", "bf16"],
|
|
551
|
-
"sort": {
|
|
552
|
-
"by": "price",
|
|
553
|
-
"partition": "model"
|
|
554
|
-
},
|
|
555
|
-
"max_price": {
|
|
556
|
-
"prompt": 10,
|
|
557
|
-
"completion": 20
|
|
558
|
-
},
|
|
559
|
-
"preferred_min_throughput": {
|
|
560
|
-
"p50": 100,
|
|
561
|
-
"p90": 50
|
|
562
|
-
},
|
|
563
|
-
"preferred_max_latency": {
|
|
564
|
-
"p50": 1,
|
|
565
|
-
"p90": 3,
|
|
566
|
-
"p99": 5
|
|
567
|
-
}
|
|
568
|
-
}
|
|
569
|
-
}
|
|
570
|
-
}
|
|
571
|
-
]
|
|
572
|
-
}
|
|
573
|
-
}
|
|
574
|
-
}
|
|
575
|
-
```
|
|
576
|
-
|
|
577
|
-
Vercel AI Gateway example:
|
|
578
|
-
|
|
579
|
-
```json
|
|
580
|
-
{
|
|
581
|
-
"providers": {
|
|
582
|
-
"vercel-ai-gateway": {
|
|
583
|
-
"baseUrl": "https://ai-gateway.vercel.sh/v1",
|
|
584
|
-
"apiKey": "$AI_GATEWAY_API_KEY",
|
|
585
|
-
"api": "openai-completions",
|
|
586
|
-
"models": [
|
|
587
|
-
{
|
|
588
|
-
"id": "moonshotai/kimi-k2.5",
|
|
589
|
-
"name": "Kimi K2.5 (Fireworks via Vercel)",
|
|
590
|
-
"reasoning": true,
|
|
591
|
-
"input": ["text", "image"],
|
|
592
|
-
"cost": { "input": 0.6, "output": 3, "cacheRead": 0, "cacheWrite": 0 },
|
|
593
|
-
"contextWindow": 262144,
|
|
594
|
-
"maxTokens": 262144,
|
|
595
|
-
"compat": {
|
|
596
|
-
"vercelGatewayRouting": {
|
|
597
|
-
"only": ["fireworks", "novita"],
|
|
598
|
-
"order": ["fireworks", "novita"]
|
|
599
|
-
}
|
|
600
|
-
}
|
|
601
|
-
}
|
|
602
|
-
]
|
|
603
|
-
}
|
|
604
|
-
}
|
|
605
|
-
}
|
|
606
|
-
```
|
|
123
|
+
Check its API type and compatibility settings in `models.json`. The upstream server must support the corresponding request fields and behavior.
|