@knightcodeai/cli-win32-x64 0.9.0 → 0.9.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/CHANGELOG.md +88 -0
- package/bin/README.md +52 -19
- package/bin/docs/cli-integration.md +106 -0
- package/bin/docs/cli.md +270 -0
- package/bin/docs/compaction.md +56 -37
- package/bin/docs/configuration.md +46 -0
- package/bin/docs/containerization.md +86 -54
- package/bin/docs/custom-provider.md +132 -782
- package/bin/docs/docs.json +143 -103
- package/bin/docs/environment-variables.md +5 -3
- package/bin/docs/extensions.md +134 -2937
- package/bin/docs/how-knightcode-works.md +49 -0
- package/bin/docs/index.md +24 -69
- package/bin/docs/json.md +193 -65
- package/bin/docs/keybindings.md +56 -101
- package/bin/docs/llama-cpp.md +3 -3
- package/bin/docs/message-types.md +261 -0
- package/bin/docs/models.md +65 -517
- package/bin/docs/packages.md +66 -167
- package/bin/docs/prompt-templates.md +31 -68
- package/bin/docs/providers.md +103 -233
- package/bin/docs/quickstart.md +61 -106
- package/bin/docs/rpc-commands.md +854 -0
- package/bin/docs/rpc-extension-ui.md +200 -0
- package/bin/docs/rpc.md +129 -1556
- package/bin/docs/sdk.md +76 -1160
- package/bin/docs/security.md +70 -32
- package/bin/docs/session-format.md +39 -216
- package/bin/docs/sessions.md +43 -121
- package/bin/docs/settings.md +112 -367
- package/bin/docs/shell-aliases.md +85 -5
- package/bin/docs/skills.md +51 -189
- package/bin/docs/slash-commands.md +63 -0
- package/bin/docs/terminal-setup.md +107 -79
- package/bin/docs/termux.md +74 -83
- package/bin/docs/themes.md +68 -280
- package/bin/docs/tmux.md +31 -39
- package/bin/docs/tui.md +69 -923
- package/bin/docs/usage.md +79 -285
- package/bin/docs/windows.md +43 -17
- package/bin/export-html/template.js +6 -1
- package/bin/knightcode.exe +2 -2
- package/bin/package.json +6 -6
- package/package.json +1 -1
- package/bin/docs/development.md +0 -71
package/bin/docs/models.md
CHANGED
|
@@ -1,575 +1,123 @@
|
|
|
1
|
-
#
|
|
1
|
+
# Choose a Model
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
For a built-in provider, start with `/login`, then choose a model with `/model`. Use custom model configuration only when KnightCode does not already include the provider or endpoint you need.
|
|
4
4
|
|
|
5
|
-
##
|
|
5
|
+
## Choose a connection
|
|
6
6
|
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
- [Anthropic Messages Compatibility](#anthropic-messages-compatibility)
|
|
15
|
-
- [OpenAI Compatibility](#openai-compatibility)
|
|
7
|
+
| What you have | Recommended setup |
|
|
8
|
+
|---|---|
|
|
9
|
+
| A supported subscription | Sign in through `/login` |
|
|
10
|
+
| A provider API key | Store it through `/login` or set its environment variable |
|
|
11
|
+
| A local GGUF model | Connect KnightCode to the llama.cpp router |
|
|
12
|
+
| An OpenAI-, Anthropic-, or Google-compatible endpoint | Add it to `models.json` |
|
|
13
|
+
| A provider with a custom protocol or authentication flow | Build or install a provider extension |
|
|
16
14
|
|
|
17
|
-
|
|
15
|
+
Browse the [model catalog](https://knightcode.dev/models) for current providers, model IDs, capabilities, context limits, and pricing. KnightCode starts with its bundled catalog and can overlay newer catalog data from knightcode.dev. Cached catalog data remains available offline; run `knightcode update --models` to force a refresh.
|
|
18
16
|
|
|
19
|
-
|
|
17
|
+
## Authenticate
|
|
20
18
|
|
|
21
|
-
|
|
22
|
-
{
|
|
23
|
-
"providers": {
|
|
24
|
-
"ollama": {
|
|
25
|
-
"baseUrl": "http://localhost:11434/v1",
|
|
26
|
-
"api": "openai-completions",
|
|
27
|
-
"apiKey": "ollama",
|
|
28
|
-
"models": [
|
|
29
|
-
{ "id": "llama3.1:8b" },
|
|
30
|
-
{ "id": "qwen2.5-coder:7b" }
|
|
31
|
-
]
|
|
32
|
-
}
|
|
33
|
-
}
|
|
34
|
-
}
|
|
35
|
-
```
|
|
19
|
+
Run `/login` and select a provider. KnightCode stores credentials in [`auth.json`](configuration.md#agent-directory). Run `/logout` to remove stored credentials for a provider.
|
|
36
20
|
|
|
37
|
-
|
|
21
|
+
You can instead provide an API key through the provider's environment variable. This is useful in CI and other environments where KnightCode should not write credentials. [Provider Authentication](providers.md) lists the variables and cloud-provider setup.
|
|
38
22
|
|
|
39
|
-
|
|
23
|
+
When several credential sources are configured, KnightCode uses a runtime `--api-key` first, then a stored `auth.json` credential, an `apiKey` from `models.json`, and finally the provider's environment variables or ambient cloud credentials. Provider extensions can define their own authentication behavior.
|
|
40
24
|
|
|
41
|
-
|
|
25
|
+
Keep `auth.json` and any credential commands private. Project settings and extensions can execute inside the KnightCode process after you trust a project. Review [Security](security.md) before loading configuration from an untrusted directory.
|
|
42
26
|
|
|
43
|
-
|
|
44
|
-
{
|
|
45
|
-
"providers": {
|
|
46
|
-
"ollama": {
|
|
47
|
-
"baseUrl": "http://localhost:11434/v1",
|
|
48
|
-
"api": "openai-completions",
|
|
49
|
-
"apiKey": "ollama",
|
|
50
|
-
"compat": {
|
|
51
|
-
"supportsDeveloperRole": false,
|
|
52
|
-
"supportsReasoningEffort": false
|
|
53
|
-
},
|
|
54
|
-
"models": [
|
|
55
|
-
{
|
|
56
|
-
"id": "gpt-oss:20b",
|
|
57
|
-
"reasoning": true
|
|
58
|
-
}
|
|
59
|
-
]
|
|
60
|
-
}
|
|
61
|
-
}
|
|
62
|
-
}
|
|
63
|
-
```
|
|
27
|
+
## Select a model
|
|
64
28
|
|
|
65
|
-
|
|
29
|
+
Run `/model` to search available models. The picker shows models whose providers have usable authentication. Press `Ctrl+S` (or `Ctrl+D`, if your terminal swallows `Ctrl+S`) on a model to save it as the default for new sessions.
|
|
66
30
|
|
|
67
|
-
|
|
31
|
+
Run `/thinking` to select the thinking level for the current model. Press `Ctrl+S` or `Ctrl+D` there to save the startup level. KnightCode limits the choices to levels supported by the selected model.
|
|
68
32
|
|
|
69
|
-
|
|
70
|
-
{
|
|
71
|
-
"providers": {
|
|
72
|
-
"ollama": {
|
|
73
|
-
"baseUrl": "http://localhost:11434/v1",
|
|
74
|
-
"api": "openai-completions",
|
|
75
|
-
"apiKey": "ollama",
|
|
76
|
-
"models": [
|
|
77
|
-
{
|
|
78
|
-
"id": "llama3.1:8b",
|
|
79
|
-
"name": "Llama 3.1 8B (Local)",
|
|
80
|
-
"reasoning": false,
|
|
81
|
-
"input": ["text"],
|
|
82
|
-
"contextWindow": 128000,
|
|
83
|
-
"maxTokens": 32000,
|
|
84
|
-
"cost": { "input": 0, "output": 0, "cacheRead": 0, "cacheWrite": 0 }
|
|
85
|
-
}
|
|
86
|
-
]
|
|
87
|
-
}
|
|
88
|
-
}
|
|
89
|
-
}
|
|
90
|
-
```
|
|
33
|
+
`Ctrl+P` cycles through available models. Use `/scoped-models` to control that cycle and save the selection, or configure model patterns through [Settings](settings.md#model-cycling).
|
|
91
34
|
|
|
92
|
-
|
|
35
|
+
A session records model and thinking-level changes. Resuming the session restores them without changing defaults for new sessions.
|
|
93
36
|
|
|
94
|
-
##
|
|
37
|
+
## Connect local models
|
|
95
38
|
|
|
96
|
-
|
|
39
|
+
KnightCode integrates directly with the llama.cpp router. The router discovers GGUF files and loads models on demand. KnightCode's `/llama` command manages the router, while `/model` selects one of its loaded models.
|
|
97
40
|
|
|
98
|
-
|
|
99
|
-
{
|
|
100
|
-
"providers": {
|
|
101
|
-
"my-google": {
|
|
102
|
-
"baseUrl": "https://generativelanguage.googleapis.com/v1beta",
|
|
103
|
-
"api": "google-generative-ai",
|
|
104
|
-
"apiKey": "$GEMINI_API_KEY",
|
|
105
|
-
"models": [
|
|
106
|
-
{
|
|
107
|
-
"id": "gemma-4-31b-it",
|
|
108
|
-
"name": "Gemma 4 31B",
|
|
109
|
-
"input": ["text", "image"],
|
|
110
|
-
"contextWindow": 262144,
|
|
111
|
-
"reasoning": true
|
|
112
|
-
}
|
|
113
|
-
]
|
|
114
|
-
}
|
|
115
|
-
}
|
|
116
|
-
}
|
|
117
|
-
```
|
|
118
|
-
|
|
119
|
-
The `baseUrl` is required when adding custom models to the `google-generative-ai` API type.
|
|
120
|
-
|
|
121
|
-
## Supported APIs
|
|
122
|
-
|
|
123
|
-
| API | Description |
|
|
124
|
-
|-----|-------------|
|
|
125
|
-
| `openai-completions` | OpenAI Chat Completions (most compatible) |
|
|
126
|
-
| `openai-responses` | OpenAI Responses API |
|
|
127
|
-
| `anthropic-messages` | Anthropic Messages API |
|
|
128
|
-
| `google-generative-ai` | Google Generative AI |
|
|
129
|
-
|
|
130
|
-
Set `api` at provider level (default for all models) or model level (override per model).
|
|
131
|
-
|
|
132
|
-
## Provider Configuration
|
|
133
|
-
|
|
134
|
-
| Field | Description |
|
|
135
|
-
|-------|-------------|
|
|
136
|
-
| `baseUrl` | API endpoint URL |
|
|
137
|
-
| `api` | API type (see above) |
|
|
138
|
-
| `apiKey` | Optional API key config (see value resolution below). Omit it when auth is provided by `/login`/`auth.json` or CLI `--api-key`. |
|
|
139
|
-
| `oauth` | Dynamic OAuth provider type. Currently supports `"radius"`; requires the gateway `baseUrl`. |
|
|
140
|
-
| `headers` | Custom headers (see value resolution below) |
|
|
141
|
-
| `authHeader` | Set `true` to add `Authorization: Bearer <apiKey>` automatically |
|
|
142
|
-
| `models` | Array of model configurations |
|
|
143
|
-
| `modelOverrides` | Per-model overrides for built-in or extension-registered models on this provider |
|
|
144
|
-
|
|
145
|
-
For providers with `models`, non-built-in provider configs need `baseUrl` and an `api` value at either provider or model level. `apiKey` is not required to load the file: models become available when auth is configured through `/login`/`auth.json`, CLI `--api-key`, or provider `apiKey`. If no auth is configured, the models load but stay unavailable in `/model` and `--list-models`.
|
|
41
|
+
Follow [Local Models with llama.cpp](llama-cpp.md) for server startup, model layout, downloads, and connection troubleshooting.
|
|
146
42
|
|
|
147
|
-
|
|
43
|
+
For Ollama, LM Studio, vLLM, SGLang, and other compatible servers, [configure a compatible endpoint](#configure-a-compatible-endpoint) in `models.json`.
|
|
148
44
|
|
|
149
|
-
|
|
45
|
+
## Configure a compatible endpoint
|
|
150
46
|
|
|
151
|
-
-
|
|
152
|
-
```json
|
|
153
|
-
"apiKey": "!security find-generic-password -ws 'anthropic'"
|
|
154
|
-
"apiKey": "!op read 'op://vault/item/credential'"
|
|
155
|
-
```
|
|
156
|
-
- **Environment interpolation:** `"$ENV_VAR"` or `"${ENV_VAR}"` uses the value of the named variable. Interpolation works inside larger literals.
|
|
157
|
-
```json
|
|
158
|
-
"apiKey": "$MY_API_KEY"
|
|
159
|
-
"apiKey": "${KEY_PREFIX}_${KEY_SUFFIX}"
|
|
160
|
-
```
|
|
161
|
-
`$FOO_BAR` is the variable `FOO_BAR`; use `${FOO}_BAR` when `BAR` is literal text. Missing environment variables make the value unresolved.
|
|
162
|
-
- **Escapes:** `"$$"` emits a literal `"$"`; `"$!"` emits a literal `"!"` without triggering command execution.
|
|
163
|
-
```json
|
|
164
|
-
"apiKey": "$$literal-dollar-prefix"
|
|
165
|
-
"apiKey": "$!literal-bang-prefix"
|
|
166
|
-
```
|
|
167
|
-
- **Literal value:** Used directly. Plain uppercase strings such as `MY_API_KEY` are literals; use `$MY_API_KEY` for environment variables.
|
|
168
|
-
```json
|
|
169
|
-
"apiKey": "sk-..."
|
|
170
|
-
```
|
|
171
|
-
|
|
172
|
-
For `models.json`, shell commands are resolved at request time. knightcode intentionally does not apply built-in TTL, stale reuse, or recovery logic for arbitrary commands. Different commands need different caching and failure strategies, and knightcode cannot infer the right one.
|
|
173
|
-
|
|
174
|
-
If your command is slow, expensive, rate-limited, or should keep using a previous value on transient failures, wrap it in your own script or command that implements the caching or TTL behavior you want.
|
|
175
|
-
|
|
176
|
-
`/model` availability checks use configured auth presence and do not execute shell commands.
|
|
177
|
-
|
|
178
|
-
### Custom Headers
|
|
47
|
+
Use [`models.json`](configuration.md#agent-directory) when an endpoint speaks an API KnightCode already supports. This includes most Ollama, LM Studio, vLLM, SGLang, and proxy deployments.
|
|
179
48
|
|
|
180
49
|
```json
|
|
181
50
|
{
|
|
182
51
|
"providers": {
|
|
183
|
-
"
|
|
184
|
-
"baseUrl": "
|
|
185
|
-
"
|
|
186
|
-
"
|
|
187
|
-
"
|
|
188
|
-
"
|
|
189
|
-
|
|
190
|
-
},
|
|
191
|
-
"models": [...]
|
|
192
|
-
}
|
|
193
|
-
}
|
|
194
|
-
}
|
|
195
|
-
```
|
|
196
|
-
|
|
197
|
-
## Model Configuration
|
|
198
|
-
|
|
199
|
-
| Field | Required | Default | Description |
|
|
200
|
-
|-------|----------|---------|-------------|
|
|
201
|
-
| `id` | Yes | — | Model identifier (passed to the API) |
|
|
202
|
-
| `name` | No | `id` | Human-readable model label. Used for matching (`--model` patterns) and shown as secondary model detail text. |
|
|
203
|
-
| `api` | No | provider's `api` | Override provider's API for this model |
|
|
204
|
-
| `reasoning` | No | `false` | Supports extended thinking |
|
|
205
|
-
| `thinkingLevelMap` | No | omitted | Maps knightcode thinking levels to provider values and marks unsupported levels (see below) |
|
|
206
|
-
| `input` | No | `["text"]` | Input types: `["text"]` or `["text", "image"]` |
|
|
207
|
-
| `contextWindow` | No | `128000` | Context window size in tokens |
|
|
208
|
-
| `maxTokens` | No | `16384` | Maximum output tokens |
|
|
209
|
-
| `samplingParams` | No | omitted | Sampling parameters merged verbatim into every request body (see below) |
|
|
210
|
-
| `cost` | No | all zeros | Per-million-token rates with optional request-wide input pricing tiers |
|
|
211
|
-
| `compat` | No | provider `compat` | Provider compatibility overrides. Merged with provider-level `compat` when both are set. |
|
|
212
|
-
|
|
213
|
-
A cost tier supplies a complete alternate rate set and applies to the full request when total input usage (`input + cacheRead + cacheWrite`) exceeds `inputTokensAbove`. When multiple tiers match, the highest threshold wins.
|
|
214
|
-
|
|
215
|
-
```json
|
|
216
|
-
{
|
|
217
|
-
"cost": {
|
|
218
|
-
"input": 5,
|
|
219
|
-
"output": 30,
|
|
220
|
-
"cacheRead": 0.5,
|
|
221
|
-
"cacheWrite": 6.25,
|
|
222
|
-
"tiers": [
|
|
223
|
-
{
|
|
224
|
-
"inputTokensAbove": 272000,
|
|
225
|
-
"input": 10,
|
|
226
|
-
"output": 45,
|
|
227
|
-
"cacheRead": 1,
|
|
228
|
-
"cacheWrite": 12.5
|
|
229
|
-
}
|
|
230
|
-
]
|
|
231
|
-
}
|
|
232
|
-
}
|
|
233
|
-
```
|
|
234
|
-
|
|
235
|
-
Current behavior:
|
|
236
|
-
- `/model`, `--list-models`, and the interactive footer display entries by model `id`.
|
|
237
|
-
- The configured `name` is used for model matching and secondary model detail text. It does not replace the footer/status-bar model id.
|
|
238
|
-
|
|
239
|
-
### Sampling Parameters
|
|
240
|
-
|
|
241
|
-
`samplingParams` is a free-form object merged verbatim into every request body for the model, after the fields knightcode sets itself, so its keys win. Use it to send sampling parameters knightcode does not model — including server-specific ones like llama.cpp's `min_p` or vLLM's `top_k`:
|
|
242
|
-
|
|
243
|
-
```json
|
|
244
|
-
{
|
|
245
|
-
"id": "deepseek-v4-flash",
|
|
246
|
-
"samplingParams": {
|
|
247
|
-
"temperature": 1.0,
|
|
248
|
-
"top_p": 0.95,
|
|
249
|
-
"top_k": 0,
|
|
250
|
-
"min_p": 0.0
|
|
251
|
-
}
|
|
252
|
-
}
|
|
253
|
-
```
|
|
254
|
-
|
|
255
|
-
Only OpenAI-compatible APIs apply it (`openai-completions`, `openai-responses`, `azure-openai-responses`); other APIs ignore it. Keys override knightcode's named request fields (for example a `temperature` key here beats the request-level temperature), so prefer it as the single source of sampling truth for a model. In `modelOverrides`, `samplingParams` merges per key with the base model's value.
|
|
256
|
-
|
|
257
|
-
A constant thinking-token cap can go here too, but it will not follow `thinkingBudgets` or leave room for the answer. Prefer `compat.thinkingTokenBudgetField` (or the `supportsThinkingTokenBudget` alias) for that.
|
|
258
|
-
|
|
259
|
-
### Thinking Level Map
|
|
260
|
-
|
|
261
|
-
Use `thinkingLevelMap` on a model to describe model-specific thinking controls. Keys are knightcode thinking levels: `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, `max`. Maps may contain holes; for example, a model can expose `high` and `max` without exposing `xhigh`.
|
|
262
|
-
|
|
263
|
-
Values are tristate:
|
|
264
|
-
|
|
265
|
-
| Value | Meaning |
|
|
266
|
-
|-------|---------|
|
|
267
|
-
| omitted | Standard levels through `high` use the provider's default mapping; extended `xhigh` and `max` levels are unsupported |
|
|
268
|
-
| string | Level is supported and this value is sent to the provider |
|
|
269
|
-
| `null` | Level is unsupported and hidden/skipped/clamped away |
|
|
270
|
-
|
|
271
|
-
Example for a model that only supports off, high, and max reasoning:
|
|
272
|
-
|
|
273
|
-
```json
|
|
274
|
-
{
|
|
275
|
-
"id": "deepseek-v4-pro",
|
|
276
|
-
"reasoning": true,
|
|
277
|
-
"thinkingLevelMap": {
|
|
278
|
-
"minimal": null,
|
|
279
|
-
"low": null,
|
|
280
|
-
"medium": null,
|
|
281
|
-
"high": "high",
|
|
282
|
-
"xhigh": null,
|
|
283
|
-
"max": "max"
|
|
284
|
-
}
|
|
285
|
-
}
|
|
286
|
-
```
|
|
287
|
-
|
|
288
|
-
Example for a model where thinking cannot be disabled:
|
|
289
|
-
|
|
290
|
-
```json
|
|
291
|
-
{
|
|
292
|
-
"id": "always-thinking-model",
|
|
293
|
-
"reasoning": true,
|
|
294
|
-
"thinkingLevelMap": {
|
|
295
|
-
"off": null
|
|
296
|
-
}
|
|
297
|
-
}
|
|
298
|
-
```
|
|
299
|
-
|
|
300
|
-
Migration: older configs that used `compat.reasoningEffortMap` should move that mapping to model-level `thinkingLevelMap`. Use `null` for levels that should not appear in the UI.
|
|
301
|
-
|
|
302
|
-
## Overriding Built-in Providers
|
|
303
|
-
|
|
304
|
-
Route a built-in provider through a proxy without redefining models:
|
|
305
|
-
|
|
306
|
-
```json
|
|
307
|
-
{
|
|
308
|
-
"providers": {
|
|
309
|
-
"anthropic": {
|
|
310
|
-
"baseUrl": "https://my-proxy.example.com/v1"
|
|
52
|
+
"ollama": {
|
|
53
|
+
"baseUrl": "http://localhost:11434/v1",
|
|
54
|
+
"api": "openai-completions",
|
|
55
|
+
"apiKey": "ollama",
|
|
56
|
+
"models": [
|
|
57
|
+
{ "id": "qwen2.5-coder:7b" }
|
|
58
|
+
]
|
|
311
59
|
}
|
|
312
60
|
}
|
|
313
61
|
}
|
|
314
62
|
```
|
|
315
63
|
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
To merge custom models into a built-in provider, include the `models` array:
|
|
319
|
-
|
|
320
|
-
```json
|
|
321
|
-
{
|
|
322
|
-
"providers": {
|
|
323
|
-
"anthropic": {
|
|
324
|
-
"baseUrl": "https://my-proxy.example.com/v1",
|
|
325
|
-
"apiKey": "$ANTHROPIC_API_KEY",
|
|
326
|
-
"api": "anthropic-messages",
|
|
327
|
-
"models": [...]
|
|
328
|
-
}
|
|
329
|
-
}
|
|
330
|
-
}
|
|
331
|
-
```
|
|
64
|
+
The dummy key makes the model available to KnightCode; Ollama ignores it. For an authenticated endpoint, `apiKey` and header values can use `$NAME` or `${NAME}` environment interpolation, a literal value, or a leading `!command`. Commands in `models.json` run at request time and are not cached by KnightCode.
|
|
332
65
|
|
|
333
|
-
|
|
334
|
-
- Built-in models are kept.
|
|
335
|
-
- Custom models are upserted by `id` within the provider.
|
|
336
|
-
- If a custom model `id` matches a built-in model `id`, the custom model replaces that built-in model.
|
|
337
|
-
- If a custom model `id` is new, it is added alongside built-in models.
|
|
66
|
+
Opening `/model` reloads the file. A `models` entry adds or replaces a model with the same ID on that provider. Use `modelOverrides` to change metadata for an existing built-in or extension-provided model without replacing the provider's model list. Unknown override IDs are ignored.
|
|
338
67
|
|
|
339
|
-
|
|
68
|
+
### Describe model input and caching
|
|
340
69
|
|
|
341
|
-
Use `
|
|
70
|
+
Use `inputLimits.images.resize` to control how KnightCode encodes new image attachments, `read` results, and tool-result images before storing them in conversation history:
|
|
342
71
|
|
|
343
72
|
```json
|
|
344
73
|
{
|
|
345
|
-
"
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
}
|
|
355
|
-
}
|
|
74
|
+
"id": "vision-model",
|
|
75
|
+
"input": ["text", "image"],
|
|
76
|
+
"inputLimits": {
|
|
77
|
+
"images": {
|
|
78
|
+
"resize": {
|
|
79
|
+
"maxWidth": 1568,
|
|
80
|
+
"maxHeight": 1568,
|
|
81
|
+
"maxBytes": 524288,
|
|
82
|
+
"jpegQuality": 75
|
|
356
83
|
}
|
|
357
84
|
}
|
|
358
85
|
}
|
|
359
86
|
}
|
|
360
87
|
```
|
|
361
88
|
|
|
362
|
-
`
|
|
89
|
+
`maxBytes` limits the base64-encoded payload. Omitted resize fields use conservative defaults of 2000 by 2000 pixels, 4.5 MiB encoded, and JPEG quality 80. Images are encoded once; changing models does not rewrite historical images. The catalog can also describe hard request limits with `inputLimits.maxRequestBytes`, `images.maxPerMessage`, and `images.maxPerRequest`, but KnightCode does not yet rewrite or reject history based on them.
|
|
90
|
+
|
|
91
|
+
<a id="prompt-cache-lifetimes"></a>
|
|
363
92
|
|
|
364
|
-
|
|
93
|
+
Use `promptCache` to declare the provider's best-effort cache lifetime in seconds for the `short` or `long` retention tier:
|
|
365
94
|
|
|
366
95
|
```json
|
|
367
|
-
{
|
|
368
|
-
"providers": {
|
|
369
|
-
"openai": {
|
|
370
|
-
"modelOverrides": {
|
|
371
|
-
"gpt-5.6-sol": {
|
|
372
|
-
"contextWindow": 1050000
|
|
373
|
-
}
|
|
374
|
-
}
|
|
375
|
-
}
|
|
376
|
-
}
|
|
377
|
-
}
|
|
96
|
+
{ "id": "claude-sonnet-5", "promptCache": { "short": 300, "long": 3600 } }
|
|
378
97
|
```
|
|
379
98
|
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
Behavior notes:
|
|
383
|
-
- `modelOverrides` are applied to built-in provider models and matching extension-registered provider models.
|
|
384
|
-
- Unknown model IDs are ignored.
|
|
385
|
-
- You can combine provider-level `baseUrl`/`headers` with `modelOverrides`.
|
|
386
|
-
- Overriding `name` changes model matching and secondary detail text only; the footer and primary model lists continue to show the model `id`.
|
|
387
|
-
- If `models` is also defined for a provider, custom models are merged after built-in overrides. A custom model with the same `id` replaces the overridden built-in model entry.
|
|
388
|
-
|
|
389
|
-
## Anthropic Messages Compatibility
|
|
390
|
-
|
|
391
|
-
For providers or proxies using `api: "anthropic-messages"`, use `compat` to control Anthropic-specific request compatibility.
|
|
392
|
-
|
|
393
|
-
By default knightcode sends per-tool `eager_input_streaming: true`. If a proxy or Anthropic-compatible backend rejects that field, set `supportsEagerToolInputStreaming` to `false`. KnightCode will omit `tools[].eager_input_streaming` and send the legacy `fine-grained-tool-streaming-2025-05-14` beta header for tool-enabled requests instead.
|
|
394
|
-
|
|
395
|
-
Some Anthropic models require adaptive thinking (`thinking.type: "adaptive"` plus `output_config.effort`) instead of the legacy budget-based thinking payload. Built-in models set this automatically. For custom providers or aliases that route to those models, set `forceAdaptiveThinking` to `true`.
|
|
99
|
+
Choose the conservative end of any published range. A model without a lifetime for the active tier is not eligible for cache warming. A `modelOverrides` entry can set `inputLimits` or `promptCache` for a built-in or extension model, including a model accessed through a validated proxy. See [`cacheWarming`](settings.md#model-and-thinking).
|
|
396
100
|
|
|
397
|
-
|
|
101
|
+
Compatibility settings should describe verified differences in the endpoint's request or response behavior. Do not enable them based only on an endpoint advertising OpenAI or Anthropic compatibility.
|
|
398
102
|
|
|
399
|
-
|
|
103
|
+
## Add a custom provider
|
|
400
104
|
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
```json
|
|
404
|
-
{
|
|
405
|
-
"providers": {
|
|
406
|
-
"anthropic-proxy": {
|
|
407
|
-
"baseUrl": "https://proxy.example.com",
|
|
408
|
-
"api": "anthropic-messages",
|
|
409
|
-
"apiKey": "$ANTHROPIC_PROXY_KEY",
|
|
410
|
-
"compat": {
|
|
411
|
-
"supportsEagerToolInputStreaming": false,
|
|
412
|
-
"supportsLongCacheRetention": true,
|
|
413
|
-
"forceAdaptiveThinking": true,
|
|
414
|
-
"allowEmptySignature": true
|
|
415
|
-
},
|
|
416
|
-
"models": [
|
|
417
|
-
{
|
|
418
|
-
"id": "claude-opus-4-7",
|
|
419
|
-
"reasoning": true,
|
|
420
|
-
"input": ["text", "image"]
|
|
421
|
-
}
|
|
422
|
-
]
|
|
423
|
-
}
|
|
424
|
-
}
|
|
425
|
-
}
|
|
426
|
-
```
|
|
105
|
+
Use an extension when the provider needs custom streaming, model discovery, or authentication behavior. See [Custom Providers](custom-provider.md) for the extension workflow.
|
|
427
106
|
|
|
428
|
-
|
|
429
|
-
|-------|-------------|
|
|
430
|
-
| `supportsEagerToolInputStreaming` | Whether the provider accepts per-tool `eager_input_streaming`. Default: `true`. Set to `false` to omit that field and use the legacy fine-grained tool streaming beta header on tool-enabled requests. |
|
|
431
|
-
| `supportsLongCacheRetention` | Whether the provider accepts Anthropic long cache retention (`cache_control.ttl: "1h"`) when cache retention is `long`. Default: `true`. |
|
|
432
|
-
| `sendSessionAffinityHeaders` | Whether to send a session-affinity header from the session id when caching is enabled. Default: `true` for OpenRouter endpoints, auto-detected for other known providers. |
|
|
433
|
-
| `sessionAffinityFormat` | Session-affinity header name: `openrouter` sends `x-session-id`. When unset, `x-session-affinity` is sent. Default: `openrouter` for OpenRouter endpoints. |
|
|
434
|
-
| `supportsCacheControlOnTools` | Whether the provider accepts Anthropic-style `cache_control` markers on tool definitions. Default: `true`. |
|
|
435
|
-
| `forceAdaptiveThinking` | Whether to send adaptive thinking (`thinking.type: "adaptive"` plus `output_config.effort`) for this model. Built-in adaptive models set this automatically. Default: `false`. |
|
|
436
|
-
| `supportsMidConvoEffort` | Whether the exact Claude model transport supports per-turn effort system messages and thinking binding controls. KnightCode persists native effort levels and always sends `drop_block` when enabled. Default: `false`. |
|
|
437
|
-
| `allowEmptySignature` | Whether to replay empty thinking signatures as `signature: ""` instead of converting thinking to text. Default: `false`. |
|
|
438
|
-
| `supportsStrictTools` | Whether the provider accepts strict JSON-schema tool definitions. Default: `false`; built-in Anthropic models enable it in generated metadata. |
|
|
439
|
-
| `allowedFallbackModels` | Up to three server-side fallback models, each with `provider`, `model`, and complete `cost` metadata. An empty array disables fallback. |
|
|
107
|
+
## Troubleshooting
|
|
440
108
|
|
|
441
|
-
|
|
109
|
+
### A model does not appear
|
|
442
110
|
|
|
443
|
-
|
|
111
|
+
Confirm that its provider has usable authentication. Custom models can load from `models.json` but remain unavailable in `/model` until KnightCode can resolve credentials. For llama.cpp, only models currently loaded by the router appear.
|
|
444
112
|
|
|
445
|
-
|
|
446
|
-
- Model-level `compat` overrides provider-level values for that model.
|
|
113
|
+
### Authentication works in one shell only
|
|
447
114
|
|
|
448
|
-
|
|
449
|
-
{
|
|
450
|
-
"providers": {
|
|
451
|
-
"local-llm": {
|
|
452
|
-
"baseUrl": "http://localhost:8080/v1",
|
|
453
|
-
"api": "openai-completions",
|
|
454
|
-
"compat": {
|
|
455
|
-
"supportsUsageInStreaming": false,
|
|
456
|
-
"maxTokensField": "max_tokens"
|
|
457
|
-
},
|
|
458
|
-
"models": [...]
|
|
459
|
-
}
|
|
460
|
-
}
|
|
461
|
-
}
|
|
462
|
-
```
|
|
115
|
+
Check whether the key came from an environment variable rather than `auth.json`. Environment variables must be present in the process that starts KnightCode.
|
|
463
116
|
|
|
464
|
-
|
|
465
|
-
|-------|-------------|
|
|
466
|
-
| `supportsStore` | Provider supports `store` field |
|
|
467
|
-
| `supportsDeveloperRole` | Use `developer` vs `system` role |
|
|
468
|
-
| `supportsReasoningEffort` | Support for `reasoning_effort` parameter |
|
|
469
|
-
| `supportsUsageInStreaming` | Supports `stream_options: { include_usage: true }` (default: `true`) |
|
|
470
|
-
| `supportsFinishReason` | Whether streamed responses include `finish_reason`. When `false`, knightcode infers `stop` or `toolUse` when the stream ends. Default: `true`. |
|
|
471
|
-
| `maxTokensField` | Use `max_completion_tokens` or `max_tokens` |
|
|
472
|
-
| `requiresToolResultName` | Include `name` on tool result messages |
|
|
473
|
-
| `requiresAssistantAfterToolResult` | Insert an assistant message before a user message after tool results |
|
|
474
|
-
| `requiresThinkingAsText` | Convert thinking blocks to plain text |
|
|
475
|
-
| `requiresReasoningContentOnAssistantMessages` | Include empty `reasoning_content` on all replayed assistant messages when reasoning is enabled |
|
|
476
|
-
| `thinkingFormat` | Use `reasoning_effort`, `openrouter`, `deepseek`, `together`, `baseten`, `zai`, `qwen`, `chat-template`, or `qwen-chat-template` thinking parameters |
|
|
477
|
-
| `chatTemplateKwargs` | `chat_template_kwargs` values for `thinkingFormat: "chat-template"`; use `{ "$var": "thinking.enabled" }`, `{ "$var": "thinking.effort" }`, or `{ "$var": "thinking.budget" }` for knightcode-controlled thinking values |
|
|
478
|
-
| `chatTemplateArgs` | `chat_template_args` values for `thinkingFormat: "baseten"`; use `{ "$var": "thinking.enabled" }`, `{ "$var": "thinking.effort" }`, or `{ "$var": "thinking.budget" }` for knightcode-controlled thinking values |
|
|
479
|
-
| `thinkingTokenBudgetField` | Top-level request field used to cap reasoning tokens from `thinkingBudgets`, clamped so at least 1024 tokens remain for the answer. `"thinking_token_budget"` (vLLM), `"thinking_budget"` (Qwen/DashScope/SGLang), `"thinking_budget_tokens"` (llama.cpp). Off by default; not set on the generated catalog. |
|
|
480
|
-
| `supportsThinkingTokenBudget` | Alias for `thinkingTokenBudgetField: "thinking_token_budget"` (vLLM). Prefer `thinkingTokenBudgetField`. Default: `false`. |
|
|
481
|
-
| `cacheControlFormat` | Use Anthropic-style `cache_control` markers on the system prompt, last tool definition, and last user, assistant, or tool-result text content. Currently only `anthropic` is supported. |
|
|
482
|
-
| `sendSessionAffinityHeaders` | For `openai-completions`, send session-affinity headers from the session id when caching is enabled. Default: `true` for OpenRouter endpoints, `false` otherwise. |
|
|
483
|
-
| `sessionAffinityFormat` | For `openai-completions` and `openai-responses`, the session-affinity header format: `openai` sends `session_id`/`x-client-request-id` (completions also `x-session-affinity`), `openai-nosession` omits the underscore-containing `session_id` header, `openrouter` sends `x-session-id`. Does not affect the `prompt_cache_key` body param. Default: auto-detected. |
|
|
484
|
-
| `supportsStrictMode` | Whether the provider accepts strict JSON-schema function tool definitions. Defaults depend on the API; built-in OpenAI models carry explicit capability metadata. |
|
|
485
|
-
| `supportsOpenAIGrammarTools` | Whether OpenAI-compatible APIs emit custom Lark/regex grammar tools. When `false`, grammar-constrained tools fall back to normal function tools. Default: `false`; the built-in model catalog enables it for GPT-5+ models on OpenAI, OpenAI Codex, Azure OpenAI, GitHub Copilot, opencode, and Cloudflare AI Gateway. |
|
|
486
|
-
| `supportsLongCacheRetention` | Whether the provider accepts long cache retention when cache retention is `long`: `prompt_cache_options.ttl: "30m"` for GPT-5.6+ Responses models, `prompt_cache_retention: "24h"` for earlier OpenAI models, or `cache_control.ttl: "1h"` when `cacheControlFormat` is `anthropic`. Default: `true`. |
|
|
487
|
-
| `openRouterRouting` | OpenRouter provider routing preferences. This object is sent as-is in the `provider` field of the [OpenRouter API request](https://openrouter.ai/docs/guides/routing/provider-selection). |
|
|
488
|
-
| `vercelGatewayRouting` | Vercel AI Gateway routing config for provider selection (`only`, `order`) |
|
|
489
|
-
|
|
490
|
-
`openrouter` uses `reasoning: { effort }`. `together` uses `reasoning: { enabled }` and also `reasoning_effort` when `supportsReasoningEffort` is enabled. `qwen` uses top-level `enable_thinking`. Use `qwen-chat-template` for local Qwen-compatible servers that require `chat_template_kwargs.enable_thinking` and `preserve_thinking`. Use `chat-template` for vLLM/Hugging Face chat templates that need configurable `chat_template_kwargs`, such as `chatTemplateKwargs: { "thinking": { "$var": "thinking.enabled" } }` for DeepSeek V3.x templates. Use `thinkingFormat: "baseten"` with `chatTemplateArgs` for providers that expose toggle controls through `chat_template_args` and optionally support top-level `reasoning_effort`.
|
|
491
|
-
|
|
492
|
-
`thinkingTokenBudgetField` is independent of `thinkingFormat`. Do not enable it on the generated Qwen catalog: those models already send `reasoning_effort`, and DashScope rejects `thinking_budget` together with `reasoning_effort`.
|
|
493
|
-
|
|
494
|
-
`cacheControlFormat: "anthropic"` is for OpenAI-compatible providers that expose Anthropic-style prompt caching through `cache_control` markers on text content and tool definitions.
|
|
495
|
-
|
|
496
|
-
Example:
|
|
117
|
+
### Sign-in opens a browser on a remote machine
|
|
497
118
|
|
|
498
|
-
|
|
499
|
-
{
|
|
500
|
-
"providers": {
|
|
501
|
-
"openrouter": {
|
|
502
|
-
"baseUrl": "https://openrouter.ai/api/v1",
|
|
503
|
-
"apiKey": "$OPENROUTER_API_KEY",
|
|
504
|
-
"api": "openai-completions",
|
|
505
|
-
"models": [
|
|
506
|
-
{
|
|
507
|
-
"id": "openrouter/anthropic/claude-3.5-sonnet",
|
|
508
|
-
"name": "OpenRouter Claude 3.5 Sonnet",
|
|
509
|
-
"compat": {
|
|
510
|
-
"openRouterRouting": {
|
|
511
|
-
"allow_fallbacks": true,
|
|
512
|
-
"require_parameters": false,
|
|
513
|
-
"data_collection": "deny",
|
|
514
|
-
"zdr": true,
|
|
515
|
-
"enforce_distillable_text": false,
|
|
516
|
-
"order": ["anthropic", "amazon-bedrock", "google-vertex"],
|
|
517
|
-
"only": ["anthropic", "amazon-bedrock"],
|
|
518
|
-
"ignore": ["gmicloud", "friendli"],
|
|
519
|
-
"quantizations": ["fp16", "bf16"],
|
|
520
|
-
"sort": {
|
|
521
|
-
"by": "price",
|
|
522
|
-
"partition": "model"
|
|
523
|
-
},
|
|
524
|
-
"max_price": {
|
|
525
|
-
"prompt": 10,
|
|
526
|
-
"completion": 20
|
|
527
|
-
},
|
|
528
|
-
"preferred_min_throughput": {
|
|
529
|
-
"p50": 100,
|
|
530
|
-
"p90": 50
|
|
531
|
-
},
|
|
532
|
-
"preferred_max_latency": {
|
|
533
|
-
"p50": 1,
|
|
534
|
-
"p90": 3,
|
|
535
|
-
"p99": 5
|
|
536
|
-
}
|
|
537
|
-
}
|
|
538
|
-
}
|
|
539
|
-
}
|
|
540
|
-
]
|
|
541
|
-
}
|
|
542
|
-
}
|
|
543
|
-
}
|
|
544
|
-
```
|
|
119
|
+
Complete the provider's headless authentication flow when available. Some providers let you paste the final redirect URL or authorization code back into KnightCode. See [Authenticate interactively](providers.md#authenticate-interactively).
|
|
545
120
|
|
|
546
|
-
|
|
121
|
+
### A compatible endpoint rejects requests
|
|
547
122
|
|
|
548
|
-
|
|
549
|
-
{
|
|
550
|
-
"providers": {
|
|
551
|
-
"vercel-ai-gateway": {
|
|
552
|
-
"baseUrl": "https://ai-gateway.vercel.sh/v1",
|
|
553
|
-
"apiKey": "$AI_GATEWAY_API_KEY",
|
|
554
|
-
"api": "openai-completions",
|
|
555
|
-
"models": [
|
|
556
|
-
{
|
|
557
|
-
"id": "moonshotai/kimi-k2.5",
|
|
558
|
-
"name": "Kimi K2.5 (Fireworks via Vercel)",
|
|
559
|
-
"reasoning": true,
|
|
560
|
-
"input": ["text", "image"],
|
|
561
|
-
"cost": { "input": 0.6, "output": 3, "cacheRead": 0, "cacheWrite": 0 },
|
|
562
|
-
"contextWindow": 262144,
|
|
563
|
-
"maxTokens": 262144,
|
|
564
|
-
"compat": {
|
|
565
|
-
"vercelGatewayRouting": {
|
|
566
|
-
"only": ["fireworks", "novita"],
|
|
567
|
-
"order": ["fireworks", "novita"]
|
|
568
|
-
}
|
|
569
|
-
}
|
|
570
|
-
}
|
|
571
|
-
]
|
|
572
|
-
}
|
|
573
|
-
}
|
|
574
|
-
}
|
|
575
|
-
```
|
|
123
|
+
Check its API type and compatibility settings in `models.json`. The upstream server must support the corresponding request fields and behavior.
|