@knightcodeai/cli-linux-x64 0.9.0 → 0.9.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/CHANGELOG.md +88 -0
- package/bin/README.md +52 -19
- package/bin/docs/cli-integration.md +106 -0
- package/bin/docs/cli.md +270 -0
- package/bin/docs/compaction.md +56 -37
- package/bin/docs/configuration.md +46 -0
- package/bin/docs/containerization.md +86 -54
- package/bin/docs/custom-provider.md +132 -782
- package/bin/docs/docs.json +143 -103
- package/bin/docs/environment-variables.md +5 -3
- package/bin/docs/extensions.md +134 -2937
- package/bin/docs/how-knightcode-works.md +49 -0
- package/bin/docs/index.md +24 -69
- package/bin/docs/json.md +193 -65
- package/bin/docs/keybindings.md +56 -101
- package/bin/docs/llama-cpp.md +3 -3
- package/bin/docs/message-types.md +261 -0
- package/bin/docs/models.md +65 -517
- package/bin/docs/packages.md +66 -167
- package/bin/docs/prompt-templates.md +31 -68
- package/bin/docs/providers.md +103 -233
- package/bin/docs/quickstart.md +61 -106
- package/bin/docs/rpc-commands.md +854 -0
- package/bin/docs/rpc-extension-ui.md +200 -0
- package/bin/docs/rpc.md +129 -1556
- package/bin/docs/sdk.md +76 -1160
- package/bin/docs/security.md +70 -32
- package/bin/docs/session-format.md +39 -216
- package/bin/docs/sessions.md +43 -121
- package/bin/docs/settings.md +112 -367
- package/bin/docs/shell-aliases.md +85 -5
- package/bin/docs/skills.md +51 -189
- package/bin/docs/slash-commands.md +63 -0
- package/bin/docs/terminal-setup.md +107 -79
- package/bin/docs/termux.md +74 -83
- package/bin/docs/themes.md +68 -280
- package/bin/docs/tmux.md +31 -39
- package/bin/docs/tui.md +69 -923
- package/bin/docs/usage.md +79 -285
- package/bin/docs/windows.md +43 -17
- package/bin/export-html/template.js +6 -1
- package/bin/knightcode +2 -2
- package/bin/package.json +6 -6
- package/package.json +1 -1
- package/bin/docs/development.md +0 -71
|
@@ -1,784 +1,134 @@
|
|
|
1
1
|
# Custom Providers
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
-
|
|
26
|
-
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
name?: string;
|
|
136
|
-
context_window?: number;
|
|
137
|
-
max_tokens?: number;
|
|
138
|
-
}>;
|
|
139
|
-
};
|
|
140
|
-
|
|
141
|
-
knightcode.registerProvider("local-openai", {
|
|
142
|
-
baseUrl: "http://localhost:1234/v1",
|
|
143
|
-
apiKey: "$LOCAL_OPENAI_API_KEY",
|
|
144
|
-
api: "openai-completions",
|
|
145
|
-
models: payload.data.map((model) => ({
|
|
146
|
-
id: model.id,
|
|
147
|
-
name: model.name ?? model.id,
|
|
148
|
-
reasoning: false,
|
|
149
|
-
input: ["text"],
|
|
150
|
-
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
|
151
|
-
contextWindow: model.context_window ?? 128000,
|
|
152
|
-
maxTokens: model.max_tokens ?? 4096,
|
|
153
|
-
})),
|
|
154
|
-
});
|
|
155
|
-
}
|
|
156
|
-
```
|
|
157
|
-
|
|
158
|
-
This registers the fetched models before startup finishes.
|
|
159
|
-
|
|
160
|
-
```typescript
|
|
161
|
-
knightcode.registerProvider("my-llm", {
|
|
162
|
-
baseUrl: "https://api.my-llm.com/v1",
|
|
163
|
-
apiKey: "$MY_LLM_API_KEY", // env var reference
|
|
164
|
-
api: "openai-completions", // which streaming API to use
|
|
165
|
-
models: [
|
|
166
|
-
{
|
|
167
|
-
id: "my-llm-large",
|
|
168
|
-
name: "My LLM Large",
|
|
169
|
-
reasoning: true, // supports extended thinking
|
|
170
|
-
input: ["text", "image"],
|
|
171
|
-
cost: {
|
|
172
|
-
input: 3.0, // $/million tokens
|
|
173
|
-
output: 15.0,
|
|
174
|
-
cacheRead: 0.3,
|
|
175
|
-
cacheWrite: 3.75
|
|
176
|
-
},
|
|
177
|
-
contextWindow: 200000,
|
|
178
|
-
maxTokens: 16384
|
|
179
|
-
}
|
|
180
|
-
]
|
|
181
|
-
});
|
|
182
|
-
```
|
|
183
|
-
|
|
184
|
-
When `models` is provided, it **replaces** all existing models for that provider.
|
|
185
|
-
|
|
186
|
-
`apiKey` and custom header values use the same config value syntax as `models.json`: `!command` at the start executes a command for the whole value, `$ENV_VAR` and `${ENV_VAR}` interpolate environment variables, `$$` emits a literal `$`, and `$!` emits a literal `!`.
|
|
187
|
-
|
|
188
|
-
## Unregister Provider
|
|
189
|
-
|
|
190
|
-
Use `knightcode.unregisterProvider(name)` to remove a provider that was previously registered via `knightcode.registerProvider(name, ...)`:
|
|
191
|
-
|
|
192
|
-
```typescript
|
|
193
|
-
// Register
|
|
194
|
-
knightcode.registerProvider("my-llm", {
|
|
195
|
-
baseUrl: "https://api.my-llm.com/v1",
|
|
196
|
-
apiKey: "$MY_LLM_API_KEY",
|
|
197
|
-
api: "openai-completions",
|
|
198
|
-
models: [
|
|
199
|
-
{
|
|
200
|
-
id: "my-llm-large",
|
|
201
|
-
name: "My LLM Large",
|
|
202
|
-
reasoning: true,
|
|
203
|
-
input: ["text", "image"],
|
|
204
|
-
cost: { input: 3.0, output: 15.0, cacheRead: 0.3, cacheWrite: 3.75 },
|
|
205
|
-
contextWindow: 200000,
|
|
206
|
-
maxTokens: 16384
|
|
207
|
-
}
|
|
208
|
-
]
|
|
209
|
-
});
|
|
210
|
-
|
|
211
|
-
// Later, remove it
|
|
212
|
-
knightcode.unregisterProvider("my-llm");
|
|
213
|
-
```
|
|
214
|
-
|
|
215
|
-
Unregistering removes that provider's dynamic models, API key fallback, OAuth provider registration, and custom stream handler registrations. Any built-in models or provider behavior that were overridden are restored.
|
|
216
|
-
|
|
217
|
-
Calls made after the initial extension load phase are applied immediately, so no `/reload` is required.
|
|
218
|
-
|
|
219
|
-
### API Types
|
|
220
|
-
|
|
221
|
-
The `api` field determines which streaming implementation is used:
|
|
222
|
-
|
|
223
|
-
| API | Use for |
|
|
224
|
-
|-----|---------|
|
|
225
|
-
| `anthropic-messages` | Anthropic Claude API and compatibles |
|
|
226
|
-
| `openai-completions` | OpenAI Chat Completions API and compatibles |
|
|
227
|
-
| `openai-responses` | OpenAI Responses API |
|
|
228
|
-
| `azure-openai-responses` | Azure OpenAI Responses API |
|
|
229
|
-
| `openai-codex-responses` | OpenAI Codex Responses API |
|
|
230
|
-
| `mistral-conversations` | Native Mistral Chat Completions streaming |
|
|
231
|
-
| `google-generative-ai` | Google Generative AI API |
|
|
232
|
-
| `google-vertex` | Google Vertex AI API |
|
|
233
|
-
| `bedrock-converse-stream` | Amazon Bedrock Converse API |
|
|
234
|
-
|
|
235
|
-
Most OpenAI-compatible providers work with `openai-completions`. Use model-level `thinkingLevelMap` for model-specific thinking levels, and `compat` for provider quirks. The `xhigh` and `max` levels are opt-in, require non-null map entries, and may be separated by unsupported holes:
|
|
236
|
-
|
|
237
|
-
```typescript
|
|
238
|
-
models: [{
|
|
239
|
-
id: "custom-model",
|
|
240
|
-
// ...
|
|
241
|
-
reasoning: true,
|
|
242
|
-
thinkingLevelMap: { // map knightcode levels to provider values; null hides unsupported levels
|
|
243
|
-
minimal: null,
|
|
244
|
-
low: null,
|
|
245
|
-
medium: null,
|
|
246
|
-
high: "default",
|
|
247
|
-
xhigh: null,
|
|
248
|
-
max: "max"
|
|
249
|
-
},
|
|
250
|
-
compat: {
|
|
251
|
-
supportsDeveloperRole: false, // use "system" instead of "developer"
|
|
252
|
-
supportsReasoningEffort: true,
|
|
253
|
-
maxTokensField: "max_tokens", // instead of "max_completion_tokens"
|
|
254
|
-
requiresToolResultName: true, // tool results need name field
|
|
255
|
-
thinkingFormat: "qwen", // top-level enable_thinking: true
|
|
256
|
-
cacheControlFormat: "anthropic" // Anthropic-style cache_control markers
|
|
257
|
-
}
|
|
258
|
-
}]
|
|
259
|
-
```
|
|
260
|
-
|
|
261
|
-
Use `openrouter` for OpenRouter-style `reasoning: { effort }` controls. Use `together` for Together-style `reasoning: { enabled }` controls; with `supportsReasoningEffort`, it also sends `reasoning_effort`. Use `qwen-chat-template` for local Qwen-compatible servers that read `chat_template_kwargs.enable_thinking` and need `preserve_thinking`.
|
|
262
|
-
Use `cacheControlFormat: "anthropic"` for OpenAI-compatible providers that expose Anthropic-style prompt caching via `cache_control` on the system prompt, last tool definition, and last user, assistant, or tool-result text content.
|
|
263
|
-
|
|
264
|
-
For Anthropic-compatible providers using `api: "anthropic-messages"`, set `compat.forceAdaptiveThinking: true` on models or providers whose upstream model requires adaptive thinking (`thinking.type: "adaptive"` plus `output_config.effort`). Built-in adaptive Claude models set this automatically. Set `compat.allowEmptySignature: true` only for providers that emit empty thinking signatures and expect `signature: ""` on replay.
|
|
265
|
-
|
|
266
|
-
> Migration note: Mistral moved from `openai-completions` to `mistral-conversations`.
|
|
267
|
-
> Use `mistral-conversations` for native Mistral models.
|
|
268
|
-
> If you intentionally route Mistral-compatible/custom endpoints through `openai-completions`, set `compat` flags explicitly as needed.
|
|
269
|
-
|
|
270
|
-
### Auth Header
|
|
271
|
-
|
|
272
|
-
If your provider expects `Authorization: Bearer <key>` but doesn't use a standard API, set `authHeader: true`:
|
|
273
|
-
|
|
274
|
-
```typescript
|
|
275
|
-
knightcode.registerProvider("custom-api", {
|
|
276
|
-
baseUrl: "https://api.example.com",
|
|
277
|
-
apiKey: "$MY_API_KEY",
|
|
278
|
-
authHeader: true, // adds Authorization: Bearer header
|
|
279
|
-
api: "openai-completions",
|
|
280
|
-
models: [...]
|
|
281
|
-
});
|
|
282
|
-
```
|
|
283
|
-
|
|
284
|
-
The key is resolved for each request. An explicit request `Authorization` header takes precedence over the generated value.
|
|
285
|
-
|
|
286
|
-
## OAuth Support
|
|
287
|
-
|
|
288
|
-
Add OAuth/SSO authentication that integrates with `/login`:
|
|
289
|
-
|
|
290
|
-
```typescript
|
|
291
|
-
import type { OAuthCredentials, OAuthLoginCallbacks } from "@knightcode/ai";
|
|
292
|
-
|
|
293
|
-
knightcode.registerProvider("corporate-ai", {
|
|
294
|
-
baseUrl: "https://ai.corp.com/v1",
|
|
295
|
-
api: "openai-responses",
|
|
296
|
-
models: [...],
|
|
297
|
-
oauth: {
|
|
298
|
-
name: "Corporate AI (SSO)",
|
|
299
|
-
|
|
300
|
-
async login(callbacks: OAuthLoginCallbacks): Promise<OAuthCredentials> {
|
|
301
|
-
const method = await callbacks.onSelect({
|
|
302
|
-
message: "Select login method:",
|
|
303
|
-
options: [
|
|
304
|
-
{ id: "browser", label: "Browser OAuth" },
|
|
305
|
-
{ id: "device", label: "Device code" }
|
|
306
|
-
]
|
|
307
|
-
});
|
|
308
|
-
if (!method) throw new Error("Login cancelled");
|
|
309
|
-
|
|
310
|
-
let code: string;
|
|
311
|
-
if (method === "device") {
|
|
312
|
-
callbacks.onDeviceCode({
|
|
313
|
-
userCode: "ABCD-1234",
|
|
314
|
-
verificationUri: "https://sso.corp.com/device",
|
|
315
|
-
intervalSeconds: 5,
|
|
316
|
-
expiresInSeconds: 900
|
|
317
|
-
});
|
|
318
|
-
code = await pollDeviceCodeUntilComplete();
|
|
319
|
-
} else {
|
|
320
|
-
callbacks.onAuth({ url: "https://sso.corp.com/authorize?..." });
|
|
321
|
-
code = await callbacks.onPrompt({ message: "Enter SSO code:" });
|
|
322
|
-
}
|
|
323
|
-
|
|
324
|
-
// Exchange for tokens (your implementation)
|
|
325
|
-
const tokens = await exchangeCodeForTokens(code);
|
|
326
|
-
|
|
327
|
-
return {
|
|
328
|
-
refresh: tokens.refreshToken,
|
|
329
|
-
access: tokens.accessToken,
|
|
330
|
-
expires: Date.now() + tokens.expiresIn * 1000
|
|
331
|
-
};
|
|
332
|
-
},
|
|
333
|
-
|
|
334
|
-
async refreshToken(credentials: OAuthCredentials, signal: AbortSignal): Promise<OAuthCredentials> {
|
|
335
|
-
const tokens = await refreshAccessToken(credentials.refresh, signal);
|
|
336
|
-
return {
|
|
337
|
-
refresh: tokens.refreshToken ?? credentials.refresh,
|
|
338
|
-
access: tokens.accessToken,
|
|
339
|
-
expires: Date.now() + tokens.expiresIn * 1000
|
|
340
|
-
};
|
|
341
|
-
},
|
|
342
|
-
|
|
343
|
-
getApiKey(credentials: OAuthCredentials): string {
|
|
344
|
-
return credentials.access;
|
|
345
|
-
}
|
|
346
|
-
}
|
|
347
|
-
});
|
|
348
|
-
```
|
|
349
|
-
|
|
350
|
-
After registration, users can authenticate via `/login corporate-ai`.
|
|
351
|
-
|
|
352
|
-
### OAuthLoginCallbacks
|
|
353
|
-
|
|
354
|
-
The `callbacks` object provides UI-neutral interactions for the provider-owned flow:
|
|
355
|
-
|
|
356
|
-
```typescript
|
|
357
|
-
interface OAuthLoginCallbacks {
|
|
358
|
-
// Open URL in browser (for OAuth redirects)
|
|
359
|
-
onAuth(params: { url: string }): void;
|
|
360
|
-
|
|
361
|
-
// Show device code (for device authorization flow)
|
|
362
|
-
onDeviceCode(params: {
|
|
363
|
-
userCode: string;
|
|
364
|
-
verificationUri: string;
|
|
365
|
-
intervalSeconds?: number;
|
|
366
|
-
expiresInSeconds?: number;
|
|
367
|
-
}): void;
|
|
368
|
-
|
|
369
|
-
// Show transient progress
|
|
370
|
-
onProgress?(message: string): void;
|
|
371
|
-
|
|
372
|
-
// Prompt user for input (for manual token entry)
|
|
373
|
-
onPrompt(params: { message: string }): Promise<string>;
|
|
374
|
-
|
|
375
|
-
// Show an interactive selector, e.g. to choose browser OAuth vs device code
|
|
376
|
-
onSelect(params: {
|
|
377
|
-
message: string;
|
|
378
|
-
options: { id: string; label: string }[];
|
|
379
|
-
}): Promise<string | undefined>;
|
|
380
|
-
}
|
|
381
|
-
```
|
|
382
|
-
|
|
383
|
-
### OAuthCredentials
|
|
384
|
-
|
|
385
|
-
Credentials are persisted in `~/.knightcode/agent/auth.json`:
|
|
386
|
-
|
|
387
|
-
```typescript
|
|
388
|
-
interface OAuthCredentials {
|
|
389
|
-
refresh: string; // Refresh token (for refreshToken())
|
|
390
|
-
access: string; // Access token (returned by getApiKey())
|
|
391
|
-
expires: number; // Expiration timestamp in milliseconds
|
|
392
|
-
}
|
|
393
|
-
```
|
|
394
|
-
|
|
395
|
-
## Custom Streaming API
|
|
396
|
-
|
|
397
|
-
For providers with non-standard APIs, implement `streamSimple`. Study the existing API implementations before writing your own:
|
|
398
|
-
|
|
399
|
-
**Reference implementations:**
|
|
400
|
-
- [anthropic-messages.ts](https://github.com/KnightCodeAI/knightcode/blob/main/packages/ai/src/api/anthropic-messages.ts) - Anthropic Messages API
|
|
401
|
-
- [mistral-conversations.ts](https://github.com/KnightCodeAI/knightcode/blob/main/packages/ai/src/api/mistral-conversations.ts) - Mistral Conversations API
|
|
402
|
-
- [openai-completions.ts](https://github.com/KnightCodeAI/knightcode/blob/main/packages/ai/src/api/openai-completions.ts) - OpenAI Chat Completions
|
|
403
|
-
- [openai-responses.ts](https://github.com/KnightCodeAI/knightcode/blob/main/packages/ai/src/api/openai-responses.ts) - OpenAI Responses API
|
|
404
|
-
- [google-generative-ai.ts](https://github.com/KnightCodeAI/knightcode/blob/main/packages/ai/src/api/google-generative-ai.ts) - Google Generative AI
|
|
405
|
-
- [bedrock-converse-stream.ts](https://github.com/KnightCodeAI/knightcode/blob/main/packages/ai/src/api/bedrock-converse-stream.ts) - AWS Bedrock
|
|
406
|
-
|
|
407
|
-
### Stream Pattern
|
|
408
|
-
|
|
409
|
-
All providers follow the same pattern. The context is a normalized transcript: the system prompt and tool declarations live in its system messages, so read them with `getCurrentSystemPrompt(context.messages)` and `getCurrentTools(context.messages)` rather than expecting `context.systemPrompt` or `context.tools`. Models that accept system messages mid-conversation can send them in place; otherwise call `collapseSystemMessages(context)` first to fold later system messages into the leading one.
|
|
410
|
-
|
|
411
|
-
```typescript
|
|
412
|
-
import {
|
|
413
|
-
type AssistantMessage,
|
|
414
|
-
type AssistantMessageEventStream,
|
|
415
|
-
type Model,
|
|
416
|
-
type SimpleStreamOptions,
|
|
417
|
-
type TranscriptContext,
|
|
418
|
-
calculateCost,
|
|
419
|
-
collapseSystemMessages,
|
|
420
|
-
createAssistantMessageEventStream,
|
|
421
|
-
getCurrentSystemPrompt,
|
|
422
|
-
getCurrentTools,
|
|
423
|
-
} from "@knightcode/ai";
|
|
424
|
-
|
|
425
|
-
function streamMyProvider(
|
|
426
|
-
model: Model<any>,
|
|
427
|
-
context: TranscriptContext,
|
|
428
|
-
options?: SimpleStreamOptions
|
|
429
|
-
): AssistantMessageEventStream {
|
|
430
|
-
const stream = createAssistantMessageEventStream();
|
|
431
|
-
const transcript = collapseSystemMessages(context);
|
|
432
|
-
const systemPrompt = getCurrentSystemPrompt(transcript.messages);
|
|
433
|
-
const tools = getCurrentTools(transcript.messages);
|
|
434
|
-
|
|
435
|
-
(async () => {
|
|
436
|
-
// Initialize output message
|
|
437
|
-
const output: AssistantMessage = {
|
|
438
|
-
role: "assistant",
|
|
439
|
-
content: [],
|
|
440
|
-
api: model.api,
|
|
441
|
-
provider: model.provider,
|
|
442
|
-
model: model.id,
|
|
443
|
-
usage: {
|
|
444
|
-
input: 0,
|
|
445
|
-
output: 0,
|
|
446
|
-
cacheRead: 0,
|
|
447
|
-
cacheWrite: 0,
|
|
448
|
-
totalTokens: 0,
|
|
449
|
-
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
|
450
|
-
},
|
|
451
|
-
stopReason: "pending",
|
|
452
|
-
timestamp: Date.now(),
|
|
453
|
-
};
|
|
454
|
-
|
|
455
|
-
try {
|
|
456
|
-
// Push start event
|
|
457
|
-
stream.push({ type: "start", partial: output });
|
|
458
|
-
|
|
459
|
-
// Make API request and process response...
|
|
460
|
-
// Push content events as they arrive and set stopReason from the terminal event.
|
|
461
|
-
if (output.stopReason === "pending") {
|
|
462
|
-
throw new Error("Provider stream ended without a stop reason");
|
|
463
|
-
}
|
|
464
|
-
if (output.stopReason === "error" || output.stopReason === "aborted") {
|
|
465
|
-
throw new Error(output.errorMessage || "An unknown error occurred");
|
|
466
|
-
}
|
|
467
|
-
|
|
468
|
-
// Push done event
|
|
469
|
-
stream.push({
|
|
470
|
-
type: "done",
|
|
471
|
-
reason: output.stopReason,
|
|
472
|
-
message: output
|
|
473
|
-
});
|
|
474
|
-
stream.end();
|
|
475
|
-
} catch (error) {
|
|
476
|
-
output.stopReason = options?.signal?.aborted ? "aborted" : "error";
|
|
477
|
-
output.errorMessage = error instanceof Error ? error.message : String(error);
|
|
478
|
-
stream.push({ type: "error", reason: output.stopReason, error: output });
|
|
479
|
-
stream.end();
|
|
480
|
-
}
|
|
481
|
-
})();
|
|
482
|
-
|
|
483
|
-
return stream;
|
|
484
|
-
}
|
|
485
|
-
```
|
|
486
|
-
|
|
487
|
-
### Event Types
|
|
488
|
-
|
|
489
|
-
Push events via `stream.push()` in this order:
|
|
490
|
-
|
|
491
|
-
1. `{ type: "start", partial: output }` - Stream started
|
|
492
|
-
|
|
493
|
-
2. Content events (repeatable, track `contentIndex` for each block):
|
|
494
|
-
- `{ type: "text_start", contentIndex, partial }` - Text block started
|
|
495
|
-
- `{ type: "text_delta", contentIndex, delta, partial }` - Text chunk
|
|
496
|
-
- `{ type: "text_end", contentIndex, content, partial }` - Text block ended
|
|
497
|
-
- `{ type: "thinking_start", contentIndex, partial }` - Thinking started
|
|
498
|
-
- `{ type: "thinking_delta", contentIndex, delta, partial }` - Thinking chunk
|
|
499
|
-
- `{ type: "thinking_end", contentIndex, content, partial }` - Thinking ended
|
|
500
|
-
- `{ type: "toolcall_start", contentIndex, partial }` - Tool call started
|
|
501
|
-
- `{ type: "toolcall_delta", contentIndex, delta, partial }` - Tool call JSON chunk
|
|
502
|
-
- `{ type: "toolcall_end", contentIndex, toolCall, partial }` - Tool call ended
|
|
503
|
-
|
|
504
|
-
3. `{ type: "done", reason, message }` or `{ type: "error", reason, error }` - Stream ended
|
|
505
|
-
|
|
506
|
-
The `partial` field in each event contains the current `AssistantMessage` state. Update `output.content` as you receive data, then include `output` as the `partial`.
|
|
507
|
-
|
|
508
|
-
### Content Blocks
|
|
509
|
-
|
|
510
|
-
Add content blocks to `output.content` as they arrive:
|
|
511
|
-
|
|
512
|
-
```typescript
|
|
513
|
-
// Text block
|
|
514
|
-
output.content.push({ type: "text", text: "" });
|
|
515
|
-
stream.push({ type: "text_start", contentIndex: output.content.length - 1, partial: output });
|
|
516
|
-
|
|
517
|
-
// As text arrives
|
|
518
|
-
const block = output.content[contentIndex];
|
|
519
|
-
if (block.type === "text") {
|
|
520
|
-
block.text += delta;
|
|
521
|
-
stream.push({ type: "text_delta", contentIndex, delta, partial: output });
|
|
522
|
-
}
|
|
523
|
-
|
|
524
|
-
// When block completes
|
|
525
|
-
stream.push({ type: "text_end", contentIndex, content: block.text, partial: output });
|
|
526
|
-
```
|
|
527
|
-
|
|
528
|
-
### Tool Calls
|
|
529
|
-
|
|
530
|
-
Tool calls require accumulating JSON and parsing:
|
|
531
|
-
|
|
532
|
-
```typescript
|
|
533
|
-
// Start tool call
|
|
534
|
-
output.content.push({
|
|
535
|
-
type: "toolCall",
|
|
536
|
-
id: toolCallId,
|
|
537
|
-
name: toolName,
|
|
538
|
-
arguments: {}
|
|
539
|
-
});
|
|
540
|
-
stream.push({ type: "toolcall_start", contentIndex: output.content.length - 1, partial: output });
|
|
541
|
-
|
|
542
|
-
// Accumulate JSON
|
|
543
|
-
let partialJson = "";
|
|
544
|
-
partialJson += jsonDelta;
|
|
545
|
-
try {
|
|
546
|
-
block.arguments = JSON.parse(partialJson);
|
|
547
|
-
} catch {}
|
|
548
|
-
stream.push({ type: "toolcall_delta", contentIndex, delta: jsonDelta, partial: output });
|
|
549
|
-
|
|
550
|
-
// Complete
|
|
551
|
-
stream.push({
|
|
552
|
-
type: "toolcall_end",
|
|
553
|
-
contentIndex,
|
|
554
|
-
toolCall: { type: "toolCall", id, name, arguments: block.arguments },
|
|
555
|
-
partial: output
|
|
556
|
-
});
|
|
557
|
-
```
|
|
558
|
-
|
|
559
|
-
### Usage and Cost
|
|
560
|
-
|
|
561
|
-
Update usage from API response and calculate cost:
|
|
562
|
-
|
|
563
|
-
```typescript
|
|
564
|
-
output.usage.input = response.usage.input_tokens;
|
|
565
|
-
output.usage.output = response.usage.output_tokens;
|
|
566
|
-
output.usage.cacheRead = response.usage.cache_read_tokens ?? 0;
|
|
567
|
-
output.usage.cacheWrite = response.usage.cache_write_tokens ?? 0;
|
|
568
|
-
output.usage.totalTokens = output.usage.input + output.usage.output +
|
|
569
|
-
output.usage.cacheRead + output.usage.cacheWrite;
|
|
570
|
-
calculateCost(model, output.usage);
|
|
571
|
-
```
|
|
572
|
-
|
|
573
|
-
### Context Overflow Errors
|
|
574
|
-
|
|
575
|
-
When a request exceeds the model's context window, knightcode can recover automatically by compacting the conversation and retrying. This recovery only kicks in if knightcode recognizes the failure as an overflow.
|
|
576
|
-
|
|
577
|
-
Detection runs on the finalized assistant message:
|
|
578
|
-
|
|
579
|
-
- `stopReason === "error"`
|
|
580
|
-
- `errorMessage` matches one of knightcode's known overflow patterns (see [`packages/ai/src/utils/overflow.ts`](https://github.com/KnightCodeAI/knightcode/blob/main/packages/ai/src/utils/overflow.ts))
|
|
581
|
-
|
|
582
|
-
If your provider returns overflow errors with a message knightcode does not recognize, normalize the error from the same extension that registers the provider. Use a `message_end` handler to rewrite the assistant message so its `errorMessage` starts with a phrase knightcode recognizes. The generic fallback `context_length_exceeded` is the safest choice.
|
|
583
|
-
|
|
584
|
-
```typescript
|
|
585
|
-
const MY_PROVIDER_OVERFLOW_PATTERN = /your provider's overflow phrase/i;
|
|
586
|
-
|
|
587
|
-
export default function (knightcode: ExtensionAPI) {
|
|
588
|
-
knightcode.registerProvider("my-provider", { /* ... */ });
|
|
589
|
-
|
|
590
|
-
knightcode.on("message_end", (event, ctx) => {
|
|
591
|
-
const message = event.message;
|
|
592
|
-
if (message.role !== "assistant") return;
|
|
593
|
-
if (message.stopReason !== "error") return;
|
|
594
|
-
if (
|
|
595
|
-
message.provider !== "my-provider" &&
|
|
596
|
-
ctx.model?.provider !== "my-provider"
|
|
597
|
-
)
|
|
598
|
-
return;
|
|
599
|
-
|
|
600
|
-
const errorMessage = message.errorMessage ?? "";
|
|
601
|
-
if (errorMessage.includes("context_length_exceeded")) return;
|
|
602
|
-
if (!MY_PROVIDER_OVERFLOW_PATTERN.test(errorMessage)) return;
|
|
603
|
-
|
|
604
|
-
return {
|
|
605
|
-
message: {
|
|
606
|
-
...message,
|
|
607
|
-
errorMessage: `context_length_exceeded: ${errorMessage}`,
|
|
608
|
-
},
|
|
609
|
-
};
|
|
610
|
-
});
|
|
611
|
-
}
|
|
612
|
-
```
|
|
613
|
-
|
|
614
|
-
`message_end` runs before knightcode tracks the assistant message for auto-compaction, so the rewritten `errorMessage` is what knightcode checks. With this in place, knightcode will:
|
|
615
|
-
|
|
616
|
-
1. Detect the overflow from `errorMessage`.
|
|
617
|
-
2. Drop the failed assistant message from live context.
|
|
618
|
-
3. Run compaction.
|
|
619
|
-
4. Retry the request once.
|
|
620
|
-
|
|
621
|
-
Guard the rewrite carefully:
|
|
622
|
-
|
|
623
|
-
- Scope it to your provider (`message.provider` and `ctx.model?.provider`) so unrelated errors from other providers are untouched.
|
|
624
|
-
- Match a provider-specific pattern, not knightcode's generic overflow patterns. Rewriting rate-limit or throttling errors (`rate limit`, `too many requests`) would falsely trigger compaction instead of knightcode's normal retry-with-backoff path.
|
|
625
|
-
- Skip when `errorMessage` already includes `context_length_exceeded` so the handler is idempotent.
|
|
626
|
-
|
|
627
|
-
### Registration
|
|
628
|
-
|
|
629
|
-
Register your stream function:
|
|
630
|
-
|
|
631
|
-
```typescript
|
|
632
|
-
knightcode.registerProvider("my-provider", {
|
|
633
|
-
baseUrl: "https://api.example.com",
|
|
634
|
-
apiKey: "$MY_API_KEY",
|
|
635
|
-
api: "my-custom-api",
|
|
636
|
-
models: [...],
|
|
637
|
-
streamSimple: streamMyProvider
|
|
638
|
-
});
|
|
639
|
-
```
|
|
640
|
-
|
|
641
|
-
## Testing Your Implementation
|
|
642
|
-
|
|
643
|
-
Test your provider against the same test suites used by built-in providers. Copy and adapt these test files from [packages/ai/test/](https://github.com/KnightCodeAI/knightcode/tree/main/packages/ai/test):
|
|
644
|
-
|
|
645
|
-
| Test | Purpose |
|
|
646
|
-
|------|---------|
|
|
647
|
-
| `stream.test.ts` | Basic streaming, text output |
|
|
648
|
-
| `tokens.test.ts` | Token counting and usage |
|
|
649
|
-
| `abort.test.ts` | AbortSignal handling |
|
|
650
|
-
| `empty.test.ts` | Empty/minimal responses |
|
|
651
|
-
| `context-overflow.test.ts` | Context window limits |
|
|
652
|
-
| `image-limits.test.ts` | Image input handling |
|
|
653
|
-
| `unicode-surrogate.test.ts` | Unicode edge cases |
|
|
654
|
-
| `tool-call-without-result.test.ts` | Tool call edge cases |
|
|
655
|
-
| `image-tool-result.test.ts` | Images in tool results |
|
|
656
|
-
| `total-tokens.test.ts` | Total token calculation |
|
|
657
|
-
| `cross-provider-handoff.test.ts` | Context handoff between providers |
|
|
658
|
-
|
|
659
|
-
Run tests with your provider/model pairs to verify compatibility.
|
|
660
|
-
|
|
661
|
-
## Config Reference
|
|
662
|
-
|
|
663
|
-
```typescript
|
|
664
|
-
interface ProviderConfig {
|
|
665
|
-
/** Display name for the provider in UI such as /login. */
|
|
666
|
-
name?: string;
|
|
667
|
-
|
|
668
|
-
/** API endpoint URL. Required when defining models. */
|
|
669
|
-
baseUrl?: string;
|
|
670
|
-
|
|
671
|
-
/** API key literal, env interpolation ($ENV_VAR or ${ENV_VAR}), or !command. Required when defining models (unless oauth). */
|
|
672
|
-
apiKey?: string;
|
|
673
|
-
|
|
674
|
-
/** API type for streaming. Required at provider or model level when defining models. */
|
|
675
|
-
api?: Api;
|
|
676
|
-
|
|
677
|
-
/** Custom streaming implementation for non-standard APIs. Receives a normalized transcript. */
|
|
678
|
-
streamSimple?: (
|
|
679
|
-
model: Model<Api>,
|
|
680
|
-
context: TranscriptContext,
|
|
681
|
-
options?: SimpleStreamOptions
|
|
682
|
-
) => AssistantMessageEventStream;
|
|
683
|
-
|
|
684
|
-
/** Custom headers to include in requests. Values use the same resolution syntax as apiKey. */
|
|
685
|
-
headers?: Record<string, string>;
|
|
686
|
-
|
|
687
|
-
/** If true, adds Authorization: Bearer header with the resolved API key. */
|
|
688
|
-
authHeader?: boolean;
|
|
689
|
-
|
|
690
|
-
/** Models to register. If provided, replaces all existing models for this provider. */
|
|
691
|
-
models?: ProviderModelConfig[];
|
|
692
|
-
|
|
693
|
-
/** OAuth provider for /login support. */
|
|
694
|
-
oauth?: {
|
|
695
|
-
name: string;
|
|
696
|
-
login(callbacks: OAuthLoginCallbacks): Promise<OAuthCredentials>;
|
|
697
|
-
refreshToken(credentials: OAuthCredentials, signal: AbortSignal): Promise<OAuthCredentials>;
|
|
698
|
-
getApiKey(credentials: OAuthCredentials): string;
|
|
699
|
-
};
|
|
700
|
-
}
|
|
701
|
-
```
|
|
702
|
-
|
|
703
|
-
## Model Definition Reference
|
|
704
|
-
|
|
705
|
-
```typescript
|
|
706
|
-
interface ProviderModelConfig {
|
|
707
|
-
/** Model ID (e.g., "claude-sonnet-4-20250514"). */
|
|
708
|
-
id: string;
|
|
709
|
-
|
|
710
|
-
/** Display name (e.g., "Claude 4 Sonnet"). */
|
|
711
|
-
name: string;
|
|
712
|
-
|
|
713
|
-
/** API type override for this specific model. */
|
|
714
|
-
api?: Api;
|
|
715
|
-
|
|
716
|
-
/** API endpoint URL override for this specific model. */
|
|
717
|
-
baseUrl?: string;
|
|
718
|
-
|
|
719
|
-
/** Whether the model supports extended thinking. */
|
|
720
|
-
reasoning: boolean;
|
|
721
|
-
|
|
722
|
-
/** Maps knightcode thinking levels to provider/model-specific values; null marks a level unsupported. */
|
|
723
|
-
thinkingLevelMap?: Partial<Record<"off" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max", string | null>>;
|
|
724
|
-
|
|
725
|
-
/** Supported input types. */
|
|
726
|
-
input: ("text" | "image")[];
|
|
727
|
-
|
|
728
|
-
/** Cost per million tokens (for usage tracking). */
|
|
729
|
-
cost: {
|
|
730
|
-
input: number;
|
|
731
|
-
output: number;
|
|
732
|
-
cacheRead: number;
|
|
733
|
-
cacheWrite: number;
|
|
734
|
-
};
|
|
735
|
-
|
|
736
|
-
/** Maximum context window size in tokens. */
|
|
737
|
-
contextWindow: number;
|
|
738
|
-
|
|
739
|
-
/** Maximum output tokens. */
|
|
740
|
-
maxTokens: number;
|
|
741
|
-
|
|
742
|
-
/** Custom headers for this specific model. */
|
|
743
|
-
headers?: Record<string, string>;
|
|
744
|
-
|
|
745
|
-
/** Compatibility settings for the selected API. */
|
|
746
|
-
compat?: {
|
|
747
|
-
// openai-completions
|
|
748
|
-
supportsStore?: boolean;
|
|
749
|
-
supportsDeveloperRole?: boolean;
|
|
750
|
-
supportsReasoningEffort?: boolean;
|
|
751
|
-
supportsUsageInStreaming?: boolean;
|
|
752
|
-
supportsFinishReason?: boolean;
|
|
753
|
-
supportsStrictMode?: boolean;
|
|
754
|
-
supportsOpenAIGrammarTools?: boolean; // openai-completions/openai-responses; false falls back to normal function tools
|
|
755
|
-
maxTokensField?: "max_completion_tokens" | "max_tokens";
|
|
756
|
-
requiresToolResultName?: boolean;
|
|
757
|
-
requiresAssistantAfterToolResult?: boolean;
|
|
758
|
-
requiresThinkingAsText?: boolean;
|
|
759
|
-
requiresReasoningContentOnAssistantMessages?: boolean;
|
|
760
|
-
thinkingFormat?: "openai" | "openrouter" | "deepseek" | "together" | "baseten" | "zai" | "qwen" | "chat-template" | "qwen-chat-template" | "string-thinking" | "ant-ling";
|
|
761
|
-
chatTemplateKwargs?: Record<string, string | number | boolean | null | { "$var": "thinking.enabled" | "thinking.effort" | "thinking.budget"; omitWhenOff?: boolean }>;
|
|
762
|
-
chatTemplateArgs?: Record<string, string | number | boolean | null | { "$var": "thinking.enabled" | "thinking.effort" | "thinking.budget"; omitWhenOff?: boolean }>;
|
|
763
|
-
thinkingTokenBudgetField?: "thinking_token_budget" | "thinking_budget" | "thinking_budget_tokens";
|
|
764
|
-
supportsThinkingTokenBudget?: boolean;
|
|
765
|
-
cacheControlFormat?: "anthropic";
|
|
766
|
-
sessionAffinityFormat?: "openai" | "openai-nosession" | "openrouter";
|
|
767
|
-
sendSessionAffinityHeaders?: boolean;
|
|
768
|
-
|
|
769
|
-
// anthropic-messages
|
|
770
|
-
supportsEagerToolInputStreaming?: boolean;
|
|
771
|
-
supportsLongCacheRetention?: boolean;
|
|
772
|
-
sendSessionAffinityHeaders?: boolean;
|
|
773
|
-
sessionAffinityFormat?: "openrouter";
|
|
774
|
-
supportsCacheControlOnTools?: boolean;
|
|
775
|
-
forceAdaptiveThinking?: boolean;
|
|
776
|
-
allowEmptySignature?: boolean;
|
|
777
|
-
supportsStrictTools?: boolean;
|
|
778
|
-
};
|
|
779
|
-
}
|
|
780
|
-
```
|
|
781
|
-
|
|
782
|
-
`openrouter` sends `reasoning: { effort }`. `deepseek` sends `thinking: { type: "enabled" | "disabled" }` and `reasoning_effort` when enabled. `together` sends `reasoning: { enabled }` and also `reasoning_effort` when `supportsReasoningEffort` is enabled. `qwen` is for DashScope-style top-level `enable_thinking`. Use `qwen-chat-template` for local Qwen-compatible servers that read `chat_template_kwargs.enable_thinking` and need `preserve_thinking`. Use `chat-template` for configurable `chat_template_kwargs`, for example DeepSeek V3.x behind vLLM with `chatTemplateKwargs: { "thinking": { "$var": "thinking.enabled" } }`. Use `thinkingFormat: "baseten"` with `chatTemplateArgs` when the provider expects toggle values under `chat_template_args` and optionally supports top-level `reasoning_effort`.
|
|
783
|
-
`thinkingTokenBudgetField` sends a clamped per-level thinking budget as a top-level request field (`thinking_token_budget` on vLLM, `thinking_budget` on Qwen/SGLang, `thinking_budget_tokens` on llama.cpp). `supportsThinkingTokenBudget: true` is an alias for the vLLM field name. Do not combine it with `reasoning_effort` on DashScope Qwen models.
|
|
784
|
-
`cacheControlFormat: "anthropic"` applies Anthropic-style `cache_control` markers to the system prompt, last tool definition, and last user, assistant, or tool-result text content.
|
|
3
|
+
A provider extension connects KnightCode to a model service that needs custom authentication, model discovery, request handling, or streaming. If the service already speaks a supported API, configure it in `models.json` instead.
|
|
4
|
+
|
|
5
|
+
Provider extensions run inside KnightCode and can inspect credentials, prompts, tool definitions, model responses, and usage. Treat them as trusted code and avoid logging secrets or provider payloads.
|
|
6
|
+
|
|
7
|
+
## Choose the smallest integration
|
|
8
|
+
|
|
9
|
+
| Requirement | Use |
|
|
10
|
+
|---|---|
|
|
11
|
+
| Add models behind a supported API | [`models.json`](models.md#configure-a-compatible-endpoint) |
|
|
12
|
+
| Change an existing provider endpoint or headers | `models.json` or a small provider extension |
|
|
13
|
+
| Discover models dynamically | A provider with `refreshModels` |
|
|
14
|
+
| Add a `/login` flow | A provider with native or legacy OAuth configuration |
|
|
15
|
+
| Implement an unsupported wire protocol | A provider with `stream` or `streamSimple` |
|
|
16
|
+
|
|
17
|
+
A provider extension is an [extension](extensions.md), so it follows the same loading, trust, reload, and error behavior.
|
|
18
|
+
|
|
19
|
+
## Register a provider
|
|
20
|
+
|
|
21
|
+
Call `knightcode.registerProvider()` from the extension factory. KnightCode waits for asynchronous factories before startup continues, so providers registered there are available to startup model selection and `knightcode --list-models`.
|
|
22
|
+
|
|
23
|
+
There are two registration forms:
|
|
24
|
+
|
|
25
|
+
- Register a complete `Provider` from `@knightcode/ai` for native authentication, filtering, discovery, refresh, and streaming behavior.
|
|
26
|
+
- Register a provider name with `ProviderConfig` for the legacy configuration form used by existing extensions.
|
|
27
|
+
|
|
28
|
+
Prefer a complete provider for new integrations that own more than static endpoint and model metadata. KnightCode composes `models.json` overrides above a registered native provider.
|
|
29
|
+
|
|
30
|
+
Registering only `baseUrl` or `headers` for an existing provider preserves its built-in models. Supplying `models` in the legacy form replaces the models supplied by that registration.
|
|
31
|
+
|
|
32
|
+
Calls made after initial extension loading take effect immediately. Use `knightcode.unregisterProvider()` to remove the dynamic provider and restore built-in behavior that it replaced.
|
|
33
|
+
|
|
34
|
+
See the checked [GitLab Duo provider](../examples/extensions/custom-provider-gitlab-duo/) for a complete registration that delegates streaming to built-in API implementations.
|
|
35
|
+
|
|
36
|
+
## Provide authentication
|
|
37
|
+
|
|
38
|
+
Static providers can resolve an API key from a literal, environment interpolation, or a command. These values use the same syntax as `models.json`:
|
|
39
|
+
|
|
40
|
+
- `$NAME` and `${NAME}` read environment variables.
|
|
41
|
+
- A leading `!command` uses command output.
|
|
42
|
+
- `$$` emits a literal `$`.
|
|
43
|
+
- `$!` emits a literal leading `!`.
|
|
44
|
+
|
|
45
|
+
Use native provider authentication when the integration needs stored credentials, custom resolution, provider-scoped environment, or multiple login methods.
|
|
46
|
+
|
|
47
|
+
An OAuth provider supplies a display name, login flow, token refresh, and access-token resolution. After registration it appears in `/login`, and KnightCode stores returned credentials in `~/.knightcode/agent/auth.json`.
|
|
48
|
+
|
|
49
|
+
OAuth callbacks are UI-neutral. They can open an authorization URL, show a device code, report progress, request input, or ask the user to choose a login method. Honor cancellation and the supplied abort signal during network requests.
|
|
50
|
+
|
|
51
|
+
Never write access tokens, refresh tokens, authorization headers, or complete provider responses to ordinary logs.
|
|
52
|
+
|
|
53
|
+
## Supply and refresh models
|
|
54
|
+
|
|
55
|
+
Every model needs an ID, display name, input capabilities, context window, output limit, reasoning support, and cost metadata. Choose the API implementation at the provider level unless one model requires an override.
|
|
56
|
+
|
|
57
|
+
Set `promptCache.short` or `promptCache.long` to the provider's best-effort cache lifetime in seconds when KnightCode should keep an idle prompt cache warm. Leave them unset to disable cache warming for that retention tier.
|
|
58
|
+
|
|
59
|
+
Compatibility flags describe verified differences in an otherwise supported API. Do not enable them based only on an endpoint claiming compatibility.
|
|
60
|
+
|
|
61
|
+
Confirm the request fields and response behavior against the actual server.
|
|
62
|
+
|
|
63
|
+
Use `refreshModels` when the available catalog comes from a live service. Pass `context.signal` to blocking I/O so callers can cancel refreshes.
|
|
64
|
+
|
|
65
|
+
The two registration forms have different refresh contracts:
|
|
66
|
+
|
|
67
|
+
- A complete `Provider` returns nothing. It calls `context.publish({ update })` to install provider-owned model state, after which its synchronous `getModels()` exposes the latest list.
|
|
68
|
+
- Legacy `ProviderConfig.refreshModels` returns model definitions. KnightCode replaces that registration’s live models with the returned list and applies any requested persistence.
|
|
69
|
+
|
|
70
|
+
Publish persisted catalog data only when it should survive across runs. A live service such as llama.cpp can update its in-memory list without persisting it; a remote catalog can retain a snapshot for offline startup.
|
|
71
|
+
|
|
72
|
+
## Reuse a supported streaming API
|
|
73
|
+
|
|
74
|
+
Use one of KnightCode AI’s API implementations whenever the provider protocol matches it.
|
|
75
|
+
|
|
76
|
+
Supported implementations cover Anthropic Messages, OpenAI Chat Completions and Responses, Google Generative AI and Vertex, Azure OpenAI Responses, Mistral Conversations, and Bedrock Converse.
|
|
77
|
+
|
|
78
|
+
The provider can still customize authentication, base URLs, headers, model filtering, and discovery while delegating request conversion and streaming to an existing API implementation.
|
|
79
|
+
|
|
80
|
+
This is safer than copying a stream implementation because it preserves KnightCode’s message conversion, tool handling, usage accounting, cancellation, and compatibility behavior.
|
|
81
|
+
|
|
82
|
+
## Implement custom streaming
|
|
83
|
+
|
|
84
|
+
Implement `streamSimple` only when no existing API implementation can represent the service. Study the implementations under [`packages/ai/src/api`](https://github.com/KnightCodeAI/knightcode/tree/main/packages/ai/src/api) first.
|
|
85
|
+
|
|
86
|
+
The stream receives a normalized `TranscriptContext`. System prompts and tool declarations live in transcript system messages, so read them with `getCurrentSystemPrompt(context.messages)` and `getCurrentTools(context.messages)` rather than expecting `context.systemPrompt` or `context.tools`. A model that supports mid-conversation system messages can receive them in place; otherwise call `collapseSystemMessages(context)` to fold later system messages into the leading one.
|
|
87
|
+
|
|
88
|
+
A custom stream must:
|
|
89
|
+
|
|
90
|
+
1. Create an assistant message with provider, model, timestamp, pending stop reason, content, and zeroed usage.
|
|
91
|
+
2. After request setup succeeds, emit one `start` event before content events.
|
|
92
|
+
3. Update the message while emitting balanced text, thinking, and tool-call events.
|
|
93
|
+
4. Finalize usage, cost, content, and stop reason.
|
|
94
|
+
5. Emit exactly one terminal `done` or `error` event and close the stream.
|
|
95
|
+
6. Convert cancellation into an aborted result.
|
|
96
|
+
|
|
97
|
+
Request setup can fail before `start`; in that case the stream can terminate directly with `error`. Missing request authentication may also throw synchronously before a stream is returned.
|
|
98
|
+
|
|
99
|
+
Content indexes refer to blocks in the assistant message. Update each block before emitting the event whose `partial` field exposes that state. Tool-call arguments must contain valid parsed input by `toolcall_end`.
|
|
100
|
+
|
|
101
|
+
The stream must also honor request instrumentation supplied through `SimpleStreamOptions`:
|
|
102
|
+
|
|
103
|
+
- Call `options.onPayload` before sending the provider request and use any replacement payload it returns.
|
|
104
|
+
- Call `options.onResponse` after receiving the response but before consuming its body.
|
|
105
|
+
- Pass through the abort signal and provider-scoped environment.
|
|
106
|
+
|
|
107
|
+
These hooks power extension request inspection and response-header events. Omitting them makes the provider behave differently from KnightCode’s built-in providers.
|
|
108
|
+
|
|
109
|
+
## Report failures and usage
|
|
110
|
+
|
|
111
|
+
Set a concrete terminal stop reason. Error and aborted messages need an `errorMessage`; successful messages need accurate input, output, cache, total-token, and cost values.
|
|
112
|
+
|
|
113
|
+
KnightCode can compact and retry after recognized context-overflow errors. If the service uses an unknown message, normalize only that provider’s overflow response to `context_length_exceeded` in a guarded `message_end` handler.
|
|
114
|
+
|
|
115
|
+
Do not rewrite rate limits or transient provider failures as context overflow. Those failures use KnightCode’s normal retry behavior instead.
|
|
116
|
+
|
|
117
|
+
## Test the integration
|
|
118
|
+
|
|
119
|
+
Test at least:
|
|
120
|
+
|
|
121
|
+
- ordinary and empty text responses
|
|
122
|
+
- tool calls and tool results
|
|
123
|
+
- image input and image tool results when supported
|
|
124
|
+
- usage and cost accounting
|
|
125
|
+
- abort behavior
|
|
126
|
+
- context overflow
|
|
127
|
+
- malformed or partial streams
|
|
128
|
+
- Unicode boundaries
|
|
129
|
+
- cross-provider session handoff
|
|
130
|
+
- authentication refresh and cancellation
|
|
131
|
+
|
|
132
|
+
The provider tests under [`packages/ai/test`](https://github.com/KnightCodeAI/knightcode/tree/main/packages/ai/test) define the behavior expected from built-in providers. Adapt the relevant suites rather than relying only on manual prompts.
|
|
133
|
+
|
|
134
|
+
Run the extension directly while developing, then move it to a discovered extension location or distribute it through a [KnightCode package](packages.md). Use `/reload` after changing a discovered provider extension in an active session.
|