@orkestrel/ollama 0.0.13 → 0.0.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -9
- package/dist/src/server/index.cjs +387 -164
- package/dist/src/server/index.cjs.map +1 -1
- package/dist/src/server/index.d.cts +405 -154
- package/dist/src/server/index.d.ts +405 -154
- package/dist/src/server/index.js +379 -165
- package/dist/src/server/index.js.map +1 -1
- package/package.json +20 -22
|
@@ -1,54 +1,84 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
3
|
-
import { ProviderDelta } from '@orkestrel/agent';
|
|
4
|
-
import { ProviderInterface } from '@orkestrel/agent';
|
|
5
|
-
import { ProviderResult } from '@orkestrel/agent';
|
|
6
|
-
import { ProviderStreamOptions } from '@orkestrel/agent';
|
|
7
|
-
import {
|
|
8
|
-
import {
|
|
1
|
+
import type { ContextFormat } from '@orkestrel/agent';
|
|
2
|
+
import type { Message } from '@orkestrel/agent';
|
|
3
|
+
import type { ProviderDelta } from '@orkestrel/agent';
|
|
4
|
+
import type { ProviderInterface } from '@orkestrel/agent';
|
|
5
|
+
import type { ProviderResult } from '@orkestrel/agent';
|
|
6
|
+
import type { ProviderStreamOptions } from '@orkestrel/agent';
|
|
7
|
+
import type { ThinkSplitterInterface } from '@orkestrel/agent';
|
|
8
|
+
import type { TimeoutInterface } from '@orkestrel/timeout';
|
|
9
|
+
import type { TokenUsage } from '@orkestrel/budget';
|
|
10
|
+
import type { ToolCall } from '@orkestrel/tool';
|
|
11
|
+
import type { ToolDefinition } from '@orkestrel/tool';
|
|
9
12
|
|
|
10
13
|
/**
|
|
11
|
-
*
|
|
14
|
+
* Builds a `ProviderResult` from a turn's content, reasoning, tool calls, and usage.
|
|
15
|
+
*
|
|
16
|
+
* @remarks
|
|
17
|
+
* Only the present optionals are set: no empty `thinking`, no empty `tools`, and no
|
|
18
|
+
* `usage` unless the wire reported one.
|
|
19
|
+
*
|
|
20
|
+
* @param content - The clean assistant content the splitter accumulated
|
|
21
|
+
* @param thinking - The joined reasoning, empty when the turn produced none
|
|
22
|
+
* @param tools - The tool calls collected across the turn
|
|
23
|
+
* @param usage - The token usage, or `undefined` when the wire reported none
|
|
24
|
+
* @returns The result carrying only its populated fields
|
|
25
|
+
*
|
|
26
|
+
* @example
|
|
27
|
+
* ```ts
|
|
28
|
+
* buildResult('ok', '', [], undefined) // { content: 'ok' }
|
|
29
|
+
* ```
|
|
30
|
+
*/
|
|
31
|
+
export declare function buildResult(content: string, thinking: string, tools: readonly ToolCall[], usage: TokenUsage | undefined): ProviderResult;
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* Creates a local Ollama inference provider — a {@link ProviderInterface} over the
|
|
12
35
|
* daemon's `POST /api/chat`, supporting non-streaming `generate` and streaming
|
|
13
36
|
* `stream`.
|
|
14
37
|
*
|
|
15
38
|
* @remarks
|
|
16
39
|
* Only `model` is required; `url` defaults to the local daemon, `keepAlive` to `'5m'`,
|
|
17
40
|
* `timeout` to `120_000`ms, and `options` is forwarded verbatim as sampling
|
|
18
|
-
* parameters (`temperature
|
|
41
|
+
* parameters (`temperature`, `seed`, and `num_predict`). Each call takes an
|
|
19
42
|
* `AbortSignal` to bound the request; a `stream` cancelled mid-flight throws a
|
|
20
43
|
* `ProviderAbortError` carrying the partial result.
|
|
21
44
|
*
|
|
22
45
|
* The optional `fetch` + `headers` form a transport seam (see {@link OllamaOptions}):
|
|
23
46
|
* point `url` at your own server, inject a custom `fetch`, and have `headers` attach a
|
|
24
47
|
* generated/obfuscated bearer token your server validates — so a browser runtime
|
|
25
|
-
* reaches the LLM through your middleware
|
|
26
|
-
* key. Both omitted ⇒
|
|
48
|
+
* reaches the LLM through your middleware without this library ever handling the real API
|
|
49
|
+
* key. Both omitted ⇒ the global `fetch` and only a JSON content type.
|
|
27
50
|
*
|
|
28
|
-
* The optional `format` is the provider's context-framing default — the
|
|
29
|
-
* level of `AgentContext`'s format cascade (
|
|
30
|
-
*
|
|
31
|
-
* provider's models prefer context sections framed (
|
|
32
|
-
* headers). It is
|
|
33
|
-
* `/api/chat` `format` wire parameter (structured output) — the
|
|
34
|
-
* the shared word. Omitted ⇒ the provider is
|
|
51
|
+
* The optional `format` is the provider's context-framing default — the provider-default
|
|
52
|
+
* level of `AgentContext`'s format cascade (beaten by a manager-options or per-item
|
|
53
|
+
* override, beating the managers' built-in framing), declaring how this
|
|
54
|
+
* provider's models prefer context sections framed (for example XML group wrappers vs. Markdown
|
|
55
|
+
* headers). It is exposed on the provider for the Agent's `build()` and is not Ollama's
|
|
56
|
+
* `/api/chat` `format` wire parameter (structured output) — the framing default and that
|
|
57
|
+
* wire parameter are unrelated despite the shared word. Omitted ⇒ the provider is
|
|
58
|
+
* framing-agnostic (core's built-in defaults).
|
|
35
59
|
*
|
|
36
60
|
* @param options - `model` (required), and optional `url` / `keepAlive` / `timeout` /
|
|
37
61
|
* `options` / `fetch` / `headers` / `format` (see {@link OllamaOptions})
|
|
38
62
|
* @returns A working {@link ProviderInterface} backed by Ollama
|
|
39
63
|
*
|
|
40
|
-
* @example
|
|
64
|
+
* @example createOllama + generate
|
|
41
65
|
* ```ts
|
|
42
66
|
* import { createAbort } from '@orkestrel/abort'
|
|
43
|
-
* import { createOllama } from '@
|
|
67
|
+
* import { createOllama } from '@orkestrel/ollama'
|
|
44
68
|
*
|
|
45
|
-
* const provider = createOllama({ model: 'qwen3.5:2b-q4_K_M' })
|
|
69
|
+
* const provider = createOllama({ model: 'qwen3.5:2b-q4_K_M', options: { temperature: 0 } })
|
|
46
70
|
* const abort = createAbort()
|
|
71
|
+
* const messages = [
|
|
72
|
+
* { id: '1', role: 'user', content: 'Summarize the release notes for version 2.0.' },
|
|
73
|
+
* ] as const
|
|
74
|
+
*
|
|
47
75
|
* const result = await provider.generate(messages, abort.signal)
|
|
76
|
+
* console.log(result.content)
|
|
77
|
+
* if (result.usage) charge(result.usage) // fold into a token budget
|
|
48
78
|
* ```
|
|
49
79
|
*
|
|
50
80
|
* @example
|
|
51
|
-
* Route through your own server with an obfuscated token
|
|
81
|
+
* Route through your own server with an obfuscated token:
|
|
52
82
|
* ```ts
|
|
53
83
|
* const provider = createOllama({
|
|
54
84
|
* model: 'qwen3.5:2b-q4_K_M',
|
|
@@ -60,7 +90,7 @@ import { ToolDefinition } from '@orkestrel/tool';
|
|
|
60
90
|
*
|
|
61
91
|
* @example
|
|
62
92
|
* Declare a context-framing default — wrap the instructions section in an XML group (the
|
|
63
|
-
* provider-default level of `AgentContext`'s cascade;
|
|
93
|
+
* provider-default level of `AgentContext`'s cascade; not the wire `format`):
|
|
64
94
|
* ```ts
|
|
65
95
|
* const provider = createOllama({
|
|
66
96
|
* model: 'qwen3.5:2b-q4_K_M',
|
|
@@ -77,50 +107,180 @@ import { ToolDefinition } from '@orkestrel/tool';
|
|
|
77
107
|
export declare function createOllama(options: OllamaOptions): ProviderInterface;
|
|
78
108
|
|
|
79
109
|
/**
|
|
80
|
-
*
|
|
81
|
-
* omitted
|
|
110
|
+
* Names how long the model stays resident after a call — `'5m'` when
|
|
111
|
+
* `OllamaOptions.keepAlive` is omitted, Ollama's own `keep_alive` default, expressed as a
|
|
112
|
+
* duration string.
|
|
113
|
+
*
|
|
114
|
+
* @remarks
|
|
115
|
+
* The name mirrors the Ollama `/api/chat` `keep_alive` field this value is sent as, so
|
|
116
|
+
* the constant, the `OllamaOptions.keepAlive` key, and the wire member read as one term.
|
|
82
117
|
*/
|
|
83
118
|
export declare const DEFAULT_KEEP_ALIVE = "5m";
|
|
84
119
|
|
|
85
|
-
/**
|
|
120
|
+
/**
|
|
121
|
+
* Names the local Ollama daemon base URL, `'http://localhost:11434'`, assumed when
|
|
122
|
+
* `OllamaOptions.url` is omitted.
|
|
123
|
+
*/
|
|
86
124
|
export declare const DEFAULT_OLLAMA_URL = "http://localhost:11434";
|
|
87
125
|
|
|
88
126
|
/**
|
|
89
|
-
*
|
|
90
|
-
* generous enough that a cold model load does not trip it.
|
|
127
|
+
* Names the per-call deadline in milliseconds, `120_000`, when `OllamaOptions.timeout` is
|
|
128
|
+
* omitted — generous enough that a cold model load does not trip it.
|
|
91
129
|
*/
|
|
92
130
|
export declare const DEFAULT_PROVIDER_TIMEOUT = 120000;
|
|
93
131
|
|
|
94
132
|
/**
|
|
95
|
-
*
|
|
133
|
+
* Extracts a wire `arguments` value as a record.
|
|
134
|
+
*
|
|
135
|
+
* @remarks
|
|
136
|
+
* Total: an object passes through, a JSON string is parsed when it yields a record, and
|
|
137
|
+
* a malformed string yields `{}` rather than throwing.
|
|
138
|
+
*
|
|
139
|
+
* @param value - The wire's `function.arguments` value, of unknown shape
|
|
140
|
+
* @returns The argument record, or `{}` when the value carries none
|
|
141
|
+
*
|
|
142
|
+
* @example
|
|
143
|
+
* ```ts
|
|
144
|
+
* extractArguments('{"city":"Oslo"}') // { city: 'Oslo' }
|
|
145
|
+
* ```
|
|
146
|
+
*/
|
|
147
|
+
export declare function extractArguments(value: unknown): Readonly<Record<string, unknown>>;
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Extracts the assistant text of one wire record.
|
|
151
|
+
*
|
|
152
|
+
* @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
|
|
153
|
+
* @returns The record's `message.content` when it is a string, else `''`
|
|
154
|
+
*
|
|
155
|
+
* @example
|
|
156
|
+
* ```ts
|
|
157
|
+
* extractContent({ message: { content: 'ok' } }) // 'ok'
|
|
158
|
+
* ```
|
|
159
|
+
*/
|
|
160
|
+
export declare function extractContent(record: Readonly<Record<string, unknown>>): string;
|
|
161
|
+
|
|
162
|
+
/**
|
|
163
|
+
* Extracts the daemon-side reasoning of one wire record.
|
|
164
|
+
*
|
|
165
|
+
* @remarks
|
|
166
|
+
* `message.thinking` is the `think: true` wire shape. It is read whatever the configured
|
|
167
|
+
* flag says, because a daemon may separate reasoning on its own.
|
|
168
|
+
*
|
|
169
|
+
* @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
|
|
170
|
+
* @returns The record's `message.thinking` when it is a string, else `''`
|
|
171
|
+
*
|
|
172
|
+
* @example
|
|
173
|
+
* ```ts
|
|
174
|
+
* extractThinking({ message: { thinking: 'weighing it' } }) // 'weighing it'
|
|
175
|
+
* ```
|
|
176
|
+
*/
|
|
177
|
+
export declare function extractThinking(record: Readonly<Record<string, unknown>>): string;
|
|
178
|
+
|
|
179
|
+
/**
|
|
180
|
+
* Extracts the tool calls of one wire record's `message.tool_calls`.
|
|
181
|
+
*
|
|
182
|
+
* @remarks
|
|
183
|
+
* Each entry narrows to `{ id, name, arguments }`: the entry and its `function` must be
|
|
184
|
+
* records and `name` a string, else the entry is dropped. An id is minted when the wire
|
|
185
|
+
* omits one.
|
|
186
|
+
*
|
|
187
|
+
* @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
|
|
188
|
+
* @returns The narrowed tool calls, empty when the record carries none
|
|
189
|
+
*
|
|
190
|
+
* @example
|
|
191
|
+
* ```ts
|
|
192
|
+
* extractTools({ message: { tool_calls: [{ function: { name: 'weather' } }] } })
|
|
193
|
+
* // [{ id: '…', name: 'weather', arguments: {} }]
|
|
194
|
+
* ```
|
|
195
|
+
*/
|
|
196
|
+
export declare function extractTools(record: Readonly<Record<string, unknown>>): readonly ToolCall[];
|
|
197
|
+
|
|
198
|
+
/**
|
|
199
|
+
* Extracts the token usage of one wire record.
|
|
200
|
+
*
|
|
201
|
+
* @remarks
|
|
202
|
+
* Both counts must be numbers, which is true of the non-stream body and the stream's
|
|
203
|
+
* `done: true` line. A delta line carries neither, so it yields `undefined`.
|
|
204
|
+
*
|
|
205
|
+
* @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
|
|
206
|
+
* @returns The `TokenUsage` shape, or `undefined` when either count is absent
|
|
207
|
+
*
|
|
208
|
+
* @example
|
|
209
|
+
* ```ts
|
|
210
|
+
* extractUsage({ prompt_eval_count: 3, eval_count: 4 })
|
|
211
|
+
* // { prompt: 3, completion: 4, total: 7 }
|
|
212
|
+
* ```
|
|
213
|
+
*/
|
|
214
|
+
export declare function extractUsage(record: Readonly<Record<string, unknown>>): TokenUsage | undefined;
|
|
215
|
+
|
|
216
|
+
/**
|
|
217
|
+
* Checks whether a value is an {@link OllamaHTTPError}.
|
|
218
|
+
*
|
|
219
|
+
* @remarks
|
|
220
|
+
* The check is an `instanceof` test, so it narrows a caught `unknown` to the error class
|
|
221
|
+
* without parsing the thrown message.
|
|
96
222
|
*
|
|
97
223
|
* @param value - The value to test
|
|
98
|
-
* @returns
|
|
224
|
+
* @returns True if `value` is an `OllamaHTTPError`; false otherwise
|
|
99
225
|
*/
|
|
100
226
|
export declare function isOllamaHTTPError(value: unknown): value is OllamaHTTPError;
|
|
101
227
|
|
|
102
228
|
/**
|
|
103
|
-
*
|
|
229
|
+
* Joins a call's reasoning carriers — the splitter's separated in-content spans and the
|
|
230
|
+
* accumulated wire-side `message.thinking` — into the result's `thinking`.
|
|
231
|
+
*
|
|
232
|
+
* @param splitter - The per-call splitter holding the separated in-content spans
|
|
233
|
+
* @param wired - The accumulated wire-side `message.thinking` text
|
|
234
|
+
* @returns The carriers separated by a blank line, or whichever one is non-empty
|
|
235
|
+
*
|
|
236
|
+
* @example
|
|
237
|
+
* ```ts
|
|
238
|
+
* joinThinking(createThinkSplitter(), 'from the wire') // 'from the wire'
|
|
239
|
+
* ```
|
|
240
|
+
*/
|
|
241
|
+
export declare function joinThinking(splitter: ThinkSplitterInterface, wired: string): string;
|
|
242
|
+
|
|
243
|
+
/**
|
|
244
|
+
* Maps conversation turns onto the `/api/chat` wire's minimal message shape.
|
|
245
|
+
*
|
|
246
|
+
* @remarks
|
|
247
|
+
* `tool_calls` is emitted only on a turn that replays them and `images` only on a
|
|
248
|
+
* multimodal turn, so an empty optional never reaches the wire.
|
|
249
|
+
*
|
|
250
|
+
* @param messages - The conversation turns to send
|
|
251
|
+
* @returns The wire `messages` array, one entry per turn, in order
|
|
252
|
+
*
|
|
253
|
+
* @example
|
|
254
|
+
* ```ts
|
|
255
|
+
* mapMessages([{ id: '1', role: 'user', content: 'Say hello.' }])
|
|
256
|
+
* // [{ role: 'user', content: 'Say hello.' }]
|
|
257
|
+
* ```
|
|
258
|
+
*/
|
|
259
|
+
export declare function mapMessages(messages: readonly Message[]): WireChatRequest['messages'];
|
|
260
|
+
|
|
261
|
+
/**
|
|
262
|
+
* Names the character cap, `2048`, on how much of a non-OK response body is
|
|
104
263
|
* incorporated into a thrown {@link OllamaHTTPError}'s message.
|
|
105
264
|
*
|
|
106
265
|
* @remarks
|
|
107
266
|
* Bounds the excerpt so a defensive proxy or a misbehaving daemon handing
|
|
108
267
|
* back an unbounded response body cannot inflate the thrown error's message
|
|
109
|
-
* without limit
|
|
110
|
-
*
|
|
111
|
-
* concern.
|
|
268
|
+
* without limit, while the cap stays generous enough to carry a useful
|
|
269
|
+
* diagnostic snippet.
|
|
112
270
|
*/
|
|
113
271
|
export declare const MAX_ERROR_BODY_LENGTH = 2048;
|
|
114
272
|
|
|
115
273
|
/**
|
|
116
|
-
*
|
|
274
|
+
* Represents an error thrown when the Ollama `/api/chat` HTTP transport fails.
|
|
117
275
|
*
|
|
118
276
|
* @remarks
|
|
119
|
-
* Carries the response `status` (0 when no
|
|
120
|
-
*
|
|
121
|
-
* failure sites — the non-OK status branch and the
|
|
122
|
-
* caller can branch on `error.
|
|
123
|
-
*
|
|
277
|
+
* Carries the machine-readable `code` `'HTTP'` and the response `status` (0 when no
|
|
278
|
+
* HTTP response was received at all, for example a `null` body). Thrown by
|
|
279
|
+
* {@link OllamaProvider} at its HTTP failure sites — the non-OK status branch and the
|
|
280
|
+
* null-body branch — so a caller can branch on `error.code` and read `error.status`
|
|
281
|
+
* for the HTTP number instead of parsing the message. The message carries a body excerpt
|
|
282
|
+
* bounded to {@link MAX_ERROR_BODY_LENGTH} — `2048` characters. Narrow a caught value with
|
|
283
|
+
* {@link isOllamaHTTPError}.
|
|
124
284
|
*
|
|
125
285
|
* @example
|
|
126
286
|
* ```ts
|
|
@@ -134,117 +294,142 @@ export declare const MAX_ERROR_BODY_LENGTH = 2048;
|
|
|
134
294
|
* ```
|
|
135
295
|
*/
|
|
136
296
|
export declare class OllamaHTTPError extends Error {
|
|
297
|
+
/**
|
|
298
|
+
* Names the machine-readable condition this error reports — `'HTTP'`: an `/api/chat`
|
|
299
|
+
* transport, status, or body failure.
|
|
300
|
+
*/
|
|
301
|
+
readonly code: "HTTP";
|
|
137
302
|
readonly status: number;
|
|
138
|
-
constructor(message: string, status: number, options?:
|
|
139
|
-
readonly cause?: unknown;
|
|
140
|
-
});
|
|
303
|
+
constructor(message: string, status: number, options?: OllamaHTTPErrorOptions);
|
|
141
304
|
}
|
|
142
305
|
|
|
143
306
|
/**
|
|
144
|
-
*
|
|
307
|
+
* Represents the options a thrown {@link OllamaHTTPError} accepts beside its message and status —
|
|
308
|
+
* the standard error `cause` link, named so a consumer can reference the shape.
|
|
309
|
+
*
|
|
310
|
+
* @remarks
|
|
311
|
+
* `cause` is the underlying value that produced the HTTP failure: the transport or
|
|
312
|
+
* body-read rejection the provider caught before rethrowing. It is `unknown` because a
|
|
313
|
+
* thrown value is unconstrained. Omitted ⇒ the error carries no cause.
|
|
314
|
+
*/
|
|
315
|
+
export declare interface OllamaHTTPErrorOptions {
|
|
316
|
+
readonly cause?: unknown;
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
/**
|
|
320
|
+
* Represents the configuration `createOllama` accepts for the local Ollama backend.
|
|
145
321
|
*
|
|
146
322
|
* @remarks
|
|
147
323
|
* Only `model` is required. `url` defaults to the local daemon, `keepAlive` controls
|
|
148
324
|
* how long the model stays resident after a call, `timeout` is the per-call deadline
|
|
149
325
|
* in milliseconds, and `options` is a passthrough bag of sampling parameters
|
|
150
|
-
* (`temperature
|
|
326
|
+
* (`temperature`, `seed`, and `num_predict`) forwarded verbatim to the wire.
|
|
151
327
|
*
|
|
152
328
|
* The optional `fetch` + `headers` form a **transport seam**: by default the provider
|
|
153
329
|
* talks straight to a local daemon over `globalThis.fetch` with only a JSON content
|
|
154
|
-
* type, but a browser-side runtime can inject a custom transport
|
|
155
|
-
* (
|
|
330
|
+
* type, but a browser-side runtime can inject both a custom transport and a dynamic header
|
|
331
|
+
* (for example an obfuscated bearer token) so requests route through the developer's own
|
|
156
332
|
* server, which validates that header and forwards to the real LLM. Your app never
|
|
157
333
|
* holds a real API key — the real key lives only on the developer's server; the
|
|
158
334
|
* `headers` hook supplies whatever short-lived/obfuscated token that server expects.
|
|
159
335
|
*/
|
|
160
336
|
export declare interface OllamaOptions {
|
|
161
337
|
readonly model: string;
|
|
162
|
-
/**
|
|
338
|
+
/** Sets the daemon base URL; defaults to `'http://localhost:11434'`. */
|
|
163
339
|
readonly url?: string;
|
|
164
|
-
/**
|
|
340
|
+
/**
|
|
341
|
+
* Sets how long the model stays resident after a call; defaults to `'5m'`. Mirrors the
|
|
342
|
+
* Ollama `/api/chat` `keep_alive` field, whose value this key carries verbatim onto
|
|
343
|
+
* {@link WireChatRequest.keep_alive}.
|
|
344
|
+
*/
|
|
165
345
|
readonly keepAlive?: string | number;
|
|
166
|
-
/**
|
|
346
|
+
/** Sets the per-call deadline in milliseconds; defaults to `120_000`. */
|
|
167
347
|
readonly timeout?: number;
|
|
168
|
-
/**
|
|
348
|
+
/**
|
|
349
|
+
* Carries passthrough sampling parameters (`temperature`, `seed`, and `num_predict`).
|
|
350
|
+
* Mirrors the Ollama `/api/chat` `options` field, whose value this key carries verbatim
|
|
351
|
+
* onto {@link WireChatRequest.options}.
|
|
352
|
+
*/
|
|
169
353
|
readonly options?: Readonly<Record<string, unknown>>;
|
|
170
354
|
/**
|
|
171
|
-
*
|
|
172
|
-
* model (
|
|
355
|
+
* Sets the `/api/chat` `think` wire flag; defaults to `false`. When `true`, a thinking-capable
|
|
356
|
+
* model (for example `qwen3`) separates its reasoning natively at the wire — the daemon returns it
|
|
173
357
|
* on the distinct `message.thinking` channel (surfaced on `ProviderResult.thinking`) rather
|
|
174
|
-
* than inline in `message.content`. The default
|
|
175
|
-
*
|
|
358
|
+
* than inline in `message.content`. The default is `false`, so a non-thinking model needs no
|
|
359
|
+
* configuration and answers immediately; the per-call ThinkSplitter
|
|
176
360
|
* remains the defensive fallback for daemons/models that still inline `<think>` tags either
|
|
177
|
-
* way. Set it `true` for a thinking model whose reasoning you intend to
|
|
361
|
+
* way. Set it `true` for a thinking model whose reasoning you intend to display separately.
|
|
178
362
|
*/
|
|
179
363
|
readonly think?: boolean;
|
|
180
364
|
/**
|
|
181
|
-
*
|
|
365
|
+
* Sets a custom `fetch` implementation for every request; defaults to
|
|
182
366
|
* `globalThis.fetch`. Lets a runtime inject its own transport (a browser fetch
|
|
183
|
-
* pointed at the developer's server, an instrumented wrapper
|
|
367
|
+
* pointed at the developer's server, an instrumented wrapper) without changing
|
|
184
368
|
* the wire protocol. Omitted ⇒ the global `fetch`.
|
|
185
369
|
*/
|
|
186
370
|
readonly fetch?: typeof globalThis.fetch;
|
|
187
371
|
/**
|
|
188
|
-
*
|
|
372
|
+
* Sets a dynamic, possibly-async header injector called once per request; its returned
|
|
189
373
|
* headers are merged into the request on top of the base `Content-Type`. Use it to
|
|
190
|
-
* attach an authorization header —
|
|
374
|
+
* attach an authorization header — for example an obfuscated/generated bearer token the
|
|
191
375
|
* developer's server validates before relaying to the real LLM — so a browser
|
|
192
|
-
* runtime can authenticate
|
|
376
|
+
* runtime can authenticate without your app ever handling a real API key. Async so a
|
|
193
377
|
* token can be refreshed/fetched per call. A returned `Content-Type` overrides the
|
|
194
378
|
* default; other headers add to it. Omitted ⇒ only `Content-Type: application/json`.
|
|
195
379
|
*/
|
|
196
|
-
readonly headers?: () => Record<string, string
|
|
380
|
+
readonly headers?: () => Readonly<Record<string, string>> | Promise<Readonly<Record<string, string>>>;
|
|
197
381
|
/**
|
|
198
|
-
*
|
|
382
|
+
* Sets the provider's optional context-framing default — the provider-default level of
|
|
199
383
|
* `AgentContext`'s format cascade (beaten by a manager-options or per-item override,
|
|
200
384
|
* beating the managers' built-in framing). Declares how this provider's models prefer
|
|
201
|
-
* context sections framed (
|
|
202
|
-
* the provider is framing-agnostic and core's built-in defaults apply unchanged.
|
|
203
|
-
*
|
|
385
|
+
* context sections framed (for example XML group wrappers vs. Markdown headers). Omitted ⇒
|
|
386
|
+
* the provider is framing-agnostic and core's built-in defaults apply unchanged. This
|
|
387
|
+
* is the prompt-context framing consumed by `AgentContext.build()` — it is not
|
|
204
388
|
* Ollama's `/api/chat` `format` wire parameter (structured-output / JSON schema),
|
|
205
|
-
* which this provider
|
|
206
|
-
* shared word.
|
|
389
|
+
* which this provider sends only when a call supplies a `schema`; the two are unrelated
|
|
390
|
+
* despite the shared word.
|
|
207
391
|
*/
|
|
208
|
-
readonly format?:
|
|
392
|
+
readonly format?: ContextFormat;
|
|
209
393
|
}
|
|
210
394
|
|
|
211
395
|
/**
|
|
212
|
-
*
|
|
396
|
+
* Implements the local Ollama inference boundary — a {@link ProviderInterface} over Ollama's
|
|
213
397
|
* `POST /api/chat`, both non-streaming (`generate`) and streaming NDJSON (`stream`).
|
|
214
398
|
*
|
|
215
399
|
* @remarks
|
|
216
400
|
* - **Wire protocol.** Posts `{ model, messages, stream, keep_alive, think }` plus
|
|
217
401
|
* passthrough sampling `options` and mapped function `tools`. The `think` flag is
|
|
218
|
-
*
|
|
402
|
+
* configurable through {@link OllamaOptions.think} (default `false`). Non-stream parses
|
|
219
403
|
* one JSON body; stream consumes NDJSON (one JSON object per `\n`-terminated line) —
|
|
220
404
|
* deltas carry `message.content`, the final `done: true` line carries the token usage.
|
|
221
|
-
* - **Think separation
|
|
405
|
+
* - **Think separation.** The wire `think` flag is configurable
|
|
222
406
|
* ({@link OllamaOptions.think}, default `false`). With `think: true` a thinking model's
|
|
223
|
-
* daemon separates reasoning
|
|
224
|
-
* channel (read here
|
|
407
|
+
* daemon separates reasoning natively — returning it on the distinct `message.thinking`
|
|
408
|
+
* channel (read here through `extractThinking`) instead of inline in `message.content`. Either
|
|
225
409
|
* way the per-call {@link ThinkSplitterInterface} is the defensive guarantee: a daemon
|
|
226
410
|
* may ignore `think: false` for a thinking model and inline `<think>` tags, so every
|
|
227
|
-
* content delta routes through the splitter, only
|
|
411
|
+
* content delta routes through the splitter, only clean content is yielded / assembled,
|
|
228
412
|
* and the separated reasoning (plus any daemon-side `message.thinking` deltas) lands on
|
|
229
413
|
* `ProviderResult.thinking`, never in the conversation.
|
|
230
|
-
* - **Boundary narrowing
|
|
414
|
+
* - **Boundary narrowing.** Every wire value arrives as `unknown` and is
|
|
231
415
|
* narrowed through guards (`isRecord` / `isString` / `isNumber`) — never `as`. A
|
|
232
416
|
* missing / malformed field degrades to a sensible default (empty content, no
|
|
233
417
|
* usage, `{}` arguments), never a throw.
|
|
234
418
|
* - **Bounded.** Each call arms a {@link Timeout} for `OllamaOptions.timeout` and
|
|
235
419
|
* passes `AbortSignal.any([timeout.signal, signal])` to `fetch`, so the caller's
|
|
236
|
-
* signal
|
|
420
|
+
* signal and the deadline both cancel the request. The timeout is always cleared —
|
|
237
421
|
* in `#fetch` if the request fails/aborts, otherwise in the consuming call's `finally`.
|
|
238
422
|
* - **Abort recovers partial.** A `stream` cancelled mid-flight throws a
|
|
239
423
|
* `ProviderAbortError` carrying the partial result assembled so far; pairing the
|
|
240
|
-
* `TextDecoder({ stream: true })` with the
|
|
424
|
+
* `TextDecoder({ stream: true })` with the `createNDJSONParser` parser keeps multi-byte
|
|
241
425
|
* UTF-8 splits and partial lines honest.
|
|
242
426
|
* - **Event-free.** A pure functional boundary — no Emitter, no events.
|
|
243
427
|
* - **Transport seam.** {@link OllamaOptions.fetch} swaps the transport (default
|
|
244
428
|
* `globalThis.fetch`) and {@link OllamaOptions.headers} is a per-request, possibly
|
|
245
429
|
* async header injector merged over the base `Content-Type` — so a browser runtime
|
|
246
430
|
* can route through the developer's own server with an obfuscated bearer token,
|
|
247
|
-
* without this library ever handling a real API key. Both omitted ⇒
|
|
431
|
+
* without this library ever handling a real API key. Both omitted ⇒ the global `fetch`
|
|
432
|
+
* and only a JSON content type.
|
|
248
433
|
* Orthogonal to the deadline: the hook is awaited inside `#fetch`'s try, so a hook
|
|
249
434
|
* rejection clears the armed timer like any other request failure.
|
|
250
435
|
*
|
|
@@ -256,91 +441,157 @@ export declare interface OllamaOptions {
|
|
|
256
441
|
*/
|
|
257
442
|
export declare class OllamaProvider implements ProviderInterface {
|
|
258
443
|
#private;
|
|
259
|
-
readonly id: `${string}-${string}-${string}-${string}-${string}`;
|
|
260
444
|
readonly name = "ollama";
|
|
261
445
|
constructor(options: OllamaOptions);
|
|
262
446
|
/**
|
|
263
|
-
*
|
|
264
|
-
* {@link
|
|
265
|
-
*
|
|
266
|
-
*
|
|
447
|
+
* Exposes this instance's identity — a fresh `crypto.randomUUID()` minted at
|
|
448
|
+
* construction, satisfying the {@link ProviderInterface.id} contract member. A second
|
|
449
|
+
* provider built from identical options carries a distinct id.
|
|
450
|
+
*
|
|
451
|
+
* @returns The instance's minted identifier
|
|
452
|
+
*/
|
|
453
|
+
get id(): string;
|
|
454
|
+
/**
|
|
455
|
+
* Exposes the provider's context-framing default — the provider-default level of
|
|
456
|
+
* {@link import('@orkestrel/agent').AgentContextInterface.build}'s format cascade (it beats
|
|
457
|
+
* the managers' built-in framing, is beaten by a manager-options or per-item override).
|
|
458
|
+
* Satisfies the optional {@link ProviderInterface.format} contract member: `undefined`
|
|
267
459
|
* when {@link OllamaOptions.format} was omitted (the framing-agnostic default ⇒ core's
|
|
268
460
|
* built-in framing applies unchanged), else the exact configured framing the Agent
|
|
269
461
|
* threads into `build()`.
|
|
270
462
|
*
|
|
271
463
|
* @remarks
|
|
272
|
-
*
|
|
273
|
-
* on the `/api/chat` wire (it is absent from `#body` / the request). This is
|
|
274
|
-
* structured-output `format` wire parameter — that one
|
|
464
|
+
* Expose-only — read by the Agent loop and consumed by core's cascade; it is never sent
|
|
465
|
+
* on the `/api/chat` wire (it is absent from `#body` / the request). This is not Ollama's
|
|
466
|
+
* structured-output `format` wire parameter — that one is sent in `#body`, but only when
|
|
275
467
|
* a per-call `ProviderStreamOptions.schema` is supplied; only the word collides.
|
|
276
468
|
*
|
|
277
|
-
* @returns The configured {@link
|
|
469
|
+
* @returns The configured {@link ContextFormat}, or `undefined` when none
|
|
278
470
|
*/
|
|
279
|
-
get format():
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
471
|
+
get format(): ContextFormat | undefined;
|
|
472
|
+
/**
|
|
473
|
+
* Generates one complete turn and resolves the assembled result — the clean content,
|
|
474
|
+
* any separated reasoning, any tool calls, and any usage the wire reported.
|
|
475
|
+
*
|
|
476
|
+
* @remarks
|
|
477
|
+
* Sends `stream: false` and parses one JSON body. Content routes through a per-call
|
|
478
|
+
* think splitter, so the assembled content stays clean even where the daemon renders a
|
|
479
|
+
* thinking model's reasoning inline; the separated spans and any daemon-side
|
|
480
|
+
* `message.thinking` land on `thinking`. The caller's signal and the armed deadline
|
|
481
|
+
* both cancel the request, and the deadline is cleared once the body is read.
|
|
482
|
+
*
|
|
483
|
+
* @param messages - The conversation turns to send
|
|
484
|
+
* @param signal - The caller's bounding signal, folded with the armed deadline
|
|
485
|
+
* @param tools - The callable tools to advertise for this turn, when the caller passes any
|
|
486
|
+
* @param options - The per-call overrides, `think` and `schema` among them
|
|
487
|
+
* @returns The assembled result of the turn
|
|
488
|
+
* @throws {@link OllamaHTTPError} When the daemon answers a non-OK status.
|
|
489
|
+
*/
|
|
490
|
+
generate(messages: readonly Message[], signal: AbortSignal, tools?: readonly ToolDefinition[], options?: ProviderStreamOptions): Promise<ProviderResult>;
|
|
491
|
+
/**
|
|
492
|
+
* Streams one turn, yielding a channel-tagged delta per non-empty content or reasoning
|
|
493
|
+
* span and returning the assembled result when the stream completes.
|
|
494
|
+
*
|
|
495
|
+
* @remarks
|
|
496
|
+
* Sends `stream: true` and consumes NDJSON — one JSON object per newline-terminated
|
|
497
|
+
* line — pairing a streaming `TextDecoder` with the `NDJSONParser` so a record split
|
|
498
|
+
* across byte reads is reassembled. The returned result's content is the splitter's
|
|
499
|
+
* clean accumulation, beside any tool calls collected across lines and the usage the
|
|
500
|
+
* `done` line carries. A cancel mid-flight throws a `ProviderAbortError` carrying the
|
|
501
|
+
* partial assembled so far.
|
|
502
|
+
*
|
|
503
|
+
* @param messages - The conversation turns to send
|
|
504
|
+
* @param signal - The caller's bounding signal, folded with the armed deadline
|
|
505
|
+
* @param tools - The callable tools to advertise for this turn, when the caller passes any
|
|
506
|
+
* @param options - The per-call overrides, `think` and `schema` among them
|
|
507
|
+
* @returns The assembled result of the turn, after the last delta
|
|
508
|
+
* @throws {@link OllamaHTTPError} When the daemon answers a non-OK status or a `null` body.
|
|
509
|
+
*/
|
|
510
|
+
stream(messages: readonly Message[], signal: AbortSignal, tools?: readonly ToolDefinition[], options?: ProviderStreamOptions): AsyncGenerator<ProviderDelta, ProviderResult>;
|
|
511
|
+
}
|
|
283
512
|
|
|
284
|
-
/**
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
export declare interface OllamaResponse {
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
}
|
|
513
|
+
/**
|
|
514
|
+
* Represents an open `POST /api/chat` response together with the deadline and the
|
|
515
|
+
* combined signal that bound the request.
|
|
516
|
+
*
|
|
517
|
+
* @remarks
|
|
518
|
+
* The `response` is the open `POST /api/chat` `Response`; `timeout` is the armed
|
|
519
|
+
* {@link TimeoutInterface} the consuming call clears once it finishes reading the body
|
|
520
|
+
* (or that the provider clears on a failed or aborted request); `combined` is the
|
|
521
|
+
* `AbortSignal.any([timeout.signal, callerSignal])` the request was issued under, which
|
|
522
|
+
* the streaming path checks to tell a mid-stream cancel apart from any other error.
|
|
523
|
+
*/
|
|
524
|
+
export declare interface OllamaResponse {
|
|
525
|
+
readonly response: Response;
|
|
526
|
+
readonly timeout: TimeoutInterface;
|
|
527
|
+
readonly combined: AbortSignal;
|
|
528
|
+
}
|
|
300
529
|
|
|
301
|
-
/**
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
}
|
|
530
|
+
/**
|
|
531
|
+
* Parses a non-stream `/api/chat` response body into a wire record.
|
|
532
|
+
*
|
|
533
|
+
* @remarks
|
|
534
|
+
* Total by construction: an empty body, a body that is not JSON, and a body whose JSON is
|
|
535
|
+
* not an object all yield `undefined`, so a malformed daemon response never escapes as a
|
|
536
|
+
* `SyntaxError`. The call site supplies the empty-record default that reads as empty
|
|
537
|
+
* content and no usage.
|
|
538
|
+
*
|
|
539
|
+
* @param response - The 200-OK `/api/chat` response whose body is read as text
|
|
540
|
+
* @returns The parsed record, or `undefined` when the body is empty or malformed
|
|
541
|
+
*
|
|
542
|
+
* @example
|
|
543
|
+
* ```ts
|
|
544
|
+
* await parseBody(new Response('{"message":{"content":"ok"}}'))
|
|
545
|
+
* // { message: { content: 'ok' } }
|
|
546
|
+
* ```
|
|
547
|
+
*/
|
|
548
|
+
export declare function parseBody(response: Response): Promise<Readonly<Record<string, unknown>> | undefined>;
|
|
549
|
+
|
|
550
|
+
/**
|
|
551
|
+
* Represents the exact `POST /api/chat` request body `OllamaProvider` sends — the internal typed
|
|
552
|
+
* wire contract.
|
|
553
|
+
*
|
|
554
|
+
* @remarks
|
|
555
|
+
* This is the typed wire shape asserted against the official `ollama` client's
|
|
556
|
+
* `ChatRequest` by the compile-time parity test; `src/` never imports `ollama` itself.
|
|
557
|
+
* `messages` mirrors the minimal turn shape `mapMessages` builds (`role` / `content`, plus
|
|
558
|
+
* `tool_calls` only on a turn that replays them and `images` only on a multimodal
|
|
559
|
+
* turn); `options` and `tools` are only present when configured. `format` carries the
|
|
560
|
+
* `/api/chat` structured-output constraint, forwarded verbatim from the per-call
|
|
561
|
+
* `ProviderStreamOptions.schema` and absent when no schema is supplied.
|
|
562
|
+
*/
|
|
563
|
+
export declare interface WireChatRequest {
|
|
564
|
+
readonly model: string;
|
|
565
|
+
readonly messages: ReadonlyArray<{
|
|
566
|
+
readonly role: string;
|
|
567
|
+
readonly content: string;
|
|
568
|
+
readonly tool_calls?: ReadonlyArray<{
|
|
569
|
+
readonly function: {
|
|
570
|
+
readonly name: string;
|
|
571
|
+
readonly arguments: Readonly<Record<string, unknown>>;
|
|
572
|
+
};
|
|
573
|
+
}>;
|
|
574
|
+
readonly images?: readonly string[];
|
|
575
|
+
}>;
|
|
576
|
+
readonly stream: boolean;
|
|
577
|
+
readonly keep_alive: string | number;
|
|
578
|
+
readonly think: boolean;
|
|
579
|
+
readonly options?: Readonly<Record<string, unknown>>;
|
|
580
|
+
readonly tools?: ReadonlyArray<{
|
|
581
|
+
readonly type: 'function';
|
|
582
|
+
readonly function: {
|
|
583
|
+
readonly name: string;
|
|
584
|
+
readonly description?: string;
|
|
585
|
+
readonly parameters?: Readonly<Record<string, unknown>>;
|
|
586
|
+
};
|
|
587
|
+
}>;
|
|
588
|
+
/**
|
|
589
|
+
* Holds the `/api/chat` structured-output constraint — a JSON-Schema object forwarded
|
|
590
|
+
* verbatim from the per-call `ProviderStreamOptions.schema`. This is not
|
|
591
|
+
* `OllamaOptions.format` (the unrelated prompt-context framing); only present
|
|
592
|
+
* when a call supplies a `schema`.
|
|
593
|
+
*/
|
|
594
|
+
readonly format?: Readonly<Record<string, unknown>>;
|
|
595
|
+
}
|
|
345
596
|
|
|
346
|
-
export { }
|
|
597
|
+
export { }
|