@orkestrel/ollama 0.0.13 → 0.0.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,54 +1,84 @@
1
- import { ContextFormatInterface } from '@orkestrel/agent';
2
- import { MessageInterface } from '@orkestrel/agent';
3
- import { ProviderDelta } from '@orkestrel/agent';
4
- import { ProviderInterface } from '@orkestrel/agent';
5
- import { ProviderResult } from '@orkestrel/agent';
6
- import { ProviderStreamOptions } from '@orkestrel/agent';
7
- import { TimeoutInterface } from '@orkestrel/timeout';
8
- import { ToolDefinition } from '@orkestrel/tool';
1
+ import type { ContextFormat } from '@orkestrel/agent';
2
+ import type { Message } from '@orkestrel/agent';
3
+ import type { ProviderDelta } from '@orkestrel/agent';
4
+ import type { ProviderInterface } from '@orkestrel/agent';
5
+ import type { ProviderResult } from '@orkestrel/agent';
6
+ import type { ProviderStreamOptions } from '@orkestrel/agent';
7
+ import type { ThinkSplitterInterface } from '@orkestrel/agent';
8
+ import type { TimeoutInterface } from '@orkestrel/timeout';
9
+ import type { TokenUsage } from '@orkestrel/budget';
10
+ import type { ToolCall } from '@orkestrel/tool';
11
+ import type { ToolDefinition } from '@orkestrel/tool';
9
12
 
10
13
  /**
11
- * Create a local Ollama inference provider — a {@link ProviderInterface} over the
14
+ * Builds a `ProviderResult` from a turn's content, reasoning, tool calls, and usage.
15
+ *
16
+ * @remarks
17
+ * Only the present optionals are set: no empty `thinking`, no empty `tools`, and no
18
+ * `usage` unless the wire reported one.
19
+ *
20
+ * @param content - The clean assistant content the splitter accumulated
21
+ * @param thinking - The joined reasoning, empty when the turn produced none
22
+ * @param tools - The tool calls collected across the turn
23
+ * @param usage - The token usage, or `undefined` when the wire reported none
24
+ * @returns The result carrying only its populated fields
25
+ *
26
+ * @example
27
+ * ```ts
28
+ * buildResult('ok', '', [], undefined) // { content: 'ok' }
29
+ * ```
30
+ */
31
+ export declare function buildResult(content: string, thinking: string, tools: readonly ToolCall[], usage: TokenUsage | undefined): ProviderResult;
32
+
33
+ /**
34
+ * Creates a local Ollama inference provider — a {@link ProviderInterface} over the
12
35
  * daemon's `POST /api/chat`, supporting non-streaming `generate` and streaming
13
36
  * `stream`.
14
37
  *
15
38
  * @remarks
16
39
  * Only `model` is required; `url` defaults to the local daemon, `keepAlive` to `'5m'`,
17
40
  * `timeout` to `120_000`ms, and `options` is forwarded verbatim as sampling
18
- * parameters (`temperature` / `seed` / `num_predict` / …). Both calls take an
41
+ * parameters (`temperature`, `seed`, and `num_predict`). Each call takes an
19
42
  * `AbortSignal` to bound the request; a `stream` cancelled mid-flight throws a
20
43
  * `ProviderAbortError` carrying the partial result.
21
44
  *
22
45
  * The optional `fetch` + `headers` form a transport seam (see {@link OllamaOptions}):
23
46
  * point `url` at your own server, inject a custom `fetch`, and have `headers` attach a
24
47
  * generated/obfuscated bearer token your server validates — so a browser runtime
25
- * reaches the LLM through your middleware WITHOUT this library ever handling the real API
26
- * key. Both omitted ⇒ today's behaviour (the global `fetch`, only a JSON content type).
48
+ * reaches the LLM through your middleware without this library ever handling the real API
49
+ * key. Both omitted ⇒ the global `fetch` and only a JSON content type.
27
50
  *
28
- * The optional `format` is the provider's context-framing default — the PROVIDER-DEFAULT
29
- * level of `AgentContext`'s format cascade (see [agents.md]; beaten by a manager-options
30
- * or per-item override, beating the managers' built-in framing), declaring how this
31
- * provider's models prefer context sections framed (e.g. XML group wrappers vs. Markdown
32
- * headers). It is EXPOSED on the provider for the Agent's `build()` and is NOT Ollama's
33
- * `/api/chat` `format` wire parameter (structured output) — the two are unrelated despite
34
- * the shared word. Omitted ⇒ the provider is framing-agnostic (core's built-in defaults).
51
+ * The optional `format` is the provider's context-framing default — the provider-default
52
+ * level of `AgentContext`'s format cascade (beaten by a manager-options or per-item
53
+ * override, beating the managers' built-in framing), declaring how this
54
+ * provider's models prefer context sections framed (for example XML group wrappers vs. Markdown
55
+ * headers). It is exposed on the provider for the Agent's `build()` and is not Ollama's
56
+ * `/api/chat` `format` wire parameter (structured output) — the framing default and that
57
+ * wire parameter are unrelated despite the shared word. Omitted ⇒ the provider is
58
+ * framing-agnostic (core's built-in defaults).
35
59
  *
36
60
  * @param options - `model` (required), and optional `url` / `keepAlive` / `timeout` /
37
61
  * `options` / `fetch` / `headers` / `format` (see {@link OllamaOptions})
38
62
  * @returns A working {@link ProviderInterface} backed by Ollama
39
63
  *
40
- * @example
64
+ * @example createOllama + generate
41
65
  * ```ts
42
66
  * import { createAbort } from '@orkestrel/abort'
43
- * import { createOllama } from '@src/server'
67
+ * import { createOllama } from '@orkestrel/ollama'
44
68
  *
45
- * const provider = createOllama({ model: 'qwen3.5:2b-q4_K_M' })
69
+ * const provider = createOllama({ model: 'qwen3.5:2b-q4_K_M', options: { temperature: 0 } })
46
70
  * const abort = createAbort()
71
+ * const messages = [
72
+ * { id: '1', role: 'user', content: 'Summarize the release notes for version 2.0.' },
73
+ * ] as const
74
+ *
47
75
  * const result = await provider.generate(messages, abort.signal)
76
+ * console.log(result.content)
77
+ * if (result.usage) charge(result.usage) // fold into a token budget
48
78
  * ```
49
79
  *
50
80
  * @example
51
- * Route through your own server with an obfuscated token (deployment scenario S2):
81
+ * Route through your own server with an obfuscated token:
52
82
  * ```ts
53
83
  * const provider = createOllama({
54
84
  * model: 'qwen3.5:2b-q4_K_M',
@@ -60,7 +90,7 @@ import { ToolDefinition } from '@orkestrel/tool';
60
90
  *
61
91
  * @example
62
92
  * Declare a context-framing default — wrap the instructions section in an XML group (the
63
- * provider-default level of `AgentContext`'s cascade; NOT the wire `format`):
93
+ * provider-default level of `AgentContext`'s cascade; not the wire `format`):
64
94
  * ```ts
65
95
  * const provider = createOllama({
66
96
  * model: 'qwen3.5:2b-q4_K_M',
@@ -77,50 +107,180 @@ import { ToolDefinition } from '@orkestrel/tool';
77
107
  export declare function createOllama(options: OllamaOptions): ProviderInterface;
78
108
 
79
109
  /**
80
- * How long the model stays resident after a call when `OllamaOptions.keepAlive` is
81
- * omitted — Ollama's own `keep_alive` default, expressed as a duration string.
110
+ * Names how long the model stays resident after a call — `'5m'` when
111
+ * `OllamaOptions.keepAlive` is omitted, Ollama's own `keep_alive` default, expressed as a
112
+ * duration string.
113
+ *
114
+ * @remarks
115
+ * The name mirrors the Ollama `/api/chat` `keep_alive` field this value is sent as, so
116
+ * the constant, the `OllamaOptions.keepAlive` key, and the wire member read as one term.
82
117
  */
83
118
  export declare const DEFAULT_KEEP_ALIVE = "5m";
84
119
 
85
- /** The local Ollama daemon base URL assumed when `OllamaOptions.url` is omitted. */
120
+ /**
121
+ * Names the local Ollama daemon base URL, `'http://localhost:11434'`, assumed when
122
+ * `OllamaOptions.url` is omitted.
123
+ */
86
124
  export declare const DEFAULT_OLLAMA_URL = "http://localhost:11434";
87
125
 
88
126
  /**
89
- * The per-call deadline in milliseconds when `OllamaOptions.timeout` is omitted —
90
- * generous enough that a cold model load does not trip it.
127
+ * Names the per-call deadline in milliseconds, `120_000`, when `OllamaOptions.timeout` is
128
+ * omitted — generous enough that a cold model load does not trip it.
91
129
  */
92
130
  export declare const DEFAULT_PROVIDER_TIMEOUT = 120000;
93
131
 
94
132
  /**
95
- * Whether a value is an {@link OllamaHTTPError}.
133
+ * Extracts a wire `arguments` value as a record.
134
+ *
135
+ * @remarks
136
+ * Total: an object passes through, a JSON string is parsed when it yields a record, and
137
+ * a malformed string yields `{}` rather than throwing.
138
+ *
139
+ * @param value - The wire's `function.arguments` value, of unknown shape
140
+ * @returns The argument record, or `{}` when the value carries none
141
+ *
142
+ * @example
143
+ * ```ts
144
+ * extractArguments('{"city":"Oslo"}') // { city: 'Oslo' }
145
+ * ```
146
+ */
147
+ export declare function extractArguments(value: unknown): Readonly<Record<string, unknown>>;
148
+
149
+ /**
150
+ * Extracts the assistant text of one wire record.
151
+ *
152
+ * @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
153
+ * @returns The record's `message.content` when it is a string, else `''`
154
+ *
155
+ * @example
156
+ * ```ts
157
+ * extractContent({ message: { content: 'ok' } }) // 'ok'
158
+ * ```
159
+ */
160
+ export declare function extractContent(record: Readonly<Record<string, unknown>>): string;
161
+
162
+ /**
163
+ * Extracts the daemon-side reasoning of one wire record.
164
+ *
165
+ * @remarks
166
+ * `message.thinking` is the `think: true` wire shape. It is read whatever the configured
167
+ * flag says, because a daemon may separate reasoning on its own.
168
+ *
169
+ * @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
170
+ * @returns The record's `message.thinking` when it is a string, else `''`
171
+ *
172
+ * @example
173
+ * ```ts
174
+ * extractThinking({ message: { thinking: 'weighing it' } }) // 'weighing it'
175
+ * ```
176
+ */
177
+ export declare function extractThinking(record: Readonly<Record<string, unknown>>): string;
178
+
179
+ /**
180
+ * Extracts the tool calls of one wire record's `message.tool_calls`.
181
+ *
182
+ * @remarks
183
+ * Each entry narrows to `{ id, name, arguments }`: the entry and its `function` must be
184
+ * records and `name` a string, else the entry is dropped. An id is minted when the wire
185
+ * omits one.
186
+ *
187
+ * @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
188
+ * @returns The narrowed tool calls, empty when the record carries none
189
+ *
190
+ * @example
191
+ * ```ts
192
+ * extractTools({ message: { tool_calls: [{ function: { name: 'weather' } }] } })
193
+ * // [{ id: '…', name: 'weather', arguments: {} }]
194
+ * ```
195
+ */
196
+ export declare function extractTools(record: Readonly<Record<string, unknown>>): readonly ToolCall[];
197
+
198
+ /**
199
+ * Extracts the token usage of one wire record.
200
+ *
201
+ * @remarks
202
+ * Both counts must be numbers, which is true of the non-stream body and the stream's
203
+ * `done: true` line. A delta line carries neither, so it yields `undefined`.
204
+ *
205
+ * @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
206
+ * @returns The `TokenUsage` shape, or `undefined` when either count is absent
207
+ *
208
+ * @example
209
+ * ```ts
210
+ * extractUsage({ prompt_eval_count: 3, eval_count: 4 })
211
+ * // { prompt: 3, completion: 4, total: 7 }
212
+ * ```
213
+ */
214
+ export declare function extractUsage(record: Readonly<Record<string, unknown>>): TokenUsage | undefined;
215
+
216
+ /**
217
+ * Checks whether a value is an {@link OllamaHTTPError}.
218
+ *
219
+ * @remarks
220
+ * The check is an `instanceof` test, so it narrows a caught `unknown` to the error class
221
+ * without parsing the thrown message.
96
222
  *
97
223
  * @param value - The value to test
98
- * @returns `true` when `value` is an `OllamaHTTPError`
224
+ * @returns True if `value` is an `OllamaHTTPError`; false otherwise
99
225
  */
100
226
  export declare function isOllamaHTTPError(value: unknown): value is OllamaHTTPError;
101
227
 
102
228
  /**
103
- * The cap, in characters, on how much of a non-OK response body is
229
+ * Joins a call's reasoning carriers — the splitter's separated in-content spans and the
230
+ * accumulated wire-side `message.thinking` — into the result's `thinking`.
231
+ *
232
+ * @param splitter - The per-call splitter holding the separated in-content spans
233
+ * @param wired - The accumulated wire-side `message.thinking` text
234
+ * @returns The carriers separated by a blank line, or whichever one is non-empty
235
+ *
236
+ * @example
237
+ * ```ts
238
+ * joinThinking(createThinkSplitter(), 'from the wire') // 'from the wire'
239
+ * ```
240
+ */
241
+ export declare function joinThinking(splitter: ThinkSplitterInterface, wired: string): string;
242
+
243
+ /**
244
+ * Maps conversation turns onto the `/api/chat` wire's minimal message shape.
245
+ *
246
+ * @remarks
247
+ * `tool_calls` is emitted only on a turn that replays them and `images` only on a
248
+ * multimodal turn, so an empty optional never reaches the wire.
249
+ *
250
+ * @param messages - The conversation turns to send
251
+ * @returns The wire `messages` array, one entry per turn, in order
252
+ *
253
+ * @example
254
+ * ```ts
255
+ * mapMessages([{ id: '1', role: 'user', content: 'Say hello.' }])
256
+ * // [{ role: 'user', content: 'Say hello.' }]
257
+ * ```
258
+ */
259
+ export declare function mapMessages(messages: readonly Message[]): WireChatRequest['messages'];
260
+
261
+ /**
262
+ * Names the character cap, `2048`, on how much of a non-OK response body is
104
263
  * incorporated into a thrown {@link OllamaHTTPError}'s message.
105
264
  *
106
265
  * @remarks
107
266
  * Bounds the excerpt so a defensive proxy or a misbehaving daemon handing
108
267
  * back an unbounded response body cannot inflate the thrown error's message
109
- * without limit (§14). `2048` characters is generous enough to carry a
110
- * useful diagnostic snippet while staying well short of any practical size
111
- * concern.
268
+ * without limit, while the cap stays generous enough to carry a useful
269
+ * diagnostic snippet.
112
270
  */
113
271
  export declare const MAX_ERROR_BODY_LENGTH = 2048;
114
272
 
115
273
  /**
116
- * An error thrown when the Ollama `/api/chat` HTTP transport fails.
274
+ * Represents an error thrown when the Ollama `/api/chat` HTTP transport fails.
117
275
  *
118
276
  * @remarks
119
- * Carries the response `status` (0 when no HTTP response was received at all,
120
- * e.g. a `null` body). Thrown by {@link OllamaProvider} at its two HTTP
121
- * failure sites — the non-OK status branch and the null-body branch — so a
122
- * caller can branch on `error.status` instead of parsing the message. Narrow
123
- * a caught value with {@link isOllamaHTTPError}.
277
+ * Carries the machine-readable `code` `'HTTP'` and the response `status` (0 when no
278
+ * HTTP response was received at all, for example a `null` body). Thrown by
279
+ * {@link OllamaProvider} at its HTTP failure sites — the non-OK status branch and the
280
+ * null-body branch — so a caller can branch on `error.code` and read `error.status`
281
+ * for the HTTP number instead of parsing the message. The message carries a body excerpt
282
+ * bounded to {@link MAX_ERROR_BODY_LENGTH} — `2048` characters. Narrow a caught value with
283
+ * {@link isOllamaHTTPError}.
124
284
  *
125
285
  * @example
126
286
  * ```ts
@@ -134,117 +294,142 @@ export declare const MAX_ERROR_BODY_LENGTH = 2048;
134
294
  * ```
135
295
  */
136
296
  export declare class OllamaHTTPError extends Error {
297
+ /**
298
+ * Names the machine-readable condition this error reports — `'HTTP'`: an `/api/chat`
299
+ * transport, status, or body failure.
300
+ */
301
+ readonly code: "HTTP";
137
302
  readonly status: number;
138
- constructor(message: string, status: number, options?: {
139
- readonly cause?: unknown;
140
- });
303
+ constructor(message: string, status: number, options?: OllamaHTTPErrorOptions);
141
304
  }
142
305
 
143
306
  /**
144
- * Options for `createOllama` — the local Ollama backend's configuration.
307
+ * Represents the options a thrown {@link OllamaHTTPError} accepts beside its message and status —
308
+ * the standard error `cause` link, named so a consumer can reference the shape.
309
+ *
310
+ * @remarks
311
+ * `cause` is the underlying value that produced the HTTP failure: the transport or
312
+ * body-read rejection the provider caught before rethrowing. It is `unknown` because a
313
+ * thrown value is unconstrained. Omitted ⇒ the error carries no cause.
314
+ */
315
+ export declare interface OllamaHTTPErrorOptions {
316
+ readonly cause?: unknown;
317
+ }
318
+
319
+ /**
320
+ * Represents the configuration `createOllama` accepts for the local Ollama backend.
145
321
  *
146
322
  * @remarks
147
323
  * Only `model` is required. `url` defaults to the local daemon, `keepAlive` controls
148
324
  * how long the model stays resident after a call, `timeout` is the per-call deadline
149
325
  * in milliseconds, and `options` is a passthrough bag of sampling parameters
150
- * (`temperature` / `seed` / `num_predict` / …) forwarded verbatim to the wire.
326
+ * (`temperature`, `seed`, and `num_predict`) forwarded verbatim to the wire.
151
327
  *
152
328
  * The optional `fetch` + `headers` form a **transport seam**: by default the provider
153
329
  * talks straight to a local daemon over `globalThis.fetch` with only a JSON content
154
- * type, but a browser-side runtime can inject a custom transport AND a dynamic header
155
- * (e.g. an obfuscated bearer token) so requests route through the developer's OWN
330
+ * type, but a browser-side runtime can inject both a custom transport and a dynamic header
331
+ * (for example an obfuscated bearer token) so requests route through the developer's own
156
332
  * server, which validates that header and forwards to the real LLM. Your app never
157
333
  * holds a real API key — the real key lives only on the developer's server; the
158
334
  * `headers` hook supplies whatever short-lived/obfuscated token that server expects.
159
335
  */
160
336
  export declare interface OllamaOptions {
161
337
  readonly model: string;
162
- /** The daemon base URL; defaults to `'http://localhost:11434'`. */
338
+ /** Sets the daemon base URL; defaults to `'http://localhost:11434'`. */
163
339
  readonly url?: string;
164
- /** How long the model stays resident after a call; defaults to `'5m'`. */
340
+ /**
341
+ * Sets how long the model stays resident after a call; defaults to `'5m'`. Mirrors the
342
+ * Ollama `/api/chat` `keep_alive` field, whose value this key carries verbatim onto
343
+ * {@link WireChatRequest.keep_alive}.
344
+ */
165
345
  readonly keepAlive?: string | number;
166
- /** The per-call deadline in milliseconds; defaults to `120_000`. */
346
+ /** Sets the per-call deadline in milliseconds; defaults to `120_000`. */
167
347
  readonly timeout?: number;
168
- /** Passthrough sampling options (`temperature` / `seed` / `num_predict` / …). */
348
+ /**
349
+ * Carries passthrough sampling parameters (`temperature`, `seed`, and `num_predict`).
350
+ * Mirrors the Ollama `/api/chat` `options` field, whose value this key carries verbatim
351
+ * onto {@link WireChatRequest.options}.
352
+ */
169
353
  readonly options?: Readonly<Record<string, unknown>>;
170
354
  /**
171
- * The `/api/chat` `think` wire flag; defaults to `false`. When `true`, a thinking-capable
172
- * model (e.g. `qwen3`) separates its reasoning NATIVELY at the wire — the daemon returns it
355
+ * Sets the `/api/chat` `think` wire flag; defaults to `false`. When `true`, a thinking-capable
356
+ * model (for example `qwen3`) separates its reasoning natively at the wire — the daemon returns it
173
357
  * on the distinct `message.thinking` channel (surfaced on `ProviderResult.thinking`) rather
174
- * than inline in `message.content`. The default stays `false` so a general-purpose provider
175
- * is backward-compatible and immediate for non-thinking models; the per-call ThinkSplitter
358
+ * than inline in `message.content`. The default is `false`, so a non-thinking model needs no
359
+ * configuration and answers immediately; the per-call ThinkSplitter
176
360
  * remains the defensive fallback for daemons/models that still inline `<think>` tags either
177
- * way. Set it `true` for a thinking model whose reasoning you intend to DISPLAY separately.
361
+ * way. Set it `true` for a thinking model whose reasoning you intend to display separately.
178
362
  */
179
363
  readonly think?: boolean;
180
364
  /**
181
- * A custom `fetch` implementation for every request; defaults to
365
+ * Sets a custom `fetch` implementation for every request; defaults to
182
366
  * `globalThis.fetch`. Lets a runtime inject its own transport (a browser fetch
183
- * pointed at the developer's server, an instrumented wrapper, …) without changing
367
+ * pointed at the developer's server, an instrumented wrapper) without changing
184
368
  * the wire protocol. Omitted ⇒ the global `fetch`.
185
369
  */
186
370
  readonly fetch?: typeof globalThis.fetch;
187
371
  /**
188
- * A dynamic, possibly-async header injector called once per request; its returned
372
+ * Sets a dynamic, possibly-async header injector called once per request; its returned
189
373
  * headers are merged into the request on top of the base `Content-Type`. Use it to
190
- * attach an authorization header — e.g. an obfuscated/generated bearer token the
374
+ * attach an authorization header — for example an obfuscated/generated bearer token the
191
375
  * developer's server validates before relaying to the real LLM — so a browser
192
- * runtime can authenticate WITHOUT your app ever handling a real API key. Async so a
376
+ * runtime can authenticate without your app ever handling a real API key. Async so a
193
377
  * token can be refreshed/fetched per call. A returned `Content-Type` overrides the
194
378
  * default; other headers add to it. Omitted ⇒ only `Content-Type: application/json`.
195
379
  */
196
- readonly headers?: () => Record<string, string> | Promise<Record<string, string>>;
380
+ readonly headers?: () => Readonly<Record<string, string>> | Promise<Readonly<Record<string, string>>>;
197
381
  /**
198
- * The provider's OPTIONAL context-framing default — the PROVIDER-DEFAULT level of
382
+ * Sets the provider's optional context-framing default — the provider-default level of
199
383
  * `AgentContext`'s format cascade (beaten by a manager-options or per-item override,
200
384
  * beating the managers' built-in framing). Declares how this provider's models prefer
201
- * context sections framed (e.g. XML group wrappers vs. Markdown headers). Omitted ⇒
202
- * the provider is framing-agnostic and core's built-in defaults apply unchanged. NOTE:
203
- * this is the prompt-CONTEXT framing consumed by `AgentContext.build()` — it is NOT
385
+ * context sections framed (for example XML group wrappers vs. Markdown headers). Omitted ⇒
386
+ * the provider is framing-agnostic and core's built-in defaults apply unchanged. This
387
+ * is the prompt-context framing consumed by `AgentContext.build()` — it is not
204
388
  * Ollama's `/api/chat` `format` wire parameter (structured-output / JSON schema),
205
- * which this provider does not currently send; the two are unrelated despite the
206
- * shared word.
389
+ * which this provider sends only when a call supplies a `schema`; the two are unrelated
390
+ * despite the shared word.
207
391
  */
208
- readonly format?: ContextFormatInterface;
392
+ readonly format?: ContextFormat;
209
393
  }
210
394
 
211
395
  /**
212
- * The local Ollama inference boundary — a {@link ProviderInterface} over Ollama's
396
+ * Implements the local Ollama inference boundary — a {@link ProviderInterface} over Ollama's
213
397
  * `POST /api/chat`, both non-streaming (`generate`) and streaming NDJSON (`stream`).
214
398
  *
215
399
  * @remarks
216
400
  * - **Wire protocol.** Posts `{ model, messages, stream, keep_alive, think }` plus
217
401
  * passthrough sampling `options` and mapped function `tools`. The `think` flag is
218
- * CONFIGURABLE via {@link OllamaOptions.think} (default `false`). Non-stream parses
402
+ * configurable through {@link OllamaOptions.think} (default `false`). Non-stream parses
219
403
  * one JSON body; stream consumes NDJSON (one JSON object per `\n`-terminated line) —
220
404
  * deltas carry `message.content`, the final `done: true` line carries the token usage.
221
- * - **Think separation (H4).** The wire `think` flag is configurable
405
+ * - **Think separation.** The wire `think` flag is configurable
222
406
  * ({@link OllamaOptions.think}, default `false`). With `think: true` a thinking model's
223
- * daemon separates reasoning NATIVELY — returning it on the distinct `message.thinking`
224
- * channel (read here via `#thinking`) instead of inline in `message.content`. EITHER
407
+ * daemon separates reasoning natively — returning it on the distinct `message.thinking`
408
+ * channel (read here through `extractThinking`) instead of inline in `message.content`. Either
225
409
  * way the per-call {@link ThinkSplitterInterface} is the defensive guarantee: a daemon
226
410
  * may ignore `think: false` for a thinking model and inline `<think>` tags, so every
227
- * content delta routes through the splitter, only CLEAN content is yielded / assembled,
411
+ * content delta routes through the splitter, only clean content is yielded / assembled,
228
412
  * and the separated reasoning (plus any daemon-side `message.thinking` deltas) lands on
229
413
  * `ProviderResult.thinking`, never in the conversation.
230
- * - **Boundary narrowing (§14).** Every wire value arrives as `unknown` and is
414
+ * - **Boundary narrowing.** Every wire value arrives as `unknown` and is
231
415
  * narrowed through guards (`isRecord` / `isString` / `isNumber`) — never `as`. A
232
416
  * missing / malformed field degrades to a sensible default (empty content, no
233
417
  * usage, `{}` arguments), never a throw.
234
418
  * - **Bounded.** Each call arms a {@link Timeout} for `OllamaOptions.timeout` and
235
419
  * passes `AbortSignal.any([timeout.signal, signal])` to `fetch`, so the caller's
236
- * signal AND the deadline both cancel the request. The timeout is always cleared —
420
+ * signal and the deadline both cancel the request. The timeout is always cleared —
237
421
  * in `#fetch` if the request fails/aborts, otherwise in the consuming call's `finally`.
238
422
  * - **Abort recovers partial.** A `stream` cancelled mid-flight throws a
239
423
  * `ProviderAbortError` carrying the partial result assembled so far; pairing the
240
- * `TextDecoder({ stream: true })` with the {@link NDJSONParser} parser keeps multi-byte
424
+ * `TextDecoder({ stream: true })` with the `createNDJSONParser` parser keeps multi-byte
241
425
  * UTF-8 splits and partial lines honest.
242
426
  * - **Event-free.** A pure functional boundary — no Emitter, no events.
243
427
  * - **Transport seam.** {@link OllamaOptions.fetch} swaps the transport (default
244
428
  * `globalThis.fetch`) and {@link OllamaOptions.headers} is a per-request, possibly
245
429
  * async header injector merged over the base `Content-Type` — so a browser runtime
246
430
  * can route through the developer's own server with an obfuscated bearer token,
247
- * without this library ever handling a real API key. Both omitted ⇒ today's behaviour.
431
+ * without this library ever handling a real API key. Both omitted ⇒ the global `fetch`
432
+ * and only a JSON content type.
248
433
  * Orthogonal to the deadline: the hook is awaited inside `#fetch`'s try, so a hook
249
434
  * rejection clears the armed timer like any other request failure.
250
435
  *
@@ -256,91 +441,157 @@ export declare interface OllamaOptions {
256
441
  */
257
442
  export declare class OllamaProvider implements ProviderInterface {
258
443
  #private;
259
- readonly id: `${string}-${string}-${string}-${string}-${string}`;
260
444
  readonly name = "ollama";
261
445
  constructor(options: OllamaOptions);
262
446
  /**
263
- * The provider's context-framing default — the PROVIDER-DEFAULT level of
264
- * {@link import('@orkestrel/agent').AgentContextInterface.build}'s format cascade (it BEATS
265
- * the managers' built-in framing, is BEATEN by a manager-options or per-item override).
266
- * Satisfies the OPTIONAL {@link ProviderInterface.format} contract member: `undefined`
447
+ * Exposes this instance's identity — a fresh `crypto.randomUUID()` minted at
448
+ * construction, satisfying the {@link ProviderInterface.id} contract member. A second
449
+ * provider built from identical options carries a distinct id.
450
+ *
451
+ * @returns The instance's minted identifier
452
+ */
453
+ get id(): string;
454
+ /**
455
+ * Exposes the provider's context-framing default — the provider-default level of
456
+ * {@link import('@orkestrel/agent').AgentContextInterface.build}'s format cascade (it beats
457
+ * the managers' built-in framing, is beaten by a manager-options or per-item override).
458
+ * Satisfies the optional {@link ProviderInterface.format} contract member: `undefined`
267
459
  * when {@link OllamaOptions.format} was omitted (the framing-agnostic default ⇒ core's
268
460
  * built-in framing applies unchanged), else the exact configured framing the Agent
269
461
  * threads into `build()`.
270
462
  *
271
463
  * @remarks
272
- * EXPOSE-ONLY — read by the Agent loop and consumed by core's cascade; it is NEVER sent
273
- * on the `/api/chat` wire (it is absent from `#body` / the request). This is NOT Ollama's
274
- * structured-output `format` wire parameter — that one IS sent in `#body`, but only when
464
+ * Expose-only — read by the Agent loop and consumed by core's cascade; it is never sent
465
+ * on the `/api/chat` wire (it is absent from `#body` / the request). This is not Ollama's
466
+ * structured-output `format` wire parameter — that one is sent in `#body`, but only when
275
467
  * a per-call `ProviderStreamOptions.schema` is supplied; only the word collides.
276
468
  *
277
- * @returns The configured {@link ContextFormatInterface}, or `undefined` when none
469
+ * @returns The configured {@link ContextFormat}, or `undefined` when none
278
470
  */
279
- get format(): ContextFormatInterface | undefined;
280
- generate(messages: readonly MessageInterface[], signal: AbortSignal, tools?: readonly ToolDefinition[], options?: ProviderStreamOptions): Promise<ProviderResult>;
281
- stream(messages: readonly MessageInterface[], signal: AbortSignal, tools?: readonly ToolDefinition[], options?: ProviderStreamOptions): AsyncGenerator<ProviderDelta, ProviderResult>;
282
- }
471
+ get format(): ContextFormat | undefined;
472
+ /**
473
+ * Generates one complete turn and resolves the assembled result — the clean content,
474
+ * any separated reasoning, any tool calls, and any usage the wire reported.
475
+ *
476
+ * @remarks
477
+ * Sends `stream: false` and parses one JSON body. Content routes through a per-call
478
+ * think splitter, so the assembled content stays clean even where the daemon renders a
479
+ * thinking model's reasoning inline; the separated spans and any daemon-side
480
+ * `message.thinking` land on `thinking`. The caller's signal and the armed deadline
481
+ * both cancel the request, and the deadline is cleared once the body is read.
482
+ *
483
+ * @param messages - The conversation turns to send
484
+ * @param signal - The caller's bounding signal, folded with the armed deadline
485
+ * @param tools - The callable tools to advertise for this turn, when the caller passes any
486
+ * @param options - The per-call overrides, `think` and `schema` among them
487
+ * @returns The assembled result of the turn
488
+ * @throws {@link OllamaHTTPError} When the daemon answers a non-OK status.
489
+ */
490
+ generate(messages: readonly Message[], signal: AbortSignal, tools?: readonly ToolDefinition[], options?: ProviderStreamOptions): Promise<ProviderResult>;
491
+ /**
492
+ * Streams one turn, yielding a channel-tagged delta per non-empty content or reasoning
493
+ * span and returning the assembled result when the stream completes.
494
+ *
495
+ * @remarks
496
+ * Sends `stream: true` and consumes NDJSON — one JSON object per newline-terminated
497
+ * line — pairing a streaming `TextDecoder` with the `NDJSONParser` so a record split
498
+ * across byte reads is reassembled. The returned result's content is the splitter's
499
+ * clean accumulation, beside any tool calls collected across lines and the usage the
500
+ * `done` line carries. A cancel mid-flight throws a `ProviderAbortError` carrying the
501
+ * partial assembled so far.
502
+ *
503
+ * @param messages - The conversation turns to send
504
+ * @param signal - The caller's bounding signal, folded with the armed deadline
505
+ * @param tools - The callable tools to advertise for this turn, when the caller passes any
506
+ * @param options - The per-call overrides, `think` and `schema` among them
507
+ * @returns The assembled result of the turn, after the last delta
508
+ * @throws {@link OllamaHTTPError} When the daemon answers a non-OK status or a `null` body.
509
+ */
510
+ stream(messages: readonly Message[], signal: AbortSignal, tools?: readonly ToolDefinition[], options?: ProviderStreamOptions): AsyncGenerator<ProviderDelta, ProviderResult>;
511
+ }
283
512
 
284
- /**
285
- * A live `fetch` to `/api/chat` with the deadline + combined signal that bound it —
286
- * the internal wire-shape `OllamaProvider.#fetch` hands back to a consuming call.
287
- *
288
- * @remarks
289
- * The `response` is the open `POST /api/chat` `Response`; `timeout` is the armed
290
- * {@link TimeoutInterface} the consuming call clears once it finishes reading the body
291
- * (or that `#fetch` itself clears on a failed/aborted request); `combined` is the
292
- * `AbortSignal.any([timeout.signal, callerSignal])` the request was issued under, which
293
- * the streaming path checks to tell a mid-stream cancel apart from any other error.
294
- */
295
- export declare interface OllamaResponse {
296
- readonly response: Response;
297
- readonly timeout: TimeoutInterface;
298
- readonly combined: AbortSignal;
299
- }
513
+ /**
514
+ * Represents an open `POST /api/chat` response together with the deadline and the
515
+ * combined signal that bound the request.
516
+ *
517
+ * @remarks
518
+ * The `response` is the open `POST /api/chat` `Response`; `timeout` is the armed
519
+ * {@link TimeoutInterface} the consuming call clears once it finishes reading the body
520
+ * (or that the provider clears on a failed or aborted request); `combined` is the
521
+ * `AbortSignal.any([timeout.signal, callerSignal])` the request was issued under, which
522
+ * the streaming path checks to tell a mid-stream cancel apart from any other error.
523
+ */
524
+ export declare interface OllamaResponse {
525
+ readonly response: Response;
526
+ readonly timeout: TimeoutInterface;
527
+ readonly combined: AbortSignal;
528
+ }
300
529
 
301
- /**
302
- * The exact `POST /api/chat` request body `OllamaProvider` sends — the internal typed
303
- * wire contract.
304
- *
305
- * @remarks
306
- * This is the typed wire shape asserted against the official `ollama` client's
307
- * `ChatRequest` by the compile-time parity test; `src/` never imports `ollama` itself.
308
- * `messages` mirrors the minimal turn shape `#plain` builds (`role` / `content`, plus
309
- * `tool_calls` only on a turn that replays them and `images` only on a multimodal
310
- * turn); `options` and `tools` are only present when configured.
311
- */
312
- export declare interface WireChatRequest {
313
- readonly model: string;
314
- readonly messages: ReadonlyArray<{
315
- readonly role: string;
316
- readonly content: string;
317
- readonly tool_calls?: ReadonlyArray<{
318
- readonly function: {
319
- readonly name: string;
320
- readonly arguments: Readonly<Record<string, unknown>>;
321
- };
322
- }>;
323
- readonly images?: readonly string[];
324
- }>;
325
- readonly stream: boolean;
326
- readonly keep_alive: string | number;
327
- readonly think: boolean;
328
- readonly options?: Readonly<Record<string, unknown>>;
329
- readonly tools?: ReadonlyArray<{
330
- readonly type: 'function';
331
- readonly function: {
332
- readonly name: string;
333
- readonly description?: string;
334
- readonly parameters?: Readonly<Record<string, unknown>>;
335
- };
336
- }>;
337
- /**
338
- * The `/api/chat` structured-output constraint — a JSON-Schema object forwarded
339
- * verbatim from the per-call `ProviderStreamOptions.schema`. This is NOT
340
- * `OllamaOptions.format` (the unrelated prompt-context framing); only present
341
- * when a call supplies a `schema`.
342
- */
343
- readonly format?: Readonly<Record<string, unknown>>;
344
- }
530
+ /**
531
+ * Parses a non-stream `/api/chat` response body into a wire record.
532
+ *
533
+ * @remarks
534
+ * Total by construction: an empty body, a body that is not JSON, and a body whose JSON is
535
+ * not an object all yield `undefined`, so a malformed daemon response never escapes as a
536
+ * `SyntaxError`. The call site supplies the empty-record default that reads as empty
537
+ * content and no usage.
538
+ *
539
+ * @param response - The 200-OK `/api/chat` response whose body is read as text
540
+ * @returns The parsed record, or `undefined` when the body is empty or malformed
541
+ *
542
+ * @example
543
+ * ```ts
544
+ * await parseBody(new Response('{"message":{"content":"ok"}}'))
545
+ * // { message: { content: 'ok' } }
546
+ * ```
547
+ */
548
+ export declare function parseBody(response: Response): Promise<Readonly<Record<string, unknown>> | undefined>;
549
+
550
+ /**
551
+ * Represents the exact `POST /api/chat` request body `OllamaProvider` sends — the internal typed
552
+ * wire contract.
553
+ *
554
+ * @remarks
555
+ * This is the typed wire shape asserted against the official `ollama` client's
556
+ * `ChatRequest` by the compile-time parity test; `src/` never imports `ollama` itself.
557
+ * `messages` mirrors the minimal turn shape `mapMessages` builds (`role` / `content`, plus
558
+ * `tool_calls` only on a turn that replays them and `images` only on a multimodal
559
+ * turn); `options` and `tools` are only present when configured. `format` carries the
560
+ * `/api/chat` structured-output constraint, forwarded verbatim from the per-call
561
+ * `ProviderStreamOptions.schema` and absent when no schema is supplied.
562
+ */
563
+ export declare interface WireChatRequest {
564
+ readonly model: string;
565
+ readonly messages: ReadonlyArray<{
566
+ readonly role: string;
567
+ readonly content: string;
568
+ readonly tool_calls?: ReadonlyArray<{
569
+ readonly function: {
570
+ readonly name: string;
571
+ readonly arguments: Readonly<Record<string, unknown>>;
572
+ };
573
+ }>;
574
+ readonly images?: readonly string[];
575
+ }>;
576
+ readonly stream: boolean;
577
+ readonly keep_alive: string | number;
578
+ readonly think: boolean;
579
+ readonly options?: Readonly<Record<string, unknown>>;
580
+ readonly tools?: ReadonlyArray<{
581
+ readonly type: 'function';
582
+ readonly function: {
583
+ readonly name: string;
584
+ readonly description?: string;
585
+ readonly parameters?: Readonly<Record<string, unknown>>;
586
+ };
587
+ }>;
588
+ /**
589
+ * Holds the `/api/chat` structured-output constraint — a JSON-Schema object forwarded
590
+ * verbatim from the per-call `ProviderStreamOptions.schema`. This is not
591
+ * `OllamaOptions.format` (the unrelated prompt-context framing); only present
592
+ * when a call supplies a `schema`.
593
+ */
594
+ readonly format?: Readonly<Record<string, unknown>>;
595
+ }
345
596
 
346
- export { }
597
+ export { }