@orkestrel/ollama 0.0.13 → 0.0.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,43 +1,52 @@
1
+ import { isNumber, isRecord, isString, parseJSONAs } from "@orkestrel/contract";
1
2
  import { ProviderAbortError, createThinkSplitter } from "@orkestrel/agent";
2
- import { isNumber, isRecord, isString } from "@orkestrel/contract";
3
3
  import { createNDJSONParser } from "@orkestrel/ndjson";
4
4
  import { Timeout } from "@orkestrel/timeout";
5
5
  //#region src/server/constants.ts
6
- /** The local Ollama daemon base URL assumed when `OllamaOptions.url` is omitted. */
6
+ /**
7
+ * Names the local Ollama daemon base URL, `'http://localhost:11434'`, assumed when
8
+ * `OllamaOptions.url` is omitted.
9
+ */
7
10
  var DEFAULT_OLLAMA_URL = "http://localhost:11434";
8
11
  /**
9
- * How long the model stays resident after a call when `OllamaOptions.keepAlive` is
10
- * omitted — Ollama's own `keep_alive` default, expressed as a duration string.
12
+ * Names how long the model stays resident after a call — `'5m'` when
13
+ * `OllamaOptions.keepAlive` is omitted, Ollama's own `keep_alive` default, expressed as a
14
+ * duration string.
15
+ *
16
+ * @remarks
17
+ * The name mirrors the Ollama `/api/chat` `keep_alive` field this value is sent as, so
18
+ * the constant, the `OllamaOptions.keepAlive` key, and the wire member read as one term.
11
19
  */
12
20
  var DEFAULT_KEEP_ALIVE = "5m";
13
21
  /**
14
- * The per-call deadline in milliseconds when `OllamaOptions.timeout` is omitted —
15
- * generous enough that a cold model load does not trip it.
22
+ * Names the per-call deadline in milliseconds, `120_000`, when `OllamaOptions.timeout` is
23
+ * omitted — generous enough that a cold model load does not trip it.
16
24
  */
17
25
  var DEFAULT_PROVIDER_TIMEOUT = 12e4;
18
26
  /**
19
- * The cap, in characters, on how much of a non-OK response body is
27
+ * Names the character cap, `2048`, on how much of a non-OK response body is
20
28
  * incorporated into a thrown {@link OllamaHTTPError}'s message.
21
29
  *
22
30
  * @remarks
23
31
  * Bounds the excerpt so a defensive proxy or a misbehaving daemon handing
24
32
  * back an unbounded response body cannot inflate the thrown error's message
25
- * without limit (§14). `2048` characters is generous enough to carry a
26
- * useful diagnostic snippet while staying well short of any practical size
27
- * concern.
33
+ * without limit, while the cap stays generous enough to carry a useful
34
+ * diagnostic snippet.
28
35
  */
29
36
  var MAX_ERROR_BODY_LENGTH = 2048;
30
37
  //#endregion
31
38
  //#region src/server/errors.ts
32
39
  /**
33
- * An error thrown when the Ollama `/api/chat` HTTP transport fails.
40
+ * Represents an error thrown when the Ollama `/api/chat` HTTP transport fails.
34
41
  *
35
42
  * @remarks
36
- * Carries the response `status` (0 when no HTTP response was received at all,
37
- * e.g. a `null` body). Thrown by {@link OllamaProvider} at its two HTTP
38
- * failure sites — the non-OK status branch and the null-body branch — so a
39
- * caller can branch on `error.status` instead of parsing the message. Narrow
40
- * a caught value with {@link isOllamaHTTPError}.
43
+ * Carries the machine-readable `code` `'HTTP'` and the response `status` (0 when no
44
+ * HTTP response was received at all, for example a `null` body). Thrown by
45
+ * {@link OllamaProvider} at its HTTP failure sites — the non-OK status branch and the
46
+ * null-body branch — so a caller can branch on `error.code` and read `error.status`
47
+ * for the HTTP number instead of parsing the message. The message carries a body excerpt
48
+ * bounded to {@link MAX_ERROR_BODY_LENGTH} — `2048` characters. Narrow a caught value with
49
+ * {@link isOllamaHTTPError}.
41
50
  *
42
51
  * @example
43
52
  * ```ts
@@ -51,6 +60,11 @@ var MAX_ERROR_BODY_LENGTH = 2048;
51
60
  * ```
52
61
  */
53
62
  var OllamaHTTPError = class extends Error {
63
+ /**
64
+ * Names the machine-readable condition this error reports — `'HTTP'`: an `/api/chat`
65
+ * transport, status, or body failure.
66
+ */
67
+ code = "HTTP";
54
68
  status;
55
69
  constructor(message, status, options) {
56
70
  super(message, options);
@@ -59,53 +73,275 @@ var OllamaHTTPError = class extends Error {
59
73
  }
60
74
  };
61
75
  /**
62
- * Whether a value is an {@link OllamaHTTPError}.
76
+ * Checks whether a value is an {@link OllamaHTTPError}.
77
+ *
78
+ * @remarks
79
+ * The check is an `instanceof` test, so it narrows a caught `unknown` to the error class
80
+ * without parsing the thrown message.
63
81
  *
64
82
  * @param value - The value to test
65
- * @returns `true` when `value` is an `OllamaHTTPError`
83
+ * @returns True if `value` is an `OllamaHTTPError`; false otherwise
66
84
  */
67
85
  function isOllamaHTTPError(value) {
68
86
  return value instanceof OllamaHTTPError;
69
87
  }
70
88
  //#endregion
89
+ //#region src/server/helpers.ts
90
+ /**
91
+ * Maps conversation turns onto the `/api/chat` wire's minimal message shape.
92
+ *
93
+ * @remarks
94
+ * `tool_calls` is emitted only on a turn that replays them and `images` only on a
95
+ * multimodal turn, so an empty optional never reaches the wire.
96
+ *
97
+ * @param messages - The conversation turns to send
98
+ * @returns The wire `messages` array, one entry per turn, in order
99
+ *
100
+ * @example
101
+ * ```ts
102
+ * mapMessages([{ id: '1', role: 'user', content: 'Say hello.' }])
103
+ * // [{ role: 'user', content: 'Say hello.' }]
104
+ * ```
105
+ */
106
+ function mapMessages(messages) {
107
+ return messages.map((message) => ({
108
+ role: message.role,
109
+ content: message.content,
110
+ ...message.calls !== void 0 && message.calls.length > 0 ? { tool_calls: message.calls.map((call) => ({ function: {
111
+ name: call.name,
112
+ arguments: call.arguments
113
+ } })) } : {},
114
+ ...message.images !== void 0 && message.images.length > 0 ? { images: [...message.images] } : {}
115
+ }));
116
+ }
117
+ /**
118
+ * Builds a `ProviderResult` from a turn's content, reasoning, tool calls, and usage.
119
+ *
120
+ * @remarks
121
+ * Only the present optionals are set: no empty `thinking`, no empty `tools`, and no
122
+ * `usage` unless the wire reported one.
123
+ *
124
+ * @param content - The clean assistant content the splitter accumulated
125
+ * @param thinking - The joined reasoning, empty when the turn produced none
126
+ * @param tools - The tool calls collected across the turn
127
+ * @param usage - The token usage, or `undefined` when the wire reported none
128
+ * @returns The result carrying only its populated fields
129
+ *
130
+ * @example
131
+ * ```ts
132
+ * buildResult('ok', '', [], undefined) // { content: 'ok' }
133
+ * ```
134
+ */
135
+ function buildResult(content, thinking, tools, usage) {
136
+ const result = { content };
137
+ if (thinking.length > 0) result.thinking = thinking;
138
+ if (tools.length > 0) result.tools = tools;
139
+ if (usage !== void 0) result.usage = usage;
140
+ return result;
141
+ }
142
+ /**
143
+ * Extracts the assistant text of one wire record.
144
+ *
145
+ * @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
146
+ * @returns The record's `message.content` when it is a string, else `''`
147
+ *
148
+ * @example
149
+ * ```ts
150
+ * extractContent({ message: { content: 'ok' } }) // 'ok'
151
+ * ```
152
+ */
153
+ function extractContent(record) {
154
+ const message = Reflect.get(record, "message");
155
+ if (!isRecord(message)) return "";
156
+ const content = Reflect.get(message, "content");
157
+ return isString(content) ? content : "";
158
+ }
159
+ /**
160
+ * Extracts the daemon-side reasoning of one wire record.
161
+ *
162
+ * @remarks
163
+ * `message.thinking` is the `think: true` wire shape. It is read whatever the configured
164
+ * flag says, because a daemon may separate reasoning on its own.
165
+ *
166
+ * @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
167
+ * @returns The record's `message.thinking` when it is a string, else `''`
168
+ *
169
+ * @example
170
+ * ```ts
171
+ * extractThinking({ message: { thinking: 'weighing it' } }) // 'weighing it'
172
+ * ```
173
+ */
174
+ function extractThinking(record) {
175
+ const message = Reflect.get(record, "message");
176
+ if (!isRecord(message)) return "";
177
+ const thinking = Reflect.get(message, "thinking");
178
+ return isString(thinking) ? thinking : "";
179
+ }
180
+ /**
181
+ * Joins a call's reasoning carriers — the splitter's separated in-content spans and the
182
+ * accumulated wire-side `message.thinking` — into the result's `thinking`.
183
+ *
184
+ * @param splitter - The per-call splitter holding the separated in-content spans
185
+ * @param wired - The accumulated wire-side `message.thinking` text
186
+ * @returns The carriers separated by a blank line, or whichever one is non-empty
187
+ *
188
+ * @example
189
+ * ```ts
190
+ * joinThinking(createThinkSplitter(), 'from the wire') // 'from the wire'
191
+ * ```
192
+ */
193
+ function joinThinking(splitter, wired) {
194
+ if (splitter.thinking.length === 0) return wired;
195
+ if (wired.length === 0) return splitter.thinking;
196
+ return `${splitter.thinking}\n\n${wired}`;
197
+ }
198
+ /**
199
+ * Extracts the token usage of one wire record.
200
+ *
201
+ * @remarks
202
+ * Both counts must be numbers, which is true of the non-stream body and the stream's
203
+ * `done: true` line. A delta line carries neither, so it yields `undefined`.
204
+ *
205
+ * @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
206
+ * @returns The `TokenUsage` shape, or `undefined` when either count is absent
207
+ *
208
+ * @example
209
+ * ```ts
210
+ * extractUsage({ prompt_eval_count: 3, eval_count: 4 })
211
+ * // { prompt: 3, completion: 4, total: 7 }
212
+ * ```
213
+ */
214
+ function extractUsage(record) {
215
+ const prompt = Reflect.get(record, "prompt_eval_count");
216
+ const completion = Reflect.get(record, "eval_count");
217
+ if (!isNumber(prompt) || !isNumber(completion)) return void 0;
218
+ return {
219
+ prompt,
220
+ completion,
221
+ total: prompt + completion
222
+ };
223
+ }
224
+ /**
225
+ * Extracts the tool calls of one wire record's `message.tool_calls`.
226
+ *
227
+ * @remarks
228
+ * Each entry narrows to `{ id, name, arguments }`: the entry and its `function` must be
229
+ * records and `name` a string, else the entry is dropped. An id is minted when the wire
230
+ * omits one.
231
+ *
232
+ * @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
233
+ * @returns The narrowed tool calls, empty when the record carries none
234
+ *
235
+ * @example
236
+ * ```ts
237
+ * extractTools({ message: { tool_calls: [{ function: { name: 'weather' } }] } })
238
+ * // [{ id: '…', name: 'weather', arguments: {} }]
239
+ * ```
240
+ */
241
+ function extractTools(record) {
242
+ const message = Reflect.get(record, "message");
243
+ if (!isRecord(message)) return [];
244
+ const calls = Reflect.get(message, "tool_calls");
245
+ if (!Array.isArray(calls)) return [];
246
+ const out = [];
247
+ for (const entry of calls) {
248
+ if (!isRecord(entry)) continue;
249
+ const callable = Reflect.get(entry, "function");
250
+ if (!isRecord(callable)) continue;
251
+ const name = Reflect.get(callable, "name");
252
+ if (!isString(name)) continue;
253
+ const id = Reflect.get(entry, "id");
254
+ out.push({
255
+ id: isString(id) ? id : crypto.randomUUID(),
256
+ name,
257
+ arguments: extractArguments(Reflect.get(callable, "arguments"))
258
+ });
259
+ }
260
+ return out;
261
+ }
262
+ /**
263
+ * Extracts a wire `arguments` value as a record.
264
+ *
265
+ * @remarks
266
+ * Total: an object passes through, a JSON string is parsed when it yields a record, and
267
+ * a malformed string yields `{}` rather than throwing.
268
+ *
269
+ * @param value - The wire's `function.arguments` value, of unknown shape
270
+ * @returns The argument record, or `{}` when the value carries none
271
+ *
272
+ * @example
273
+ * ```ts
274
+ * extractArguments('{"city":"Oslo"}') // { city: 'Oslo' }
275
+ * ```
276
+ */
277
+ function extractArguments(value) {
278
+ if (isRecord(value)) return value;
279
+ if (isString(value)) return parseJSONAs(value, isRecord) ?? {};
280
+ return {};
281
+ }
282
+ //#endregion
283
+ //#region src/server/parsers.ts
284
+ /**
285
+ * Parses a non-stream `/api/chat` response body into a wire record.
286
+ *
287
+ * @remarks
288
+ * Total by construction: an empty body, a body that is not JSON, and a body whose JSON is
289
+ * not an object all yield `undefined`, so a malformed daemon response never escapes as a
290
+ * `SyntaxError`. The call site supplies the empty-record default that reads as empty
291
+ * content and no usage.
292
+ *
293
+ * @param response - The 200-OK `/api/chat` response whose body is read as text
294
+ * @returns The parsed record, or `undefined` when the body is empty or malformed
295
+ *
296
+ * @example
297
+ * ```ts
298
+ * await parseBody(new Response('{"message":{"content":"ok"}}'))
299
+ * // { message: { content: 'ok' } }
300
+ * ```
301
+ */
302
+ async function parseBody(response) {
303
+ return parseJSONAs(await response.text(), isRecord);
304
+ }
305
+ //#endregion
71
306
  //#region src/server/OllamaProvider.ts
72
307
  /**
73
- * The local Ollama inference boundary — a {@link ProviderInterface} over Ollama's
308
+ * Implements the local Ollama inference boundary — a {@link ProviderInterface} over Ollama's
74
309
  * `POST /api/chat`, both non-streaming (`generate`) and streaming NDJSON (`stream`).
75
310
  *
76
311
  * @remarks
77
312
  * - **Wire protocol.** Posts `{ model, messages, stream, keep_alive, think }` plus
78
313
  * passthrough sampling `options` and mapped function `tools`. The `think` flag is
79
- * CONFIGURABLE via {@link OllamaOptions.think} (default `false`). Non-stream parses
314
+ * configurable through {@link OllamaOptions.think} (default `false`). Non-stream parses
80
315
  * one JSON body; stream consumes NDJSON (one JSON object per `\n`-terminated line) —
81
316
  * deltas carry `message.content`, the final `done: true` line carries the token usage.
82
- * - **Think separation (H4).** The wire `think` flag is configurable
317
+ * - **Think separation.** The wire `think` flag is configurable
83
318
  * ({@link OllamaOptions.think}, default `false`). With `think: true` a thinking model's
84
- * daemon separates reasoning NATIVELY — returning it on the distinct `message.thinking`
85
- * channel (read here via `#thinking`) instead of inline in `message.content`. EITHER
319
+ * daemon separates reasoning natively — returning it on the distinct `message.thinking`
320
+ * channel (read here through `extractThinking`) instead of inline in `message.content`. Either
86
321
  * way the per-call {@link ThinkSplitterInterface} is the defensive guarantee: a daemon
87
322
  * may ignore `think: false` for a thinking model and inline `<think>` tags, so every
88
- * content delta routes through the splitter, only CLEAN content is yielded / assembled,
323
+ * content delta routes through the splitter, only clean content is yielded / assembled,
89
324
  * and the separated reasoning (plus any daemon-side `message.thinking` deltas) lands on
90
325
  * `ProviderResult.thinking`, never in the conversation.
91
- * - **Boundary narrowing (§14).** Every wire value arrives as `unknown` and is
326
+ * - **Boundary narrowing.** Every wire value arrives as `unknown` and is
92
327
  * narrowed through guards (`isRecord` / `isString` / `isNumber`) — never `as`. A
93
328
  * missing / malformed field degrades to a sensible default (empty content, no
94
329
  * usage, `{}` arguments), never a throw.
95
330
  * - **Bounded.** Each call arms a {@link Timeout} for `OllamaOptions.timeout` and
96
331
  * passes `AbortSignal.any([timeout.signal, signal])` to `fetch`, so the caller's
97
- * signal AND the deadline both cancel the request. The timeout is always cleared —
332
+ * signal and the deadline both cancel the request. The timeout is always cleared —
98
333
  * in `#fetch` if the request fails/aborts, otherwise in the consuming call's `finally`.
99
334
  * - **Abort recovers partial.** A `stream` cancelled mid-flight throws a
100
335
  * `ProviderAbortError` carrying the partial result assembled so far; pairing the
101
- * `TextDecoder({ stream: true })` with the {@link NDJSONParser} parser keeps multi-byte
336
+ * `TextDecoder({ stream: true })` with the `createNDJSONParser` parser keeps multi-byte
102
337
  * UTF-8 splits and partial lines honest.
103
338
  * - **Event-free.** A pure functional boundary — no Emitter, no events.
104
339
  * - **Transport seam.** {@link OllamaOptions.fetch} swaps the transport (default
105
340
  * `globalThis.fetch`) and {@link OllamaOptions.headers} is a per-request, possibly
106
341
  * async header injector merged over the base `Content-Type` — so a browser runtime
107
342
  * can route through the developer's own server with an obfuscated bearer token,
108
- * without this library ever handling a real API key. Both omitted ⇒ today's behaviour.
343
+ * without this library ever handling a real API key. Both omitted ⇒ the global `fetch`
344
+ * and only a JSON content type.
109
345
  * Orthogonal to the deadline: the hook is awaited inside `#fetch`'s try, so a hook
110
346
  * rejection clears the armed timer like any other request failure.
111
347
  *
@@ -116,8 +352,8 @@ function isOllamaHTTPError(value) {
116
352
  * ```
117
353
  */
118
354
  var OllamaProvider = class {
119
- id = crypto.randomUUID();
120
355
  name = "ollama";
356
+ #id;
121
357
  #model;
122
358
  #url;
123
359
  #keepAlive;
@@ -128,6 +364,7 @@ var OllamaProvider = class {
128
364
  #headers;
129
365
  #format;
130
366
  constructor(options) {
367
+ this.#id = crypto.randomUUID();
131
368
  this.#model = options.model;
132
369
  this.#url = options.url ?? "http://localhost:11434";
133
370
  this.#keepAlive = options.keepAlive ?? "5m";
@@ -139,38 +376,85 @@ var OllamaProvider = class {
139
376
  this.#format = options.format;
140
377
  }
141
378
  /**
142
- * The provider's context-framing default — the PROVIDER-DEFAULT level of
143
- * {@link import('@orkestrel/agent').AgentContextInterface.build}'s format cascade (it BEATS
144
- * the managers' built-in framing, is BEATEN by a manager-options or per-item override).
145
- * Satisfies the OPTIONAL {@link ProviderInterface.format} contract member: `undefined`
379
+ * Exposes this instance's identity — a fresh `crypto.randomUUID()` minted at
380
+ * construction, satisfying the {@link ProviderInterface.id} contract member. A second
381
+ * provider built from identical options carries a distinct id.
382
+ *
383
+ * @returns The instance's minted identifier
384
+ */
385
+ get id() {
386
+ return this.#id;
387
+ }
388
+ /**
389
+ * Exposes the provider's context-framing default — the provider-default level of
390
+ * {@link import('@orkestrel/agent').AgentContextInterface.build}'s format cascade (it beats
391
+ * the managers' built-in framing, is beaten by a manager-options or per-item override).
392
+ * Satisfies the optional {@link ProviderInterface.format} contract member: `undefined`
146
393
  * when {@link OllamaOptions.format} was omitted (the framing-agnostic default ⇒ core's
147
394
  * built-in framing applies unchanged), else the exact configured framing the Agent
148
395
  * threads into `build()`.
149
396
  *
150
397
  * @remarks
151
- * EXPOSE-ONLY — read by the Agent loop and consumed by core's cascade; it is NEVER sent
152
- * on the `/api/chat` wire (it is absent from `#body` / the request). This is NOT Ollama's
153
- * structured-output `format` wire parameter — that one IS sent in `#body`, but only when
398
+ * Expose-only — read by the Agent loop and consumed by core's cascade; it is never sent
399
+ * on the `/api/chat` wire (it is absent from `#body` / the request). This is not Ollama's
400
+ * structured-output `format` wire parameter — that one is sent in `#body`, but only when
154
401
  * a per-call `ProviderStreamOptions.schema` is supplied; only the word collides.
155
402
  *
156
- * @returns The configured {@link ContextFormatInterface}, or `undefined` when none
403
+ * @returns The configured {@link ContextFormat}, or `undefined` when none
157
404
  */
158
405
  get format() {
159
406
  return this.#format;
160
407
  }
408
+ /**
409
+ * Generates one complete turn and resolves the assembled result — the clean content,
410
+ * any separated reasoning, any tool calls, and any usage the wire reported.
411
+ *
412
+ * @remarks
413
+ * Sends `stream: false` and parses one JSON body. Content routes through a per-call
414
+ * think splitter, so the assembled content stays clean even where the daemon renders a
415
+ * thinking model's reasoning inline; the separated spans and any daemon-side
416
+ * `message.thinking` land on `thinking`. The caller's signal and the armed deadline
417
+ * both cancel the request, and the deadline is cleared once the body is read.
418
+ *
419
+ * @param messages - The conversation turns to send
420
+ * @param signal - The caller's bounding signal, folded with the armed deadline
421
+ * @param tools - The callable tools to advertise for this turn, when the caller passes any
422
+ * @param options - The per-call overrides, `think` and `schema` among them
423
+ * @returns The assembled result of the turn
424
+ * @throws {@link OllamaHTTPError} When the daemon answers a non-OK status.
425
+ */
161
426
  async generate(messages, signal, tools, options) {
162
427
  const { response, timeout } = await this.#fetch(messages, false, signal, tools, options);
163
428
  try {
164
- const record = await this.#parseBody(response);
429
+ const record = await parseBody(response) ?? {};
165
430
  const splitter = createThinkSplitter();
166
- splitter.split(this.#content(record));
431
+ splitter.split(extractContent(record));
167
432
  splitter.flush();
168
- const thinking = this.#thought(splitter, this.#thinking(record));
169
- return this.#result(splitter.content, thinking, this.#tools(record), this.#usage(record));
433
+ const thinking = joinThinking(splitter, extractThinking(record));
434
+ return buildResult(splitter.content, thinking, extractTools(record), extractUsage(record));
170
435
  } finally {
171
436
  timeout.clear();
172
437
  }
173
438
  }
439
+ /**
440
+ * Streams one turn, yielding a channel-tagged delta per non-empty content or reasoning
441
+ * span and returning the assembled result when the stream completes.
442
+ *
443
+ * @remarks
444
+ * Sends `stream: true` and consumes NDJSON — one JSON object per newline-terminated
445
+ * line — pairing a streaming `TextDecoder` with the `NDJSONParser` so a record split
446
+ * across byte reads is reassembled. The returned result's content is the splitter's
447
+ * clean accumulation, beside any tool calls collected across lines and the usage the
448
+ * `done` line carries. A cancel mid-flight throws a `ProviderAbortError` carrying the
449
+ * partial assembled so far.
450
+ *
451
+ * @param messages - The conversation turns to send
452
+ * @param signal - The caller's bounding signal, folded with the armed deadline
453
+ * @param tools - The callable tools to advertise for this turn, when the caller passes any
454
+ * @param options - The per-call overrides, `think` and `schema` among them
455
+ * @returns The assembled result of the turn, after the last delta
456
+ * @throws {@link OllamaHTTPError} When the daemon answers a non-OK status or a `null` body.
457
+ */
174
458
  async *stream(messages, signal, tools, options) {
175
459
  const { response, timeout, combined } = await this.#fetch(messages, true, signal, tools, options);
176
460
  const body = response.body;
@@ -182,54 +466,63 @@ var OllamaProvider = class {
182
466
  const decoder = new TextDecoder();
183
467
  const parser = createNDJSONParser();
184
468
  const splitter = createThinkSplitter();
185
- const state = {
186
- splitter,
187
- wired: "",
188
- calls: [],
189
- usage: void 0
190
- };
469
+ let wired = "";
470
+ const calls = [];
471
+ let usage;
191
472
  try {
192
473
  for (;;) {
193
474
  const { value, done } = await reader.read();
194
475
  if (done) break;
195
- for (const record of parser.parse(decoder.decode(value, { stream: true }))) yield* this.#deltas(record, state);
476
+ for (const record of parser.parse(decoder.decode(value, { stream: true }))) {
477
+ const increment = yield* this.#deltas(record, splitter, usage);
478
+ wired += increment.thinking;
479
+ calls.push(...increment.calls);
480
+ usage = increment.usage;
481
+ }
196
482
  }
197
483
  const decoderTail = decoder.decode();
198
- for (const record of parser.parse(decoderTail.length > 0 ? `${decoderTail}\n` : "\n")) yield* this.#deltas(record, state);
484
+ for (const record of parser.parse(decoderTail.length > 0 ? `${decoderTail}\n` : "\n")) {
485
+ const increment = yield* this.#deltas(record, splitter, usage);
486
+ wired += increment.thinking;
487
+ calls.push(...increment.calls);
488
+ usage = increment.usage;
489
+ }
199
490
  const tail = splitter.flush();
200
491
  if (tail.length > 0) yield {
201
- type: "content",
492
+ channel: "content",
202
493
  text: tail
203
494
  };
204
495
  } catch (error) {
205
496
  if (combined.aborted) {
206
497
  splitter.flush();
207
- throw new ProviderAbortError(this.#result(splitter.content, this.#thought(splitter, state.wired), state.calls, state.usage));
498
+ throw new ProviderAbortError(buildResult(splitter.content, joinThinking(splitter, wired), calls, usage));
208
499
  }
209
500
  throw error;
210
501
  } finally {
211
502
  try {
212
503
  await reader.cancel();
213
504
  } catch {}
214
- parser.reset();
505
+ parser.clear();
215
506
  timeout.clear();
216
507
  }
217
- return this.#result(splitter.content, this.#thought(splitter, state.wired), state.calls, state.usage);
508
+ return buildResult(splitter.content, joinThinking(splitter, wired), calls, usage);
218
509
  }
219
- *#deltas(record, state) {
220
- const delta = state.splitter.split(this.#content(record));
510
+ *#deltas(record, splitter, usage) {
511
+ const delta = splitter.split(extractContent(record));
221
512
  if (delta.length > 0) yield {
222
- type: "content",
513
+ channel: "content",
223
514
  text: delta
224
515
  };
225
- const thinking = this.#thinking(record);
516
+ const thinking = extractThinking(record);
226
517
  if (thinking.length > 0) yield {
227
- type: "thinking",
518
+ channel: "thinking",
228
519
  text: thinking
229
520
  };
230
- state.wired += thinking;
231
- state.calls.push(...this.#tools(record));
232
- if (Reflect.get(record, "done") === true) state.usage = this.#usage(record);
521
+ return {
522
+ thinking,
523
+ calls: extractTools(record),
524
+ usage: Reflect.get(record, "done") === true ? extractUsage(record) : usage
525
+ };
233
526
  }
234
527
  async #fetch(messages, stream, signal, tools, options) {
235
528
  const timeout = new Timeout({ ms: this.#timeout });
@@ -262,16 +555,6 @@ var OllamaProvider = class {
262
555
  throw error;
263
556
  }
264
557
  }
265
- async #parseBody(response) {
266
- const text = await response.text();
267
- if (text.length === 0) return {};
268
- try {
269
- const data = JSON.parse(text);
270
- return isRecord(data) ? data : {};
271
- } catch {
272
- return {};
273
- }
274
- }
275
558
  async #requestHeaders() {
276
559
  const headers = { "Content-Type": "application/json" };
277
560
  if (this.#headers !== void 0) for (const [key, value] of Object.entries(await this.#headers())) headers[key] = value;
@@ -280,7 +563,7 @@ var OllamaProvider = class {
280
563
  #body(messages, stream, tools, options) {
281
564
  return {
282
565
  model: this.#model,
283
- messages: this.#plain(messages),
566
+ messages: mapMessages(messages),
284
567
  stream,
285
568
  keep_alive: this.#keepAlive,
286
569
  think: options?.think ?? this.#think,
@@ -296,127 +579,58 @@ var OllamaProvider = class {
296
579
  })) } : {}
297
580
  };
298
581
  }
299
- #plain(messages) {
300
- return messages.map((message) => ({
301
- role: message.role,
302
- content: message.content,
303
- ...message.calls !== void 0 && message.calls.length > 0 ? { tool_calls: message.calls.map((call) => ({ function: {
304
- name: call.name,
305
- arguments: call.arguments
306
- } })) } : {},
307
- ...message.images !== void 0 && message.images.length > 0 ? { images: [...message.images] } : {}
308
- }));
309
- }
310
- #result(content, thinking, tools, usage) {
311
- const result = { content };
312
- if (thinking.length > 0) result.thinking = thinking;
313
- if (tools.length > 0) result.tools = tools;
314
- if (usage !== void 0) result.usage = usage;
315
- return result;
316
- }
317
- #content(record) {
318
- const message = Reflect.get(record, "message");
319
- if (!isRecord(message)) return "";
320
- const content = Reflect.get(message, "content");
321
- return isString(content) ? content : "";
322
- }
323
- #thinking(record) {
324
- const message = Reflect.get(record, "message");
325
- if (!isRecord(message)) return "";
326
- const thinking = Reflect.get(message, "thinking");
327
- return isString(thinking) ? thinking : "";
328
- }
329
- #thought(splitter, wired) {
330
- if (splitter.thinking.length === 0) return wired;
331
- if (wired.length === 0) return splitter.thinking;
332
- return `${splitter.thinking}\n\n${wired}`;
333
- }
334
- #usage(record) {
335
- const prompt = Reflect.get(record, "prompt_eval_count");
336
- const completion = Reflect.get(record, "eval_count");
337
- if (!isNumber(prompt) || !isNumber(completion)) return void 0;
338
- return {
339
- prompt,
340
- completion,
341
- total: prompt + completion
342
- };
343
- }
344
- #tools(record) {
345
- const message = Reflect.get(record, "message");
346
- if (!isRecord(message)) return [];
347
- const calls = Reflect.get(message, "tool_calls");
348
- if (!Array.isArray(calls)) return [];
349
- const out = [];
350
- for (const entry of calls) {
351
- if (!isRecord(entry)) continue;
352
- const callable = Reflect.get(entry, "function");
353
- if (!isRecord(callable)) continue;
354
- const name = Reflect.get(callable, "name");
355
- if (!isString(name)) continue;
356
- const id = Reflect.get(entry, "id");
357
- out.push({
358
- id: isString(id) ? id : crypto.randomUUID(),
359
- name,
360
- arguments: this.#arguments(Reflect.get(callable, "arguments"))
361
- });
362
- }
363
- return out;
364
- }
365
- #arguments(value) {
366
- if (isRecord(value)) return value;
367
- if (isString(value)) try {
368
- const parsed = JSON.parse(value);
369
- if (isRecord(parsed)) return parsed;
370
- } catch {
371
- return {};
372
- }
373
- return {};
374
- }
375
582
  };
376
583
  //#endregion
377
584
  //#region src/server/factories.ts
378
585
  /**
379
- * Create a local Ollama inference provider — a {@link ProviderInterface} over the
586
+ * Creates a local Ollama inference provider — a {@link ProviderInterface} over the
380
587
  * daemon's `POST /api/chat`, supporting non-streaming `generate` and streaming
381
588
  * `stream`.
382
589
  *
383
590
  * @remarks
384
591
  * Only `model` is required; `url` defaults to the local daemon, `keepAlive` to `'5m'`,
385
592
  * `timeout` to `120_000`ms, and `options` is forwarded verbatim as sampling
386
- * parameters (`temperature` / `seed` / `num_predict` / …). Both calls take an
593
+ * parameters (`temperature`, `seed`, and `num_predict`). Each call takes an
387
594
  * `AbortSignal` to bound the request; a `stream` cancelled mid-flight throws a
388
595
  * `ProviderAbortError` carrying the partial result.
389
596
  *
390
597
  * The optional `fetch` + `headers` form a transport seam (see {@link OllamaOptions}):
391
598
  * point `url` at your own server, inject a custom `fetch`, and have `headers` attach a
392
599
  * generated/obfuscated bearer token your server validates — so a browser runtime
393
- * reaches the LLM through your middleware WITHOUT this library ever handling the real API
394
- * key. Both omitted ⇒ today's behaviour (the global `fetch`, only a JSON content type).
600
+ * reaches the LLM through your middleware without this library ever handling the real API
601
+ * key. Both omitted ⇒ the global `fetch` and only a JSON content type.
395
602
  *
396
- * The optional `format` is the provider's context-framing default — the PROVIDER-DEFAULT
397
- * level of `AgentContext`'s format cascade (see [agents.md]; beaten by a manager-options
398
- * or per-item override, beating the managers' built-in framing), declaring how this
399
- * provider's models prefer context sections framed (e.g. XML group wrappers vs. Markdown
400
- * headers). It is EXPOSED on the provider for the Agent's `build()` and is NOT Ollama's
401
- * `/api/chat` `format` wire parameter (structured output) — the two are unrelated despite
402
- * the shared word. Omitted ⇒ the provider is framing-agnostic (core's built-in defaults).
603
+ * The optional `format` is the provider's context-framing default — the provider-default
604
+ * level of `AgentContext`'s format cascade (beaten by a manager-options or per-item
605
+ * override, beating the managers' built-in framing), declaring how this
606
+ * provider's models prefer context sections framed (for example XML group wrappers vs. Markdown
607
+ * headers). It is exposed on the provider for the Agent's `build()` and is not Ollama's
608
+ * `/api/chat` `format` wire parameter (structured output) — the framing default and that
609
+ * wire parameter are unrelated despite the shared word. Omitted ⇒ the provider is
610
+ * framing-agnostic (core's built-in defaults).
403
611
  *
404
612
  * @param options - `model` (required), and optional `url` / `keepAlive` / `timeout` /
405
613
  * `options` / `fetch` / `headers` / `format` (see {@link OllamaOptions})
406
614
  * @returns A working {@link ProviderInterface} backed by Ollama
407
615
  *
408
- * @example
616
+ * @example createOllama + generate
409
617
  * ```ts
410
618
  * import { createAbort } from '@orkestrel/abort'
411
- * import { createOllama } from '@src/server'
619
+ * import { createOllama } from '@orkestrel/ollama'
412
620
  *
413
- * const provider = createOllama({ model: 'qwen3.5:2b-q4_K_M' })
621
+ * const provider = createOllama({ model: 'qwen3.5:2b-q4_K_M', options: { temperature: 0 } })
414
622
  * const abort = createAbort()
623
+ * const messages = [
624
+ * { id: '1', role: 'user', content: 'Summarize the release notes for version 2.0.' },
625
+ * ] as const
626
+ *
415
627
  * const result = await provider.generate(messages, abort.signal)
628
+ * console.log(result.content)
629
+ * if (result.usage) charge(result.usage) // fold into a token budget
416
630
  * ```
417
631
  *
418
632
  * @example
419
- * Route through your own server with an obfuscated token (deployment scenario S2):
633
+ * Route through your own server with an obfuscated token:
420
634
  * ```ts
421
635
  * const provider = createOllama({
422
636
  * model: 'qwen3.5:2b-q4_K_M',
@@ -428,7 +642,7 @@ var OllamaProvider = class {
428
642
  *
429
643
  * @example
430
644
  * Declare a context-framing default — wrap the instructions section in an XML group (the
431
- * provider-default level of `AgentContext`'s cascade; NOT the wire `format`):
645
+ * provider-default level of `AgentContext`'s cascade; not the wire `format`):
432
646
  * ```ts
433
647
  * const provider = createOllama({
434
648
  * model: 'qwen3.5:2b-q4_K_M',
@@ -446,6 +660,6 @@ function createOllama(options) {
446
660
  return new OllamaProvider(options);
447
661
  }
448
662
  //#endregion
449
- export { DEFAULT_KEEP_ALIVE, DEFAULT_OLLAMA_URL, DEFAULT_PROVIDER_TIMEOUT, MAX_ERROR_BODY_LENGTH, OllamaHTTPError, OllamaProvider, createOllama, isOllamaHTTPError };
663
+ export { DEFAULT_KEEP_ALIVE, DEFAULT_OLLAMA_URL, DEFAULT_PROVIDER_TIMEOUT, MAX_ERROR_BODY_LENGTH, OllamaHTTPError, OllamaProvider, buildResult, createOllama, extractArguments, extractContent, extractThinking, extractTools, extractUsage, isOllamaHTTPError, joinThinking, mapMessages, parseBody };
450
664
 
451
665
  //# sourceMappingURL=index.js.map