@orkestrel/ollama 0.0.13 → 0.0.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,44 +1,53 @@
1
1
  Object.defineProperty(exports, Symbol.toStringTag, { value: "Module" });
2
- let _orkestrel_agent = require("@orkestrel/agent");
3
2
  let _orkestrel_contract = require("@orkestrel/contract");
3
+ let _orkestrel_agent = require("@orkestrel/agent");
4
4
  let _orkestrel_ndjson = require("@orkestrel/ndjson");
5
5
  let _orkestrel_timeout = require("@orkestrel/timeout");
6
6
  //#region src/server/constants.ts
7
- /** The local Ollama daemon base URL assumed when `OllamaOptions.url` is omitted. */
7
+ /**
8
+ * Names the local Ollama daemon base URL, `'http://localhost:11434'`, assumed when
9
+ * `OllamaOptions.url` is omitted.
10
+ */
8
11
  var DEFAULT_OLLAMA_URL = "http://localhost:11434";
9
12
  /**
10
- * How long the model stays resident after a call when `OllamaOptions.keepAlive` is
11
- * omitted — Ollama's own `keep_alive` default, expressed as a duration string.
13
+ * Names how long the model stays resident after a call — `'5m'` when
14
+ * `OllamaOptions.keepAlive` is omitted, Ollama's own `keep_alive` default, expressed as a
15
+ * duration string.
16
+ *
17
+ * @remarks
18
+ * The name mirrors the Ollama `/api/chat` `keep_alive` field this value is sent as, so
19
+ * the constant, the `OllamaOptions.keepAlive` key, and the wire member read as one term.
12
20
  */
13
21
  var DEFAULT_KEEP_ALIVE = "5m";
14
22
  /**
15
- * The per-call deadline in milliseconds when `OllamaOptions.timeout` is omitted —
16
- * generous enough that a cold model load does not trip it.
23
+ * Names the per-call deadline in milliseconds, `120_000`, when `OllamaOptions.timeout` is
24
+ * omitted — generous enough that a cold model load does not trip it.
17
25
  */
18
26
  var DEFAULT_PROVIDER_TIMEOUT = 12e4;
19
27
  /**
20
- * The cap, in characters, on how much of a non-OK response body is
28
+ * Names the character cap, `2048`, on how much of a non-OK response body is
21
29
  * incorporated into a thrown {@link OllamaHTTPError}'s message.
22
30
  *
23
31
  * @remarks
24
32
  * Bounds the excerpt so a defensive proxy or a misbehaving daemon handing
25
33
  * back an unbounded response body cannot inflate the thrown error's message
26
- * without limit (§14). `2048` characters is generous enough to carry a
27
- * useful diagnostic snippet while staying well short of any practical size
28
- * concern.
34
+ * without limit, while the cap stays generous enough to carry a useful
35
+ * diagnostic snippet.
29
36
  */
30
37
  var MAX_ERROR_BODY_LENGTH = 2048;
31
38
  //#endregion
32
39
  //#region src/server/errors.ts
33
40
  /**
34
- * An error thrown when the Ollama `/api/chat` HTTP transport fails.
41
+ * Represents an error thrown when the Ollama `/api/chat` HTTP transport fails.
35
42
  *
36
43
  * @remarks
37
- * Carries the response `status` (0 when no HTTP response was received at all,
38
- * e.g. a `null` body). Thrown by {@link OllamaProvider} at its two HTTP
39
- * failure sites — the non-OK status branch and the null-body branch — so a
40
- * caller can branch on `error.status` instead of parsing the message. Narrow
41
- * a caught value with {@link isOllamaHTTPError}.
44
+ * Carries the machine-readable `code` `'HTTP'` and the response `status` (0 when no
45
+ * HTTP response was received at all, for example a `null` body). Thrown by
46
+ * {@link OllamaProvider} at its HTTP failure sites — the non-OK status branch and the
47
+ * null-body branch — so a caller can branch on `error.code` and read `error.status`
48
+ * for the HTTP number instead of parsing the message. The message carries a body excerpt
49
+ * bounded to {@link MAX_ERROR_BODY_LENGTH} — `2048` characters. Narrow a caught value with
50
+ * {@link isOllamaHTTPError}.
42
51
  *
43
52
  * @example
44
53
  * ```ts
@@ -52,6 +61,11 @@ var MAX_ERROR_BODY_LENGTH = 2048;
52
61
  * ```
53
62
  */
54
63
  var OllamaHTTPError = class extends Error {
64
+ /**
65
+ * Names the machine-readable condition this error reports — `'HTTP'`: an `/api/chat`
66
+ * transport, status, or body failure.
67
+ */
68
+ code = "HTTP";
55
69
  status;
56
70
  constructor(message, status, options) {
57
71
  super(message, options);
@@ -60,53 +74,275 @@ var OllamaHTTPError = class extends Error {
60
74
  }
61
75
  };
62
76
  /**
63
- * Whether a value is an {@link OllamaHTTPError}.
77
+ * Checks whether a value is an {@link OllamaHTTPError}.
78
+ *
79
+ * @remarks
80
+ * The check is an `instanceof` test, so it narrows a caught `unknown` to the error class
81
+ * without parsing the thrown message.
64
82
  *
65
83
  * @param value - The value to test
66
- * @returns `true` when `value` is an `OllamaHTTPError`
84
+ * @returns True if `value` is an `OllamaHTTPError`; false otherwise
67
85
  */
68
86
  function isOllamaHTTPError(value) {
69
87
  return value instanceof OllamaHTTPError;
70
88
  }
71
89
  //#endregion
90
+ //#region src/server/helpers.ts
91
+ /**
92
+ * Maps conversation turns onto the `/api/chat` wire's minimal message shape.
93
+ *
94
+ * @remarks
95
+ * `tool_calls` is emitted only on a turn that replays them and `images` only on a
96
+ * multimodal turn, so an empty optional never reaches the wire.
97
+ *
98
+ * @param messages - The conversation turns to send
99
+ * @returns The wire `messages` array, one entry per turn, in order
100
+ *
101
+ * @example
102
+ * ```ts
103
+ * mapMessages([{ id: '1', role: 'user', content: 'Say hello.' }])
104
+ * // [{ role: 'user', content: 'Say hello.' }]
105
+ * ```
106
+ */
107
+ function mapMessages(messages) {
108
+ return messages.map((message) => ({
109
+ role: message.role,
110
+ content: message.content,
111
+ ...message.calls !== void 0 && message.calls.length > 0 ? { tool_calls: message.calls.map((call) => ({ function: {
112
+ name: call.name,
113
+ arguments: call.arguments
114
+ } })) } : {},
115
+ ...message.images !== void 0 && message.images.length > 0 ? { images: [...message.images] } : {}
116
+ }));
117
+ }
118
+ /**
119
+ * Builds a `ProviderResult` from a turn's content, reasoning, tool calls, and usage.
120
+ *
121
+ * @remarks
122
+ * Only the present optionals are set: no empty `thinking`, no empty `tools`, and no
123
+ * `usage` unless the wire reported one.
124
+ *
125
+ * @param content - The clean assistant content the splitter accumulated
126
+ * @param thinking - The joined reasoning, empty when the turn produced none
127
+ * @param tools - The tool calls collected across the turn
128
+ * @param usage - The token usage, or `undefined` when the wire reported none
129
+ * @returns The result carrying only its populated fields
130
+ *
131
+ * @example
132
+ * ```ts
133
+ * buildResult('ok', '', [], undefined) // { content: 'ok' }
134
+ * ```
135
+ */
136
+ function buildResult(content, thinking, tools, usage) {
137
+ const result = { content };
138
+ if (thinking.length > 0) result.thinking = thinking;
139
+ if (tools.length > 0) result.tools = tools;
140
+ if (usage !== void 0) result.usage = usage;
141
+ return result;
142
+ }
143
+ /**
144
+ * Extracts the assistant text of one wire record.
145
+ *
146
+ * @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
147
+ * @returns The record's `message.content` when it is a string, else `''`
148
+ *
149
+ * @example
150
+ * ```ts
151
+ * extractContent({ message: { content: 'ok' } }) // 'ok'
152
+ * ```
153
+ */
154
+ function extractContent(record) {
155
+ const message = Reflect.get(record, "message");
156
+ if (!(0, _orkestrel_contract.isRecord)(message)) return "";
157
+ const content = Reflect.get(message, "content");
158
+ return (0, _orkestrel_contract.isString)(content) ? content : "";
159
+ }
160
+ /**
161
+ * Extracts the daemon-side reasoning of one wire record.
162
+ *
163
+ * @remarks
164
+ * `message.thinking` is the `think: true` wire shape. It is read whatever the configured
165
+ * flag says, because a daemon may separate reasoning on its own.
166
+ *
167
+ * @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
168
+ * @returns The record's `message.thinking` when it is a string, else `''`
169
+ *
170
+ * @example
171
+ * ```ts
172
+ * extractThinking({ message: { thinking: 'weighing it' } }) // 'weighing it'
173
+ * ```
174
+ */
175
+ function extractThinking(record) {
176
+ const message = Reflect.get(record, "message");
177
+ if (!(0, _orkestrel_contract.isRecord)(message)) return "";
178
+ const thinking = Reflect.get(message, "thinking");
179
+ return (0, _orkestrel_contract.isString)(thinking) ? thinking : "";
180
+ }
181
+ /**
182
+ * Joins a call's reasoning carriers — the splitter's separated in-content spans and the
183
+ * accumulated wire-side `message.thinking` — into the result's `thinking`.
184
+ *
185
+ * @param splitter - The per-call splitter holding the separated in-content spans
186
+ * @param wired - The accumulated wire-side `message.thinking` text
187
+ * @returns The carriers separated by a blank line, or whichever one is non-empty
188
+ *
189
+ * @example
190
+ * ```ts
191
+ * joinThinking(createThinkSplitter(), 'from the wire') // 'from the wire'
192
+ * ```
193
+ */
194
+ function joinThinking(splitter, wired) {
195
+ if (splitter.thinking.length === 0) return wired;
196
+ if (wired.length === 0) return splitter.thinking;
197
+ return `${splitter.thinking}\n\n${wired}`;
198
+ }
199
+ /**
200
+ * Extracts the token usage of one wire record.
201
+ *
202
+ * @remarks
203
+ * Both counts must be numbers, which is true of the non-stream body and the stream's
204
+ * `done: true` line. A delta line carries neither, so it yields `undefined`.
205
+ *
206
+ * @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
207
+ * @returns The `TokenUsage` shape, or `undefined` when either count is absent
208
+ *
209
+ * @example
210
+ * ```ts
211
+ * extractUsage({ prompt_eval_count: 3, eval_count: 4 })
212
+ * // { prompt: 3, completion: 4, total: 7 }
213
+ * ```
214
+ */
215
+ function extractUsage(record) {
216
+ const prompt = Reflect.get(record, "prompt_eval_count");
217
+ const completion = Reflect.get(record, "eval_count");
218
+ if (!(0, _orkestrel_contract.isNumber)(prompt) || !(0, _orkestrel_contract.isNumber)(completion)) return void 0;
219
+ return {
220
+ prompt,
221
+ completion,
222
+ total: prompt + completion
223
+ };
224
+ }
225
+ /**
226
+ * Extracts the tool calls of one wire record's `message.tool_calls`.
227
+ *
228
+ * @remarks
229
+ * Each entry narrows to `{ id, name, arguments }`: the entry and its `function` must be
230
+ * records and `name` a string, else the entry is dropped. An id is minted when the wire
231
+ * omits one.
232
+ *
233
+ * @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
234
+ * @returns The narrowed tool calls, empty when the record carries none
235
+ *
236
+ * @example
237
+ * ```ts
238
+ * extractTools({ message: { tool_calls: [{ function: { name: 'weather' } }] } })
239
+ * // [{ id: '…', name: 'weather', arguments: {} }]
240
+ * ```
241
+ */
242
+ function extractTools(record) {
243
+ const message = Reflect.get(record, "message");
244
+ if (!(0, _orkestrel_contract.isRecord)(message)) return [];
245
+ const calls = Reflect.get(message, "tool_calls");
246
+ if (!Array.isArray(calls)) return [];
247
+ const out = [];
248
+ for (const entry of calls) {
249
+ if (!(0, _orkestrel_contract.isRecord)(entry)) continue;
250
+ const callable = Reflect.get(entry, "function");
251
+ if (!(0, _orkestrel_contract.isRecord)(callable)) continue;
252
+ const name = Reflect.get(callable, "name");
253
+ if (!(0, _orkestrel_contract.isString)(name)) continue;
254
+ const id = Reflect.get(entry, "id");
255
+ out.push({
256
+ id: (0, _orkestrel_contract.isString)(id) ? id : crypto.randomUUID(),
257
+ name,
258
+ arguments: extractArguments(Reflect.get(callable, "arguments"))
259
+ });
260
+ }
261
+ return out;
262
+ }
263
+ /**
264
+ * Extracts a wire `arguments` value as a record.
265
+ *
266
+ * @remarks
267
+ * Total: an object passes through, a JSON string is parsed when it yields a record, and
268
+ * a malformed string yields `{}` rather than throwing.
269
+ *
270
+ * @param value - The wire's `function.arguments` value, of unknown shape
271
+ * @returns The argument record, or `{}` when the value carries none
272
+ *
273
+ * @example
274
+ * ```ts
275
+ * extractArguments('{"city":"Oslo"}') // { city: 'Oslo' }
276
+ * ```
277
+ */
278
+ function extractArguments(value) {
279
+ if ((0, _orkestrel_contract.isRecord)(value)) return value;
280
+ if ((0, _orkestrel_contract.isString)(value)) return (0, _orkestrel_contract.parseJSONAs)(value, _orkestrel_contract.isRecord) ?? {};
281
+ return {};
282
+ }
283
+ //#endregion
284
+ //#region src/server/parsers.ts
285
+ /**
286
+ * Parses a non-stream `/api/chat` response body into a wire record.
287
+ *
288
+ * @remarks
289
+ * Total by construction: an empty body, a body that is not JSON, and a body whose JSON is
290
+ * not an object all yield `undefined`, so a malformed daemon response never escapes as a
291
+ * `SyntaxError`. The call site supplies the empty-record default that reads as empty
292
+ * content and no usage.
293
+ *
294
+ * @param response - The 200-OK `/api/chat` response whose body is read as text
295
+ * @returns The parsed record, or `undefined` when the body is empty or malformed
296
+ *
297
+ * @example
298
+ * ```ts
299
+ * await parseBody(new Response('{"message":{"content":"ok"}}'))
300
+ * // { message: { content: 'ok' } }
301
+ * ```
302
+ */
303
+ async function parseBody(response) {
304
+ return (0, _orkestrel_contract.parseJSONAs)(await response.text(), _orkestrel_contract.isRecord);
305
+ }
306
+ //#endregion
72
307
  //#region src/server/OllamaProvider.ts
73
308
  /**
74
- * The local Ollama inference boundary — a {@link ProviderInterface} over Ollama's
309
+ * Implements the local Ollama inference boundary — a {@link ProviderInterface} over Ollama's
75
310
  * `POST /api/chat`, both non-streaming (`generate`) and streaming NDJSON (`stream`).
76
311
  *
77
312
  * @remarks
78
313
  * - **Wire protocol.** Posts `{ model, messages, stream, keep_alive, think }` plus
79
314
  * passthrough sampling `options` and mapped function `tools`. The `think` flag is
80
- * CONFIGURABLE via {@link OllamaOptions.think} (default `false`). Non-stream parses
315
+ * configurable through {@link OllamaOptions.think} (default `false`). Non-stream parses
81
316
  * one JSON body; stream consumes NDJSON (one JSON object per `\n`-terminated line) —
82
317
  * deltas carry `message.content`, the final `done: true` line carries the token usage.
83
- * - **Think separation (H4).** The wire `think` flag is configurable
318
+ * - **Think separation.** The wire `think` flag is configurable
84
319
  * ({@link OllamaOptions.think}, default `false`). With `think: true` a thinking model's
85
- * daemon separates reasoning NATIVELY — returning it on the distinct `message.thinking`
86
- * channel (read here via `#thinking`) instead of inline in `message.content`. EITHER
320
+ * daemon separates reasoning natively — returning it on the distinct `message.thinking`
321
+ * channel (read here through `extractThinking`) instead of inline in `message.content`. Either
87
322
  * way the per-call {@link ThinkSplitterInterface} is the defensive guarantee: a daemon
88
323
  * may ignore `think: false` for a thinking model and inline `<think>` tags, so every
89
- * content delta routes through the splitter, only CLEAN content is yielded / assembled,
324
+ * content delta routes through the splitter, only clean content is yielded / assembled,
90
325
  * and the separated reasoning (plus any daemon-side `message.thinking` deltas) lands on
91
326
  * `ProviderResult.thinking`, never in the conversation.
92
- * - **Boundary narrowing (§14).** Every wire value arrives as `unknown` and is
327
+ * - **Boundary narrowing.** Every wire value arrives as `unknown` and is
93
328
  * narrowed through guards (`isRecord` / `isString` / `isNumber`) — never `as`. A
94
329
  * missing / malformed field degrades to a sensible default (empty content, no
95
330
  * usage, `{}` arguments), never a throw.
96
331
  * - **Bounded.** Each call arms a {@link Timeout} for `OllamaOptions.timeout` and
97
332
  * passes `AbortSignal.any([timeout.signal, signal])` to `fetch`, so the caller's
98
- * signal AND the deadline both cancel the request. The timeout is always cleared —
333
+ * signal and the deadline both cancel the request. The timeout is always cleared —
99
334
  * in `#fetch` if the request fails/aborts, otherwise in the consuming call's `finally`.
100
335
  * - **Abort recovers partial.** A `stream` cancelled mid-flight throws a
101
336
  * `ProviderAbortError` carrying the partial result assembled so far; pairing the
102
- * `TextDecoder({ stream: true })` with the {@link NDJSONParser} parser keeps multi-byte
337
+ * `TextDecoder({ stream: true })` with the `createNDJSONParser` parser keeps multi-byte
103
338
  * UTF-8 splits and partial lines honest.
104
339
  * - **Event-free.** A pure functional boundary — no Emitter, no events.
105
340
  * - **Transport seam.** {@link OllamaOptions.fetch} swaps the transport (default
106
341
  * `globalThis.fetch`) and {@link OllamaOptions.headers} is a per-request, possibly
107
342
  * async header injector merged over the base `Content-Type` — so a browser runtime
108
343
  * can route through the developer's own server with an obfuscated bearer token,
109
- * without this library ever handling a real API key. Both omitted ⇒ today's behaviour.
344
+ * without this library ever handling a real API key. Both omitted ⇒ the global `fetch`
345
+ * and only a JSON content type.
110
346
  * Orthogonal to the deadline: the hook is awaited inside `#fetch`'s try, so a hook
111
347
  * rejection clears the armed timer like any other request failure.
112
348
  *
@@ -117,8 +353,8 @@ function isOllamaHTTPError(value) {
117
353
  * ```
118
354
  */
119
355
  var OllamaProvider = class {
120
- id = crypto.randomUUID();
121
356
  name = "ollama";
357
+ #id;
122
358
  #model;
123
359
  #url;
124
360
  #keepAlive;
@@ -129,6 +365,7 @@ var OllamaProvider = class {
129
365
  #headers;
130
366
  #format;
131
367
  constructor(options) {
368
+ this.#id = crypto.randomUUID();
132
369
  this.#model = options.model;
133
370
  this.#url = options.url ?? "http://localhost:11434";
134
371
  this.#keepAlive = options.keepAlive ?? "5m";
@@ -140,38 +377,85 @@ var OllamaProvider = class {
140
377
  this.#format = options.format;
141
378
  }
142
379
  /**
143
- * The provider's context-framing default — the PROVIDER-DEFAULT level of
144
- * {@link import('@orkestrel/agent').AgentContextInterface.build}'s format cascade (it BEATS
145
- * the managers' built-in framing, is BEATEN by a manager-options or per-item override).
146
- * Satisfies the OPTIONAL {@link ProviderInterface.format} contract member: `undefined`
380
+ * Exposes this instance's identity — a fresh `crypto.randomUUID()` minted at
381
+ * construction, satisfying the {@link ProviderInterface.id} contract member. A second
382
+ * provider built from identical options carries a distinct id.
383
+ *
384
+ * @returns The instance's minted identifier
385
+ */
386
+ get id() {
387
+ return this.#id;
388
+ }
389
+ /**
390
+ * Exposes the provider's context-framing default — the provider-default level of
391
+ * {@link import('@orkestrel/agent').AgentContextInterface.build}'s format cascade (it beats
392
+ * the managers' built-in framing, is beaten by a manager-options or per-item override).
393
+ * Satisfies the optional {@link ProviderInterface.format} contract member: `undefined`
147
394
  * when {@link OllamaOptions.format} was omitted (the framing-agnostic default ⇒ core's
148
395
  * built-in framing applies unchanged), else the exact configured framing the Agent
149
396
  * threads into `build()`.
150
397
  *
151
398
  * @remarks
152
- * EXPOSE-ONLY — read by the Agent loop and consumed by core's cascade; it is NEVER sent
153
- * on the `/api/chat` wire (it is absent from `#body` / the request). This is NOT Ollama's
154
- * structured-output `format` wire parameter — that one IS sent in `#body`, but only when
399
+ * Expose-only — read by the Agent loop and consumed by core's cascade; it is never sent
400
+ * on the `/api/chat` wire (it is absent from `#body` / the request). This is not Ollama's
401
+ * structured-output `format` wire parameter — that one is sent in `#body`, but only when
155
402
  * a per-call `ProviderStreamOptions.schema` is supplied; only the word collides.
156
403
  *
157
- * @returns The configured {@link ContextFormatInterface}, or `undefined` when none
404
+ * @returns The configured {@link ContextFormat}, or `undefined` when none
158
405
  */
159
406
  get format() {
160
407
  return this.#format;
161
408
  }
409
+ /**
410
+ * Generates one complete turn and resolves the assembled result — the clean content,
411
+ * any separated reasoning, any tool calls, and any usage the wire reported.
412
+ *
413
+ * @remarks
414
+ * Sends `stream: false` and parses one JSON body. Content routes through a per-call
415
+ * think splitter, so the assembled content stays clean even where the daemon renders a
416
+ * thinking model's reasoning inline; the separated spans and any daemon-side
417
+ * `message.thinking` land on `thinking`. The caller's signal and the armed deadline
418
+ * both cancel the request, and the deadline is cleared once the body is read.
419
+ *
420
+ * @param messages - The conversation turns to send
421
+ * @param signal - The caller's bounding signal, folded with the armed deadline
422
+ * @param tools - The callable tools to advertise for this turn, when the caller passes any
423
+ * @param options - The per-call overrides, `think` and `schema` among them
424
+ * @returns The assembled result of the turn
425
+ * @throws {@link OllamaHTTPError} When the daemon answers a non-OK status.
426
+ */
162
427
  async generate(messages, signal, tools, options) {
163
428
  const { response, timeout } = await this.#fetch(messages, false, signal, tools, options);
164
429
  try {
165
- const record = await this.#parseBody(response);
430
+ const record = await parseBody(response) ?? {};
166
431
  const splitter = (0, _orkestrel_agent.createThinkSplitter)();
167
- splitter.split(this.#content(record));
432
+ splitter.split(extractContent(record));
168
433
  splitter.flush();
169
- const thinking = this.#thought(splitter, this.#thinking(record));
170
- return this.#result(splitter.content, thinking, this.#tools(record), this.#usage(record));
434
+ const thinking = joinThinking(splitter, extractThinking(record));
435
+ return buildResult(splitter.content, thinking, extractTools(record), extractUsage(record));
171
436
  } finally {
172
437
  timeout.clear();
173
438
  }
174
439
  }
440
+ /**
441
+ * Streams one turn, yielding a channel-tagged delta per non-empty content or reasoning
442
+ * span and returning the assembled result when the stream completes.
443
+ *
444
+ * @remarks
445
+ * Sends `stream: true` and consumes NDJSON — one JSON object per newline-terminated
446
+ * line — pairing a streaming `TextDecoder` with the `NDJSONParser` so a record split
447
+ * across byte reads is reassembled. The returned result's content is the splitter's
448
+ * clean accumulation, beside any tool calls collected across lines and the usage the
449
+ * `done` line carries. A cancel mid-flight throws a `ProviderAbortError` carrying the
450
+ * partial assembled so far.
451
+ *
452
+ * @param messages - The conversation turns to send
453
+ * @param signal - The caller's bounding signal, folded with the armed deadline
454
+ * @param tools - The callable tools to advertise for this turn, when the caller passes any
455
+ * @param options - The per-call overrides, `think` and `schema` among them
456
+ * @returns The assembled result of the turn, after the last delta
457
+ * @throws {@link OllamaHTTPError} When the daemon answers a non-OK status or a `null` body.
458
+ */
175
459
  async *stream(messages, signal, tools, options) {
176
460
  const { response, timeout, combined } = await this.#fetch(messages, true, signal, tools, options);
177
461
  const body = response.body;
@@ -183,54 +467,63 @@ var OllamaProvider = class {
183
467
  const decoder = new TextDecoder();
184
468
  const parser = (0, _orkestrel_ndjson.createNDJSONParser)();
185
469
  const splitter = (0, _orkestrel_agent.createThinkSplitter)();
186
- const state = {
187
- splitter,
188
- wired: "",
189
- calls: [],
190
- usage: void 0
191
- };
470
+ let wired = "";
471
+ const calls = [];
472
+ let usage;
192
473
  try {
193
474
  for (;;) {
194
475
  const { value, done } = await reader.read();
195
476
  if (done) break;
196
- for (const record of parser.parse(decoder.decode(value, { stream: true }))) yield* this.#deltas(record, state);
477
+ for (const record of parser.parse(decoder.decode(value, { stream: true }))) {
478
+ const increment = yield* this.#deltas(record, splitter, usage);
479
+ wired += increment.thinking;
480
+ calls.push(...increment.calls);
481
+ usage = increment.usage;
482
+ }
197
483
  }
198
484
  const decoderTail = decoder.decode();
199
- for (const record of parser.parse(decoderTail.length > 0 ? `${decoderTail}\n` : "\n")) yield* this.#deltas(record, state);
485
+ for (const record of parser.parse(decoderTail.length > 0 ? `${decoderTail}\n` : "\n")) {
486
+ const increment = yield* this.#deltas(record, splitter, usage);
487
+ wired += increment.thinking;
488
+ calls.push(...increment.calls);
489
+ usage = increment.usage;
490
+ }
200
491
  const tail = splitter.flush();
201
492
  if (tail.length > 0) yield {
202
- type: "content",
493
+ channel: "content",
203
494
  text: tail
204
495
  };
205
496
  } catch (error) {
206
497
  if (combined.aborted) {
207
498
  splitter.flush();
208
- throw new _orkestrel_agent.ProviderAbortError(this.#result(splitter.content, this.#thought(splitter, state.wired), state.calls, state.usage));
499
+ throw new _orkestrel_agent.ProviderAbortError(buildResult(splitter.content, joinThinking(splitter, wired), calls, usage));
209
500
  }
210
501
  throw error;
211
502
  } finally {
212
503
  try {
213
504
  await reader.cancel();
214
505
  } catch {}
215
- parser.reset();
506
+ parser.clear();
216
507
  timeout.clear();
217
508
  }
218
- return this.#result(splitter.content, this.#thought(splitter, state.wired), state.calls, state.usage);
509
+ return buildResult(splitter.content, joinThinking(splitter, wired), calls, usage);
219
510
  }
220
- *#deltas(record, state) {
221
- const delta = state.splitter.split(this.#content(record));
511
+ *#deltas(record, splitter, usage) {
512
+ const delta = splitter.split(extractContent(record));
222
513
  if (delta.length > 0) yield {
223
- type: "content",
514
+ channel: "content",
224
515
  text: delta
225
516
  };
226
- const thinking = this.#thinking(record);
517
+ const thinking = extractThinking(record);
227
518
  if (thinking.length > 0) yield {
228
- type: "thinking",
519
+ channel: "thinking",
229
520
  text: thinking
230
521
  };
231
- state.wired += thinking;
232
- state.calls.push(...this.#tools(record));
233
- if (Reflect.get(record, "done") === true) state.usage = this.#usage(record);
522
+ return {
523
+ thinking,
524
+ calls: extractTools(record),
525
+ usage: Reflect.get(record, "done") === true ? extractUsage(record) : usage
526
+ };
234
527
  }
235
528
  async #fetch(messages, stream, signal, tools, options) {
236
529
  const timeout = new _orkestrel_timeout.Timeout({ ms: this.#timeout });
@@ -263,16 +556,6 @@ var OllamaProvider = class {
263
556
  throw error;
264
557
  }
265
558
  }
266
- async #parseBody(response) {
267
- const text = await response.text();
268
- if (text.length === 0) return {};
269
- try {
270
- const data = JSON.parse(text);
271
- return (0, _orkestrel_contract.isRecord)(data) ? data : {};
272
- } catch {
273
- return {};
274
- }
275
- }
276
559
  async #requestHeaders() {
277
560
  const headers = { "Content-Type": "application/json" };
278
561
  if (this.#headers !== void 0) for (const [key, value] of Object.entries(await this.#headers())) headers[key] = value;
@@ -281,7 +564,7 @@ var OllamaProvider = class {
281
564
  #body(messages, stream, tools, options) {
282
565
  return {
283
566
  model: this.#model,
284
- messages: this.#plain(messages),
567
+ messages: mapMessages(messages),
285
568
  stream,
286
569
  keep_alive: this.#keepAlive,
287
570
  think: options?.think ?? this.#think,
@@ -297,127 +580,58 @@ var OllamaProvider = class {
297
580
  })) } : {}
298
581
  };
299
582
  }
300
- #plain(messages) {
301
- return messages.map((message) => ({
302
- role: message.role,
303
- content: message.content,
304
- ...message.calls !== void 0 && message.calls.length > 0 ? { tool_calls: message.calls.map((call) => ({ function: {
305
- name: call.name,
306
- arguments: call.arguments
307
- } })) } : {},
308
- ...message.images !== void 0 && message.images.length > 0 ? { images: [...message.images] } : {}
309
- }));
310
- }
311
- #result(content, thinking, tools, usage) {
312
- const result = { content };
313
- if (thinking.length > 0) result.thinking = thinking;
314
- if (tools.length > 0) result.tools = tools;
315
- if (usage !== void 0) result.usage = usage;
316
- return result;
317
- }
318
- #content(record) {
319
- const message = Reflect.get(record, "message");
320
- if (!(0, _orkestrel_contract.isRecord)(message)) return "";
321
- const content = Reflect.get(message, "content");
322
- return (0, _orkestrel_contract.isString)(content) ? content : "";
323
- }
324
- #thinking(record) {
325
- const message = Reflect.get(record, "message");
326
- if (!(0, _orkestrel_contract.isRecord)(message)) return "";
327
- const thinking = Reflect.get(message, "thinking");
328
- return (0, _orkestrel_contract.isString)(thinking) ? thinking : "";
329
- }
330
- #thought(splitter, wired) {
331
- if (splitter.thinking.length === 0) return wired;
332
- if (wired.length === 0) return splitter.thinking;
333
- return `${splitter.thinking}\n\n${wired}`;
334
- }
335
- #usage(record) {
336
- const prompt = Reflect.get(record, "prompt_eval_count");
337
- const completion = Reflect.get(record, "eval_count");
338
- if (!(0, _orkestrel_contract.isNumber)(prompt) || !(0, _orkestrel_contract.isNumber)(completion)) return void 0;
339
- return {
340
- prompt,
341
- completion,
342
- total: prompt + completion
343
- };
344
- }
345
- #tools(record) {
346
- const message = Reflect.get(record, "message");
347
- if (!(0, _orkestrel_contract.isRecord)(message)) return [];
348
- const calls = Reflect.get(message, "tool_calls");
349
- if (!Array.isArray(calls)) return [];
350
- const out = [];
351
- for (const entry of calls) {
352
- if (!(0, _orkestrel_contract.isRecord)(entry)) continue;
353
- const callable = Reflect.get(entry, "function");
354
- if (!(0, _orkestrel_contract.isRecord)(callable)) continue;
355
- const name = Reflect.get(callable, "name");
356
- if (!(0, _orkestrel_contract.isString)(name)) continue;
357
- const id = Reflect.get(entry, "id");
358
- out.push({
359
- id: (0, _orkestrel_contract.isString)(id) ? id : crypto.randomUUID(),
360
- name,
361
- arguments: this.#arguments(Reflect.get(callable, "arguments"))
362
- });
363
- }
364
- return out;
365
- }
366
- #arguments(value) {
367
- if ((0, _orkestrel_contract.isRecord)(value)) return value;
368
- if ((0, _orkestrel_contract.isString)(value)) try {
369
- const parsed = JSON.parse(value);
370
- if ((0, _orkestrel_contract.isRecord)(parsed)) return parsed;
371
- } catch {
372
- return {};
373
- }
374
- return {};
375
- }
376
583
  };
377
584
  //#endregion
378
585
  //#region src/server/factories.ts
379
586
  /**
380
- * Create a local Ollama inference provider — a {@link ProviderInterface} over the
587
+ * Creates a local Ollama inference provider — a {@link ProviderInterface} over the
381
588
  * daemon's `POST /api/chat`, supporting non-streaming `generate` and streaming
382
589
  * `stream`.
383
590
  *
384
591
  * @remarks
385
592
  * Only `model` is required; `url` defaults to the local daemon, `keepAlive` to `'5m'`,
386
593
  * `timeout` to `120_000`ms, and `options` is forwarded verbatim as sampling
387
- * parameters (`temperature` / `seed` / `num_predict` / …). Both calls take an
594
+ * parameters (`temperature`, `seed`, and `num_predict`). Each call takes an
388
595
  * `AbortSignal` to bound the request; a `stream` cancelled mid-flight throws a
389
596
  * `ProviderAbortError` carrying the partial result.
390
597
  *
391
598
  * The optional `fetch` + `headers` form a transport seam (see {@link OllamaOptions}):
392
599
  * point `url` at your own server, inject a custom `fetch`, and have `headers` attach a
393
600
  * generated/obfuscated bearer token your server validates — so a browser runtime
394
- * reaches the LLM through your middleware WITHOUT this library ever handling the real API
395
- * key. Both omitted ⇒ today's behaviour (the global `fetch`, only a JSON content type).
601
+ * reaches the LLM through your middleware without this library ever handling the real API
602
+ * key. Both omitted ⇒ the global `fetch` and only a JSON content type.
396
603
  *
397
- * The optional `format` is the provider's context-framing default — the PROVIDER-DEFAULT
398
- * level of `AgentContext`'s format cascade (see [agents.md]; beaten by a manager-options
399
- * or per-item override, beating the managers' built-in framing), declaring how this
400
- * provider's models prefer context sections framed (e.g. XML group wrappers vs. Markdown
401
- * headers). It is EXPOSED on the provider for the Agent's `build()` and is NOT Ollama's
402
- * `/api/chat` `format` wire parameter (structured output) — the two are unrelated despite
403
- * the shared word. Omitted ⇒ the provider is framing-agnostic (core's built-in defaults).
604
+ * The optional `format` is the provider's context-framing default — the provider-default
605
+ * level of `AgentContext`'s format cascade (beaten by a manager-options or per-item
606
+ * override, beating the managers' built-in framing), declaring how this
607
+ * provider's models prefer context sections framed (for example XML group wrappers vs. Markdown
608
+ * headers). It is exposed on the provider for the Agent's `build()` and is not Ollama's
609
+ * `/api/chat` `format` wire parameter (structured output) — the framing default and that
610
+ * wire parameter are unrelated despite the shared word. Omitted ⇒ the provider is
611
+ * framing-agnostic (core's built-in defaults).
404
612
  *
405
613
  * @param options - `model` (required), and optional `url` / `keepAlive` / `timeout` /
406
614
  * `options` / `fetch` / `headers` / `format` (see {@link OllamaOptions})
407
615
  * @returns A working {@link ProviderInterface} backed by Ollama
408
616
  *
409
- * @example
617
+ * @example createOllama + generate
410
618
  * ```ts
411
619
  * import { createAbort } from '@orkestrel/abort'
412
- * import { createOllama } from '@src/server'
620
+ * import { createOllama } from '@orkestrel/ollama'
413
621
  *
414
- * const provider = createOllama({ model: 'qwen3.5:2b-q4_K_M' })
622
+ * const provider = createOllama({ model: 'qwen3.5:2b-q4_K_M', options: { temperature: 0 } })
415
623
  * const abort = createAbort()
624
+ * const messages = [
625
+ * { id: '1', role: 'user', content: 'Summarize the release notes for version 2.0.' },
626
+ * ] as const
627
+ *
416
628
  * const result = await provider.generate(messages, abort.signal)
629
+ * console.log(result.content)
630
+ * if (result.usage) charge(result.usage) // fold into a token budget
417
631
  * ```
418
632
  *
419
633
  * @example
420
- * Route through your own server with an obfuscated token (deployment scenario S2):
634
+ * Route through your own server with an obfuscated token:
421
635
  * ```ts
422
636
  * const provider = createOllama({
423
637
  * model: 'qwen3.5:2b-q4_K_M',
@@ -429,7 +643,7 @@ var OllamaProvider = class {
429
643
  *
430
644
  * @example
431
645
  * Declare a context-framing default — wrap the instructions section in an XML group (the
432
- * provider-default level of `AgentContext`'s cascade; NOT the wire `format`):
646
+ * provider-default level of `AgentContext`'s cascade; not the wire `format`):
433
647
  * ```ts
434
648
  * const provider = createOllama({
435
649
  * model: 'qwen3.5:2b-q4_K_M',
@@ -453,7 +667,16 @@ exports.DEFAULT_PROVIDER_TIMEOUT = DEFAULT_PROVIDER_TIMEOUT;
453
667
  exports.MAX_ERROR_BODY_LENGTH = MAX_ERROR_BODY_LENGTH;
454
668
  exports.OllamaHTTPError = OllamaHTTPError;
455
669
  exports.OllamaProvider = OllamaProvider;
670
+ exports.buildResult = buildResult;
456
671
  exports.createOllama = createOllama;
672
+ exports.extractArguments = extractArguments;
673
+ exports.extractContent = extractContent;
674
+ exports.extractThinking = extractThinking;
675
+ exports.extractTools = extractTools;
676
+ exports.extractUsage = extractUsage;
457
677
  exports.isOllamaHTTPError = isOllamaHTTPError;
678
+ exports.joinThinking = joinThinking;
679
+ exports.mapMessages = mapMessages;
680
+ exports.parseBody = parseBody;
458
681
 
459
682
  //# sourceMappingURL=index.cjs.map