@orkestrel/ollama 0.0.15 → 0.0.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,682 +0,0 @@
1
- Object.defineProperty(exports, Symbol.toStringTag, { value: "Module" });
2
- let _orkestrel_contract = require("@orkestrel/contract");
3
- let _orkestrel_agent = require("@orkestrel/agent");
4
- let _orkestrel_ndjson = require("@orkestrel/ndjson");
5
- let _orkestrel_timeout = require("@orkestrel/timeout");
6
- //#region src/server/constants.ts
7
- /**
8
- * Names the local Ollama daemon base URL, `'http://localhost:11434'`, assumed when
9
- * `OllamaOptions.url` is omitted.
10
- */
11
- var DEFAULT_OLLAMA_URL = "http://localhost:11434";
12
- /**
13
- * Names how long the model stays resident after a call — `'5m'` when
14
- * `OllamaOptions.keepAlive` is omitted, Ollama's own `keep_alive` default, expressed as a
15
- * duration string.
16
- *
17
- * @remarks
18
- * The name mirrors the Ollama `/api/chat` `keep_alive` field this value is sent as, so
19
- * the constant, the `OllamaOptions.keepAlive` key, and the wire member read as one term.
20
- */
21
- var DEFAULT_KEEP_ALIVE = "5m";
22
- /**
23
- * Names the per-call deadline in milliseconds, `120_000`, when `OllamaOptions.timeout` is
24
- * omitted — generous enough that a cold model load does not trip it.
25
- */
26
- var DEFAULT_PROVIDER_TIMEOUT = 12e4;
27
- /**
28
- * Names the character cap, `2048`, on how much of a non-OK response body is
29
- * incorporated into a thrown {@link OllamaHTTPError}'s message.
30
- *
31
- * @remarks
32
- * Bounds the excerpt so a defensive proxy or a misbehaving daemon handing
33
- * back an unbounded response body cannot inflate the thrown error's message
34
- * without limit, while the cap stays generous enough to carry a useful
35
- * diagnostic snippet.
36
- */
37
- var MAX_ERROR_BODY_LENGTH = 2048;
38
- //#endregion
39
- //#region src/server/errors.ts
40
- /**
41
- * Represents an error thrown when the Ollama `/api/chat` HTTP transport fails.
42
- *
43
- * @remarks
44
- * Carries the machine-readable `code` `'HTTP'` and the response `status` (0 when no
45
- * HTTP response was received at all, for example a `null` body). Thrown by
46
- * {@link OllamaProvider} at its HTTP failure sites — the non-OK status branch and the
47
- * null-body branch — so a caller can branch on `error.code` and read `error.status`
48
- * for the HTTP number instead of parsing the message. The message carries a body excerpt
49
- * bounded to {@link MAX_ERROR_BODY_LENGTH} — `2048` characters. Narrow a caught value with
50
- * {@link isOllamaHTTPError}.
51
- *
52
- * @example
53
- * ```ts
54
- * try {
55
- * await provider.generate(messages, signal)
56
- * } catch (error) {
57
- * if (isOllamaHTTPError(error) && error.status === 404) {
58
- * // the configured model isn't pulled
59
- * }
60
- * }
61
- * ```
62
- */
63
- var OllamaHTTPError = class extends Error {
64
- /**
65
- * Names the machine-readable condition this error reports — `'HTTP'`: an `/api/chat`
66
- * transport, status, or body failure.
67
- */
68
- code = "HTTP";
69
- status;
70
- constructor(message, status, options) {
71
- super(message, options);
72
- this.name = "OllamaHTTPError";
73
- this.status = status;
74
- }
75
- };
76
- /**
77
- * Checks whether a value is an {@link OllamaHTTPError}.
78
- *
79
- * @remarks
80
- * The check is an `instanceof` test, so it narrows a caught `unknown` to the error class
81
- * without parsing the thrown message.
82
- *
83
- * @param value - The value to test
84
- * @returns True if `value` is an `OllamaHTTPError`; false otherwise
85
- */
86
- function isOllamaHTTPError(value) {
87
- return value instanceof OllamaHTTPError;
88
- }
89
- //#endregion
90
- //#region src/server/helpers.ts
91
- /**
92
- * Maps conversation turns onto the `/api/chat` wire's minimal message shape.
93
- *
94
- * @remarks
95
- * `tool_calls` is emitted only on a turn that replays them and `images` only on a
96
- * multimodal turn, so an empty optional never reaches the wire.
97
- *
98
- * @param messages - The conversation turns to send
99
- * @returns The wire `messages` array, one entry per turn, in order
100
- *
101
- * @example
102
- * ```ts
103
- * mapMessages([{ id: '1', role: 'user', content: 'Say hello.' }])
104
- * // [{ role: 'user', content: 'Say hello.' }]
105
- * ```
106
- */
107
- function mapMessages(messages) {
108
- return messages.map((message) => ({
109
- role: message.role,
110
- content: message.content,
111
- ...message.calls !== void 0 && message.calls.length > 0 ? { tool_calls: message.calls.map((call) => ({ function: {
112
- name: call.name,
113
- arguments: call.arguments
114
- } })) } : {},
115
- ...message.images !== void 0 && message.images.length > 0 ? { images: [...message.images] } : {}
116
- }));
117
- }
118
- /**
119
- * Builds a `ProviderResult` from a turn's content, reasoning, tool calls, and usage.
120
- *
121
- * @remarks
122
- * Only the present optionals are set: no empty `thinking`, no empty `tools`, and no
123
- * `usage` unless the wire reported one.
124
- *
125
- * @param content - The clean assistant content the splitter accumulated
126
- * @param thinking - The joined reasoning, empty when the turn produced none
127
- * @param tools - The tool calls collected across the turn
128
- * @param usage - The token usage, or `undefined` when the wire reported none
129
- * @returns The result carrying only its populated fields
130
- *
131
- * @example
132
- * ```ts
133
- * buildResult('ok', '', [], undefined) // { content: 'ok' }
134
- * ```
135
- */
136
- function buildResult(content, thinking, tools, usage) {
137
- const result = { content };
138
- if (thinking.length > 0) result.thinking = thinking;
139
- if (tools.length > 0) result.tools = tools;
140
- if (usage !== void 0) result.usage = usage;
141
- return result;
142
- }
143
- /**
144
- * Extracts the assistant text of one wire record.
145
- *
146
- * @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
147
- * @returns The record's `message.content` when it is a string, else `''`
148
- *
149
- * @example
150
- * ```ts
151
- * extractContent({ message: { content: 'ok' } }) // 'ok'
152
- * ```
153
- */
154
- function extractContent(record) {
155
- const message = Reflect.get(record, "message");
156
- if (!(0, _orkestrel_contract.isRecord)(message)) return "";
157
- const content = Reflect.get(message, "content");
158
- return (0, _orkestrel_contract.isString)(content) ? content : "";
159
- }
160
- /**
161
- * Extracts the daemon-side reasoning of one wire record.
162
- *
163
- * @remarks
164
- * `message.thinking` is the `think: true` wire shape. It is read whatever the configured
165
- * flag says, because a daemon may separate reasoning on its own.
166
- *
167
- * @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
168
- * @returns The record's `message.thinking` when it is a string, else `''`
169
- *
170
- * @example
171
- * ```ts
172
- * extractThinking({ message: { thinking: 'weighing it' } }) // 'weighing it'
173
- * ```
174
- */
175
- function extractThinking(record) {
176
- const message = Reflect.get(record, "message");
177
- if (!(0, _orkestrel_contract.isRecord)(message)) return "";
178
- const thinking = Reflect.get(message, "thinking");
179
- return (0, _orkestrel_contract.isString)(thinking) ? thinking : "";
180
- }
181
- /**
182
- * Joins a call's reasoning carriers — the splitter's separated in-content spans and the
183
- * accumulated wire-side `message.thinking` — into the result's `thinking`.
184
- *
185
- * @param splitter - The per-call splitter holding the separated in-content spans
186
- * @param wired - The accumulated wire-side `message.thinking` text
187
- * @returns The carriers separated by a blank line, or whichever one is non-empty
188
- *
189
- * @example
190
- * ```ts
191
- * joinThinking(createThinkSplitter(), 'from the wire') // 'from the wire'
192
- * ```
193
- */
194
- function joinThinking(splitter, wired) {
195
- if (splitter.thinking.length === 0) return wired;
196
- if (wired.length === 0) return splitter.thinking;
197
- return `${splitter.thinking}\n\n${wired}`;
198
- }
199
- /**
200
- * Extracts the token usage of one wire record.
201
- *
202
- * @remarks
203
- * Both counts must be numbers, which is true of the non-stream body and the stream's
204
- * `done: true` line. A delta line carries neither, so it yields `undefined`.
205
- *
206
- * @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
207
- * @returns The `TokenUsage` shape, or `undefined` when either count is absent
208
- *
209
- * @example
210
- * ```ts
211
- * extractUsage({ prompt_eval_count: 3, eval_count: 4 })
212
- * // { prompt: 3, completion: 4, total: 7 }
213
- * ```
214
- */
215
- function extractUsage(record) {
216
- const prompt = Reflect.get(record, "prompt_eval_count");
217
- const completion = Reflect.get(record, "eval_count");
218
- if (!(0, _orkestrel_contract.isNumber)(prompt) || !(0, _orkestrel_contract.isNumber)(completion)) return void 0;
219
- return {
220
- prompt,
221
- completion,
222
- total: prompt + completion
223
- };
224
- }
225
- /**
226
- * Extracts the tool calls of one wire record's `message.tool_calls`.
227
- *
228
- * @remarks
229
- * Each entry narrows to `{ id, name, arguments }`: the entry and its `function` must be
230
- * records and `name` a string, else the entry is dropped. An id is minted when the wire
231
- * omits one.
232
- *
233
- * @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
234
- * @returns The narrowed tool calls, empty when the record carries none
235
- *
236
- * @example
237
- * ```ts
238
- * extractTools({ message: { tool_calls: [{ function: { name: 'weather' } }] } })
239
- * // [{ id: '…', name: 'weather', arguments: {} }]
240
- * ```
241
- */
242
- function extractTools(record) {
243
- const message = Reflect.get(record, "message");
244
- if (!(0, _orkestrel_contract.isRecord)(message)) return [];
245
- const calls = Reflect.get(message, "tool_calls");
246
- if (!Array.isArray(calls)) return [];
247
- const out = [];
248
- for (const entry of calls) {
249
- if (!(0, _orkestrel_contract.isRecord)(entry)) continue;
250
- const callable = Reflect.get(entry, "function");
251
- if (!(0, _orkestrel_contract.isRecord)(callable)) continue;
252
- const name = Reflect.get(callable, "name");
253
- if (!(0, _orkestrel_contract.isString)(name)) continue;
254
- const id = Reflect.get(entry, "id");
255
- out.push({
256
- id: (0, _orkestrel_contract.isString)(id) ? id : crypto.randomUUID(),
257
- name,
258
- arguments: extractArguments(Reflect.get(callable, "arguments"))
259
- });
260
- }
261
- return out;
262
- }
263
- /**
264
- * Extracts a wire `arguments` value as a record.
265
- *
266
- * @remarks
267
- * Total: an object passes through, a JSON string is parsed when it yields a record, and
268
- * a malformed string yields `{}` rather than throwing.
269
- *
270
- * @param value - The wire's `function.arguments` value, of unknown shape
271
- * @returns The argument record, or `{}` when the value carries none
272
- *
273
- * @example
274
- * ```ts
275
- * extractArguments('{"city":"Oslo"}') // { city: 'Oslo' }
276
- * ```
277
- */
278
- function extractArguments(value) {
279
- if ((0, _orkestrel_contract.isRecord)(value)) return value;
280
- if ((0, _orkestrel_contract.isString)(value)) return (0, _orkestrel_contract.parseJSONAs)(value, _orkestrel_contract.isRecord) ?? {};
281
- return {};
282
- }
283
- //#endregion
284
- //#region src/server/parsers.ts
285
- /**
286
- * Parses a non-stream `/api/chat` response body into a wire record.
287
- *
288
- * @remarks
289
- * Total by construction: an empty body, a body that is not JSON, and a body whose JSON is
290
- * not an object all yield `undefined`, so a malformed daemon response never escapes as a
291
- * `SyntaxError`. The call site supplies the empty-record default that reads as empty
292
- * content and no usage.
293
- *
294
- * @param response - The 200-OK `/api/chat` response whose body is read as text
295
- * @returns The parsed record, or `undefined` when the body is empty or malformed
296
- *
297
- * @example
298
- * ```ts
299
- * await parseBody(new Response('{"message":{"content":"ok"}}'))
300
- * // { message: { content: 'ok' } }
301
- * ```
302
- */
303
- async function parseBody(response) {
304
- return (0, _orkestrel_contract.parseJSONAs)(await response.text(), _orkestrel_contract.isRecord);
305
- }
306
- //#endregion
307
- //#region src/server/OllamaProvider.ts
308
- /**
309
- * Implements the local Ollama inference boundary — a {@link ProviderInterface} over Ollama's
310
- * `POST /api/chat`, both non-streaming (`generate`) and streaming NDJSON (`stream`).
311
- *
312
- * @remarks
313
- * - **Wire protocol.** Posts `{ model, messages, stream, keep_alive, think }` plus
314
- * passthrough sampling `options` and mapped function `tools`. The `think` flag is
315
- * configurable through {@link OllamaOptions.think} (default `false`). Non-stream parses
316
- * one JSON body; stream consumes NDJSON (one JSON object per `\n`-terminated line) —
317
- * deltas carry `message.content`, the final `done: true` line carries the token usage.
318
- * - **Think separation.** The wire `think` flag is configurable
319
- * ({@link OllamaOptions.think}, default `false`). With `think: true` a thinking model's
320
- * daemon separates reasoning natively — returning it on the distinct `message.thinking`
321
- * channel (read here through `extractThinking`) instead of inline in `message.content`. Either
322
- * way the per-call {@link ThinkSplitterInterface} is the defensive guarantee: a daemon
323
- * may ignore `think: false` for a thinking model and inline `<think>` tags, so every
324
- * content delta routes through the splitter, only clean content is yielded / assembled,
325
- * and the separated reasoning (plus any daemon-side `message.thinking` deltas) lands on
326
- * `ProviderResult.thinking`, never in the conversation.
327
- * - **Boundary narrowing.** Every wire value arrives as `unknown` and is
328
- * narrowed through guards (`isRecord` / `isString` / `isNumber`) — never `as`. A
329
- * missing / malformed field degrades to a sensible default (empty content, no
330
- * usage, `{}` arguments), never a throw.
331
- * - **Bounded.** Each call arms a {@link Timeout} for `OllamaOptions.timeout` and
332
- * passes `AbortSignal.any([timeout.signal, signal])` to `fetch`, so the caller's
333
- * signal and the deadline both cancel the request. The timeout is always cleared —
334
- * in `#fetch` if the request fails/aborts, otherwise in the consuming call's `finally`.
335
- * - **Abort recovers partial.** A `stream` cancelled mid-flight throws a
336
- * `ProviderAbortError` carrying the partial result assembled so far; pairing the
337
- * `TextDecoder({ stream: true })` with the `createNDJSONParser` parser keeps multi-byte
338
- * UTF-8 splits and partial lines honest.
339
- * - **Event-free.** A pure functional boundary — no Emitter, no events.
340
- * - **Transport seam.** {@link OllamaOptions.fetch} swaps the transport (default
341
- * `globalThis.fetch`) and {@link OllamaOptions.headers} is a per-request, possibly
342
- * async header injector merged over the base `Content-Type` — so a browser runtime
343
- * can route through the developer's own server with an obfuscated bearer token,
344
- * without this library ever handling a real API key. Both omitted ⇒ the global `fetch`
345
- * and only a JSON content type.
346
- * Orthogonal to the deadline: the hook is awaited inside `#fetch`'s try, so a hook
347
- * rejection clears the armed timer like any other request failure.
348
- *
349
- * @example
350
- * ```ts
351
- * const provider = new OllamaProvider({ model: 'qwen3.5:2b-q4_K_M' })
352
- * const result = await provider.generate(messages, abort.signal)
353
- * ```
354
- */
355
- var OllamaProvider = class {
356
- name = "ollama";
357
- #id;
358
- #model;
359
- #url;
360
- #keepAlive;
361
- #timeout;
362
- #think;
363
- #options;
364
- #transport;
365
- #headers;
366
- #format;
367
- constructor(options) {
368
- this.#id = crypto.randomUUID();
369
- this.#model = options.model;
370
- this.#url = options.url ?? "http://localhost:11434";
371
- this.#keepAlive = options.keepAlive ?? "5m";
372
- this.#timeout = options.timeout ?? 12e4;
373
- this.#think = options.think ?? false;
374
- this.#options = options.options;
375
- this.#transport = options.fetch ?? globalThis.fetch.bind(globalThis);
376
- this.#headers = options.headers;
377
- this.#format = options.format;
378
- }
379
- /**
380
- * Exposes this instance's identity — a fresh `crypto.randomUUID()` minted at
381
- * construction, satisfying the {@link ProviderInterface.id} contract member. A second
382
- * provider built from identical options carries a distinct id.
383
- *
384
- * @returns The instance's minted identifier
385
- */
386
- get id() {
387
- return this.#id;
388
- }
389
- /**
390
- * Exposes the provider's context-framing default — the provider-default level of
391
- * {@link import('@orkestrel/agent').AgentContextInterface.build}'s format cascade (it beats
392
- * the managers' built-in framing, is beaten by a manager-options or per-item override).
393
- * Satisfies the optional {@link ProviderInterface.format} contract member: `undefined`
394
- * when {@link OllamaOptions.format} was omitted (the framing-agnostic default ⇒ core's
395
- * built-in framing applies unchanged), else the exact configured framing the Agent
396
- * threads into `build()`.
397
- *
398
- * @remarks
399
- * Expose-only — read by the Agent loop and consumed by core's cascade; it is never sent
400
- * on the `/api/chat` wire (it is absent from `#body` / the request). This is not Ollama's
401
- * structured-output `format` wire parameter — that one is sent in `#body`, but only when
402
- * a per-call `ProviderStreamOptions.schema` is supplied; only the word collides.
403
- *
404
- * @returns The configured {@link ContextFormat}, or `undefined` when none
405
- */
406
- get format() {
407
- return this.#format;
408
- }
409
- /**
410
- * Generates one complete turn and resolves the assembled result — the clean content,
411
- * any separated reasoning, any tool calls, and any usage the wire reported.
412
- *
413
- * @remarks
414
- * Sends `stream: false` and parses one JSON body. Content routes through a per-call
415
- * think splitter, so the assembled content stays clean even where the daemon renders a
416
- * thinking model's reasoning inline; the separated spans and any daemon-side
417
- * `message.thinking` land on `thinking`. The caller's signal and the armed deadline
418
- * both cancel the request, and the deadline is cleared once the body is read.
419
- *
420
- * @param messages - The conversation turns to send
421
- * @param signal - The caller's bounding signal, folded with the armed deadline
422
- * @param tools - The callable tools to advertise for this turn, when the caller passes any
423
- * @param options - The per-call overrides, `think` and `schema` among them
424
- * @returns The assembled result of the turn
425
- * @throws {@link OllamaHTTPError} When the daemon answers a non-OK status.
426
- */
427
- async generate(messages, signal, tools, options) {
428
- const { response, timeout } = await this.#fetch(messages, false, signal, tools, options);
429
- try {
430
- const record = await parseBody(response) ?? {};
431
- const splitter = (0, _orkestrel_agent.createThinkSplitter)();
432
- splitter.split(extractContent(record));
433
- splitter.flush();
434
- const thinking = joinThinking(splitter, extractThinking(record));
435
- return buildResult(splitter.content, thinking, extractTools(record), extractUsage(record));
436
- } finally {
437
- timeout.clear();
438
- }
439
- }
440
- /**
441
- * Streams one turn, yielding a channel-tagged delta per non-empty content or reasoning
442
- * span and returning the assembled result when the stream completes.
443
- *
444
- * @remarks
445
- * Sends `stream: true` and consumes NDJSON — one JSON object per newline-terminated
446
- * line — pairing a streaming `TextDecoder` with the `NDJSONParser` so a record split
447
- * across byte reads is reassembled. The returned result's content is the splitter's
448
- * clean accumulation, beside any tool calls collected across lines and the usage the
449
- * `done` line carries. A cancel mid-flight throws a `ProviderAbortError` carrying the
450
- * partial assembled so far.
451
- *
452
- * @param messages - The conversation turns to send
453
- * @param signal - The caller's bounding signal, folded with the armed deadline
454
- * @param tools - The callable tools to advertise for this turn, when the caller passes any
455
- * @param options - The per-call overrides, `think` and `schema` among them
456
- * @returns The assembled result of the turn, after the last delta
457
- * @throws {@link OllamaHTTPError} When the daemon answers a non-OK status or a `null` body.
458
- */
459
- async *stream(messages, signal, tools, options) {
460
- const { response, timeout, combined } = await this.#fetch(messages, true, signal, tools, options);
461
- const body = response.body;
462
- if (body === null) {
463
- timeout.clear();
464
- throw new OllamaHTTPError("Ollama API error: no response body", 0);
465
- }
466
- const reader = body.getReader();
467
- const decoder = new TextDecoder();
468
- const parser = (0, _orkestrel_ndjson.createNDJSONParser)();
469
- const splitter = (0, _orkestrel_agent.createThinkSplitter)();
470
- let wired = "";
471
- const calls = [];
472
- let usage;
473
- try {
474
- for (;;) {
475
- const { value, done } = await reader.read();
476
- if (done) break;
477
- for (const record of parser.parse(decoder.decode(value, { stream: true }))) {
478
- const increment = yield* this.#deltas(record, splitter, usage);
479
- wired += increment.thinking;
480
- calls.push(...increment.calls);
481
- usage = increment.usage;
482
- }
483
- }
484
- const decoderTail = decoder.decode();
485
- for (const record of parser.parse(decoderTail.length > 0 ? `${decoderTail}\n` : "\n")) {
486
- const increment = yield* this.#deltas(record, splitter, usage);
487
- wired += increment.thinking;
488
- calls.push(...increment.calls);
489
- usage = increment.usage;
490
- }
491
- const tail = splitter.flush();
492
- if (tail.length > 0) yield {
493
- channel: "content",
494
- text: tail
495
- };
496
- } catch (error) {
497
- if (combined.aborted) {
498
- splitter.flush();
499
- throw new _orkestrel_agent.ProviderAbortError(buildResult(splitter.content, joinThinking(splitter, wired), calls, usage));
500
- }
501
- throw error;
502
- } finally {
503
- try {
504
- await reader.cancel();
505
- } catch {}
506
- parser.clear();
507
- timeout.clear();
508
- }
509
- return buildResult(splitter.content, joinThinking(splitter, wired), calls, usage);
510
- }
511
- *#deltas(record, splitter, usage) {
512
- const delta = splitter.split(extractContent(record));
513
- if (delta.length > 0) yield {
514
- channel: "content",
515
- text: delta
516
- };
517
- const thinking = extractThinking(record);
518
- if (thinking.length > 0) yield {
519
- channel: "thinking",
520
- text: thinking
521
- };
522
- return {
523
- thinking,
524
- calls: extractTools(record),
525
- usage: Reflect.get(record, "done") === true ? extractUsage(record) : usage
526
- };
527
- }
528
- async #fetch(messages, stream, signal, tools, options) {
529
- const timeout = new _orkestrel_timeout.Timeout({ ms: this.#timeout });
530
- timeout.start();
531
- const combined = AbortSignal.any([timeout.signal, signal]);
532
- try {
533
- const response = await this.#transport(`${this.#url}/api/chat`, {
534
- method: "POST",
535
- headers: await this.#requestHeaders(),
536
- body: JSON.stringify(this.#body(messages, stream, tools, options)),
537
- signal: combined
538
- });
539
- if (!response.ok) {
540
- let detail;
541
- try {
542
- const text = await response.text();
543
- detail = text.length > 2048 ? text.slice(0, MAX_ERROR_BODY_LENGTH) : text;
544
- } catch (cause) {
545
- throw new OllamaHTTPError(`Ollama API error: ${response.status} - (error body unavailable)`, response.status, { cause });
546
- }
547
- throw new OllamaHTTPError(`Ollama API error: ${response.status} - ${detail}`, response.status);
548
- }
549
- return {
550
- response,
551
- timeout,
552
- combined
553
- };
554
- } catch (error) {
555
- timeout.clear();
556
- throw error;
557
- }
558
- }
559
- async #requestHeaders() {
560
- const headers = { "Content-Type": "application/json" };
561
- if (this.#headers !== void 0) for (const [key, value] of Object.entries(await this.#headers())) headers[key] = value;
562
- return headers;
563
- }
564
- #body(messages, stream, tools, options) {
565
- return {
566
- model: this.#model,
567
- messages: mapMessages(messages),
568
- stream,
569
- keep_alive: this.#keepAlive,
570
- think: options?.think ?? this.#think,
571
- ...this.#options !== void 0 ? { options: this.#options } : {},
572
- ...options?.schema !== void 0 ? { format: options.schema } : {},
573
- ...tools !== void 0 && tools.length > 0 ? { tools: tools.map((tool) => ({
574
- type: "function",
575
- function: {
576
- name: tool.name,
577
- ...tool.description === void 0 ? {} : { description: tool.description },
578
- ...tool.parameters === void 0 ? {} : { parameters: tool.parameters }
579
- }
580
- })) } : {}
581
- };
582
- }
583
- };
584
- //#endregion
585
- //#region src/server/factories.ts
586
- /**
587
- * Creates a local Ollama inference provider — a {@link ProviderInterface} over the
588
- * daemon's `POST /api/chat`, supporting non-streaming `generate` and streaming
589
- * `stream`.
590
- *
591
- * @remarks
592
- * Only `model` is required; `url` defaults to the local daemon, `keepAlive` to `'5m'`,
593
- * `timeout` to `120_000`ms, and `options` is forwarded verbatim as sampling
594
- * parameters (`temperature`, `seed`, and `num_predict`). Each call takes an
595
- * `AbortSignal` to bound the request; a `stream` cancelled mid-flight throws a
596
- * `ProviderAbortError` carrying the partial result.
597
- *
598
- * The optional `fetch` + `headers` form a transport seam (see {@link OllamaOptions}):
599
- * point `url` at your own server, inject a custom `fetch`, and have `headers` attach a
600
- * generated/obfuscated bearer token your server validates — so a browser runtime
601
- * reaches the LLM through your middleware without this library ever handling the real API
602
- * key. Both omitted ⇒ the global `fetch` and only a JSON content type.
603
- *
604
- * The optional `format` is the provider's context-framing default — the provider-default
605
- * level of `AgentContext`'s format cascade (beaten by a manager-options or per-item
606
- * override, beating the managers' built-in framing), declaring how this
607
- * provider's models prefer context sections framed (for example XML group wrappers vs. Markdown
608
- * headers). It is exposed on the provider for the Agent's `build()` and is not Ollama's
609
- * `/api/chat` `format` wire parameter (structured output) — the framing default and that
610
- * wire parameter are unrelated despite the shared word. Omitted ⇒ the provider is
611
- * framing-agnostic (core's built-in defaults).
612
- *
613
- * @param options - `model` (required), and optional `url` / `keepAlive` / `timeout` /
614
- * `options` / `fetch` / `headers` / `format` (see {@link OllamaOptions})
615
- * @returns A working {@link ProviderInterface} backed by Ollama
616
- *
617
- * @example createOllama + generate
618
- * ```ts
619
- * import { createAbort } from '@orkestrel/abort'
620
- * import { createOllama } from '@orkestrel/ollama'
621
- *
622
- * const provider = createOllama({ model: 'qwen3.5:2b-q4_K_M', options: { temperature: 0 } })
623
- * const abort = createAbort()
624
- * const messages = [
625
- * { id: '1', role: 'user', content: 'Summarize the release notes for version 2.0.' },
626
- * ] as const
627
- *
628
- * const result = await provider.generate(messages, abort.signal)
629
- * console.log(result.content)
630
- * if (result.usage) charge(result.usage) // fold into a token budget
631
- * ```
632
- *
633
- * @example
634
- * Route through your own server with an obfuscated token:
635
- * ```ts
636
- * const provider = createOllama({
637
- * model: 'qwen3.5:2b-q4_K_M',
638
- * url: 'https://my-app.example.com/llm', // your server, not the daemon
639
- * fetch: myFetch, // optional custom transport
640
- * headers: () => ({ authorization: `Bearer ${myToken}` }), // your server validates this
641
- * })
642
- * ```
643
- *
644
- * @example
645
- * Declare a context-framing default — wrap the instructions section in an XML group (the
646
- * provider-default level of `AgentContext`'s cascade; not the wire `format`):
647
- * ```ts
648
- * const provider = createOllama({
649
- * model: 'qwen3.5:2b-q4_K_M',
650
- * format: {
651
- * instructions: {
652
- * open: '<instructions>',
653
- * render: (i) => `<instruction>${i.content}</instruction>`,
654
- * close: '</instructions>',
655
- * },
656
- * },
657
- * })
658
- * ```
659
- */
660
- function createOllama(options) {
661
- return new OllamaProvider(options);
662
- }
663
- //#endregion
664
- exports.DEFAULT_KEEP_ALIVE = DEFAULT_KEEP_ALIVE;
665
- exports.DEFAULT_OLLAMA_URL = DEFAULT_OLLAMA_URL;
666
- exports.DEFAULT_PROVIDER_TIMEOUT = DEFAULT_PROVIDER_TIMEOUT;
667
- exports.MAX_ERROR_BODY_LENGTH = MAX_ERROR_BODY_LENGTH;
668
- exports.OllamaHTTPError = OllamaHTTPError;
669
- exports.OllamaProvider = OllamaProvider;
670
- exports.buildResult = buildResult;
671
- exports.createOllama = createOllama;
672
- exports.extractArguments = extractArguments;
673
- exports.extractContent = extractContent;
674
- exports.extractThinking = extractThinking;
675
- exports.extractTools = extractTools;
676
- exports.extractUsage = extractUsage;
677
- exports.isOllamaHTTPError = isOllamaHTTPError;
678
- exports.joinThinking = joinThinking;
679
- exports.mapMessages = mapMessages;
680
- exports.parseBody = parseBody;
681
-
682
- //# sourceMappingURL=index.cjs.map