@orkestrel/ollama 0.0.13 → 0.0.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -9
- package/dist/src/server/index.cjs +387 -164
- package/dist/src/server/index.cjs.map +1 -1
- package/dist/src/server/index.d.cts +405 -154
- package/dist/src/server/index.d.ts +405 -154
- package/dist/src/server/index.js +379 -165
- package/dist/src/server/index.js.map +1 -1
- package/package.json +20 -22
package/dist/src/server/index.js
CHANGED
|
@@ -1,43 +1,52 @@
|
|
|
1
|
+
import { isNumber, isRecord, isString, parseJSONAs } from "@orkestrel/contract";
|
|
1
2
|
import { ProviderAbortError, createThinkSplitter } from "@orkestrel/agent";
|
|
2
|
-
import { isNumber, isRecord, isString } from "@orkestrel/contract";
|
|
3
3
|
import { createNDJSONParser } from "@orkestrel/ndjson";
|
|
4
4
|
import { Timeout } from "@orkestrel/timeout";
|
|
5
5
|
//#region src/server/constants.ts
|
|
6
|
-
/**
|
|
6
|
+
/**
|
|
7
|
+
* Names the local Ollama daemon base URL, `'http://localhost:11434'`, assumed when
|
|
8
|
+
* `OllamaOptions.url` is omitted.
|
|
9
|
+
*/
|
|
7
10
|
var DEFAULT_OLLAMA_URL = "http://localhost:11434";
|
|
8
11
|
/**
|
|
9
|
-
*
|
|
10
|
-
* omitted
|
|
12
|
+
* Names how long the model stays resident after a call — `'5m'` when
|
|
13
|
+
* `OllamaOptions.keepAlive` is omitted, Ollama's own `keep_alive` default, expressed as a
|
|
14
|
+
* duration string.
|
|
15
|
+
*
|
|
16
|
+
* @remarks
|
|
17
|
+
* The name mirrors the Ollama `/api/chat` `keep_alive` field this value is sent as, so
|
|
18
|
+
* the constant, the `OllamaOptions.keepAlive` key, and the wire member read as one term.
|
|
11
19
|
*/
|
|
12
20
|
var DEFAULT_KEEP_ALIVE = "5m";
|
|
13
21
|
/**
|
|
14
|
-
*
|
|
15
|
-
* generous enough that a cold model load does not trip it.
|
|
22
|
+
* Names the per-call deadline in milliseconds, `120_000`, when `OllamaOptions.timeout` is
|
|
23
|
+
* omitted — generous enough that a cold model load does not trip it.
|
|
16
24
|
*/
|
|
17
25
|
var DEFAULT_PROVIDER_TIMEOUT = 12e4;
|
|
18
26
|
/**
|
|
19
|
-
*
|
|
27
|
+
* Names the character cap, `2048`, on how much of a non-OK response body is
|
|
20
28
|
* incorporated into a thrown {@link OllamaHTTPError}'s message.
|
|
21
29
|
*
|
|
22
30
|
* @remarks
|
|
23
31
|
* Bounds the excerpt so a defensive proxy or a misbehaving daemon handing
|
|
24
32
|
* back an unbounded response body cannot inflate the thrown error's message
|
|
25
|
-
* without limit
|
|
26
|
-
*
|
|
27
|
-
* concern.
|
|
33
|
+
* without limit, while the cap stays generous enough to carry a useful
|
|
34
|
+
* diagnostic snippet.
|
|
28
35
|
*/
|
|
29
36
|
var MAX_ERROR_BODY_LENGTH = 2048;
|
|
30
37
|
//#endregion
|
|
31
38
|
//#region src/server/errors.ts
|
|
32
39
|
/**
|
|
33
|
-
*
|
|
40
|
+
* Represents an error thrown when the Ollama `/api/chat` HTTP transport fails.
|
|
34
41
|
*
|
|
35
42
|
* @remarks
|
|
36
|
-
* Carries the response `status` (0 when no
|
|
37
|
-
*
|
|
38
|
-
* failure sites — the non-OK status branch and the
|
|
39
|
-
* caller can branch on `error.
|
|
40
|
-
*
|
|
43
|
+
* Carries the machine-readable `code` `'HTTP'` and the response `status` (0 when no
|
|
44
|
+
* HTTP response was received at all, for example a `null` body). Thrown by
|
|
45
|
+
* {@link OllamaProvider} at its HTTP failure sites — the non-OK status branch and the
|
|
46
|
+
* null-body branch — so a caller can branch on `error.code` and read `error.status`
|
|
47
|
+
* for the HTTP number instead of parsing the message. The message carries a body excerpt
|
|
48
|
+
* bounded to {@link MAX_ERROR_BODY_LENGTH} — `2048` characters. Narrow a caught value with
|
|
49
|
+
* {@link isOllamaHTTPError}.
|
|
41
50
|
*
|
|
42
51
|
* @example
|
|
43
52
|
* ```ts
|
|
@@ -51,6 +60,11 @@ var MAX_ERROR_BODY_LENGTH = 2048;
|
|
|
51
60
|
* ```
|
|
52
61
|
*/
|
|
53
62
|
var OllamaHTTPError = class extends Error {
|
|
63
|
+
/**
|
|
64
|
+
* Names the machine-readable condition this error reports — `'HTTP'`: an `/api/chat`
|
|
65
|
+
* transport, status, or body failure.
|
|
66
|
+
*/
|
|
67
|
+
code = "HTTP";
|
|
54
68
|
status;
|
|
55
69
|
constructor(message, status, options) {
|
|
56
70
|
super(message, options);
|
|
@@ -59,53 +73,275 @@ var OllamaHTTPError = class extends Error {
|
|
|
59
73
|
}
|
|
60
74
|
};
|
|
61
75
|
/**
|
|
62
|
-
*
|
|
76
|
+
* Checks whether a value is an {@link OllamaHTTPError}.
|
|
77
|
+
*
|
|
78
|
+
* @remarks
|
|
79
|
+
* The check is an `instanceof` test, so it narrows a caught `unknown` to the error class
|
|
80
|
+
* without parsing the thrown message.
|
|
63
81
|
*
|
|
64
82
|
* @param value - The value to test
|
|
65
|
-
* @returns
|
|
83
|
+
* @returns True if `value` is an `OllamaHTTPError`; false otherwise
|
|
66
84
|
*/
|
|
67
85
|
function isOllamaHTTPError(value) {
|
|
68
86
|
return value instanceof OllamaHTTPError;
|
|
69
87
|
}
|
|
70
88
|
//#endregion
|
|
89
|
+
//#region src/server/helpers.ts
|
|
90
|
+
/**
|
|
91
|
+
* Maps conversation turns onto the `/api/chat` wire's minimal message shape.
|
|
92
|
+
*
|
|
93
|
+
* @remarks
|
|
94
|
+
* `tool_calls` is emitted only on a turn that replays them and `images` only on a
|
|
95
|
+
* multimodal turn, so an empty optional never reaches the wire.
|
|
96
|
+
*
|
|
97
|
+
* @param messages - The conversation turns to send
|
|
98
|
+
* @returns The wire `messages` array, one entry per turn, in order
|
|
99
|
+
*
|
|
100
|
+
* @example
|
|
101
|
+
* ```ts
|
|
102
|
+
* mapMessages([{ id: '1', role: 'user', content: 'Say hello.' }])
|
|
103
|
+
* // [{ role: 'user', content: 'Say hello.' }]
|
|
104
|
+
* ```
|
|
105
|
+
*/
|
|
106
|
+
function mapMessages(messages) {
|
|
107
|
+
return messages.map((message) => ({
|
|
108
|
+
role: message.role,
|
|
109
|
+
content: message.content,
|
|
110
|
+
...message.calls !== void 0 && message.calls.length > 0 ? { tool_calls: message.calls.map((call) => ({ function: {
|
|
111
|
+
name: call.name,
|
|
112
|
+
arguments: call.arguments
|
|
113
|
+
} })) } : {},
|
|
114
|
+
...message.images !== void 0 && message.images.length > 0 ? { images: [...message.images] } : {}
|
|
115
|
+
}));
|
|
116
|
+
}
|
|
117
|
+
/**
|
|
118
|
+
* Builds a `ProviderResult` from a turn's content, reasoning, tool calls, and usage.
|
|
119
|
+
*
|
|
120
|
+
* @remarks
|
|
121
|
+
* Only the present optionals are set: no empty `thinking`, no empty `tools`, and no
|
|
122
|
+
* `usage` unless the wire reported one.
|
|
123
|
+
*
|
|
124
|
+
* @param content - The clean assistant content the splitter accumulated
|
|
125
|
+
* @param thinking - The joined reasoning, empty when the turn produced none
|
|
126
|
+
* @param tools - The tool calls collected across the turn
|
|
127
|
+
* @param usage - The token usage, or `undefined` when the wire reported none
|
|
128
|
+
* @returns The result carrying only its populated fields
|
|
129
|
+
*
|
|
130
|
+
* @example
|
|
131
|
+
* ```ts
|
|
132
|
+
* buildResult('ok', '', [], undefined) // { content: 'ok' }
|
|
133
|
+
* ```
|
|
134
|
+
*/
|
|
135
|
+
function buildResult(content, thinking, tools, usage) {
|
|
136
|
+
const result = { content };
|
|
137
|
+
if (thinking.length > 0) result.thinking = thinking;
|
|
138
|
+
if (tools.length > 0) result.tools = tools;
|
|
139
|
+
if (usage !== void 0) result.usage = usage;
|
|
140
|
+
return result;
|
|
141
|
+
}
|
|
142
|
+
/**
|
|
143
|
+
* Extracts the assistant text of one wire record.
|
|
144
|
+
*
|
|
145
|
+
* @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
|
|
146
|
+
* @returns The record's `message.content` when it is a string, else `''`
|
|
147
|
+
*
|
|
148
|
+
* @example
|
|
149
|
+
* ```ts
|
|
150
|
+
* extractContent({ message: { content: 'ok' } }) // 'ok'
|
|
151
|
+
* ```
|
|
152
|
+
*/
|
|
153
|
+
function extractContent(record) {
|
|
154
|
+
const message = Reflect.get(record, "message");
|
|
155
|
+
if (!isRecord(message)) return "";
|
|
156
|
+
const content = Reflect.get(message, "content");
|
|
157
|
+
return isString(content) ? content : "";
|
|
158
|
+
}
|
|
159
|
+
/**
|
|
160
|
+
* Extracts the daemon-side reasoning of one wire record.
|
|
161
|
+
*
|
|
162
|
+
* @remarks
|
|
163
|
+
* `message.thinking` is the `think: true` wire shape. It is read whatever the configured
|
|
164
|
+
* flag says, because a daemon may separate reasoning on its own.
|
|
165
|
+
*
|
|
166
|
+
* @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
|
|
167
|
+
* @returns The record's `message.thinking` when it is a string, else `''`
|
|
168
|
+
*
|
|
169
|
+
* @example
|
|
170
|
+
* ```ts
|
|
171
|
+
* extractThinking({ message: { thinking: 'weighing it' } }) // 'weighing it'
|
|
172
|
+
* ```
|
|
173
|
+
*/
|
|
174
|
+
function extractThinking(record) {
|
|
175
|
+
const message = Reflect.get(record, "message");
|
|
176
|
+
if (!isRecord(message)) return "";
|
|
177
|
+
const thinking = Reflect.get(message, "thinking");
|
|
178
|
+
return isString(thinking) ? thinking : "";
|
|
179
|
+
}
|
|
180
|
+
/**
|
|
181
|
+
* Joins a call's reasoning carriers — the splitter's separated in-content spans and the
|
|
182
|
+
* accumulated wire-side `message.thinking` — into the result's `thinking`.
|
|
183
|
+
*
|
|
184
|
+
* @param splitter - The per-call splitter holding the separated in-content spans
|
|
185
|
+
* @param wired - The accumulated wire-side `message.thinking` text
|
|
186
|
+
* @returns The carriers separated by a blank line, or whichever one is non-empty
|
|
187
|
+
*
|
|
188
|
+
* @example
|
|
189
|
+
* ```ts
|
|
190
|
+
* joinThinking(createThinkSplitter(), 'from the wire') // 'from the wire'
|
|
191
|
+
* ```
|
|
192
|
+
*/
|
|
193
|
+
function joinThinking(splitter, wired) {
|
|
194
|
+
if (splitter.thinking.length === 0) return wired;
|
|
195
|
+
if (wired.length === 0) return splitter.thinking;
|
|
196
|
+
return `${splitter.thinking}\n\n${wired}`;
|
|
197
|
+
}
|
|
198
|
+
/**
|
|
199
|
+
* Extracts the token usage of one wire record.
|
|
200
|
+
*
|
|
201
|
+
* @remarks
|
|
202
|
+
* Both counts must be numbers, which is true of the non-stream body and the stream's
|
|
203
|
+
* `done: true` line. A delta line carries neither, so it yields `undefined`.
|
|
204
|
+
*
|
|
205
|
+
* @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
|
|
206
|
+
* @returns The `TokenUsage` shape, or `undefined` when either count is absent
|
|
207
|
+
*
|
|
208
|
+
* @example
|
|
209
|
+
* ```ts
|
|
210
|
+
* extractUsage({ prompt_eval_count: 3, eval_count: 4 })
|
|
211
|
+
* // { prompt: 3, completion: 4, total: 7 }
|
|
212
|
+
* ```
|
|
213
|
+
*/
|
|
214
|
+
function extractUsage(record) {
|
|
215
|
+
const prompt = Reflect.get(record, "prompt_eval_count");
|
|
216
|
+
const completion = Reflect.get(record, "eval_count");
|
|
217
|
+
if (!isNumber(prompt) || !isNumber(completion)) return void 0;
|
|
218
|
+
return {
|
|
219
|
+
prompt,
|
|
220
|
+
completion,
|
|
221
|
+
total: prompt + completion
|
|
222
|
+
};
|
|
223
|
+
}
|
|
224
|
+
/**
|
|
225
|
+
* Extracts the tool calls of one wire record's `message.tool_calls`.
|
|
226
|
+
*
|
|
227
|
+
* @remarks
|
|
228
|
+
* Each entry narrows to `{ id, name, arguments }`: the entry and its `function` must be
|
|
229
|
+
* records and `name` a string, else the entry is dropped. An id is minted when the wire
|
|
230
|
+
* omits one.
|
|
231
|
+
*
|
|
232
|
+
* @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
|
|
233
|
+
* @returns The narrowed tool calls, empty when the record carries none
|
|
234
|
+
*
|
|
235
|
+
* @example
|
|
236
|
+
* ```ts
|
|
237
|
+
* extractTools({ message: { tool_calls: [{ function: { name: 'weather' } }] } })
|
|
238
|
+
* // [{ id: '…', name: 'weather', arguments: {} }]
|
|
239
|
+
* ```
|
|
240
|
+
*/
|
|
241
|
+
function extractTools(record) {
|
|
242
|
+
const message = Reflect.get(record, "message");
|
|
243
|
+
if (!isRecord(message)) return [];
|
|
244
|
+
const calls = Reflect.get(message, "tool_calls");
|
|
245
|
+
if (!Array.isArray(calls)) return [];
|
|
246
|
+
const out = [];
|
|
247
|
+
for (const entry of calls) {
|
|
248
|
+
if (!isRecord(entry)) continue;
|
|
249
|
+
const callable = Reflect.get(entry, "function");
|
|
250
|
+
if (!isRecord(callable)) continue;
|
|
251
|
+
const name = Reflect.get(callable, "name");
|
|
252
|
+
if (!isString(name)) continue;
|
|
253
|
+
const id = Reflect.get(entry, "id");
|
|
254
|
+
out.push({
|
|
255
|
+
id: isString(id) ? id : crypto.randomUUID(),
|
|
256
|
+
name,
|
|
257
|
+
arguments: extractArguments(Reflect.get(callable, "arguments"))
|
|
258
|
+
});
|
|
259
|
+
}
|
|
260
|
+
return out;
|
|
261
|
+
}
|
|
262
|
+
/**
|
|
263
|
+
* Extracts a wire `arguments` value as a record.
|
|
264
|
+
*
|
|
265
|
+
* @remarks
|
|
266
|
+
* Total: an object passes through, a JSON string is parsed when it yields a record, and
|
|
267
|
+
* a malformed string yields `{}` rather than throwing.
|
|
268
|
+
*
|
|
269
|
+
* @param value - The wire's `function.arguments` value, of unknown shape
|
|
270
|
+
* @returns The argument record, or `{}` when the value carries none
|
|
271
|
+
*
|
|
272
|
+
* @example
|
|
273
|
+
* ```ts
|
|
274
|
+
* extractArguments('{"city":"Oslo"}') // { city: 'Oslo' }
|
|
275
|
+
* ```
|
|
276
|
+
*/
|
|
277
|
+
function extractArguments(value) {
|
|
278
|
+
if (isRecord(value)) return value;
|
|
279
|
+
if (isString(value)) return parseJSONAs(value, isRecord) ?? {};
|
|
280
|
+
return {};
|
|
281
|
+
}
|
|
282
|
+
//#endregion
|
|
283
|
+
//#region src/server/parsers.ts
|
|
284
|
+
/**
|
|
285
|
+
* Parses a non-stream `/api/chat` response body into a wire record.
|
|
286
|
+
*
|
|
287
|
+
* @remarks
|
|
288
|
+
* Total by construction: an empty body, a body that is not JSON, and a body whose JSON is
|
|
289
|
+
* not an object all yield `undefined`, so a malformed daemon response never escapes as a
|
|
290
|
+
* `SyntaxError`. The call site supplies the empty-record default that reads as empty
|
|
291
|
+
* content and no usage.
|
|
292
|
+
*
|
|
293
|
+
* @param response - The 200-OK `/api/chat` response whose body is read as text
|
|
294
|
+
* @returns The parsed record, or `undefined` when the body is empty or malformed
|
|
295
|
+
*
|
|
296
|
+
* @example
|
|
297
|
+
* ```ts
|
|
298
|
+
* await parseBody(new Response('{"message":{"content":"ok"}}'))
|
|
299
|
+
* // { message: { content: 'ok' } }
|
|
300
|
+
* ```
|
|
301
|
+
*/
|
|
302
|
+
async function parseBody(response) {
|
|
303
|
+
return parseJSONAs(await response.text(), isRecord);
|
|
304
|
+
}
|
|
305
|
+
//#endregion
|
|
71
306
|
//#region src/server/OllamaProvider.ts
|
|
72
307
|
/**
|
|
73
|
-
*
|
|
308
|
+
* Implements the local Ollama inference boundary — a {@link ProviderInterface} over Ollama's
|
|
74
309
|
* `POST /api/chat`, both non-streaming (`generate`) and streaming NDJSON (`stream`).
|
|
75
310
|
*
|
|
76
311
|
* @remarks
|
|
77
312
|
* - **Wire protocol.** Posts `{ model, messages, stream, keep_alive, think }` plus
|
|
78
313
|
* passthrough sampling `options` and mapped function `tools`. The `think` flag is
|
|
79
|
-
*
|
|
314
|
+
* configurable through {@link OllamaOptions.think} (default `false`). Non-stream parses
|
|
80
315
|
* one JSON body; stream consumes NDJSON (one JSON object per `\n`-terminated line) —
|
|
81
316
|
* deltas carry `message.content`, the final `done: true` line carries the token usage.
|
|
82
|
-
* - **Think separation
|
|
317
|
+
* - **Think separation.** The wire `think` flag is configurable
|
|
83
318
|
* ({@link OllamaOptions.think}, default `false`). With `think: true` a thinking model's
|
|
84
|
-
* daemon separates reasoning
|
|
85
|
-
* channel (read here
|
|
319
|
+
* daemon separates reasoning natively — returning it on the distinct `message.thinking`
|
|
320
|
+
* channel (read here through `extractThinking`) instead of inline in `message.content`. Either
|
|
86
321
|
* way the per-call {@link ThinkSplitterInterface} is the defensive guarantee: a daemon
|
|
87
322
|
* may ignore `think: false` for a thinking model and inline `<think>` tags, so every
|
|
88
|
-
* content delta routes through the splitter, only
|
|
323
|
+
* content delta routes through the splitter, only clean content is yielded / assembled,
|
|
89
324
|
* and the separated reasoning (plus any daemon-side `message.thinking` deltas) lands on
|
|
90
325
|
* `ProviderResult.thinking`, never in the conversation.
|
|
91
|
-
* - **Boundary narrowing
|
|
326
|
+
* - **Boundary narrowing.** Every wire value arrives as `unknown` and is
|
|
92
327
|
* narrowed through guards (`isRecord` / `isString` / `isNumber`) — never `as`. A
|
|
93
328
|
* missing / malformed field degrades to a sensible default (empty content, no
|
|
94
329
|
* usage, `{}` arguments), never a throw.
|
|
95
330
|
* - **Bounded.** Each call arms a {@link Timeout} for `OllamaOptions.timeout` and
|
|
96
331
|
* passes `AbortSignal.any([timeout.signal, signal])` to `fetch`, so the caller's
|
|
97
|
-
* signal
|
|
332
|
+
* signal and the deadline both cancel the request. The timeout is always cleared —
|
|
98
333
|
* in `#fetch` if the request fails/aborts, otherwise in the consuming call's `finally`.
|
|
99
334
|
* - **Abort recovers partial.** A `stream` cancelled mid-flight throws a
|
|
100
335
|
* `ProviderAbortError` carrying the partial result assembled so far; pairing the
|
|
101
|
-
* `TextDecoder({ stream: true })` with the
|
|
336
|
+
* `TextDecoder({ stream: true })` with the `createNDJSONParser` parser keeps multi-byte
|
|
102
337
|
* UTF-8 splits and partial lines honest.
|
|
103
338
|
* - **Event-free.** A pure functional boundary — no Emitter, no events.
|
|
104
339
|
* - **Transport seam.** {@link OllamaOptions.fetch} swaps the transport (default
|
|
105
340
|
* `globalThis.fetch`) and {@link OllamaOptions.headers} is a per-request, possibly
|
|
106
341
|
* async header injector merged over the base `Content-Type` — so a browser runtime
|
|
107
342
|
* can route through the developer's own server with an obfuscated bearer token,
|
|
108
|
-
* without this library ever handling a real API key. Both omitted ⇒
|
|
343
|
+
* without this library ever handling a real API key. Both omitted ⇒ the global `fetch`
|
|
344
|
+
* and only a JSON content type.
|
|
109
345
|
* Orthogonal to the deadline: the hook is awaited inside `#fetch`'s try, so a hook
|
|
110
346
|
* rejection clears the armed timer like any other request failure.
|
|
111
347
|
*
|
|
@@ -116,8 +352,8 @@ function isOllamaHTTPError(value) {
|
|
|
116
352
|
* ```
|
|
117
353
|
*/
|
|
118
354
|
var OllamaProvider = class {
|
|
119
|
-
id = crypto.randomUUID();
|
|
120
355
|
name = "ollama";
|
|
356
|
+
#id;
|
|
121
357
|
#model;
|
|
122
358
|
#url;
|
|
123
359
|
#keepAlive;
|
|
@@ -128,6 +364,7 @@ var OllamaProvider = class {
|
|
|
128
364
|
#headers;
|
|
129
365
|
#format;
|
|
130
366
|
constructor(options) {
|
|
367
|
+
this.#id = crypto.randomUUID();
|
|
131
368
|
this.#model = options.model;
|
|
132
369
|
this.#url = options.url ?? "http://localhost:11434";
|
|
133
370
|
this.#keepAlive = options.keepAlive ?? "5m";
|
|
@@ -139,38 +376,85 @@ var OllamaProvider = class {
|
|
|
139
376
|
this.#format = options.format;
|
|
140
377
|
}
|
|
141
378
|
/**
|
|
142
|
-
*
|
|
143
|
-
* {@link
|
|
144
|
-
*
|
|
145
|
-
*
|
|
379
|
+
* Exposes this instance's identity — a fresh `crypto.randomUUID()` minted at
|
|
380
|
+
* construction, satisfying the {@link ProviderInterface.id} contract member. A second
|
|
381
|
+
* provider built from identical options carries a distinct id.
|
|
382
|
+
*
|
|
383
|
+
* @returns The instance's minted identifier
|
|
384
|
+
*/
|
|
385
|
+
get id() {
|
|
386
|
+
return this.#id;
|
|
387
|
+
}
|
|
388
|
+
/**
|
|
389
|
+
* Exposes the provider's context-framing default — the provider-default level of
|
|
390
|
+
* {@link import('@orkestrel/agent').AgentContextInterface.build}'s format cascade (it beats
|
|
391
|
+
* the managers' built-in framing, is beaten by a manager-options or per-item override).
|
|
392
|
+
* Satisfies the optional {@link ProviderInterface.format} contract member: `undefined`
|
|
146
393
|
* when {@link OllamaOptions.format} was omitted (the framing-agnostic default ⇒ core's
|
|
147
394
|
* built-in framing applies unchanged), else the exact configured framing the Agent
|
|
148
395
|
* threads into `build()`.
|
|
149
396
|
*
|
|
150
397
|
* @remarks
|
|
151
|
-
*
|
|
152
|
-
* on the `/api/chat` wire (it is absent from `#body` / the request). This is
|
|
153
|
-
* structured-output `format` wire parameter — that one
|
|
398
|
+
* Expose-only — read by the Agent loop and consumed by core's cascade; it is never sent
|
|
399
|
+
* on the `/api/chat` wire (it is absent from `#body` / the request). This is not Ollama's
|
|
400
|
+
* structured-output `format` wire parameter — that one is sent in `#body`, but only when
|
|
154
401
|
* a per-call `ProviderStreamOptions.schema` is supplied; only the word collides.
|
|
155
402
|
*
|
|
156
|
-
* @returns The configured {@link
|
|
403
|
+
* @returns The configured {@link ContextFormat}, or `undefined` when none
|
|
157
404
|
*/
|
|
158
405
|
get format() {
|
|
159
406
|
return this.#format;
|
|
160
407
|
}
|
|
408
|
+
/**
|
|
409
|
+
* Generates one complete turn and resolves the assembled result — the clean content,
|
|
410
|
+
* any separated reasoning, any tool calls, and any usage the wire reported.
|
|
411
|
+
*
|
|
412
|
+
* @remarks
|
|
413
|
+
* Sends `stream: false` and parses one JSON body. Content routes through a per-call
|
|
414
|
+
* think splitter, so the assembled content stays clean even where the daemon renders a
|
|
415
|
+
* thinking model's reasoning inline; the separated spans and any daemon-side
|
|
416
|
+
* `message.thinking` land on `thinking`. The caller's signal and the armed deadline
|
|
417
|
+
* both cancel the request, and the deadline is cleared once the body is read.
|
|
418
|
+
*
|
|
419
|
+
* @param messages - The conversation turns to send
|
|
420
|
+
* @param signal - The caller's bounding signal, folded with the armed deadline
|
|
421
|
+
* @param tools - The callable tools to advertise for this turn, when the caller passes any
|
|
422
|
+
* @param options - The per-call overrides, `think` and `schema` among them
|
|
423
|
+
* @returns The assembled result of the turn
|
|
424
|
+
* @throws {@link OllamaHTTPError} When the daemon answers a non-OK status.
|
|
425
|
+
*/
|
|
161
426
|
async generate(messages, signal, tools, options) {
|
|
162
427
|
const { response, timeout } = await this.#fetch(messages, false, signal, tools, options);
|
|
163
428
|
try {
|
|
164
|
-
const record = await
|
|
429
|
+
const record = await parseBody(response) ?? {};
|
|
165
430
|
const splitter = createThinkSplitter();
|
|
166
|
-
splitter.split(
|
|
431
|
+
splitter.split(extractContent(record));
|
|
167
432
|
splitter.flush();
|
|
168
|
-
const thinking =
|
|
169
|
-
return
|
|
433
|
+
const thinking = joinThinking(splitter, extractThinking(record));
|
|
434
|
+
return buildResult(splitter.content, thinking, extractTools(record), extractUsage(record));
|
|
170
435
|
} finally {
|
|
171
436
|
timeout.clear();
|
|
172
437
|
}
|
|
173
438
|
}
|
|
439
|
+
/**
|
|
440
|
+
* Streams one turn, yielding a channel-tagged delta per non-empty content or reasoning
|
|
441
|
+
* span and returning the assembled result when the stream completes.
|
|
442
|
+
*
|
|
443
|
+
* @remarks
|
|
444
|
+
* Sends `stream: true` and consumes NDJSON — one JSON object per newline-terminated
|
|
445
|
+
* line — pairing a streaming `TextDecoder` with the `NDJSONParser` so a record split
|
|
446
|
+
* across byte reads is reassembled. The returned result's content is the splitter's
|
|
447
|
+
* clean accumulation, beside any tool calls collected across lines and the usage the
|
|
448
|
+
* `done` line carries. A cancel mid-flight throws a `ProviderAbortError` carrying the
|
|
449
|
+
* partial assembled so far.
|
|
450
|
+
*
|
|
451
|
+
* @param messages - The conversation turns to send
|
|
452
|
+
* @param signal - The caller's bounding signal, folded with the armed deadline
|
|
453
|
+
* @param tools - The callable tools to advertise for this turn, when the caller passes any
|
|
454
|
+
* @param options - The per-call overrides, `think` and `schema` among them
|
|
455
|
+
* @returns The assembled result of the turn, after the last delta
|
|
456
|
+
* @throws {@link OllamaHTTPError} When the daemon answers a non-OK status or a `null` body.
|
|
457
|
+
*/
|
|
174
458
|
async *stream(messages, signal, tools, options) {
|
|
175
459
|
const { response, timeout, combined } = await this.#fetch(messages, true, signal, tools, options);
|
|
176
460
|
const body = response.body;
|
|
@@ -182,54 +466,63 @@ var OllamaProvider = class {
|
|
|
182
466
|
const decoder = new TextDecoder();
|
|
183
467
|
const parser = createNDJSONParser();
|
|
184
468
|
const splitter = createThinkSplitter();
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
calls: [],
|
|
189
|
-
usage: void 0
|
|
190
|
-
};
|
|
469
|
+
let wired = "";
|
|
470
|
+
const calls = [];
|
|
471
|
+
let usage;
|
|
191
472
|
try {
|
|
192
473
|
for (;;) {
|
|
193
474
|
const { value, done } = await reader.read();
|
|
194
475
|
if (done) break;
|
|
195
|
-
for (const record of parser.parse(decoder.decode(value, { stream: true })))
|
|
476
|
+
for (const record of parser.parse(decoder.decode(value, { stream: true }))) {
|
|
477
|
+
const increment = yield* this.#deltas(record, splitter, usage);
|
|
478
|
+
wired += increment.thinking;
|
|
479
|
+
calls.push(...increment.calls);
|
|
480
|
+
usage = increment.usage;
|
|
481
|
+
}
|
|
196
482
|
}
|
|
197
483
|
const decoderTail = decoder.decode();
|
|
198
|
-
for (const record of parser.parse(decoderTail.length > 0 ? `${decoderTail}\n` : "\n"))
|
|
484
|
+
for (const record of parser.parse(decoderTail.length > 0 ? `${decoderTail}\n` : "\n")) {
|
|
485
|
+
const increment = yield* this.#deltas(record, splitter, usage);
|
|
486
|
+
wired += increment.thinking;
|
|
487
|
+
calls.push(...increment.calls);
|
|
488
|
+
usage = increment.usage;
|
|
489
|
+
}
|
|
199
490
|
const tail = splitter.flush();
|
|
200
491
|
if (tail.length > 0) yield {
|
|
201
|
-
|
|
492
|
+
channel: "content",
|
|
202
493
|
text: tail
|
|
203
494
|
};
|
|
204
495
|
} catch (error) {
|
|
205
496
|
if (combined.aborted) {
|
|
206
497
|
splitter.flush();
|
|
207
|
-
throw new ProviderAbortError(
|
|
498
|
+
throw new ProviderAbortError(buildResult(splitter.content, joinThinking(splitter, wired), calls, usage));
|
|
208
499
|
}
|
|
209
500
|
throw error;
|
|
210
501
|
} finally {
|
|
211
502
|
try {
|
|
212
503
|
await reader.cancel();
|
|
213
504
|
} catch {}
|
|
214
|
-
parser.
|
|
505
|
+
parser.clear();
|
|
215
506
|
timeout.clear();
|
|
216
507
|
}
|
|
217
|
-
return
|
|
508
|
+
return buildResult(splitter.content, joinThinking(splitter, wired), calls, usage);
|
|
218
509
|
}
|
|
219
|
-
*#deltas(record,
|
|
220
|
-
const delta =
|
|
510
|
+
*#deltas(record, splitter, usage) {
|
|
511
|
+
const delta = splitter.split(extractContent(record));
|
|
221
512
|
if (delta.length > 0) yield {
|
|
222
|
-
|
|
513
|
+
channel: "content",
|
|
223
514
|
text: delta
|
|
224
515
|
};
|
|
225
|
-
const thinking =
|
|
516
|
+
const thinking = extractThinking(record);
|
|
226
517
|
if (thinking.length > 0) yield {
|
|
227
|
-
|
|
518
|
+
channel: "thinking",
|
|
228
519
|
text: thinking
|
|
229
520
|
};
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
521
|
+
return {
|
|
522
|
+
thinking,
|
|
523
|
+
calls: extractTools(record),
|
|
524
|
+
usage: Reflect.get(record, "done") === true ? extractUsage(record) : usage
|
|
525
|
+
};
|
|
233
526
|
}
|
|
234
527
|
async #fetch(messages, stream, signal, tools, options) {
|
|
235
528
|
const timeout = new Timeout({ ms: this.#timeout });
|
|
@@ -262,16 +555,6 @@ var OllamaProvider = class {
|
|
|
262
555
|
throw error;
|
|
263
556
|
}
|
|
264
557
|
}
|
|
265
|
-
async #parseBody(response) {
|
|
266
|
-
const text = await response.text();
|
|
267
|
-
if (text.length === 0) return {};
|
|
268
|
-
try {
|
|
269
|
-
const data = JSON.parse(text);
|
|
270
|
-
return isRecord(data) ? data : {};
|
|
271
|
-
} catch {
|
|
272
|
-
return {};
|
|
273
|
-
}
|
|
274
|
-
}
|
|
275
558
|
async #requestHeaders() {
|
|
276
559
|
const headers = { "Content-Type": "application/json" };
|
|
277
560
|
if (this.#headers !== void 0) for (const [key, value] of Object.entries(await this.#headers())) headers[key] = value;
|
|
@@ -280,7 +563,7 @@ var OllamaProvider = class {
|
|
|
280
563
|
#body(messages, stream, tools, options) {
|
|
281
564
|
return {
|
|
282
565
|
model: this.#model,
|
|
283
|
-
messages:
|
|
566
|
+
messages: mapMessages(messages),
|
|
284
567
|
stream,
|
|
285
568
|
keep_alive: this.#keepAlive,
|
|
286
569
|
think: options?.think ?? this.#think,
|
|
@@ -296,127 +579,58 @@ var OllamaProvider = class {
|
|
|
296
579
|
})) } : {}
|
|
297
580
|
};
|
|
298
581
|
}
|
|
299
|
-
#plain(messages) {
|
|
300
|
-
return messages.map((message) => ({
|
|
301
|
-
role: message.role,
|
|
302
|
-
content: message.content,
|
|
303
|
-
...message.calls !== void 0 && message.calls.length > 0 ? { tool_calls: message.calls.map((call) => ({ function: {
|
|
304
|
-
name: call.name,
|
|
305
|
-
arguments: call.arguments
|
|
306
|
-
} })) } : {},
|
|
307
|
-
...message.images !== void 0 && message.images.length > 0 ? { images: [...message.images] } : {}
|
|
308
|
-
}));
|
|
309
|
-
}
|
|
310
|
-
#result(content, thinking, tools, usage) {
|
|
311
|
-
const result = { content };
|
|
312
|
-
if (thinking.length > 0) result.thinking = thinking;
|
|
313
|
-
if (tools.length > 0) result.tools = tools;
|
|
314
|
-
if (usage !== void 0) result.usage = usage;
|
|
315
|
-
return result;
|
|
316
|
-
}
|
|
317
|
-
#content(record) {
|
|
318
|
-
const message = Reflect.get(record, "message");
|
|
319
|
-
if (!isRecord(message)) return "";
|
|
320
|
-
const content = Reflect.get(message, "content");
|
|
321
|
-
return isString(content) ? content : "";
|
|
322
|
-
}
|
|
323
|
-
#thinking(record) {
|
|
324
|
-
const message = Reflect.get(record, "message");
|
|
325
|
-
if (!isRecord(message)) return "";
|
|
326
|
-
const thinking = Reflect.get(message, "thinking");
|
|
327
|
-
return isString(thinking) ? thinking : "";
|
|
328
|
-
}
|
|
329
|
-
#thought(splitter, wired) {
|
|
330
|
-
if (splitter.thinking.length === 0) return wired;
|
|
331
|
-
if (wired.length === 0) return splitter.thinking;
|
|
332
|
-
return `${splitter.thinking}\n\n${wired}`;
|
|
333
|
-
}
|
|
334
|
-
#usage(record) {
|
|
335
|
-
const prompt = Reflect.get(record, "prompt_eval_count");
|
|
336
|
-
const completion = Reflect.get(record, "eval_count");
|
|
337
|
-
if (!isNumber(prompt) || !isNumber(completion)) return void 0;
|
|
338
|
-
return {
|
|
339
|
-
prompt,
|
|
340
|
-
completion,
|
|
341
|
-
total: prompt + completion
|
|
342
|
-
};
|
|
343
|
-
}
|
|
344
|
-
#tools(record) {
|
|
345
|
-
const message = Reflect.get(record, "message");
|
|
346
|
-
if (!isRecord(message)) return [];
|
|
347
|
-
const calls = Reflect.get(message, "tool_calls");
|
|
348
|
-
if (!Array.isArray(calls)) return [];
|
|
349
|
-
const out = [];
|
|
350
|
-
for (const entry of calls) {
|
|
351
|
-
if (!isRecord(entry)) continue;
|
|
352
|
-
const callable = Reflect.get(entry, "function");
|
|
353
|
-
if (!isRecord(callable)) continue;
|
|
354
|
-
const name = Reflect.get(callable, "name");
|
|
355
|
-
if (!isString(name)) continue;
|
|
356
|
-
const id = Reflect.get(entry, "id");
|
|
357
|
-
out.push({
|
|
358
|
-
id: isString(id) ? id : crypto.randomUUID(),
|
|
359
|
-
name,
|
|
360
|
-
arguments: this.#arguments(Reflect.get(callable, "arguments"))
|
|
361
|
-
});
|
|
362
|
-
}
|
|
363
|
-
return out;
|
|
364
|
-
}
|
|
365
|
-
#arguments(value) {
|
|
366
|
-
if (isRecord(value)) return value;
|
|
367
|
-
if (isString(value)) try {
|
|
368
|
-
const parsed = JSON.parse(value);
|
|
369
|
-
if (isRecord(parsed)) return parsed;
|
|
370
|
-
} catch {
|
|
371
|
-
return {};
|
|
372
|
-
}
|
|
373
|
-
return {};
|
|
374
|
-
}
|
|
375
582
|
};
|
|
376
583
|
//#endregion
|
|
377
584
|
//#region src/server/factories.ts
|
|
378
585
|
/**
|
|
379
|
-
*
|
|
586
|
+
* Creates a local Ollama inference provider — a {@link ProviderInterface} over the
|
|
380
587
|
* daemon's `POST /api/chat`, supporting non-streaming `generate` and streaming
|
|
381
588
|
* `stream`.
|
|
382
589
|
*
|
|
383
590
|
* @remarks
|
|
384
591
|
* Only `model` is required; `url` defaults to the local daemon, `keepAlive` to `'5m'`,
|
|
385
592
|
* `timeout` to `120_000`ms, and `options` is forwarded verbatim as sampling
|
|
386
|
-
* parameters (`temperature
|
|
593
|
+
* parameters (`temperature`, `seed`, and `num_predict`). Each call takes an
|
|
387
594
|
* `AbortSignal` to bound the request; a `stream` cancelled mid-flight throws a
|
|
388
595
|
* `ProviderAbortError` carrying the partial result.
|
|
389
596
|
*
|
|
390
597
|
* The optional `fetch` + `headers` form a transport seam (see {@link OllamaOptions}):
|
|
391
598
|
* point `url` at your own server, inject a custom `fetch`, and have `headers` attach a
|
|
392
599
|
* generated/obfuscated bearer token your server validates — so a browser runtime
|
|
393
|
-
* reaches the LLM through your middleware
|
|
394
|
-
* key. Both omitted ⇒
|
|
600
|
+
* reaches the LLM through your middleware without this library ever handling the real API
|
|
601
|
+
* key. Both omitted ⇒ the global `fetch` and only a JSON content type.
|
|
395
602
|
*
|
|
396
|
-
* The optional `format` is the provider's context-framing default — the
|
|
397
|
-
* level of `AgentContext`'s format cascade (
|
|
398
|
-
*
|
|
399
|
-
* provider's models prefer context sections framed (
|
|
400
|
-
* headers). It is
|
|
401
|
-
* `/api/chat` `format` wire parameter (structured output) — the
|
|
402
|
-
* the shared word. Omitted ⇒ the provider is
|
|
603
|
+
* The optional `format` is the provider's context-framing default — the provider-default
|
|
604
|
+
* level of `AgentContext`'s format cascade (beaten by a manager-options or per-item
|
|
605
|
+
* override, beating the managers' built-in framing), declaring how this
|
|
606
|
+
* provider's models prefer context sections framed (for example XML group wrappers vs. Markdown
|
|
607
|
+
* headers). It is exposed on the provider for the Agent's `build()` and is not Ollama's
|
|
608
|
+
* `/api/chat` `format` wire parameter (structured output) — the framing default and that
|
|
609
|
+
* wire parameter are unrelated despite the shared word. Omitted ⇒ the provider is
|
|
610
|
+
* framing-agnostic (core's built-in defaults).
|
|
403
611
|
*
|
|
404
612
|
* @param options - `model` (required), and optional `url` / `keepAlive` / `timeout` /
|
|
405
613
|
* `options` / `fetch` / `headers` / `format` (see {@link OllamaOptions})
|
|
406
614
|
* @returns A working {@link ProviderInterface} backed by Ollama
|
|
407
615
|
*
|
|
408
|
-
* @example
|
|
616
|
+
* @example createOllama + generate
|
|
409
617
|
* ```ts
|
|
410
618
|
* import { createAbort } from '@orkestrel/abort'
|
|
411
|
-
* import { createOllama } from '@
|
|
619
|
+
* import { createOllama } from '@orkestrel/ollama'
|
|
412
620
|
*
|
|
413
|
-
* const provider = createOllama({ model: 'qwen3.5:2b-q4_K_M' })
|
|
621
|
+
* const provider = createOllama({ model: 'qwen3.5:2b-q4_K_M', options: { temperature: 0 } })
|
|
414
622
|
* const abort = createAbort()
|
|
623
|
+
* const messages = [
|
|
624
|
+
* { id: '1', role: 'user', content: 'Summarize the release notes for version 2.0.' },
|
|
625
|
+
* ] as const
|
|
626
|
+
*
|
|
415
627
|
* const result = await provider.generate(messages, abort.signal)
|
|
628
|
+
* console.log(result.content)
|
|
629
|
+
* if (result.usage) charge(result.usage) // fold into a token budget
|
|
416
630
|
* ```
|
|
417
631
|
*
|
|
418
632
|
* @example
|
|
419
|
-
* Route through your own server with an obfuscated token
|
|
633
|
+
* Route through your own server with an obfuscated token:
|
|
420
634
|
* ```ts
|
|
421
635
|
* const provider = createOllama({
|
|
422
636
|
* model: 'qwen3.5:2b-q4_K_M',
|
|
@@ -428,7 +642,7 @@ var OllamaProvider = class {
|
|
|
428
642
|
*
|
|
429
643
|
* @example
|
|
430
644
|
* Declare a context-framing default — wrap the instructions section in an XML group (the
|
|
431
|
-
* provider-default level of `AgentContext`'s cascade;
|
|
645
|
+
* provider-default level of `AgentContext`'s cascade; not the wire `format`):
|
|
432
646
|
* ```ts
|
|
433
647
|
* const provider = createOllama({
|
|
434
648
|
* model: 'qwen3.5:2b-q4_K_M',
|
|
@@ -446,6 +660,6 @@ function createOllama(options) {
|
|
|
446
660
|
return new OllamaProvider(options);
|
|
447
661
|
}
|
|
448
662
|
//#endregion
|
|
449
|
-
export { DEFAULT_KEEP_ALIVE, DEFAULT_OLLAMA_URL, DEFAULT_PROVIDER_TIMEOUT, MAX_ERROR_BODY_LENGTH, OllamaHTTPError, OllamaProvider, createOllama, isOllamaHTTPError };
|
|
663
|
+
export { DEFAULT_KEEP_ALIVE, DEFAULT_OLLAMA_URL, DEFAULT_PROVIDER_TIMEOUT, MAX_ERROR_BODY_LENGTH, OllamaHTTPError, OllamaProvider, buildResult, createOllama, extractArguments, extractContent, extractThinking, extractTools, extractUsage, isOllamaHTTPError, joinThinking, mapMessages, parseBody };
|
|
450
664
|
|
|
451
665
|
//# sourceMappingURL=index.js.map
|