@orkestrel/ollama 0.0.13 → 0.0.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +6 -5
- package/dist/src/server/index.cjs +312 -142
- package/dist/src/server/index.cjs.map +1 -1
- package/dist/src/server/index.d.cts +261 -65
- package/dist/src/server/index.d.ts +261 -65
- package/dist/src/server/index.js +304 -143
- package/dist/src/server/index.js.map +1 -1
- package/package.json +19 -20
package/dist/src/server/index.js
CHANGED
|
@@ -1,28 +1,32 @@
|
|
|
1
|
+
import { isNumber, isRecord, isString, parseJSONAs } from "@orkestrel/contract";
|
|
1
2
|
import { ProviderAbortError, createThinkSplitter } from "@orkestrel/agent";
|
|
2
|
-
import { isNumber, isRecord, isString } from "@orkestrel/contract";
|
|
3
3
|
import { createNDJSONParser } from "@orkestrel/ndjson";
|
|
4
4
|
import { Timeout } from "@orkestrel/timeout";
|
|
5
5
|
//#region src/server/constants.ts
|
|
6
|
-
/**
|
|
6
|
+
/** Names the local Ollama daemon base URL assumed when `OllamaOptions.url` is omitted. */
|
|
7
7
|
var DEFAULT_OLLAMA_URL = "http://localhost:11434";
|
|
8
8
|
/**
|
|
9
|
-
*
|
|
9
|
+
* Names how long the model stays resident after a call when `OllamaOptions.keepAlive` is
|
|
10
10
|
* omitted — Ollama's own `keep_alive` default, expressed as a duration string.
|
|
11
|
+
*
|
|
12
|
+
* @remarks
|
|
13
|
+
* The name mirrors the Ollama `/api/chat` `keep_alive` field this value is sent as, so
|
|
14
|
+
* the constant, the `OllamaOptions.keepAlive` key, and the wire member read as one term.
|
|
11
15
|
*/
|
|
12
16
|
var DEFAULT_KEEP_ALIVE = "5m";
|
|
13
17
|
/**
|
|
14
|
-
*
|
|
18
|
+
* Names the per-call deadline in milliseconds when `OllamaOptions.timeout` is omitted —
|
|
15
19
|
* generous enough that a cold model load does not trip it.
|
|
16
20
|
*/
|
|
17
21
|
var DEFAULT_PROVIDER_TIMEOUT = 12e4;
|
|
18
22
|
/**
|
|
19
|
-
*
|
|
23
|
+
* Names the cap, in characters, on how much of a non-OK response body is
|
|
20
24
|
* incorporated into a thrown {@link OllamaHTTPError}'s message.
|
|
21
25
|
*
|
|
22
26
|
* @remarks
|
|
23
27
|
* Bounds the excerpt so a defensive proxy or a misbehaving daemon handing
|
|
24
28
|
* back an unbounded response body cannot inflate the thrown error's message
|
|
25
|
-
* without limit
|
|
29
|
+
* without limit. `2048` characters is generous enough to carry a
|
|
26
30
|
* useful diagnostic snippet while staying well short of any practical size
|
|
27
31
|
* concern.
|
|
28
32
|
*/
|
|
@@ -30,14 +34,15 @@ var MAX_ERROR_BODY_LENGTH = 2048;
|
|
|
30
34
|
//#endregion
|
|
31
35
|
//#region src/server/errors.ts
|
|
32
36
|
/**
|
|
33
|
-
*
|
|
37
|
+
* Represents an error thrown when the Ollama `/api/chat` HTTP transport fails.
|
|
34
38
|
*
|
|
35
39
|
* @remarks
|
|
36
|
-
* Carries the response `status` (0 when no
|
|
37
|
-
*
|
|
38
|
-
* failure sites — the non-OK status branch and the
|
|
39
|
-
* caller can branch on `error.
|
|
40
|
-
* a caught value with
|
|
40
|
+
* Carries the machine-readable `code` `'HTTP'` and the response `status` (0 when no
|
|
41
|
+
* HTTP response was received at all, for example a `null` body). Thrown by
|
|
42
|
+
* {@link OllamaProvider} at its HTTP failure sites — the non-OK status branch and the
|
|
43
|
+
* null-body branch — so a caller can branch on `error.code` and read `error.status`
|
|
44
|
+
* for the HTTP number instead of parsing the message. Narrow a caught value with
|
|
45
|
+
* {@link isOllamaHTTPError}.
|
|
41
46
|
*
|
|
42
47
|
* @example
|
|
43
48
|
* ```ts
|
|
@@ -51,6 +56,11 @@ var MAX_ERROR_BODY_LENGTH = 2048;
|
|
|
51
56
|
* ```
|
|
52
57
|
*/
|
|
53
58
|
var OllamaHTTPError = class extends Error {
|
|
59
|
+
/**
|
|
60
|
+
* Names the machine-readable condition this error reports — `'HTTP'`: an `/api/chat`
|
|
61
|
+
* transport, status, or body failure.
|
|
62
|
+
*/
|
|
63
|
+
code = "HTTP";
|
|
54
64
|
status;
|
|
55
65
|
constructor(message, status, options) {
|
|
56
66
|
super(message, options);
|
|
@@ -59,36 +69,252 @@ var OllamaHTTPError = class extends Error {
|
|
|
59
69
|
}
|
|
60
70
|
};
|
|
61
71
|
/**
|
|
62
|
-
*
|
|
72
|
+
* Checks whether a value is an {@link OllamaHTTPError}.
|
|
63
73
|
*
|
|
64
74
|
* @param value - The value to test
|
|
65
|
-
* @returns
|
|
75
|
+
* @returns True if `value` is an `OllamaHTTPError`; false otherwise
|
|
66
76
|
*/
|
|
67
77
|
function isOllamaHTTPError(value) {
|
|
68
78
|
return value instanceof OllamaHTTPError;
|
|
69
79
|
}
|
|
70
80
|
//#endregion
|
|
81
|
+
//#region src/server/helpers.ts
|
|
82
|
+
/**
|
|
83
|
+
* Maps conversation turns onto the `/api/chat` wire's minimal message shape.
|
|
84
|
+
*
|
|
85
|
+
* @remarks
|
|
86
|
+
* `tool_calls` is emitted only on a turn that replays them and `images` only on a
|
|
87
|
+
* multimodal turn, so an empty optional never reaches the wire.
|
|
88
|
+
*
|
|
89
|
+
* @param messages - The conversation turns to send
|
|
90
|
+
* @returns The wire `messages` array, one entry per turn, in order
|
|
91
|
+
*
|
|
92
|
+
* @example
|
|
93
|
+
* ```ts
|
|
94
|
+
* mapMessages([{ id: '1', role: 'user', content: 'Say hello.' }])
|
|
95
|
+
* // [{ role: 'user', content: 'Say hello.' }]
|
|
96
|
+
* ```
|
|
97
|
+
*/
|
|
98
|
+
function mapMessages(messages) {
|
|
99
|
+
return messages.map((message) => ({
|
|
100
|
+
role: message.role,
|
|
101
|
+
content: message.content,
|
|
102
|
+
...message.calls !== void 0 && message.calls.length > 0 ? { tool_calls: message.calls.map((call) => ({ function: {
|
|
103
|
+
name: call.name,
|
|
104
|
+
arguments: call.arguments
|
|
105
|
+
} })) } : {},
|
|
106
|
+
...message.images !== void 0 && message.images.length > 0 ? { images: [...message.images] } : {}
|
|
107
|
+
}));
|
|
108
|
+
}
|
|
109
|
+
/**
|
|
110
|
+
* Builds a provider result from a turn's content, reasoning, tool calls, and usage.
|
|
111
|
+
*
|
|
112
|
+
* @remarks
|
|
113
|
+
* Only the present optionals are set: no empty `thinking`, no empty `tools`, and no
|
|
114
|
+
* `usage` unless the wire reported one.
|
|
115
|
+
*
|
|
116
|
+
* @param content - The clean assistant content the splitter accumulated
|
|
117
|
+
* @param thinking - The joined reasoning, empty when the turn produced none
|
|
118
|
+
* @param tools - The tool calls collected across the turn
|
|
119
|
+
* @param usage - The token usage, or `undefined` when the wire reported none
|
|
120
|
+
* @returns The result carrying only its populated fields
|
|
121
|
+
*
|
|
122
|
+
* @example
|
|
123
|
+
* ```ts
|
|
124
|
+
* buildResult('ok', '', [], undefined) // { content: 'ok' }
|
|
125
|
+
* ```
|
|
126
|
+
*/
|
|
127
|
+
function buildResult(content, thinking, tools, usage) {
|
|
128
|
+
const result = { content };
|
|
129
|
+
if (thinking.length > 0) result.thinking = thinking;
|
|
130
|
+
if (tools.length > 0) result.tools = tools;
|
|
131
|
+
if (usage !== void 0) result.usage = usage;
|
|
132
|
+
return result;
|
|
133
|
+
}
|
|
134
|
+
/**
|
|
135
|
+
* Extracts the assistant text of one wire record.
|
|
136
|
+
*
|
|
137
|
+
* @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
|
|
138
|
+
* @returns The record's `message.content` when it is a string, else `''`
|
|
139
|
+
*
|
|
140
|
+
* @example
|
|
141
|
+
* ```ts
|
|
142
|
+
* extractContent({ message: { content: 'ok' } }) // 'ok'
|
|
143
|
+
* ```
|
|
144
|
+
*/
|
|
145
|
+
function extractContent(record) {
|
|
146
|
+
const message = Reflect.get(record, "message");
|
|
147
|
+
if (!isRecord(message)) return "";
|
|
148
|
+
const content = Reflect.get(message, "content");
|
|
149
|
+
return isString(content) ? content : "";
|
|
150
|
+
}
|
|
151
|
+
/**
|
|
152
|
+
* Extracts the daemon-side reasoning of one wire record.
|
|
153
|
+
*
|
|
154
|
+
* @remarks
|
|
155
|
+
* `message.thinking` is the `think: true` wire shape. It is read whatever the configured
|
|
156
|
+
* flag says, because a daemon may separate reasoning on its own.
|
|
157
|
+
*
|
|
158
|
+
* @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
|
|
159
|
+
* @returns The record's `message.thinking` when it is a string, else `''`
|
|
160
|
+
*
|
|
161
|
+
* @example
|
|
162
|
+
* ```ts
|
|
163
|
+
* extractThinking({ message: { thinking: 'weighing it' } }) // 'weighing it'
|
|
164
|
+
* ```
|
|
165
|
+
*/
|
|
166
|
+
function extractThinking(record) {
|
|
167
|
+
const message = Reflect.get(record, "message");
|
|
168
|
+
if (!isRecord(message)) return "";
|
|
169
|
+
const thinking = Reflect.get(message, "thinking");
|
|
170
|
+
return isString(thinking) ? thinking : "";
|
|
171
|
+
}
|
|
172
|
+
/**
|
|
173
|
+
* Joins a call's two reasoning carriers into the result's `thinking`.
|
|
174
|
+
*
|
|
175
|
+
* @param splitter - The per-call splitter holding the separated in-content spans
|
|
176
|
+
* @param wired - The accumulated wire-side `message.thinking` text
|
|
177
|
+
* @returns The two carriers separated by a blank line, or whichever one is non-empty
|
|
178
|
+
*
|
|
179
|
+
* @example
|
|
180
|
+
* ```ts
|
|
181
|
+
* joinThinking(createThinkSplitter(), 'from the wire') // 'from the wire'
|
|
182
|
+
* ```
|
|
183
|
+
*/
|
|
184
|
+
function joinThinking(splitter, wired) {
|
|
185
|
+
if (splitter.thinking.length === 0) return wired;
|
|
186
|
+
if (wired.length === 0) return splitter.thinking;
|
|
187
|
+
return `${splitter.thinking}\n\n${wired}`;
|
|
188
|
+
}
|
|
189
|
+
/**
|
|
190
|
+
* Extracts the token usage of one wire record.
|
|
191
|
+
*
|
|
192
|
+
* @remarks
|
|
193
|
+
* Both counts must be numbers, which is true of the non-stream body and the stream's
|
|
194
|
+
* `done: true` line. A delta line carries neither, so it yields `undefined`.
|
|
195
|
+
*
|
|
196
|
+
* @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
|
|
197
|
+
* @returns The `TokenUsage` shape, or `undefined` when either count is absent
|
|
198
|
+
*
|
|
199
|
+
* @example
|
|
200
|
+
* ```ts
|
|
201
|
+
* extractUsage({ prompt_eval_count: 3, eval_count: 4 })
|
|
202
|
+
* // { prompt: 3, completion: 4, total: 7 }
|
|
203
|
+
* ```
|
|
204
|
+
*/
|
|
205
|
+
function extractUsage(record) {
|
|
206
|
+
const prompt = Reflect.get(record, "prompt_eval_count");
|
|
207
|
+
const completion = Reflect.get(record, "eval_count");
|
|
208
|
+
if (!isNumber(prompt) || !isNumber(completion)) return void 0;
|
|
209
|
+
return {
|
|
210
|
+
prompt,
|
|
211
|
+
completion,
|
|
212
|
+
total: prompt + completion
|
|
213
|
+
};
|
|
214
|
+
}
|
|
215
|
+
/**
|
|
216
|
+
* Extracts the tool calls of one wire record's `message.tool_calls`.
|
|
217
|
+
*
|
|
218
|
+
* @remarks
|
|
219
|
+
* Each entry narrows to `{ id, name, arguments }`: the entry and its `function` must be
|
|
220
|
+
* records and `name` a string, else the entry is dropped. An id is minted when the wire
|
|
221
|
+
* omits one.
|
|
222
|
+
*
|
|
223
|
+
* @param record - One parsed `/api/chat` record — a non-stream body or an NDJSON line
|
|
224
|
+
* @returns The narrowed tool calls, empty when the record carries none
|
|
225
|
+
*
|
|
226
|
+
* @example
|
|
227
|
+
* ```ts
|
|
228
|
+
* extractTools({ message: { tool_calls: [{ function: { name: 'weather' } }] } })
|
|
229
|
+
* // [{ id: '…', name: 'weather', arguments: {} }]
|
|
230
|
+
* ```
|
|
231
|
+
*/
|
|
232
|
+
function extractTools(record) {
|
|
233
|
+
const message = Reflect.get(record, "message");
|
|
234
|
+
if (!isRecord(message)) return [];
|
|
235
|
+
const calls = Reflect.get(message, "tool_calls");
|
|
236
|
+
if (!Array.isArray(calls)) return [];
|
|
237
|
+
const out = [];
|
|
238
|
+
for (const entry of calls) {
|
|
239
|
+
if (!isRecord(entry)) continue;
|
|
240
|
+
const callable = Reflect.get(entry, "function");
|
|
241
|
+
if (!isRecord(callable)) continue;
|
|
242
|
+
const name = Reflect.get(callable, "name");
|
|
243
|
+
if (!isString(name)) continue;
|
|
244
|
+
const id = Reflect.get(entry, "id");
|
|
245
|
+
out.push({
|
|
246
|
+
id: isString(id) ? id : crypto.randomUUID(),
|
|
247
|
+
name,
|
|
248
|
+
arguments: extractArguments(Reflect.get(callable, "arguments"))
|
|
249
|
+
});
|
|
250
|
+
}
|
|
251
|
+
return out;
|
|
252
|
+
}
|
|
253
|
+
/**
|
|
254
|
+
* Extracts a wire `arguments` value as a record.
|
|
255
|
+
*
|
|
256
|
+
* @remarks
|
|
257
|
+
* Total: an object passes through, a JSON string is parsed when it yields a record, and
|
|
258
|
+
* a malformed string yields `{}` rather than throwing.
|
|
259
|
+
*
|
|
260
|
+
* @param value - The wire's `function.arguments` value, of unknown shape
|
|
261
|
+
* @returns The argument record, or `{}` when the value carries none
|
|
262
|
+
*
|
|
263
|
+
* @example
|
|
264
|
+
* ```ts
|
|
265
|
+
* extractArguments('{"city":"Oslo"}') // { city: 'Oslo' }
|
|
266
|
+
* ```
|
|
267
|
+
*/
|
|
268
|
+
function extractArguments(value) {
|
|
269
|
+
if (isRecord(value)) return value;
|
|
270
|
+
if (isString(value)) return parseJSONAs(value, isRecord) ?? {};
|
|
271
|
+
return {};
|
|
272
|
+
}
|
|
273
|
+
//#endregion
|
|
274
|
+
//#region src/server/parsers.ts
|
|
275
|
+
/**
|
|
276
|
+
* Parses a non-stream `/api/chat` response body into a wire record.
|
|
277
|
+
*
|
|
278
|
+
* @remarks
|
|
279
|
+
* Total by construction: an empty body, a body that is not JSON, and a body whose JSON is
|
|
280
|
+
* not an object all yield `undefined`, so a malformed daemon response never escapes as a
|
|
281
|
+
* `SyntaxError`. The call site supplies the empty-record default that reads as empty
|
|
282
|
+
* content and no usage.
|
|
283
|
+
*
|
|
284
|
+
* @param response - The 200-OK `/api/chat` response whose body is read as text
|
|
285
|
+
* @returns The parsed record, or `undefined` when the body is empty or malformed
|
|
286
|
+
*
|
|
287
|
+
* @example
|
|
288
|
+
* ```ts
|
|
289
|
+
* await parseBody(new Response('{"message":{"content":"ok"}}'))
|
|
290
|
+
* // { message: { content: 'ok' } }
|
|
291
|
+
* ```
|
|
292
|
+
*/
|
|
293
|
+
async function parseBody(response) {
|
|
294
|
+
return parseJSONAs(await response.text(), isRecord);
|
|
295
|
+
}
|
|
296
|
+
//#endregion
|
|
71
297
|
//#region src/server/OllamaProvider.ts
|
|
72
298
|
/**
|
|
73
|
-
*
|
|
299
|
+
* Implements the local Ollama inference boundary — a {@link ProviderInterface} over Ollama's
|
|
74
300
|
* `POST /api/chat`, both non-streaming (`generate`) and streaming NDJSON (`stream`).
|
|
75
301
|
*
|
|
76
302
|
* @remarks
|
|
77
303
|
* - **Wire protocol.** Posts `{ model, messages, stream, keep_alive, think }` plus
|
|
78
304
|
* passthrough sampling `options` and mapped function `tools`. The `think` flag is
|
|
79
|
-
* CONFIGURABLE
|
|
305
|
+
* CONFIGURABLE through {@link OllamaOptions.think} (default `false`). Non-stream parses
|
|
80
306
|
* one JSON body; stream consumes NDJSON (one JSON object per `\n`-terminated line) —
|
|
81
307
|
* deltas carry `message.content`, the final `done: true` line carries the token usage.
|
|
82
|
-
* - **Think separation
|
|
308
|
+
* - **Think separation.** The wire `think` flag is configurable
|
|
83
309
|
* ({@link OllamaOptions.think}, default `false`). With `think: true` a thinking model's
|
|
84
310
|
* daemon separates reasoning NATIVELY — returning it on the distinct `message.thinking`
|
|
85
|
-
* channel (read here
|
|
311
|
+
* channel (read here through `extractThinking`) instead of inline in `message.content`. EITHER
|
|
86
312
|
* way the per-call {@link ThinkSplitterInterface} is the defensive guarantee: a daemon
|
|
87
313
|
* may ignore `think: false` for a thinking model and inline `<think>` tags, so every
|
|
88
314
|
* content delta routes through the splitter, only CLEAN content is yielded / assembled,
|
|
89
315
|
* and the separated reasoning (plus any daemon-side `message.thinking` deltas) lands on
|
|
90
316
|
* `ProviderResult.thinking`, never in the conversation.
|
|
91
|
-
* - **Boundary narrowing
|
|
317
|
+
* - **Boundary narrowing.** Every wire value arrives as `unknown` and is
|
|
92
318
|
* narrowed through guards (`isRecord` / `isString` / `isNumber`) — never `as`. A
|
|
93
319
|
* missing / malformed field degrades to a sensible default (empty content, no
|
|
94
320
|
* usage, `{}` arguments), never a throw.
|
|
@@ -105,7 +331,8 @@ function isOllamaHTTPError(value) {
|
|
|
105
331
|
* `globalThis.fetch`) and {@link OllamaOptions.headers} is a per-request, possibly
|
|
106
332
|
* async header injector merged over the base `Content-Type` — so a browser runtime
|
|
107
333
|
* can route through the developer's own server with an obfuscated bearer token,
|
|
108
|
-
* without this library ever handling a real API key. Both omitted ⇒
|
|
334
|
+
* without this library ever handling a real API key. Both omitted ⇒ the global `fetch`
|
|
335
|
+
* and only a JSON content type.
|
|
109
336
|
* Orthogonal to the deadline: the hook is awaited inside `#fetch`'s try, so a hook
|
|
110
337
|
* rejection clears the armed timer like any other request failure.
|
|
111
338
|
*
|
|
@@ -116,8 +343,8 @@ function isOllamaHTTPError(value) {
|
|
|
116
343
|
* ```
|
|
117
344
|
*/
|
|
118
345
|
var OllamaProvider = class {
|
|
119
|
-
id = crypto.randomUUID();
|
|
120
346
|
name = "ollama";
|
|
347
|
+
#id;
|
|
121
348
|
#model;
|
|
122
349
|
#url;
|
|
123
350
|
#keepAlive;
|
|
@@ -128,6 +355,7 @@ var OllamaProvider = class {
|
|
|
128
355
|
#headers;
|
|
129
356
|
#format;
|
|
130
357
|
constructor(options) {
|
|
358
|
+
this.#id = crypto.randomUUID();
|
|
131
359
|
this.#model = options.model;
|
|
132
360
|
this.#url = options.url ?? "http://localhost:11434";
|
|
133
361
|
this.#keepAlive = options.keepAlive ?? "5m";
|
|
@@ -139,7 +367,17 @@ var OllamaProvider = class {
|
|
|
139
367
|
this.#format = options.format;
|
|
140
368
|
}
|
|
141
369
|
/**
|
|
142
|
-
*
|
|
370
|
+
* Exposes this instance's identity — a fresh `crypto.randomUUID()` minted at
|
|
371
|
+
* construction, satisfying the {@link ProviderInterface.id} contract member. A second
|
|
372
|
+
* provider built from identical options carries a distinct id.
|
|
373
|
+
*
|
|
374
|
+
* @returns The instance's minted identifier
|
|
375
|
+
*/
|
|
376
|
+
get id() {
|
|
377
|
+
return this.#id;
|
|
378
|
+
}
|
|
379
|
+
/**
|
|
380
|
+
* Exposes the provider's context-framing default — the PROVIDER-DEFAULT level of
|
|
143
381
|
* {@link import('@orkestrel/agent').AgentContextInterface.build}'s format cascade (it BEATS
|
|
144
382
|
* the managers' built-in framing, is BEATEN by a manager-options or per-item override).
|
|
145
383
|
* Satisfies the OPTIONAL {@link ProviderInterface.format} contract member: `undefined`
|
|
@@ -153,7 +391,7 @@ var OllamaProvider = class {
|
|
|
153
391
|
* structured-output `format` wire parameter — that one IS sent in `#body`, but only when
|
|
154
392
|
* a per-call `ProviderStreamOptions.schema` is supplied; only the word collides.
|
|
155
393
|
*
|
|
156
|
-
* @returns The configured {@link
|
|
394
|
+
* @returns The configured {@link ContextFormat}, or `undefined` when none
|
|
157
395
|
*/
|
|
158
396
|
get format() {
|
|
159
397
|
return this.#format;
|
|
@@ -161,12 +399,12 @@ var OllamaProvider = class {
|
|
|
161
399
|
async generate(messages, signal, tools, options) {
|
|
162
400
|
const { response, timeout } = await this.#fetch(messages, false, signal, tools, options);
|
|
163
401
|
try {
|
|
164
|
-
const record = await
|
|
402
|
+
const record = await parseBody(response) ?? {};
|
|
165
403
|
const splitter = createThinkSplitter();
|
|
166
|
-
splitter.split(
|
|
404
|
+
splitter.split(extractContent(record));
|
|
167
405
|
splitter.flush();
|
|
168
|
-
const thinking =
|
|
169
|
-
return
|
|
406
|
+
const thinking = joinThinking(splitter, extractThinking(record));
|
|
407
|
+
return buildResult(splitter.content, thinking, extractTools(record), extractUsage(record));
|
|
170
408
|
} finally {
|
|
171
409
|
timeout.clear();
|
|
172
410
|
}
|
|
@@ -182,54 +420,63 @@ var OllamaProvider = class {
|
|
|
182
420
|
const decoder = new TextDecoder();
|
|
183
421
|
const parser = createNDJSONParser();
|
|
184
422
|
const splitter = createThinkSplitter();
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
calls: [],
|
|
189
|
-
usage: void 0
|
|
190
|
-
};
|
|
423
|
+
let wired = "";
|
|
424
|
+
const calls = [];
|
|
425
|
+
let usage;
|
|
191
426
|
try {
|
|
192
427
|
for (;;) {
|
|
193
428
|
const { value, done } = await reader.read();
|
|
194
429
|
if (done) break;
|
|
195
|
-
for (const record of parser.parse(decoder.decode(value, { stream: true })))
|
|
430
|
+
for (const record of parser.parse(decoder.decode(value, { stream: true }))) {
|
|
431
|
+
const increment = yield* this.#deltas(record, splitter, usage);
|
|
432
|
+
wired += increment.thinking;
|
|
433
|
+
calls.push(...increment.calls);
|
|
434
|
+
usage = increment.usage;
|
|
435
|
+
}
|
|
196
436
|
}
|
|
197
437
|
const decoderTail = decoder.decode();
|
|
198
|
-
for (const record of parser.parse(decoderTail.length > 0 ? `${decoderTail}\n` : "\n"))
|
|
438
|
+
for (const record of parser.parse(decoderTail.length > 0 ? `${decoderTail}\n` : "\n")) {
|
|
439
|
+
const increment = yield* this.#deltas(record, splitter, usage);
|
|
440
|
+
wired += increment.thinking;
|
|
441
|
+
calls.push(...increment.calls);
|
|
442
|
+
usage = increment.usage;
|
|
443
|
+
}
|
|
199
444
|
const tail = splitter.flush();
|
|
200
445
|
if (tail.length > 0) yield {
|
|
201
|
-
|
|
446
|
+
channel: "content",
|
|
202
447
|
text: tail
|
|
203
448
|
};
|
|
204
449
|
} catch (error) {
|
|
205
450
|
if (combined.aborted) {
|
|
206
451
|
splitter.flush();
|
|
207
|
-
throw new ProviderAbortError(
|
|
452
|
+
throw new ProviderAbortError(buildResult(splitter.content, joinThinking(splitter, wired), calls, usage));
|
|
208
453
|
}
|
|
209
454
|
throw error;
|
|
210
455
|
} finally {
|
|
211
456
|
try {
|
|
212
457
|
await reader.cancel();
|
|
213
458
|
} catch {}
|
|
214
|
-
parser.
|
|
459
|
+
parser.clear();
|
|
215
460
|
timeout.clear();
|
|
216
461
|
}
|
|
217
|
-
return
|
|
462
|
+
return buildResult(splitter.content, joinThinking(splitter, wired), calls, usage);
|
|
218
463
|
}
|
|
219
|
-
*#deltas(record,
|
|
220
|
-
const delta =
|
|
464
|
+
*#deltas(record, splitter, usage) {
|
|
465
|
+
const delta = splitter.split(extractContent(record));
|
|
221
466
|
if (delta.length > 0) yield {
|
|
222
|
-
|
|
467
|
+
channel: "content",
|
|
223
468
|
text: delta
|
|
224
469
|
};
|
|
225
|
-
const thinking =
|
|
470
|
+
const thinking = extractThinking(record);
|
|
226
471
|
if (thinking.length > 0) yield {
|
|
227
|
-
|
|
472
|
+
channel: "thinking",
|
|
228
473
|
text: thinking
|
|
229
474
|
};
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
475
|
+
return {
|
|
476
|
+
thinking,
|
|
477
|
+
calls: extractTools(record),
|
|
478
|
+
usage: Reflect.get(record, "done") === true ? extractUsage(record) : usage
|
|
479
|
+
};
|
|
233
480
|
}
|
|
234
481
|
async #fetch(messages, stream, signal, tools, options) {
|
|
235
482
|
const timeout = new Timeout({ ms: this.#timeout });
|
|
@@ -262,16 +509,6 @@ var OllamaProvider = class {
|
|
|
262
509
|
throw error;
|
|
263
510
|
}
|
|
264
511
|
}
|
|
265
|
-
async #parseBody(response) {
|
|
266
|
-
const text = await response.text();
|
|
267
|
-
if (text.length === 0) return {};
|
|
268
|
-
try {
|
|
269
|
-
const data = JSON.parse(text);
|
|
270
|
-
return isRecord(data) ? data : {};
|
|
271
|
-
} catch {
|
|
272
|
-
return {};
|
|
273
|
-
}
|
|
274
|
-
}
|
|
275
512
|
async #requestHeaders() {
|
|
276
513
|
const headers = { "Content-Type": "application/json" };
|
|
277
514
|
if (this.#headers !== void 0) for (const [key, value] of Object.entries(await this.#headers())) headers[key] = value;
|
|
@@ -280,7 +517,7 @@ var OllamaProvider = class {
|
|
|
280
517
|
#body(messages, stream, tools, options) {
|
|
281
518
|
return {
|
|
282
519
|
model: this.#model,
|
|
283
|
-
messages:
|
|
520
|
+
messages: mapMessages(messages),
|
|
284
521
|
stream,
|
|
285
522
|
keep_alive: this.#keepAlive,
|
|
286
523
|
think: options?.think ?? this.#think,
|
|
@@ -296,94 +533,18 @@ var OllamaProvider = class {
|
|
|
296
533
|
})) } : {}
|
|
297
534
|
};
|
|
298
535
|
}
|
|
299
|
-
#plain(messages) {
|
|
300
|
-
return messages.map((message) => ({
|
|
301
|
-
role: message.role,
|
|
302
|
-
content: message.content,
|
|
303
|
-
...message.calls !== void 0 && message.calls.length > 0 ? { tool_calls: message.calls.map((call) => ({ function: {
|
|
304
|
-
name: call.name,
|
|
305
|
-
arguments: call.arguments
|
|
306
|
-
} })) } : {},
|
|
307
|
-
...message.images !== void 0 && message.images.length > 0 ? { images: [...message.images] } : {}
|
|
308
|
-
}));
|
|
309
|
-
}
|
|
310
|
-
#result(content, thinking, tools, usage) {
|
|
311
|
-
const result = { content };
|
|
312
|
-
if (thinking.length > 0) result.thinking = thinking;
|
|
313
|
-
if (tools.length > 0) result.tools = tools;
|
|
314
|
-
if (usage !== void 0) result.usage = usage;
|
|
315
|
-
return result;
|
|
316
|
-
}
|
|
317
|
-
#content(record) {
|
|
318
|
-
const message = Reflect.get(record, "message");
|
|
319
|
-
if (!isRecord(message)) return "";
|
|
320
|
-
const content = Reflect.get(message, "content");
|
|
321
|
-
return isString(content) ? content : "";
|
|
322
|
-
}
|
|
323
|
-
#thinking(record) {
|
|
324
|
-
const message = Reflect.get(record, "message");
|
|
325
|
-
if (!isRecord(message)) return "";
|
|
326
|
-
const thinking = Reflect.get(message, "thinking");
|
|
327
|
-
return isString(thinking) ? thinking : "";
|
|
328
|
-
}
|
|
329
|
-
#thought(splitter, wired) {
|
|
330
|
-
if (splitter.thinking.length === 0) return wired;
|
|
331
|
-
if (wired.length === 0) return splitter.thinking;
|
|
332
|
-
return `${splitter.thinking}\n\n${wired}`;
|
|
333
|
-
}
|
|
334
|
-
#usage(record) {
|
|
335
|
-
const prompt = Reflect.get(record, "prompt_eval_count");
|
|
336
|
-
const completion = Reflect.get(record, "eval_count");
|
|
337
|
-
if (!isNumber(prompt) || !isNumber(completion)) return void 0;
|
|
338
|
-
return {
|
|
339
|
-
prompt,
|
|
340
|
-
completion,
|
|
341
|
-
total: prompt + completion
|
|
342
|
-
};
|
|
343
|
-
}
|
|
344
|
-
#tools(record) {
|
|
345
|
-
const message = Reflect.get(record, "message");
|
|
346
|
-
if (!isRecord(message)) return [];
|
|
347
|
-
const calls = Reflect.get(message, "tool_calls");
|
|
348
|
-
if (!Array.isArray(calls)) return [];
|
|
349
|
-
const out = [];
|
|
350
|
-
for (const entry of calls) {
|
|
351
|
-
if (!isRecord(entry)) continue;
|
|
352
|
-
const callable = Reflect.get(entry, "function");
|
|
353
|
-
if (!isRecord(callable)) continue;
|
|
354
|
-
const name = Reflect.get(callable, "name");
|
|
355
|
-
if (!isString(name)) continue;
|
|
356
|
-
const id = Reflect.get(entry, "id");
|
|
357
|
-
out.push({
|
|
358
|
-
id: isString(id) ? id : crypto.randomUUID(),
|
|
359
|
-
name,
|
|
360
|
-
arguments: this.#arguments(Reflect.get(callable, "arguments"))
|
|
361
|
-
});
|
|
362
|
-
}
|
|
363
|
-
return out;
|
|
364
|
-
}
|
|
365
|
-
#arguments(value) {
|
|
366
|
-
if (isRecord(value)) return value;
|
|
367
|
-
if (isString(value)) try {
|
|
368
|
-
const parsed = JSON.parse(value);
|
|
369
|
-
if (isRecord(parsed)) return parsed;
|
|
370
|
-
} catch {
|
|
371
|
-
return {};
|
|
372
|
-
}
|
|
373
|
-
return {};
|
|
374
|
-
}
|
|
375
536
|
};
|
|
376
537
|
//#endregion
|
|
377
538
|
//#region src/server/factories.ts
|
|
378
539
|
/**
|
|
379
|
-
*
|
|
540
|
+
* Creates a local Ollama inference provider — a {@link ProviderInterface} over the
|
|
380
541
|
* daemon's `POST /api/chat`, supporting non-streaming `generate` and streaming
|
|
381
542
|
* `stream`.
|
|
382
543
|
*
|
|
383
544
|
* @remarks
|
|
384
545
|
* Only `model` is required; `url` defaults to the local daemon, `keepAlive` to `'5m'`,
|
|
385
546
|
* `timeout` to `120_000`ms, and `options` is forwarded verbatim as sampling
|
|
386
|
-
* parameters (`temperature
|
|
547
|
+
* parameters (`temperature`, `seed`, and `num_predict`). Both calls take an
|
|
387
548
|
* `AbortSignal` to bound the request; a `stream` cancelled mid-flight throws a
|
|
388
549
|
* `ProviderAbortError` carrying the partial result.
|
|
389
550
|
*
|
|
@@ -391,12 +552,12 @@ var OllamaProvider = class {
|
|
|
391
552
|
* point `url` at your own server, inject a custom `fetch`, and have `headers` attach a
|
|
392
553
|
* generated/obfuscated bearer token your server validates — so a browser runtime
|
|
393
554
|
* reaches the LLM through your middleware WITHOUT this library ever handling the real API
|
|
394
|
-
* key. Both omitted ⇒
|
|
555
|
+
* key. Both omitted ⇒ the global `fetch` and only a JSON content type.
|
|
395
556
|
*
|
|
396
557
|
* The optional `format` is the provider's context-framing default — the PROVIDER-DEFAULT
|
|
397
|
-
* level of `AgentContext`'s format cascade (
|
|
398
|
-
*
|
|
399
|
-
* provider's models prefer context sections framed (
|
|
558
|
+
* level of `AgentContext`'s format cascade (beaten by a manager-options or per-item
|
|
559
|
+
* override, beating the managers' built-in framing), declaring how this
|
|
560
|
+
* provider's models prefer context sections framed (for example XML group wrappers vs. Markdown
|
|
400
561
|
* headers). It is EXPOSED on the provider for the Agent's `build()` and is NOT Ollama's
|
|
401
562
|
* `/api/chat` `format` wire parameter (structured output) — the two are unrelated despite
|
|
402
563
|
* the shared word. Omitted ⇒ the provider is framing-agnostic (core's built-in defaults).
|
|
@@ -408,7 +569,7 @@ var OllamaProvider = class {
|
|
|
408
569
|
* @example
|
|
409
570
|
* ```ts
|
|
410
571
|
* import { createAbort } from '@orkestrel/abort'
|
|
411
|
-
* import { createOllama } from '@
|
|
572
|
+
* import { createOllama } from '@orkestrel/ollama'
|
|
412
573
|
*
|
|
413
574
|
* const provider = createOllama({ model: 'qwen3.5:2b-q4_K_M' })
|
|
414
575
|
* const abort = createAbort()
|
|
@@ -416,7 +577,7 @@ var OllamaProvider = class {
|
|
|
416
577
|
* ```
|
|
417
578
|
*
|
|
418
579
|
* @example
|
|
419
|
-
* Route through your own server with an obfuscated token
|
|
580
|
+
* Route through your own server with an obfuscated token:
|
|
420
581
|
* ```ts
|
|
421
582
|
* const provider = createOllama({
|
|
422
583
|
* model: 'qwen3.5:2b-q4_K_M',
|
|
@@ -446,6 +607,6 @@ function createOllama(options) {
|
|
|
446
607
|
return new OllamaProvider(options);
|
|
447
608
|
}
|
|
448
609
|
//#endregion
|
|
449
|
-
export { DEFAULT_KEEP_ALIVE, DEFAULT_OLLAMA_URL, DEFAULT_PROVIDER_TIMEOUT, MAX_ERROR_BODY_LENGTH, OllamaHTTPError, OllamaProvider, createOllama, isOllamaHTTPError };
|
|
610
|
+
export { DEFAULT_KEEP_ALIVE, DEFAULT_OLLAMA_URL, DEFAULT_PROVIDER_TIMEOUT, MAX_ERROR_BODY_LENGTH, OllamaHTTPError, OllamaProvider, buildResult, createOllama, extractArguments, extractContent, extractThinking, extractTools, extractUsage, isOllamaHTTPError, joinThinking, mapMessages, parseBody };
|
|
450
611
|
|
|
451
612
|
//# sourceMappingURL=index.js.map
|