@orkestrel/ollama 0.0.4 → 0.0.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -0,0 +1,346 @@
|
|
|
1
|
+
import { ContextFormatInterface } from '@orkestrel/agent';
|
|
2
|
+
import { MessageInterface } from '@orkestrel/agent';
|
|
3
|
+
import { ProviderDelta } from '@orkestrel/agent';
|
|
4
|
+
import { ProviderInterface } from '@orkestrel/agent';
|
|
5
|
+
import { ProviderResult } from '@orkestrel/agent';
|
|
6
|
+
import { ProviderStreamOptions } from '@orkestrel/agent';
|
|
7
|
+
import { TimeoutInterface } from '@orkestrel/timeout';
|
|
8
|
+
import { ToolDefinition } from '@orkestrel/agent';
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Create a local Ollama inference provider — a {@link ProviderInterface} over the
|
|
12
|
+
* daemon's `POST /api/chat`, supporting non-streaming `generate` and streaming
|
|
13
|
+
* `stream`.
|
|
14
|
+
*
|
|
15
|
+
* @remarks
|
|
16
|
+
* Only `model` is required; `url` defaults to the local daemon, `keepAlive` to `'5m'`,
|
|
17
|
+
* `timeout` to `120_000`ms, and `options` is forwarded verbatim as sampling
|
|
18
|
+
* parameters (`temperature` / `seed` / `num_predict` / …). Both calls take an
|
|
19
|
+
* `AbortSignal` to bound the request; a `stream` cancelled mid-flight throws a
|
|
20
|
+
* `ProviderAbortError` carrying the partial result.
|
|
21
|
+
*
|
|
22
|
+
* The optional `fetch` + `headers` form a transport seam (see {@link OllamaOptions}):
|
|
23
|
+
* point `url` at your own server, inject a custom `fetch`, and have `headers` attach a
|
|
24
|
+
* generated/obfuscated bearer token your server validates — so a browser runtime
|
|
25
|
+
* reaches the LLM through your middleware WITHOUT this library ever handling the real API
|
|
26
|
+
* key. Both omitted ⇒ today's behaviour (the global `fetch`, only a JSON content type).
|
|
27
|
+
*
|
|
28
|
+
* The optional `format` is the provider's context-framing default — the PROVIDER-DEFAULT
|
|
29
|
+
* level of `AgentContext`'s format cascade (see [agents.md]; beaten by a manager-options
|
|
30
|
+
* or per-item override, beating the managers' built-in framing), declaring how this
|
|
31
|
+
* provider's models prefer context sections framed (e.g. XML group wrappers vs. Markdown
|
|
32
|
+
* headers). It is EXPOSED on the provider for the Agent's `build()` and is NOT Ollama's
|
|
33
|
+
* `/api/chat` `format` wire parameter (structured output) — the two are unrelated despite
|
|
34
|
+
* the shared word. Omitted ⇒ the provider is framing-agnostic (core's built-in defaults).
|
|
35
|
+
*
|
|
36
|
+
* @param options - `model` (required), and optional `url` / `keepAlive` / `timeout` /
|
|
37
|
+
* `options` / `fetch` / `headers` / `format` (see {@link OllamaOptions})
|
|
38
|
+
* @returns A working {@link ProviderInterface} backed by Ollama
|
|
39
|
+
*
|
|
40
|
+
* @example
|
|
41
|
+
* ```ts
|
|
42
|
+
* import { createAbort } from '@orkestrel/abort'
|
|
43
|
+
* import { createOllama } from '@src/server'
|
|
44
|
+
*
|
|
45
|
+
* const provider = createOllama({ model: 'qwen3.5:2b-q4_K_M' })
|
|
46
|
+
* const abort = createAbort()
|
|
47
|
+
* const result = await provider.generate(messages, abort.signal)
|
|
48
|
+
* ```
|
|
49
|
+
*
|
|
50
|
+
* @example
|
|
51
|
+
* Route through your own server with an obfuscated token (deployment scenario S2):
|
|
52
|
+
* ```ts
|
|
53
|
+
* const provider = createOllama({
|
|
54
|
+
* model: 'qwen3.5:2b-q4_K_M',
|
|
55
|
+
* url: 'https://my-app.example.com/llm', // your server, not the daemon
|
|
56
|
+
* fetch: myFetch, // optional custom transport
|
|
57
|
+
* headers: () => ({ authorization: `Bearer ${myToken}` }), // your server validates this
|
|
58
|
+
* })
|
|
59
|
+
* ```
|
|
60
|
+
*
|
|
61
|
+
* @example
|
|
62
|
+
* Declare a context-framing default — wrap the instructions section in an XML group (the
|
|
63
|
+
* provider-default level of `AgentContext`'s cascade; NOT the wire `format`):
|
|
64
|
+
* ```ts
|
|
65
|
+
* const provider = createOllama({
|
|
66
|
+
* model: 'qwen3.5:2b-q4_K_M',
|
|
67
|
+
* format: {
|
|
68
|
+
* instructions: {
|
|
69
|
+
* open: '<instructions>',
|
|
70
|
+
* render: (i) => `<instruction>${i.content}</instruction>`,
|
|
71
|
+
* close: '</instructions>',
|
|
72
|
+
* },
|
|
73
|
+
* },
|
|
74
|
+
* })
|
|
75
|
+
* ```
|
|
76
|
+
*/
|
|
77
|
+
export declare function createOllama(options: OllamaOptions): ProviderInterface;
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* How long the model stays resident after a call when `OllamaOptions.keepAlive` is
|
|
81
|
+
* omitted — Ollama's own `keep_alive` default, expressed as a duration string.
|
|
82
|
+
*/
|
|
83
|
+
export declare const DEFAULT_KEEP_ALIVE = "5m";
|
|
84
|
+
|
|
85
|
+
/** The local Ollama daemon base URL assumed when `OllamaOptions.url` is omitted. */
|
|
86
|
+
export declare const DEFAULT_OLLAMA_URL = "http://localhost:11434";
|
|
87
|
+
|
|
88
|
+
/**
|
|
89
|
+
* The per-call deadline in milliseconds when `OllamaOptions.timeout` is omitted —
|
|
90
|
+
* generous enough that a cold model load does not trip it.
|
|
91
|
+
*/
|
|
92
|
+
export declare const DEFAULT_PROVIDER_TIMEOUT = 120000;
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Whether a value is an {@link OllamaHTTPError}.
|
|
96
|
+
*
|
|
97
|
+
* @param value - The value to test
|
|
98
|
+
* @returns `true` when `value` is an `OllamaHTTPError`
|
|
99
|
+
*/
|
|
100
|
+
export declare function isOllamaHTTPError(value: unknown): value is OllamaHTTPError;
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* The cap, in characters, on how much of a non-OK response body is
|
|
104
|
+
* incorporated into a thrown {@link OllamaHTTPError}'s message.
|
|
105
|
+
*
|
|
106
|
+
* @remarks
|
|
107
|
+
* Bounds the excerpt so a defensive proxy or a misbehaving daemon handing
|
|
108
|
+
* back an unbounded response body cannot inflate the thrown error's message
|
|
109
|
+
* without limit (§14). `2048` characters is generous enough to carry a
|
|
110
|
+
* useful diagnostic snippet while staying well short of any practical size
|
|
111
|
+
* concern.
|
|
112
|
+
*/
|
|
113
|
+
export declare const MAX_ERROR_BODY_LENGTH = 2048;
|
|
114
|
+
|
|
115
|
+
/**
|
|
116
|
+
* An error thrown when the Ollama `/api/chat` HTTP transport fails.
|
|
117
|
+
*
|
|
118
|
+
* @remarks
|
|
119
|
+
* Carries the response `status` (0 when no HTTP response was received at all,
|
|
120
|
+
* e.g. a `null` body). Thrown by {@link OllamaProvider} at its two HTTP
|
|
121
|
+
* failure sites — the non-OK status branch and the null-body branch — so a
|
|
122
|
+
* caller can branch on `error.status` instead of parsing the message. Narrow
|
|
123
|
+
* a caught value with {@link isOllamaHTTPError}.
|
|
124
|
+
*
|
|
125
|
+
* @example
|
|
126
|
+
* ```ts
|
|
127
|
+
* try {
|
|
128
|
+
* await provider.generate(messages, signal)
|
|
129
|
+
* } catch (error) {
|
|
130
|
+
* if (isOllamaHTTPError(error) && error.status === 404) {
|
|
131
|
+
* // the configured model isn't pulled
|
|
132
|
+
* }
|
|
133
|
+
* }
|
|
134
|
+
* ```
|
|
135
|
+
*/
|
|
136
|
+
export declare class OllamaHTTPError extends Error {
|
|
137
|
+
readonly status: number;
|
|
138
|
+
constructor(message: string, status: number, options?: {
|
|
139
|
+
readonly cause?: unknown;
|
|
140
|
+
});
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/**
|
|
144
|
+
* Options for `createOllama` — the local Ollama backend's configuration.
|
|
145
|
+
*
|
|
146
|
+
* @remarks
|
|
147
|
+
* Only `model` is required. `url` defaults to the local daemon, `keepAlive` controls
|
|
148
|
+
* how long the model stays resident after a call, `timeout` is the per-call deadline
|
|
149
|
+
* in milliseconds, and `options` is a passthrough bag of sampling parameters
|
|
150
|
+
* (`temperature` / `seed` / `num_predict` / …) forwarded verbatim to the wire.
|
|
151
|
+
*
|
|
152
|
+
* The optional `fetch` + `headers` form a **transport seam**: by default the provider
|
|
153
|
+
* talks straight to a local daemon over `globalThis.fetch` with only a JSON content
|
|
154
|
+
* type, but a browser-side runtime can inject a custom transport AND a dynamic header
|
|
155
|
+
* (e.g. an obfuscated bearer token) so requests route through the developer's OWN
|
|
156
|
+
* server, which validates that header and forwards to the real LLM. Your app never
|
|
157
|
+
* holds a real API key — the real key lives only on the developer's server; the
|
|
158
|
+
* `headers` hook supplies whatever short-lived/obfuscated token that server expects.
|
|
159
|
+
*/
|
|
160
|
+
export declare interface OllamaOptions {
|
|
161
|
+
readonly model: string;
|
|
162
|
+
/** The daemon base URL; defaults to `'http://localhost:11434'`. */
|
|
163
|
+
readonly url?: string;
|
|
164
|
+
/** How long the model stays resident after a call; defaults to `'5m'`. */
|
|
165
|
+
readonly keepAlive?: string | number;
|
|
166
|
+
/** The per-call deadline in milliseconds; defaults to `120_000`. */
|
|
167
|
+
readonly timeout?: number;
|
|
168
|
+
/** Passthrough sampling options (`temperature` / `seed` / `num_predict` / …). */
|
|
169
|
+
readonly options?: Readonly<Record<string, unknown>>;
|
|
170
|
+
/**
|
|
171
|
+
* The `/api/chat` `think` wire flag; defaults to `false`. When `true`, a thinking-capable
|
|
172
|
+
* model (e.g. `qwen3`) separates its reasoning NATIVELY at the wire — the daemon returns it
|
|
173
|
+
* on the distinct `message.thinking` channel (surfaced on `ProviderResult.thinking`) rather
|
|
174
|
+
* than inline in `message.content`. The default stays `false` so a general-purpose provider
|
|
175
|
+
* is backward-compatible and immediate for non-thinking models; the per-call ThinkSplitter
|
|
176
|
+
* remains the defensive fallback for daemons/models that still inline `<think>` tags either
|
|
177
|
+
* way. Set it `true` for a thinking model whose reasoning you intend to DISPLAY separately.
|
|
178
|
+
*/
|
|
179
|
+
readonly think?: boolean;
|
|
180
|
+
/**
|
|
181
|
+
* A custom `fetch` implementation for every request; defaults to
|
|
182
|
+
* `globalThis.fetch`. Lets a runtime inject its own transport (a browser fetch
|
|
183
|
+
* pointed at the developer's server, an instrumented wrapper, …) without changing
|
|
184
|
+
* the wire protocol. Omitted ⇒ the global `fetch`.
|
|
185
|
+
*/
|
|
186
|
+
readonly fetch?: typeof globalThis.fetch;
|
|
187
|
+
/**
|
|
188
|
+
* A dynamic, possibly-async header injector called once per request; its returned
|
|
189
|
+
* headers are merged into the request on top of the base `Content-Type`. Use it to
|
|
190
|
+
* attach an authorization header — e.g. an obfuscated/generated bearer token the
|
|
191
|
+
* developer's server validates before relaying to the real LLM — so a browser
|
|
192
|
+
* runtime can authenticate WITHOUT your app ever handling a real API key. Async so a
|
|
193
|
+
* token can be refreshed/fetched per call. A returned `Content-Type` overrides the
|
|
194
|
+
* default; other headers add to it. Omitted ⇒ only `Content-Type: application/json`.
|
|
195
|
+
*/
|
|
196
|
+
readonly headers?: () => Record<string, string> | Promise<Record<string, string>>;
|
|
197
|
+
/**
|
|
198
|
+
* The provider's OPTIONAL context-framing default — the PROVIDER-DEFAULT level of
|
|
199
|
+
* `AgentContext`'s format cascade (beaten by a manager-options or per-item override,
|
|
200
|
+
* beating the managers' built-in framing). Declares how this provider's models prefer
|
|
201
|
+
* context sections framed (e.g. XML group wrappers vs. Markdown headers). Omitted ⇒
|
|
202
|
+
* the provider is framing-agnostic and core's built-in defaults apply unchanged. NOTE:
|
|
203
|
+
* this is the prompt-CONTEXT framing consumed by `AgentContext.build()` — it is NOT
|
|
204
|
+
* Ollama's `/api/chat` `format` wire parameter (structured-output / JSON schema),
|
|
205
|
+
* which this provider does not currently send; the two are unrelated despite the
|
|
206
|
+
* shared word.
|
|
207
|
+
*/
|
|
208
|
+
readonly format?: ContextFormatInterface;
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
/**
|
|
212
|
+
* The local Ollama inference boundary — a {@link ProviderInterface} over Ollama's
|
|
213
|
+
* `POST /api/chat`, both non-streaming (`generate`) and streaming NDJSON (`stream`).
|
|
214
|
+
*
|
|
215
|
+
* @remarks
|
|
216
|
+
* - **Wire protocol.** Posts `{ model, messages, stream, keep_alive, think }` plus
|
|
217
|
+
* passthrough sampling `options` and mapped function `tools`. The `think` flag is
|
|
218
|
+
* CONFIGURABLE via {@link OllamaOptions.think} (default `false`). Non-stream parses
|
|
219
|
+
* one JSON body; stream consumes NDJSON (one JSON object per `\n`-terminated line) —
|
|
220
|
+
* deltas carry `message.content`, the final `done: true` line carries the token usage.
|
|
221
|
+
* - **Think separation (H4).** The wire `think` flag is configurable
|
|
222
|
+
* ({@link OllamaOptions.think}, default `false`). With `think: true` a thinking model's
|
|
223
|
+
* daemon separates reasoning NATIVELY — returning it on the distinct `message.thinking`
|
|
224
|
+
* channel (read here via `#thinking`) instead of inline in `message.content`. EITHER
|
|
225
|
+
* way the per-call {@link ThinkSplitterInterface} is the defensive guarantee: a daemon
|
|
226
|
+
* may ignore `think: false` for a thinking model and inline `<think>` tags, so every
|
|
227
|
+
* content delta routes through the splitter, only CLEAN content is yielded / assembled,
|
|
228
|
+
* and the separated reasoning (plus any daemon-side `message.thinking` deltas) lands on
|
|
229
|
+
* `ProviderResult.thinking`, never in the conversation.
|
|
230
|
+
* - **Boundary narrowing (§14).** Every wire value arrives as `unknown` and is
|
|
231
|
+
* narrowed through guards (`isRecord` / `isString` / `isNumber`) — never `as`. A
|
|
232
|
+
* missing / malformed field degrades to a sensible default (empty content, no
|
|
233
|
+
* usage, `{}` arguments), never a throw.
|
|
234
|
+
* - **Bounded.** Each call arms a {@link Timeout} for `OllamaOptions.timeout` and
|
|
235
|
+
* passes `AbortSignal.any([timeout.signal, signal])` to `fetch`, so the caller's
|
|
236
|
+
* signal AND the deadline both cancel the request. The timeout is always cleared —
|
|
237
|
+
* in `#fetch` if the request fails/aborts, otherwise in the consuming call's `finally`.
|
|
238
|
+
* - **Abort recovers partial.** A `stream` cancelled mid-flight throws a
|
|
239
|
+
* `ProviderAbortError` carrying the partial result assembled so far; pairing the
|
|
240
|
+
* `TextDecoder({ stream: true })` with the {@link NDJSONParser} parser keeps multi-byte
|
|
241
|
+
* UTF-8 splits and partial lines honest.
|
|
242
|
+
* - **Event-free.** A pure functional boundary — no Emitter, no events.
|
|
243
|
+
* - **Transport seam.** {@link OllamaOptions.fetch} swaps the transport (default
|
|
244
|
+
* `globalThis.fetch`) and {@link OllamaOptions.headers} is a per-request, possibly
|
|
245
|
+
* async header injector merged over the base `Content-Type` — so a browser runtime
|
|
246
|
+
* can route through the developer's own server with an obfuscated bearer token,
|
|
247
|
+
* without this library ever handling a real API key. Both omitted ⇒ today's behaviour.
|
|
248
|
+
* Orthogonal to the deadline: the hook is awaited inside `#fetch`'s try, so a hook
|
|
249
|
+
* rejection clears the armed timer like any other request failure.
|
|
250
|
+
*
|
|
251
|
+
* @example
|
|
252
|
+
* ```ts
|
|
253
|
+
* const provider = new OllamaProvider({ model: 'qwen3.5:2b-q4_K_M' })
|
|
254
|
+
* const result = await provider.generate(messages, abort.signal)
|
|
255
|
+
* ```
|
|
256
|
+
*/
|
|
257
|
+
export declare class OllamaProvider implements ProviderInterface {
|
|
258
|
+
#private;
|
|
259
|
+
readonly id: `${string}-${string}-${string}-${string}-${string}`;
|
|
260
|
+
readonly name = "ollama";
|
|
261
|
+
constructor(options: OllamaOptions);
|
|
262
|
+
/**
|
|
263
|
+
* The provider's context-framing default — the PROVIDER-DEFAULT level of
|
|
264
|
+
* {@link import('@orkestrel/agent').AgentContextInterface.build}'s format cascade (it BEATS
|
|
265
|
+
* the managers' built-in framing, is BEATEN by a manager-options or per-item override).
|
|
266
|
+
* Satisfies the OPTIONAL {@link ProviderInterface.format} contract member: `undefined`
|
|
267
|
+
* when {@link OllamaOptions.format} was omitted (the framing-agnostic default ⇒ core's
|
|
268
|
+
* built-in framing applies unchanged), else the exact configured framing the Agent
|
|
269
|
+
* threads into `build()`.
|
|
270
|
+
*
|
|
271
|
+
* @remarks
|
|
272
|
+
* EXPOSE-ONLY — read by the Agent loop and consumed by core's cascade; it is NEVER sent
|
|
273
|
+
* on the `/api/chat` wire (it is absent from `#body` / the request). This is NOT Ollama's
|
|
274
|
+
* structured-output `format` wire parameter — that one IS sent in `#body`, but only when
|
|
275
|
+
* a per-call `ProviderStreamOptions.schema` is supplied; only the word collides.
|
|
276
|
+
*
|
|
277
|
+
* @returns The configured {@link ContextFormatInterface}, or `undefined` when none
|
|
278
|
+
*/
|
|
279
|
+
get format(): ContextFormatInterface | undefined;
|
|
280
|
+
generate(messages: readonly MessageInterface[], signal: AbortSignal, tools?: readonly ToolDefinition[], options?: ProviderStreamOptions): Promise<ProviderResult>;
|
|
281
|
+
stream(messages: readonly MessageInterface[], signal: AbortSignal, tools?: readonly ToolDefinition[], options?: ProviderStreamOptions): AsyncGenerator<ProviderDelta, ProviderResult>;
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
/**
|
|
285
|
+
* A live `fetch` to `/api/chat` with the deadline + combined signal that bound it —
|
|
286
|
+
* the internal wire-shape `OllamaProvider.#fetch` hands back to a consuming call.
|
|
287
|
+
*
|
|
288
|
+
* @remarks
|
|
289
|
+
* The `response` is the open `POST /api/chat` `Response`; `timeout` is the armed
|
|
290
|
+
* {@link TimeoutInterface} the consuming call clears once it finishes reading the body
|
|
291
|
+
* (or that `#fetch` itself clears on a failed/aborted request); `combined` is the
|
|
292
|
+
* `AbortSignal.any([timeout.signal, callerSignal])` the request was issued under, which
|
|
293
|
+
* the streaming path checks to tell a mid-stream cancel apart from any other error.
|
|
294
|
+
*/
|
|
295
|
+
export declare interface OllamaResponse {
|
|
296
|
+
readonly response: Response;
|
|
297
|
+
readonly timeout: TimeoutInterface;
|
|
298
|
+
readonly combined: AbortSignal;
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
/**
|
|
302
|
+
* The exact `POST /api/chat` request body `OllamaProvider` sends — the internal typed
|
|
303
|
+
* wire contract.
|
|
304
|
+
*
|
|
305
|
+
* @remarks
|
|
306
|
+
* This is the typed wire shape asserted against the official `ollama` client's
|
|
307
|
+
* `ChatRequest` by the compile-time parity test; `src/` never imports `ollama` itself.
|
|
308
|
+
* `messages` mirrors the minimal turn shape `#plain` builds (`role` / `content`, plus
|
|
309
|
+
* `tool_calls` only on a turn that replays them and `images` only on a multimodal
|
|
310
|
+
* turn); `options` and `tools` are only present when configured.
|
|
311
|
+
*/
|
|
312
|
+
export declare interface WireChatRequest {
|
|
313
|
+
readonly model: string;
|
|
314
|
+
readonly messages: readonly {
|
|
315
|
+
readonly role: string;
|
|
316
|
+
readonly content: string;
|
|
317
|
+
readonly tool_calls?: readonly {
|
|
318
|
+
readonly function: {
|
|
319
|
+
readonly name: string;
|
|
320
|
+
readonly arguments: Readonly<Record<string, unknown>>;
|
|
321
|
+
};
|
|
322
|
+
}[];
|
|
323
|
+
readonly images?: readonly string[];
|
|
324
|
+
}[];
|
|
325
|
+
readonly stream: boolean;
|
|
326
|
+
readonly keep_alive: string | number;
|
|
327
|
+
readonly think: boolean;
|
|
328
|
+
readonly options?: Readonly<Record<string, unknown>>;
|
|
329
|
+
readonly tools?: readonly {
|
|
330
|
+
readonly type: 'function';
|
|
331
|
+
readonly function: {
|
|
332
|
+
readonly name: string;
|
|
333
|
+
readonly description?: string;
|
|
334
|
+
readonly parameters?: Readonly<Record<string, unknown>>;
|
|
335
|
+
};
|
|
336
|
+
}[];
|
|
337
|
+
/**
|
|
338
|
+
* The `/api/chat` structured-output constraint — a JSON-Schema object forwarded
|
|
339
|
+
* verbatim from the per-call `ProviderStreamOptions.schema`. This is NOT
|
|
340
|
+
* `OllamaOptions.format` (the unrelated prompt-context framing); only present
|
|
341
|
+
* when a call supplies a `schema`.
|
|
342
|
+
*/
|
|
343
|
+
readonly format?: Readonly<Record<string, unknown>>;
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
export { }
|