@orkestrel/ollama 0.0.15 → 0.0.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +29 -17
- package/dist/src/core/index.cjs +362 -0
- package/dist/src/core/index.cjs.map +1 -0
- package/dist/src/core/index.d.cts +351 -0
- package/dist/src/core/index.d.ts +351 -0
- package/dist/src/core/index.js +351 -0
- package/dist/src/core/index.js.map +1 -0
- package/package.json +24 -24
- package/dist/src/server/index.cjs +0 -682
- package/dist/src/server/index.cjs.map +0 -1
- package/dist/src/server/index.d.cts +0 -597
- package/dist/src/server/index.d.ts +0 -597
- package/dist/src/server/index.js +0 -665
- package/dist/src/server/index.js.map +0 -1
|
@@ -0,0 +1,351 @@
|
|
|
1
|
+
import { AgentProvider } from '@orkestrel/agent';
|
|
2
|
+
import type { AgentProviderInterface } from '@orkestrel/agent';
|
|
3
|
+
import type { Message } from '@orkestrel/agent';
|
|
4
|
+
import type { ProviderIncrement } from '@orkestrel/agent';
|
|
5
|
+
import type { ProviderInterface } from '@orkestrel/agent';
|
|
6
|
+
import type { ProviderOptions } from '@orkestrel/agent';
|
|
7
|
+
import type { ProviderParserInterface } from '@orkestrel/agent';
|
|
8
|
+
import type { ProviderRequest } from '@orkestrel/agent';
|
|
9
|
+
import type { TokenUsage } from '@orkestrel/budget';
|
|
10
|
+
import type { ToolCall } from '@orkestrel/tool';
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* Creates a local Ollama inference provider — a {@link ProviderInterface} over the
|
|
14
|
+
* daemon's `POST /api/chat`, assembling `generate` from the same NDJSON engine as `stream`.
|
|
15
|
+
*
|
|
16
|
+
* @remarks
|
|
17
|
+
* Only `model` is required; `url` defaults to the local daemon, `keepAlive` to `'5m'`,
|
|
18
|
+
* `timeout` to `120_000`ms, and `options` is forwarded verbatim as sampling
|
|
19
|
+
* parameters (`temperature`, `seed`, and `num_predict`). Each call takes an
|
|
20
|
+
* `AbortSignal` to bound the request; a `stream` cancelled mid-flight throws a
|
|
21
|
+
* `ProviderAbortError` carrying the partial result.
|
|
22
|
+
*
|
|
23
|
+
* The optional `fetch` + `headers` form a transport seam (see {@link OllamaOptions}):
|
|
24
|
+
* point `url` at your own server, inject a custom `fetch`, and have `headers` attach a
|
|
25
|
+
* generated/obfuscated bearer token your server validates — so a browser runtime
|
|
26
|
+
* reaches the LLM through your middleware without this library ever handling the real API
|
|
27
|
+
* key. Both omitted ⇒ the global `fetch` and only a JSON content type.
|
|
28
|
+
*
|
|
29
|
+
* The optional `format` is the provider's context-framing default — the provider-default
|
|
30
|
+
* level of `AgentContext`'s format cascade (beaten by a manager-options or per-item
|
|
31
|
+
* override, beating the managers' built-in framing), declaring how this
|
|
32
|
+
* provider's models prefer context sections framed (for example XML group wrappers vs. Markdown
|
|
33
|
+
* headers). It is exposed on the provider for the Agent's `build()` and is not Ollama's
|
|
34
|
+
* `/api/chat` `format` wire parameter (structured output) — the framing default and that
|
|
35
|
+
* wire parameter are unrelated despite the shared word. Omitted ⇒ the provider is
|
|
36
|
+
* framing-agnostic (core's built-in defaults).
|
|
37
|
+
*
|
|
38
|
+
* @param options - `model` (required), and optional `url` / `keepAlive` / `timeout` /
|
|
39
|
+
* `options` / `fetch` / `headers` / `format` (see {@link OllamaOptions})
|
|
40
|
+
* @returns A working {@link ProviderInterface} backed by Ollama
|
|
41
|
+
*
|
|
42
|
+
* @example createOllama + generate
|
|
43
|
+
* ```ts
|
|
44
|
+
* import { createAbort } from '@orkestrel/abort'
|
|
45
|
+
* import type { TokenUsage } from '@orkestrel/budget'
|
|
46
|
+
* import { createOllama } from '@orkestrel/ollama'
|
|
47
|
+
*
|
|
48
|
+
* declare function charge(usage: TokenUsage): void // your billing integration
|
|
49
|
+
*
|
|
50
|
+
* const provider = createOllama({ model: 'qwen3.5:2b-q4_K_M', options: { temperature: 0 } })
|
|
51
|
+
* const abort = createAbort()
|
|
52
|
+
* const messages = [
|
|
53
|
+
* { id: '1', role: 'user', content: 'Summarize the release notes for version 2.0.' },
|
|
54
|
+
* ] as const
|
|
55
|
+
*
|
|
56
|
+
* const result = await provider.generate(messages, abort.signal)
|
|
57
|
+
* console.log(result.content)
|
|
58
|
+
* if (result.usage) charge(result.usage) // fold into a token budget
|
|
59
|
+
* ```
|
|
60
|
+
*
|
|
61
|
+
* @example
|
|
62
|
+
* Route through your own server with an obfuscated token:
|
|
63
|
+
* ```ts
|
|
64
|
+
* const provider = createOllama({
|
|
65
|
+
* model: 'qwen3.5:2b-q4_K_M',
|
|
66
|
+
* url: 'https://my-app.example.com/llm', // your server, not the daemon
|
|
67
|
+
* fetch: myFetch, // optional custom transport
|
|
68
|
+
* headers: () => ({ authorization: `Bearer ${myToken}` }), // your server validates this
|
|
69
|
+
* })
|
|
70
|
+
* ```
|
|
71
|
+
*
|
|
72
|
+
* @example
|
|
73
|
+
* Declare a context-framing default — wrap the instructions section in an XML group (the
|
|
74
|
+
* provider-default level of `AgentContext`'s cascade; not the wire `format`):
|
|
75
|
+
* ```ts
|
|
76
|
+
* const provider = createOllama({
|
|
77
|
+
* model: 'qwen3.5:2b-q4_K_M',
|
|
78
|
+
* format: {
|
|
79
|
+
* instructions: {
|
|
80
|
+
* open: '<instructions>',
|
|
81
|
+
* render: (i) => `<instruction>${i.content}</instruction>`,
|
|
82
|
+
* close: '</instructions>',
|
|
83
|
+
* },
|
|
84
|
+
* },
|
|
85
|
+
* })
|
|
86
|
+
* ```
|
|
87
|
+
*/
|
|
88
|
+
export declare function createOllama(options: OllamaOptions): ProviderInterface;
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* Names how long the model stays resident after a call — `'5m'` when
|
|
92
|
+
* `OllamaOptions.keepAlive` is omitted, Ollama's own `keep_alive` default, expressed as a
|
|
93
|
+
* duration string.
|
|
94
|
+
*
|
|
95
|
+
* @remarks
|
|
96
|
+
* The name mirrors the Ollama `/api/chat` `keep_alive` field this value is sent as, so
|
|
97
|
+
* the constant, the `OllamaOptions.keepAlive` key, and the wire member read as one term.
|
|
98
|
+
*/
|
|
99
|
+
export declare const DEFAULT_KEEP_ALIVE = "5m";
|
|
100
|
+
|
|
101
|
+
/**
|
|
102
|
+
* Names the local Ollama daemon base URL, `'http://localhost:11434'`, assumed when
|
|
103
|
+
* `OllamaOptions.url` is omitted.
|
|
104
|
+
*/
|
|
105
|
+
export declare const DEFAULT_OLLAMA_URL = "http://localhost:11434";
|
|
106
|
+
|
|
107
|
+
/**
|
|
108
|
+
* Extracts a wire `arguments` value as a record.
|
|
109
|
+
*
|
|
110
|
+
* @remarks
|
|
111
|
+
* Total: an object passes through, a JSON string is parsed when it yields a record, and
|
|
112
|
+
* a malformed string yields `{}` rather than throwing.
|
|
113
|
+
*
|
|
114
|
+
* @param value - The wire's `function.arguments` value, of unknown shape
|
|
115
|
+
* @returns The argument record, or `{}` when the value carries none
|
|
116
|
+
*
|
|
117
|
+
* @example
|
|
118
|
+
* ```ts
|
|
119
|
+
* extractArguments('{"city":"Oslo"}') // { city: 'Oslo' }
|
|
120
|
+
* ```
|
|
121
|
+
*/
|
|
122
|
+
export declare function extractArguments(value: unknown): Readonly<Record<string, unknown>>;
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* Extracts the assistant text of one wire record.
|
|
126
|
+
*
|
|
127
|
+
* @param record - One parsed `/api/chat` NDJSON record
|
|
128
|
+
* @returns The record's `message.content` when it is a string, else `''`
|
|
129
|
+
*
|
|
130
|
+
* @example
|
|
131
|
+
* ```ts
|
|
132
|
+
* extractContent({ message: { content: 'ok' } }) // 'ok'
|
|
133
|
+
* ```
|
|
134
|
+
*/
|
|
135
|
+
export declare function extractContent(record: Readonly<Record<string, unknown>>): string;
|
|
136
|
+
|
|
137
|
+
/**
|
|
138
|
+
* Extracts the daemon-side reasoning of one wire record.
|
|
139
|
+
*
|
|
140
|
+
* @remarks
|
|
141
|
+
* `message.thinking` is the `think: true` wire shape. It is read whatever the configured
|
|
142
|
+
* flag says, because a daemon may separate reasoning on its own.
|
|
143
|
+
*
|
|
144
|
+
* @param record - One parsed `/api/chat` NDJSON record
|
|
145
|
+
* @returns The record's `message.thinking` when it is a string, else `''`
|
|
146
|
+
*
|
|
147
|
+
* @example
|
|
148
|
+
* ```ts
|
|
149
|
+
* extractThinking({ message: { thinking: 'weighing it' } }) // 'weighing it'
|
|
150
|
+
* ```
|
|
151
|
+
*/
|
|
152
|
+
export declare function extractThinking(record: Readonly<Record<string, unknown>>): string;
|
|
153
|
+
|
|
154
|
+
/**
|
|
155
|
+
* Extracts the tool calls of one wire record's `message.tool_calls`.
|
|
156
|
+
*
|
|
157
|
+
* @remarks
|
|
158
|
+
* Each entry narrows to `{ id, name, arguments }`: the entry and its `function` must be
|
|
159
|
+
* records and `name` a string, else the entry is dropped. An id is minted when the wire
|
|
160
|
+
* omits one.
|
|
161
|
+
*
|
|
162
|
+
* @param record - One parsed `/api/chat` NDJSON record
|
|
163
|
+
* @returns The narrowed tool calls, empty when the record carries none
|
|
164
|
+
*
|
|
165
|
+
* @example
|
|
166
|
+
* ```ts
|
|
167
|
+
* extractTools({ message: { tool_calls: [{ function: { name: 'weather' } }] } })
|
|
168
|
+
* // [{ id: '…', name: 'weather', arguments: {} }]
|
|
169
|
+
* ```
|
|
170
|
+
*/
|
|
171
|
+
export declare function extractTools(record: Readonly<Record<string, unknown>>): readonly ToolCall[];
|
|
172
|
+
|
|
173
|
+
/**
|
|
174
|
+
* Extracts the token usage of one wire record.
|
|
175
|
+
*
|
|
176
|
+
* @remarks
|
|
177
|
+
* Both counts must be numbers, which is true of the stream's `done: true` line. A
|
|
178
|
+
* delta line carries neither, so it yields `undefined`.
|
|
179
|
+
*
|
|
180
|
+
* @param record - One parsed `/api/chat` NDJSON record
|
|
181
|
+
* @returns The `TokenUsage` shape, or `undefined` when either count is absent
|
|
182
|
+
*
|
|
183
|
+
* @example
|
|
184
|
+
* ```ts
|
|
185
|
+
* extractUsage({ prompt_eval_count: 3, eval_count: 4 })
|
|
186
|
+
* // { prompt: 3, completion: 4, total: 7 }
|
|
187
|
+
* ```
|
|
188
|
+
*/
|
|
189
|
+
export declare function extractUsage(record: Readonly<Record<string, unknown>>): TokenUsage | undefined;
|
|
190
|
+
|
|
191
|
+
/**
|
|
192
|
+
* Maps conversation turns onto the `/api/chat` wire's minimal message shape.
|
|
193
|
+
*
|
|
194
|
+
* @remarks
|
|
195
|
+
* `tool_calls` is emitted only on a turn that replays them and `images` only on a
|
|
196
|
+
* multimodal turn, so an empty optional never reaches the wire.
|
|
197
|
+
*
|
|
198
|
+
* @param messages - The conversation turns to send
|
|
199
|
+
* @returns The wire `messages` array, one entry per turn, in order
|
|
200
|
+
*
|
|
201
|
+
* @example
|
|
202
|
+
* ```ts
|
|
203
|
+
* mapMessages([{ id: '1', role: 'user', content: 'Say hello.' }])
|
|
204
|
+
* // [{ role: 'user', content: 'Say hello.' }]
|
|
205
|
+
* ```
|
|
206
|
+
*/
|
|
207
|
+
export declare function mapMessages(messages: readonly Message[]): WireChatRequest['messages'];
|
|
208
|
+
|
|
209
|
+
/** Names the Ollama chat endpoint appended to the configured base URL. */
|
|
210
|
+
export declare const OLLAMA_CHAT_PATH = "/api/chat";
|
|
211
|
+
|
|
212
|
+
/**
|
|
213
|
+
* Represents the configuration `createOllama` accepts for the local Ollama backend.
|
|
214
|
+
*
|
|
215
|
+
* @remarks
|
|
216
|
+
* Only `model` is required. `url` defaults to the local daemon, `keepAlive` controls
|
|
217
|
+
* how long the model stays resident after a call, `timeout` is the per-call deadline
|
|
218
|
+
* in milliseconds, and `options` is a passthrough bag of sampling parameters
|
|
219
|
+
* (`temperature`, `seed`, and `num_predict`) forwarded verbatim to the wire.
|
|
220
|
+
*
|
|
221
|
+
* The optional `fetch` + `headers` form a **transport seam**: by default the provider
|
|
222
|
+
* talks straight to a local daemon over `globalThis.fetch` with only a JSON content
|
|
223
|
+
* type, but a browser-side runtime can inject both a custom transport and a dynamic header
|
|
224
|
+
* (for example an obfuscated bearer token) so requests route through the developer's own
|
|
225
|
+
* server, which validates that header and forwards to the real LLM. Your app never
|
|
226
|
+
* holds a real API key — the real key lives only on the developer's server; the
|
|
227
|
+
* `headers` hook supplies whatever short-lived/obfuscated token that server expects.
|
|
228
|
+
*/
|
|
229
|
+
export declare interface OllamaOptions extends ProviderOptions {
|
|
230
|
+
readonly model: string;
|
|
231
|
+
/** Sets the daemon base URL; defaults to `'http://localhost:11434'`. */
|
|
232
|
+
readonly url?: string;
|
|
233
|
+
/**
|
|
234
|
+
* Sets how long the model stays resident after a call; defaults to `'5m'`. Mirrors the
|
|
235
|
+
* Ollama `/api/chat` `keep_alive` field, whose value this key carries verbatim onto
|
|
236
|
+
* {@link WireChatRequest.keep_alive}.
|
|
237
|
+
*/
|
|
238
|
+
readonly keepAlive?: string | number;
|
|
239
|
+
/**
|
|
240
|
+
* Carries passthrough sampling parameters (`temperature`, `seed`, and `num_predict`).
|
|
241
|
+
* Mirrors the Ollama `/api/chat` `options` field, whose value this key carries verbatim
|
|
242
|
+
* onto {@link WireChatRequest.options}.
|
|
243
|
+
*/
|
|
244
|
+
readonly options?: Readonly<Record<string, unknown>>;
|
|
245
|
+
/**
|
|
246
|
+
* Sets the `/api/chat` `think` wire flag; defaults to `false`. When `true`, a thinking-capable
|
|
247
|
+
* model (for example `qwen3`) separates its reasoning natively at the wire — the daemon returns it
|
|
248
|
+
* on the distinct `message.thinking` channel (surfaced on `ProviderResult.thinking`) rather
|
|
249
|
+
* than inline in `message.content`. The default is `false`, so a non-thinking model needs no
|
|
250
|
+
* configuration and answers immediately; the per-call ThinkSplitter
|
|
251
|
+
* remains the defensive fallback for daemons/models that still inline `<think>` tags either
|
|
252
|
+
* way. Set it `true` for a thinking model whose reasoning you intend to display separately.
|
|
253
|
+
*/
|
|
254
|
+
readonly think?: boolean;
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
/**
|
|
258
|
+
* Implements the Ollama `/api/chat` wire over the shared {@link AgentProvider} engine.
|
|
259
|
+
*
|
|
260
|
+
* @remarks
|
|
261
|
+
* Every request uses NDJSON streaming. The base assembles complete turns, separates
|
|
262
|
+
* reasoning, and bounds requests; this class supplies Ollama framing and projections.
|
|
263
|
+
* Usage comes only from a `done: true` record carrying the token counts.
|
|
264
|
+
*
|
|
265
|
+
* @example
|
|
266
|
+
* ```ts
|
|
267
|
+
* const provider = new OllamaProvider({ model: 'qwen3.5:2b-q4_K_M' })
|
|
268
|
+
* const result = await provider.generate(messages, abort.signal)
|
|
269
|
+
* ```
|
|
270
|
+
*/
|
|
271
|
+
export declare class OllamaProvider extends AgentProvider implements AgentProviderInterface {
|
|
272
|
+
#private;
|
|
273
|
+
readonly name = "ollama";
|
|
274
|
+
constructor(options: OllamaOptions);
|
|
275
|
+
/**
|
|
276
|
+
* Creates fresh NDJSON framing state for a call.
|
|
277
|
+
*
|
|
278
|
+
* @returns The parser that buffers incomplete Ollama records
|
|
279
|
+
*/
|
|
280
|
+
frame(): ProviderParserInterface;
|
|
281
|
+
/**
|
|
282
|
+
* Projects conversation turns and per-call options onto the Ollama request body.
|
|
283
|
+
*
|
|
284
|
+
* @param request - The conversation, advertised tools, and per-call overrides
|
|
285
|
+
* @returns The `/api/chat` body with streaming enabled
|
|
286
|
+
*/
|
|
287
|
+
body(request: ProviderRequest): WireChatRequest;
|
|
288
|
+
/**
|
|
289
|
+
* Extracts a record's content, reasoning, tools, and completed usage report.
|
|
290
|
+
*
|
|
291
|
+
* @param record - One parsed Ollama NDJSON record
|
|
292
|
+
* @returns The turn increment, omitting usage until `done` and the counts are present
|
|
293
|
+
*/
|
|
294
|
+
read(record: Readonly<Record<string, unknown>>): ProviderIncrement;
|
|
295
|
+
/**
|
|
296
|
+
* Recovers a final NDJSON record that arrived without its line terminator.
|
|
297
|
+
*
|
|
298
|
+
* @param parser - The call's parser holding any unterminated input
|
|
299
|
+
* @returns The records completed by the final newline
|
|
300
|
+
*/
|
|
301
|
+
finish(parser: ProviderParserInterface): ReadonlyArray<Readonly<Record<string, unknown>>>;
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
/**
|
|
305
|
+
* Represents the exact `POST /api/chat` request body `OllamaProvider` sends — the internal typed
|
|
306
|
+
* wire contract.
|
|
307
|
+
*
|
|
308
|
+
* @remarks
|
|
309
|
+
* This is the typed wire shape asserted against the official `ollama` client's
|
|
310
|
+
* `ChatRequest` by the compile-time parity test; `src/` never imports `ollama` itself.
|
|
311
|
+
* `messages` mirrors the minimal turn shape `mapMessages` builds (`role` / `content`, plus
|
|
312
|
+
* `tool_calls` only on a turn that replays them and `images` only on a multimodal
|
|
313
|
+
* turn); `options` and `tools` are only present when configured. `format` carries the
|
|
314
|
+
* `/api/chat` structured-output constraint, forwarded verbatim from the per-call
|
|
315
|
+
* `ProviderStreamOptions.schema` and absent when no schema is supplied.
|
|
316
|
+
*/
|
|
317
|
+
export declare interface WireChatRequest {
|
|
318
|
+
readonly model: string;
|
|
319
|
+
readonly messages: ReadonlyArray<{
|
|
320
|
+
readonly role: string;
|
|
321
|
+
readonly content: string;
|
|
322
|
+
readonly tool_calls?: ReadonlyArray<{
|
|
323
|
+
readonly function: {
|
|
324
|
+
readonly name: string;
|
|
325
|
+
readonly arguments: Readonly<Record<string, unknown>>;
|
|
326
|
+
};
|
|
327
|
+
}>;
|
|
328
|
+
readonly images?: readonly string[];
|
|
329
|
+
}>;
|
|
330
|
+
readonly stream: boolean;
|
|
331
|
+
readonly keep_alive: string | number;
|
|
332
|
+
readonly think: boolean;
|
|
333
|
+
readonly options?: Readonly<Record<string, unknown>>;
|
|
334
|
+
readonly tools?: ReadonlyArray<{
|
|
335
|
+
readonly type: 'function';
|
|
336
|
+
readonly function: {
|
|
337
|
+
readonly name: string;
|
|
338
|
+
readonly description?: string;
|
|
339
|
+
readonly parameters?: Readonly<Record<string, unknown>>;
|
|
340
|
+
};
|
|
341
|
+
}>;
|
|
342
|
+
/**
|
|
343
|
+
* Holds the `/api/chat` structured-output constraint — a JSON-Schema object forwarded
|
|
344
|
+
* verbatim from the per-call `ProviderStreamOptions.schema`. This is not
|
|
345
|
+
* `OllamaOptions.format` (the unrelated prompt-context framing); only present
|
|
346
|
+
* when a call supplies a `schema`.
|
|
347
|
+
*/
|
|
348
|
+
readonly format?: Readonly<Record<string, unknown>>;
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
export { }
|