@orkestrel/ollama 0.0.15 → 0.0.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,355 @@
1
+ import { AgentProvider } from '@orkestrel/agent';
2
+ import type { AgentProviderInterface } from '@orkestrel/agent';
3
+ import type { Message } from '@orkestrel/agent';
4
+ import type { ProviderIncrement } from '@orkestrel/agent';
5
+ import type { ProviderInterface } from '@orkestrel/agent';
6
+ import type { ProviderOptions } from '@orkestrel/agent';
7
+ import type { ProviderParserInterface } from '@orkestrel/agent';
8
+ import type { ProviderRequest } from '@orkestrel/agent';
9
+ import type { TokenUsage } from '@orkestrel/budget';
10
+ import type { ToolCall } from '@orkestrel/tool';
11
+
12
+ /**
13
+ * Creates a local Ollama inference provider — a {@link ProviderInterface} over the
14
+ * daemon's `POST /api/chat`, assembling `generate` from the same NDJSON engine as `stream`.
15
+ *
16
+ * @remarks
17
+ * Only `model` is required; `url` defaults to the local daemon, `keepAlive` to `'5m'`,
18
+ * `timeout` to `120_000`ms, and `options` is forwarded verbatim as sampling
19
+ * parameters (`temperature`, `seed`, and `num_predict`). Each call takes an
20
+ * `AbortSignal` to bound the request; a `stream` cancelled mid-flight throws a
21
+ * `ProviderAbortError` carrying the partial result.
22
+ *
23
+ * The optional `fetch` + `headers` form a transport seam (see {@link OllamaOptions}):
24
+ * point `url` at your own server, inject a custom `fetch`, and have `headers` attach a
25
+ * generated/obfuscated bearer token your server validates — so a browser runtime
26
+ * reaches the LLM through your middleware without this library ever handling the real API
27
+ * key. Both omitted ⇒ the global `fetch` and only a JSON content type.
28
+ *
29
+ * The optional `format` is the provider's context-framing default — the provider-default
30
+ * level of `AgentContext`'s format cascade (beaten by a manager-options or per-item
31
+ * override, beating the managers' built-in framing), declaring how this
32
+ * provider's models prefer context sections framed (for example XML group wrappers vs. Markdown
33
+ * headers). It is exposed on the provider for the Agent's `build()` and is not Ollama's
34
+ * `/api/chat` `format` wire parameter (structured output) — the framing default and that
35
+ * wire parameter are unrelated despite the shared word. Omitted ⇒ the provider is
36
+ * framing-agnostic (core's built-in defaults).
37
+ *
38
+ * @param options - `model` (required), and optional `url` / `keepAlive` / `timeout` /
39
+ * `options` / `fetch` / `headers` / `format` (see {@link OllamaOptions})
40
+ * @returns A working {@link ProviderInterface} backed by Ollama
41
+ *
42
+ * @example createOllama + generate
43
+ * ```ts
44
+ * import { createAbort } from '@orkestrel/abort'
45
+ * import type { TokenUsage } from '@orkestrel/budget'
46
+ * import { createOllama } from '@orkestrel/ollama'
47
+ *
48
+ * declare function charge(usage: TokenUsage): void // your billing integration
49
+ *
50
+ * const provider = createOllama({ model: 'qwen3.5:2b-q4_K_M', options: { temperature: 0 } })
51
+ * const abort = createAbort()
52
+ * const messages = [
53
+ * { id: '1', role: 'user', content: 'Summarize the release notes for version 2.0.' },
54
+ * ] as const
55
+ *
56
+ * const result = await provider.generate(messages, abort.signal)
57
+ * console.log(result.content)
58
+ * if (result.usage) charge(result.usage) // fold into a token budget
59
+ * ```
60
+ *
61
+ * @example
62
+ * Route through your own server with an obfuscated token:
63
+ * ```ts
64
+ * const provider = createOllama({
65
+ * model: 'qwen3.5:2b-q4_K_M',
66
+ * url: 'https://my-app.example.com/llm', // your server, not the daemon
67
+ * fetch: myFetch, // optional custom transport
68
+ * headers: () => ({ authorization: `Bearer ${myToken}` }), // your server validates this
69
+ * })
70
+ * ```
71
+ *
72
+ * @example
73
+ * Declare a context-framing default — wrap the instructions section in an XML group (the
74
+ * provider-default level of `AgentContext`'s cascade; not the wire `format`):
75
+ * ```ts
76
+ * const provider = createOllama({
77
+ * model: 'qwen3.5:2b-q4_K_M',
78
+ * format: {
79
+ * instructions: {
80
+ * open: '<instructions>',
81
+ * render: (i) => `<instruction>${i.content}</instruction>`,
82
+ * close: '</instructions>',
83
+ * },
84
+ * },
85
+ * })
86
+ * ```
87
+ */
88
+ export declare function createOllama(options: OllamaOptions): ProviderInterface;
89
+
90
+ /**
91
+ * Names how long the model stays resident after a call — `'5m'` when
92
+ * `OllamaOptions.keepAlive` is omitted, Ollama's own `keep_alive` default, expressed as a
93
+ * duration string.
94
+ *
95
+ * @remarks
96
+ * The name mirrors the Ollama `/api/chat` `keep_alive` field this value is sent as, so
97
+ * the constant, the `OllamaOptions.keepAlive` key, and the wire member read as one term.
98
+ */
99
+ export declare const DEFAULT_KEEP_ALIVE = "5m";
100
+
101
+ /**
102
+ * Names the local Ollama daemon base URL, `'http://localhost:11434'`, assumed when
103
+ * `OllamaOptions.url` is omitted.
104
+ */
105
+ export declare const DEFAULT_OLLAMA_URL = "http://localhost:11434";
106
+
107
+ /**
108
+ * Extracts a wire `arguments` value as a record.
109
+ *
110
+ * @remarks
111
+ * Total: an object passes through, a JSON string is parsed when it yields a record, and
112
+ * a malformed string yields `{}` rather than throwing.
113
+ *
114
+ * @param value - The wire's `function.arguments` value, of unknown shape
115
+ * @returns The argument record, or `{}` when the value carries none
116
+ *
117
+ * @example
118
+ * ```ts
119
+ * extractArguments('{"city":"Oslo"}') // { city: 'Oslo' }
120
+ * ```
121
+ */
122
+ export declare function extractArguments(value: unknown): Readonly<Record<string, unknown>>;
123
+
124
+ /**
125
+ * Extracts the assistant text of one wire record.
126
+ *
127
+ * @param record - One parsed `/api/chat` NDJSON record
128
+ * @returns The record's `message.content` when it is a string, else `''`
129
+ *
130
+ * @example
131
+ * ```ts
132
+ * extractContent({ message: { content: 'ok' } }) // 'ok'
133
+ * ```
134
+ */
135
+ export declare function extractContent(record: Readonly<Record<string, unknown>>): string;
136
+
137
+ /**
138
+ * Extracts the daemon-side reasoning of one wire record.
139
+ *
140
+ * @remarks
141
+ * `message.thinking` is the `think: true` wire shape. It is read whatever the configured
142
+ * flag says, because a daemon may separate reasoning on its own.
143
+ *
144
+ * @param record - One parsed `/api/chat` NDJSON record
145
+ * @returns The record's `message.thinking` when it is a string, else `''`
146
+ *
147
+ * @example
148
+ * ```ts
149
+ * extractThinking({ message: { thinking: 'weighing it' } }) // 'weighing it'
150
+ * ```
151
+ */
152
+ export declare function extractThinking(record: Readonly<Record<string, unknown>>): string;
153
+
154
+ /**
155
+ * Extracts the tool calls of one wire record's `message.tool_calls`.
156
+ *
157
+ * @remarks
158
+ * Each entry narrows to `{ id, name, arguments }`: the entry and its `function` must be
159
+ * records and `name` a string, else the entry is dropped. An id is minted when the wire
160
+ * omits one.
161
+ *
162
+ * @param record - One parsed `/api/chat` NDJSON record
163
+ * @returns The narrowed tool calls, empty when the record carries none
164
+ *
165
+ * @example
166
+ * ```ts
167
+ * extractTools({ message: { tool_calls: [{ function: { name: 'weather' } }] } })
168
+ * // [{ id: '…', name: 'weather', arguments: {} }]
169
+ * ```
170
+ */
171
+ export declare function extractTools(record: Readonly<Record<string, unknown>>): readonly ToolCall[];
172
+
173
+ /**
174
+ * Extracts the token usage of one wire record.
175
+ *
176
+ * @remarks
177
+ * Both counts must be numbers, which is true of the stream's `done: true` line. A
178
+ * delta line carries neither, so it yields `undefined`.
179
+ *
180
+ * @param record - One parsed `/api/chat` NDJSON record
181
+ * @returns The `TokenUsage` shape, or `undefined` when either count is absent
182
+ *
183
+ * @example
184
+ * ```ts
185
+ * extractUsage({ prompt_eval_count: 3, eval_count: 4 })
186
+ * // { prompt: 3, completion: 4, total: 7 }
187
+ * ```
188
+ */
189
+ export declare function extractUsage(record: Readonly<Record<string, unknown>>): TokenUsage | undefined;
190
+
191
+ /**
192
+ * Maps conversation turns onto the `/api/chat` wire's minimal message shape.
193
+ *
194
+ * @remarks
195
+ * `tool_calls` is emitted only on a turn that replays them and `images` only on a
196
+ * multimodal turn, so an empty optional never reaches the wire.
197
+ *
198
+ * @param messages - The conversation turns to send
199
+ * @returns The wire `messages` array, one entry per turn, in order
200
+ *
201
+ * @example
202
+ * ```ts
203
+ * mapMessages([{ id: '1', role: 'user', content: 'Say hello.' }])
204
+ * // [{ role: 'user', content: 'Say hello.' }]
205
+ * ```
206
+ */
207
+ export declare function mapMessages(messages: readonly Message[]): WireChatRequest['messages'];
208
+
209
+ /** Names the Ollama chat endpoint appended to the configured base URL. */
210
+ export declare const OLLAMA_CHAT_PATH = "/api/chat";
211
+
212
+ /**
213
+ * Represents the configuration `createOllama` accepts for the local Ollama backend.
214
+ *
215
+ * @remarks
216
+ * Only `model` is required. `url` defaults to the local daemon, `keepAlive` controls
217
+ * how long the model stays resident after a call, `timeout` is the per-call deadline
218
+ * in milliseconds, and `options` is a passthrough bag of sampling parameters
219
+ * (`temperature`, `seed`, and `num_predict`) forwarded verbatim to the wire.
220
+ *
221
+ * The optional `fetch` + `headers` form a **transport seam**: by default the provider
222
+ * talks straight to a local daemon over `globalThis.fetch` with only a JSON content
223
+ * type, but a browser-side runtime can inject both a custom transport and a dynamic header
224
+ * (for example an obfuscated bearer token) so requests route through the developer's own
225
+ * server, which validates that header and forwards to the real LLM. Your app never
226
+ * holds a real API key — the real key lives only on the developer's server; the
227
+ * `headers` hook supplies whatever short-lived/obfuscated token that server expects.
228
+ */
229
+ export declare interface OllamaOptions extends ProviderOptions {
230
+ readonly model: string;
231
+ /** Sets the daemon base URL; defaults to `'http://localhost:11434'`. */
232
+ readonly url?: string;
233
+ /**
234
+ * Sets how long the model stays resident after a call; defaults to `'5m'`. Mirrors the
235
+ * Ollama `/api/chat` `keep_alive` field, whose value this key carries verbatim onto
236
+ * {@link WireChatRequest.keep_alive}.
237
+ */
238
+ readonly keepAlive?: string | number;
239
+ /**
240
+ * Carries passthrough sampling parameters (`temperature`, `seed`, and `num_predict`).
241
+ * Mirrors the Ollama `/api/chat` `options` field, whose value this key carries verbatim
242
+ * onto {@link WireChatRequest.options}.
243
+ */
244
+ readonly options?: Readonly<Record<string, unknown>>;
245
+ /**
246
+ * Sets the `/api/chat` `think` wire flag; defaults to `false`. When `true`, a thinking-capable
247
+ * model (for example `qwen3`) separates its reasoning natively at the wire — the daemon returns it
248
+ * on the distinct `message.thinking` channel (surfaced on `ProviderResult.thinking`) rather
249
+ * than inline in `message.content`. The default is `false`, so a non-thinking model needs no
250
+ * configuration and answers immediately; the per-call ThinkSplitter
251
+ * remains the defensive fallback for daemons/models that still inline `<think>` tags either
252
+ * way. Set it `true` for a thinking model whose reasoning you intend to display separately.
253
+ */
254
+ readonly think?: boolean;
255
+ }
256
+
257
+ /**
258
+ * Implements the Ollama `/api/chat` wire over the shared {@link AgentProvider} engine.
259
+ *
260
+ * @remarks
261
+ * Every request uses NDJSON streaming. The base assembles complete turns, separates
262
+ * reasoning, and bounds requests; this class supplies Ollama framing and projections.
263
+ * Usage comes only from a `done: true` record carrying the token counts.
264
+ *
265
+ * @example
266
+ * ```ts
267
+ * const provider = new OllamaProvider({ model: 'qwen3.5:2b-q4_K_M' })
268
+ * const result = await provider.generate(messages, abort.signal)
269
+ * ```
270
+ */
271
+ export declare class OllamaProvider extends AgentProvider implements AgentProviderInterface {
272
+ #private;
273
+ readonly name = "ollama";
274
+ constructor(options: OllamaOptions);
275
+ /**
276
+ * Creates fresh NDJSON framing state for a call.
277
+ *
278
+ * @returns The parser that buffers incomplete Ollama records
279
+ */
280
+ frame(): ProviderParserInterface;
281
+ /**
282
+ * Projects conversation turns and per-call options onto the Ollama request body.
283
+ *
284
+ * @remarks
285
+ * A tool's `title` and `annotations` are never sent: the `/api/chat` tool function
286
+ * object carries no field for either.
287
+ *
288
+ * @param request - The conversation, advertised tools, and per-call overrides
289
+ * @returns The `/api/chat` body with streaming enabled
290
+ */
291
+ body(request: ProviderRequest): WireChatRequest;
292
+ /**
293
+ * Extracts a record's content, reasoning, tools, and completed usage report.
294
+ *
295
+ * @param record - One parsed Ollama NDJSON record
296
+ * @returns The turn increment, omitting usage until `done` and the counts are present
297
+ */
298
+ read(record: Readonly<Record<string, unknown>>): ProviderIncrement;
299
+ /**
300
+ * Recovers a final NDJSON record that arrived without its line terminator.
301
+ *
302
+ * @param parser - The call's parser holding any unterminated input
303
+ * @returns The records completed by the final newline
304
+ */
305
+ finish(parser: ProviderParserInterface): ReadonlyArray<Readonly<Record<string, unknown>>>;
306
+ }
307
+
308
+ /**
309
+ * Represents the exact `POST /api/chat` request body `OllamaProvider` sends — the internal typed
310
+ * wire contract.
311
+ *
312
+ * @remarks
313
+ * This is the typed wire shape asserted against the official `ollama` client's
314
+ * `ChatRequest` by the compile-time parity test; `src/` never imports `ollama` itself.
315
+ * `messages` mirrors the minimal turn shape `mapMessages` builds (`role` / `content`, plus
316
+ * `tool_calls` only on a turn that replays them and `images` only on a multimodal
317
+ * turn); `options` and `tools` are only present when configured. `format` carries the
318
+ * `/api/chat` structured-output constraint, forwarded verbatim from the per-call
319
+ * `ProviderStreamOptions.schema` and absent when no schema is supplied.
320
+ */
321
+ export declare interface WireChatRequest {
322
+ readonly model: string;
323
+ readonly messages: ReadonlyArray<{
324
+ readonly role: string;
325
+ readonly content: string;
326
+ readonly tool_calls?: ReadonlyArray<{
327
+ readonly function: {
328
+ readonly name: string;
329
+ readonly arguments: Readonly<Record<string, unknown>>;
330
+ };
331
+ }>;
332
+ readonly images?: readonly string[];
333
+ }>;
334
+ readonly stream: boolean;
335
+ readonly keep_alive: string | number;
336
+ readonly think: boolean;
337
+ readonly options?: Readonly<Record<string, unknown>>;
338
+ readonly tools?: ReadonlyArray<{
339
+ readonly type: 'function';
340
+ readonly function: {
341
+ readonly name: string;
342
+ readonly description?: string;
343
+ readonly parameters?: Readonly<Record<string, unknown>>;
344
+ };
345
+ }>;
346
+ /**
347
+ * Holds the `/api/chat` structured-output constraint — a JSON-Schema object forwarded
348
+ * verbatim from the per-call `ProviderStreamOptions.schema`. This is not
349
+ * `OllamaOptions.format` (the unrelated prompt-context framing); only present
350
+ * when a call supplies a `schema`.
351
+ */
352
+ readonly format?: Readonly<Record<string, unknown>>;
353
+ }
354
+
355
+ export { }