dsh-local-ai 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/CHANGELOG.md +17 -0
  2. package/LICENSE +201 -0
  3. package/README.es.md +211 -0
  4. package/README.hi.md +211 -0
  5. package/README.md +211 -0
  6. package/README.pt.md +211 -0
  7. package/README.zh.md +211 -0
  8. package/THIRD_PARTY_NOTICES.md +20 -0
  9. package/cordis.patch.yml +44 -0
  10. package/lib/index.js +1334 -0
  11. package/lib/types/adapter.d.ts +39 -0
  12. package/lib/types/adapter.d.ts.map +1 -0
  13. package/lib/types/adapter.js +190 -0
  14. package/lib/types/adapter.js.map +1 -0
  15. package/lib/types/config.d.ts +109 -0
  16. package/lib/types/config.d.ts.map +1 -0
  17. package/lib/types/config.js +161 -0
  18. package/lib/types/config.js.map +1 -0
  19. package/lib/types/health.d.ts +64 -0
  20. package/lib/types/health.d.ts.map +1 -0
  21. package/lib/types/health.js +92 -0
  22. package/lib/types/health.js.map +1 -0
  23. package/lib/types/index.d.ts +43 -0
  24. package/lib/types/index.d.ts.map +1 -0
  25. package/lib/types/index.js +232 -0
  26. package/lib/types/index.js.map +1 -0
  27. package/lib/types/ollama.d.ts +91 -0
  28. package/lib/types/ollama.d.ts.map +1 -0
  29. package/lib/types/ollama.js +184 -0
  30. package/lib/types/ollama.js.map +1 -0
  31. package/lib/types/route.d.ts +52 -0
  32. package/lib/types/route.d.ts.map +1 -0
  33. package/lib/types/route.js +119 -0
  34. package/lib/types/route.js.map +1 -0
  35. package/lib/types/sanitize.d.ts +58 -0
  36. package/lib/types/sanitize.d.ts.map +1 -0
  37. package/lib/types/sanitize.js +110 -0
  38. package/lib/types/sanitize.js.map +1 -0
  39. package/lib/types/serialize.d.ts +65 -0
  40. package/lib/types/serialize.d.ts.map +1 -0
  41. package/lib/types/serialize.js +149 -0
  42. package/lib/types/serialize.js.map +1 -0
  43. package/lib/types/translate.d.ts +57 -0
  44. package/lib/types/translate.d.ts.map +1 -0
  45. package/lib/types/translate.js +169 -0
  46. package/lib/types/translate.js.map +1 -0
  47. package/lib/types/version.d.ts +6 -0
  48. package/lib/types/version.d.ts.map +1 -0
  49. package/lib/types/version.js +6 -0
  50. package/lib/types/version.js.map +1 -0
  51. package/package.json +141 -0
  52. package/src/adapter.ts +158 -0
  53. package/src/config.ts +243 -0
  54. package/src/health.ts +133 -0
  55. package/src/index.ts +279 -0
  56. package/src/ollama.ts +274 -0
  57. package/src/route.ts +127 -0
  58. package/src/sanitize.ts +114 -0
  59. package/src/serialize.ts +178 -0
  60. package/src/translate.ts +205 -0
  61. package/src/version.ts +5 -0
package/lib/index.js ADDED
@@ -0,0 +1,1334 @@
1
+ import { defineTool } from "@deepseek-ai/dsh-tools";
2
+ import z from "@deepseek-ai/schemastery";
3
+ import { CallId, EMPTY_RESPONSE_CODE, LlmAdapter, LlmError, attributionHeaders, contentHasImage, isTokenDelta } from "@deepseek-ai/dsh-llm";
4
+ import { idleWatchdog, timeoutOf } from "@deepseek-ai/dsh-timeout";
5
+ import { tmpdir } from "node:os";
6
+ //#region src/config.ts
7
+ /**
8
+ * Config schema and resolution for `dsh-local-ai`. Every tunable is a
9
+ * validated {@link Config} field changeable from cordis.yml; the resolution
10
+ * step validates URLs, numeric bounds, and route/model entries so
11
+ * misconfiguration fails loud at mount — never silently skips a rule or
12
+ * half-configures the adapter. The plugin is inert until at least one route
13
+ * rule exists or a caller selects the `ollama` provider explicitly: routing
14
+ * every request to a local model requires an explicit opt-in (privacy and
15
+ * cost default: no automatic re-routing).
16
+ * @module dsh-local-ai/config
17
+ */
18
+ /** Schemastery schema: the loader validates and fills defaults before `apply`. */
19
+ const Config = z.object({
20
+ baseURL: z.string().default("http://127.0.0.1:11434"),
21
+ requestTimeoutMs: z.number().default(3e4),
22
+ graceMs: z.number().default(15e3),
23
+ defaultContextWindow: z.number().default(8192),
24
+ maxTokens: z.number().default(4096),
25
+ temperature: z.number(),
26
+ models: z.array(z.object({
27
+ name: z.string().required(),
28
+ model: z.string(),
29
+ contextWindow: z.number(),
30
+ maxTokens: z.number(),
31
+ temperature: z.number()
32
+ })).default([]),
33
+ route: z.array(z.object({
34
+ model: z.string().required(),
35
+ purpose: z.union(["compaction", "session-title"]),
36
+ keywords: z.array(z.string()).default([]),
37
+ always: z.boolean().default(false)
38
+ })).default([])
39
+ });
40
+ /** Throw unless `value` is a positive safe integer. */
41
+ function assertPositiveInt(name, value) {
42
+ if (!Number.isSafeInteger(value) || value <= 0) throw new TypeError(`${name} must be a positive safe integer, got ${String(value)}`);
43
+ }
44
+ /** Throw unless `value` is a finite number in `[min, max]`. */
45
+ function assertFiniteRange(name, value, min, max) {
46
+ if (typeof value !== "number" || !Number.isFinite(value) || value < min || value > max) throw new TypeError(`${name} must be a finite number in [${min}, ${max}], got ${String(value)}`);
47
+ }
48
+ /**
49
+ * Validate an http(s) URL string and normalize it to a clean base (no query,
50
+ * no fragment, no trailing slash).
51
+ * @param name - config key, for the error message.
52
+ * @param value - raw URL value.
53
+ * @returns the normalized base URL.
54
+ */
55
+ function normalizeBaseUrl(name, value) {
56
+ let parsed;
57
+ try {
58
+ parsed = new URL(value);
59
+ } catch (error) {
60
+ throw new TypeError(`${name} must be a valid URL, got ${JSON.stringify(value)} (${error instanceof Error ? error.message : "invalid URL"})`);
61
+ }
62
+ if (parsed.protocol !== "http:" && parsed.protocol !== "https:") throw new TypeError(`${name} must use http(s), got ${JSON.stringify(parsed.protocol)}`);
63
+ parsed.search = "";
64
+ parsed.hash = "";
65
+ return parsed.href.replace(/\/+$/u, "");
66
+ }
67
+ /**
68
+ * Validate raw values and fill explicit defaults. Invalid URLs, numeric
69
+ * bounds, duplicate model names, or empty route/model names throw here —
70
+ * misconfiguration fails loud at mount even when the plugin is mounted
71
+ * without the Schemastery loader.
72
+ * @param config - raw (possibly partial) plugin config.
73
+ * @returns the fully resolved config.
74
+ */
75
+ function resolveConfig(config = {}) {
76
+ const baseURL = normalizeBaseUrl("baseURL", config.baseURL ?? "http://127.0.0.1:11434");
77
+ const requestTimeoutMs = config.requestTimeoutMs ?? 3e4;
78
+ assertPositiveInt("requestTimeoutMs", requestTimeoutMs);
79
+ const graceMs = config.graceMs ?? 15e3;
80
+ assertPositiveInt("graceMs", graceMs);
81
+ const defaultContextWindow = config.defaultContextWindow ?? 8192;
82
+ assertPositiveInt("defaultContextWindow", defaultContextWindow);
83
+ const maxTokens = config.maxTokens ?? 4096;
84
+ assertPositiveInt("maxTokens", maxTokens);
85
+ const temperature = config.temperature;
86
+ if (temperature !== void 0) assertFiniteRange("temperature", temperature, 0, 2);
87
+ const seenNames = /* @__PURE__ */ new Set();
88
+ const models = (config.models ?? []).map((mapping, index) => {
89
+ if (typeof mapping.name !== "string" || mapping.name.trim().length === 0) throw new TypeError(`models[${index}].name must be a non-empty string`);
90
+ const name = mapping.name.trim();
91
+ if (seenNames.has(name)) throw new TypeError(`models[${index}]: duplicate model name ${JSON.stringify(name)}`);
92
+ seenNames.add(name);
93
+ const model = (mapping.model ?? name).trim();
94
+ if (model.length === 0) throw new TypeError(`models[${index}].model must be a non-empty string`);
95
+ const contextWindow = mapping.contextWindow;
96
+ if (contextWindow !== void 0) assertPositiveInt(`models[${index}].contextWindow`, contextWindow);
97
+ const modelMaxTokens = mapping.maxTokens;
98
+ if (modelMaxTokens !== void 0) assertPositiveInt(`models[${index}].maxTokens`, modelMaxTokens);
99
+ const modelTemperature = mapping.temperature;
100
+ if (modelTemperature !== void 0) assertFiniteRange(`models[${index}].temperature`, modelTemperature, 0, 2);
101
+ return {
102
+ name,
103
+ model,
104
+ ...contextWindow === void 0 ? {} : { contextWindow },
105
+ ...modelMaxTokens === void 0 ? {} : { maxTokens: modelMaxTokens },
106
+ ...modelTemperature === void 0 ? {} : { temperature: modelTemperature }
107
+ };
108
+ });
109
+ const route = (config.route ?? []).map((rule, index) => {
110
+ if (typeof rule.model !== "string" || rule.model.trim().length === 0) throw new TypeError(`route[${index}].model must be a non-empty string`);
111
+ const keywords = (rule.keywords ?? []).map((keyword, keywordIndex) => {
112
+ if (typeof keyword !== "string" || keyword.trim().length === 0) throw new TypeError(`route[${index}].keywords[${keywordIndex}] must be a non-empty string`);
113
+ return keyword;
114
+ });
115
+ if (rule.always !== true && rule.purpose === void 0 && keywords.length === 0) throw new TypeError(`route[${index}] must declare a purpose, at least one keyword, or always: true`);
116
+ return {
117
+ model: rule.model.trim(),
118
+ ...rule.purpose === void 0 ? {} : { purpose: rule.purpose },
119
+ keywords,
120
+ always: rule.always ?? false
121
+ };
122
+ });
123
+ return {
124
+ baseURL,
125
+ requestTimeoutMs,
126
+ graceMs,
127
+ defaultContextWindow,
128
+ maxTokens,
129
+ ...temperature === void 0 ? {} : { temperature },
130
+ models,
131
+ route
132
+ };
133
+ }
134
+ //#endregion
135
+ //#region src/sanitize.ts
136
+ /**
137
+ * Display/log sanitization for `dsh-local-ai`. Every value shown to the model
138
+ * or to the user (tool results, the `/ollama` command, error messages) passes
139
+ * through one of these pure functions first, so an endpoint address or a local
140
+ * path can never leak credentials, secret query parameters, or unbounded text.
141
+ *
142
+ * All functions are pure: they depend only on their arguments, never on
143
+ * process state (the caller supplies a home directory for path redaction).
144
+ * @module dsh-local-ai/sanitize
145
+ */
146
+ /** Placeholder substituted for a redacted secret or credential. */
147
+ const REDACTED = "[REDACTED]";
148
+ /** Control characters (C0 + DEL) stripped from every sanitized value. */
149
+ const CONTROL_CHARS = /[\u0000-\u001f\u007f]/gu;
150
+ /** Secret-shaped query-parameter keys removed from endpoint URLs. */
151
+ const SECRET_KEY_PATTERN = /key|token|secret|password|credential|auth/iu;
152
+ /** Built-in secret literal patterns redacted from arbitrary text. */
153
+ const BUILTIN_SECRET_PATTERNS = [
154
+ /\bsk-[A-Za-z0-9]{16,}\b/u,
155
+ /\bghp_[A-Za-z0-9]{20,}\b/u,
156
+ /\bgho_[A-Za-z0-9]{20,}\b/u,
157
+ /\bAKIA[0-9A-Z]{16}\b/u,
158
+ /\bBearer\s+[A-Za-z0-9._~+/=-]{8,}\b/u,
159
+ /-----BEGIN [A-Z ]*PRIVATE KEY-----[A-Za-z0-9+/=\s]*-----END [A-Z ]*PRIVATE KEY-----/u
160
+ ];
161
+ /** Remove C0/DEL control characters from a string. */
162
+ function stripControl(value) {
163
+ return value.replace(CONTROL_CHARS, "");
164
+ }
165
+ /**
166
+ * Truncate a string to `maxChars`, appending an ellipsis when it was cut.
167
+ * A non-positive `maxChars` yields the empty string.
168
+ * @param value - the string to bound.
169
+ * @param maxChars - maximum returned length, including the ellipsis.
170
+ * @returns the bounded string.
171
+ */
172
+ function truncate(value, maxChars) {
173
+ if (value.length <= maxChars) return value;
174
+ if (maxChars <= 1) return maxChars <= 0 ? "" : "…";
175
+ return `${value.slice(0, maxChars - 1)}…`;
176
+ }
177
+ /**
178
+ * Sanitize an endpoint address for display: strip the URL userinfo
179
+ * (`user:pass@`), drop query parameters whose key looks like a secret, strip
180
+ * control characters, and bound the length. Values that are not parseable as
181
+ * a URL are still stripped and truncated.
182
+ * @param value - the raw endpoint (URL string or anything stringifiable).
183
+ * @param maxChars - maximum returned length.
184
+ * @returns the sanitized endpoint text.
185
+ */
186
+ function sanitizeEndpoint(value, maxChars = 2048) {
187
+ const text = stripControl(typeof value === "string" ? value : String(value));
188
+ let out;
189
+ try {
190
+ const url = new URL(text);
191
+ url.username = "";
192
+ url.password = "";
193
+ for (const key of [...url.searchParams.keys()]) if (SECRET_KEY_PATTERN.test(key)) url.searchParams.delete(key);
194
+ out = url.href;
195
+ } catch {
196
+ out = text;
197
+ }
198
+ return truncate(out, maxChars);
199
+ }
200
+ /**
201
+ * Sanitize a local path for display: strip control characters, redact a
202
+ * leading home directory to `~`, and bound the length.
203
+ * @param value - the raw path (string or anything stringifiable).
204
+ * @param home - the user's home directory to redact; omit to skip redaction.
205
+ * @param maxChars - maximum returned length.
206
+ * @returns the sanitized path text.
207
+ */
208
+ function sanitizePath(value, home = "", maxChars = 1024) {
209
+ const text = stripControl(typeof value === "string" ? value : String(value));
210
+ return truncate(home.length > 0 && text.startsWith(home) ? `~${text.slice(home.length)}` : text, maxChars);
211
+ }
212
+ /**
213
+ * Redact built-in secret literals (API keys, GitHub tokens, AWS keys, bearer
214
+ * credentials, PEM private keys) from arbitrary text. Control characters are
215
+ * stripped first.
216
+ * @param value - the raw text (string or anything stringifiable).
217
+ * @returns the text with secret literals replaced by {@link REDACTED}.
218
+ */
219
+ function redactSecrets(value) {
220
+ let out = stripControl(typeof value === "string" ? value : String(value));
221
+ for (const pattern of BUILTIN_SECRET_PATTERNS) out = out.replace(pattern, REDACTED);
222
+ return out;
223
+ }
224
+ /**
225
+ * Sanitize arbitrary display text: redact secrets, strip control characters,
226
+ * and bound the length.
227
+ * @param value - the raw text (string or anything stringifiable).
228
+ * @param maxChars - maximum returned length.
229
+ * @returns the sanitized text.
230
+ */
231
+ function sanitizeText(value, maxChars = 4e3) {
232
+ return truncate(redactSecrets(value), maxChars);
233
+ }
234
+ //#endregion
235
+ //#region src/ollama.ts
236
+ /**
237
+ * Ollama HTTP API client (zero runtime dependencies — plain `fetch`). The
238
+ * harness `LlmAdapter` streams through `/api/chat`; discovery and management
239
+ * tools call `/api/tags`, `/api/show`, `/api/pull`, `/api/delete`, and
240
+ * `/api/version`. Every request carries the harness attribution headers and
241
+ * honors the caller's abort signal; non-2xx responses fail with a normalized
242
+ * `LlmError`. The fetch implementation is injectable for tests.
243
+ * @module dsh-local-ai/ollama
244
+ */
245
+ /** Build an absolute API URL from a normalized base URL. */
246
+ function endpointUrl(baseURL, path) {
247
+ return `${baseURL}${path}`;
248
+ }
249
+ /** Map an HTTP status to a stable LlmError code. */
250
+ function httpErrorCode(status) {
251
+ if (status === 404) return "NOT_FOUND";
252
+ if (status === 400) return "INVALID_REQUEST";
253
+ if (status >= 500) return "SERVER";
254
+ return `HTTP_${status}`;
255
+ }
256
+ /** Throw a normalized LlmError from a non-2xx response, using the body's `error`. */
257
+ async function throwHttpError(response, context) {
258
+ let message = `Ollama API error (HTTP ${response.status}) from ${sanitizeEndpoint(context)}`;
259
+ try {
260
+ const body = await response.json();
261
+ if (typeof body.error === "string" && body.error.length > 0) message = body.error;
262
+ } catch {}
263
+ throw new LlmError(message, httpErrorCode(response.status), { status: response.status });
264
+ }
265
+ /** Send a GET request and parse the JSON response. */
266
+ async function requestJson(baseURL, path, fetchImpl, signal) {
267
+ const response = await fetchImpl(endpointUrl(baseURL, path), {
268
+ method: "GET",
269
+ headers: {
270
+ accept: "application/json",
271
+ ...attributionHeaders()
272
+ },
273
+ ...signal === void 0 ? {} : { signal }
274
+ });
275
+ if (!response.ok) await throwHttpError(response, endpointUrl(baseURL, path));
276
+ return response.json();
277
+ }
278
+ /** Send a POST request and parse the JSON response. */
279
+ async function postJson(baseURL, path, body, fetchImpl, signal) {
280
+ const response = await fetchImpl(endpointUrl(baseURL, path), {
281
+ method: "POST",
282
+ headers: {
283
+ "content-type": "application/json",
284
+ accept: "application/json",
285
+ ...attributionHeaders()
286
+ },
287
+ body: JSON.stringify(body),
288
+ ...signal === void 0 ? {} : { signal }
289
+ });
290
+ if (!response.ok) await throwHttpError(response, endpointUrl(baseURL, path));
291
+ return response.json();
292
+ }
293
+ /** Send a DELETE request and parse the JSON response. */
294
+ async function deleteJson(baseURL, path, body, fetchImpl, signal) {
295
+ const response = await fetchImpl(endpointUrl(baseURL, path), {
296
+ method: "DELETE",
297
+ headers: {
298
+ "content-type": "application/json",
299
+ accept: "application/json",
300
+ ...attributionHeaders()
301
+ },
302
+ body: JSON.stringify(body),
303
+ ...signal === void 0 ? {} : { signal }
304
+ });
305
+ if (!response.ok) await throwHttpError(response, endpointUrl(baseURL, path));
306
+ return response.json();
307
+ }
308
+ /**
309
+ * Send a POST and return the raw `Response` after validating 2xx. Used by the
310
+ * streaming adapter, which owns body decoding and the idle watchdog.
311
+ */
312
+ async function postStream(baseURL, path, body, fetchImpl, signal) {
313
+ const response = await fetchImpl(endpointUrl(baseURL, path), {
314
+ method: "POST",
315
+ headers: {
316
+ "content-type": "application/json",
317
+ accept: "application/x-ndjson",
318
+ ...attributionHeaders()
319
+ },
320
+ body: JSON.stringify(body),
321
+ ...signal === void 0 ? {} : { signal }
322
+ });
323
+ if (!response.ok) await throwHttpError(response, endpointUrl(baseURL, path));
324
+ return response;
325
+ }
326
+ /** List installed models from `/api/tags`. */
327
+ async function listModels(baseURL, fetchImpl, signal) {
328
+ return (await requestJson(baseURL, "/api/tags", fetchImpl, signal)).models ?? [];
329
+ }
330
+ /** List currently-loaded (running) models from `/api/ps`. */
331
+ async function listRunning(baseURL, fetchImpl, signal) {
332
+ return (await requestJson(baseURL, "/api/ps", fetchImpl, signal)).models ?? [];
333
+ }
334
+ /** Inspect one model via `/api/show`. */
335
+ async function showModel(baseURL, name, fetchImpl, signal) {
336
+ return postJson(baseURL, "/api/show", { name }, fetchImpl, signal);
337
+ }
338
+ /** Remove one model via `/api/delete`. */
339
+ async function removeModel(baseURL, name, fetchImpl, signal) {
340
+ await deleteJson(baseURL, "/api/delete", { name }, fetchImpl, signal);
341
+ }
342
+ /** Query the Ollama server version via `/api/version`. */
343
+ async function apiVersion(baseURL, fetchImpl, signal) {
344
+ return (await requestJson(baseURL, "/api/version", fetchImpl, signal)).version;
345
+ }
346
+ /**
347
+ * Pull a model via `/api/pull`, consuming the progress stream and returning the
348
+ * final status. An intermediate error status or a non-2xx response fails loud.
349
+ */
350
+ async function pullModel(baseURL, name, fetchImpl, signal) {
351
+ const response = await postStream(baseURL, "/api/pull", {
352
+ name,
353
+ stream: true
354
+ }, fetchImpl, signal);
355
+ if (!response.body) throw new LlmError("Ollama pull returned no response body", "EMPTY_RESPONSE");
356
+ let last = { status: "success" };
357
+ for await (const line of readNdjsonLines(response.body)) {
358
+ if (line.length === 0) continue;
359
+ const chunk = JSON.parse(line);
360
+ if (typeof chunk.error === "string" && chunk.error.length > 0) throw new LlmError(chunk.error, "PROVIDER");
361
+ if (typeof chunk.status === "string") last = { status: chunk.status };
362
+ }
363
+ return last;
364
+ }
365
+ /**
366
+ * Decode a `ReadableStream<Uint8Array>` into newline-delimited text lines.
367
+ * The final line is yielded even without a trailing newline; a missing body
368
+ * yields nothing.
369
+ * @param body - the response body stream.
370
+ * @returns text lines in delivery order.
371
+ */
372
+ async function* readNdjsonLines(body) {
373
+ const reader = body.getReader();
374
+ const decoder = new TextDecoder();
375
+ let buffer = "";
376
+ try {
377
+ while (true) {
378
+ const { done, value } = await reader.read();
379
+ if (done) break;
380
+ buffer += decoder.decode(value, { stream: true });
381
+ let newline = buffer.indexOf("\n");
382
+ while (newline >= 0) {
383
+ const line = buffer.slice(0, newline).replace(/\r$/u, "");
384
+ buffer = buffer.slice(newline + 1);
385
+ newline = buffer.indexOf("\n");
386
+ yield line;
387
+ }
388
+ }
389
+ } finally {
390
+ reader.releaseLock();
391
+ }
392
+ buffer += decoder.decode();
393
+ if (buffer.length > 0) yield buffer;
394
+ }
395
+ /**
396
+ * Extract the context length from an `/api/show` result by scanning
397
+ * `model_info` for a `*.context_length` or bare `context_length` entry.
398
+ * @param show - the `/api/show` result.
399
+ * @returns the context length, or `undefined` when not reported.
400
+ */
401
+ function contextLengthOf(show) {
402
+ const info = show.model_info;
403
+ if (info === void 0) return void 0;
404
+ for (const [key, value] of Object.entries(info)) if (key === "context_length" || key.endsWith(".context_length")) {
405
+ if (typeof value === "number" && Number.isInteger(value) && value > 0) return value;
406
+ }
407
+ }
408
+ //#endregion
409
+ //#region src/serialize.ts
410
+ /**
411
+ * Serialize harness messages and requests into the Ollama `/api/chat` wire
412
+ * vocabulary. User text is joined; assistant text becomes `content` and tool
413
+ * calls become `tool_calls` (with arguments parsed from the raw JSON string to
414
+ * the object Ollama expects); tool results become separate `tool` messages.
415
+ * Core image blocks are rejected explicitly because this route is text-only;
416
+ * unknown declaration-merged block types retain the documented extension
417
+ * fallback (ignored for content, retained as text where text is expected).
418
+ * @module dsh-local-ai/serialize
419
+ */
420
+ /** Join the text blocks of a message (used for user/tool-result content). */
421
+ function flattenText(blocks) {
422
+ return blocks.filter((block) => block.type === "text").map((block) => block.text).join("");
423
+ }
424
+ /** Reject core image content before any text-flattening path can silently erase it. */
425
+ function assertTextOnly(blocks) {
426
+ if (contentHasImage(blocks)) throw new LlmError("The Ollama adapter does not support image content.", "UNSUPPORTED_CONTENT");
427
+ }
428
+ /**
429
+ * Parse a tool-call argument string into the object Ollama expects. The raw
430
+ * string is guaranteed by the harness contract to be JSON; a malformed value
431
+ * from a hand-built call degrades to a single `value` field rather than
432
+ * bricking the whole session.
433
+ * @param raw - the raw JSON string produced by the model.
434
+ * @returns the parsed object, or a `{ value }` fallback.
435
+ */
436
+ function parseToolArguments(raw) {
437
+ try {
438
+ const parsed = JSON.parse(raw);
439
+ if (typeof parsed === "object" && parsed !== null && !Array.isArray(parsed)) return parsed;
440
+ return { value: raw };
441
+ } catch {
442
+ return { value: raw };
443
+ }
444
+ }
445
+ /** Serialize one assistant message (text + tool calls). */
446
+ function serializeAssistant(message) {
447
+ const text = flattenText(message.content);
448
+ const toolCalls = message.content.filter((block) => block.type === "tool-call").map((block) => ({ function: {
449
+ name: block.name,
450
+ arguments: parseToolArguments(block.arguments)
451
+ } }));
452
+ return {
453
+ role: "assistant",
454
+ content: text,
455
+ ...toolCalls.length > 0 ? { tool_calls: toolCalls } : {}
456
+ };
457
+ }
458
+ /** Resolve a tool-result block's name from the assistant tool calls that precede it. */
459
+ function toolNameOf(callId, namesByCallId) {
460
+ return namesByCallId.get(callId);
461
+ }
462
+ /**
463
+ * Serialize the conversation. `tool-result` blocks become standalone
464
+ * `{role: 'tool'}` messages; the harness puts each tool result in its own
465
+ * user-role message, so a mixed user message contributes its text first and
466
+ * its tool results as separate wire messages after. Assistant tool calls are
467
+ * indexed first so their results can carry the tool name Ollama needs.
468
+ * @param messages - the harness conversation, in order.
469
+ * @returns the wire messages; order preserved, each tool result expanded into its own entry.
470
+ */
471
+ function serializeMessages(messages) {
472
+ const namesByCallId = /* @__PURE__ */ new Map();
473
+ for (const message of messages) {
474
+ if (message.role !== "assistant") continue;
475
+ for (const block of message.content) if (block.type === "tool-call") namesByCallId.set(String(block.id), block.name);
476
+ }
477
+ const wire = [];
478
+ for (const message of messages) {
479
+ assertTextOnly(message.content);
480
+ if (message.role === "system") {
481
+ wire.push({
482
+ role: "system",
483
+ content: flattenText(message.content)
484
+ });
485
+ continue;
486
+ }
487
+ if (message.role === "assistant") {
488
+ wire.push(serializeAssistant(message));
489
+ continue;
490
+ }
491
+ const toolResults = message.content.filter((block) => block.type === "tool-result");
492
+ const text = flattenText(message.content);
493
+ if (text.length > 0 || toolResults.length === 0) wire.push({
494
+ role: "user",
495
+ content: text
496
+ });
497
+ for (const result of toolResults) {
498
+ const name = toolNameOf(String(result.toolCallId), namesByCallId);
499
+ wire.push({
500
+ role: "tool",
501
+ content: flattenText(result.content) || "(no output)",
502
+ ...name === void 0 ? {} : { tool_name: name }
503
+ });
504
+ }
505
+ }
506
+ return wire;
507
+ }
508
+ /**
509
+ * Build the full wire request. Always streaming (`stream: true`); optional
510
+ * fields are omitted rather than sent as null, so Ollama defaults apply.
511
+ * `temperature` resolves request → model mapping → plugin default; `num_predict`
512
+ * is the harness-materialized `maxTokens`.
513
+ * @param options - the harness request (model, history, system, tools, sampling).
514
+ * @param resolved - the resolved plugin config.
515
+ * @returns the `/api/chat` request body.
516
+ */
517
+ function serializeRequest(options, resolved) {
518
+ const mapping = resolved.models.find((entry) => entry.name === options.model);
519
+ const model = mapping?.model ?? options.model;
520
+ const messages = [];
521
+ if (options.system !== void 0) messages.push({
522
+ role: "system",
523
+ content: options.system
524
+ });
525
+ messages.push(...serializeMessages(options.messages));
526
+ const temperature = options.temperature ?? mapping?.temperature ?? resolved.temperature;
527
+ const ollamaOptions = {};
528
+ if (temperature !== void 0) ollamaOptions.temperature = temperature;
529
+ if (options.maxTokens !== void 0) ollamaOptions.num_predict = options.maxTokens;
530
+ if (options.stop !== void 0 && options.stop.length > 0) ollamaOptions.stop = options.stop;
531
+ const tools = options.tools?.map((tool) => ({
532
+ type: "function",
533
+ function: tool
534
+ }));
535
+ return {
536
+ model,
537
+ messages,
538
+ stream: true,
539
+ ...Object.keys(ollamaOptions).length > 0 ? { options: ollamaOptions } : {},
540
+ ...tools !== void 0 && tools.length > 0 ? { tools } : {}
541
+ };
542
+ }
543
+ //#endregion
544
+ //#region src/translate.ts
545
+ /**
546
+ * Translate Ollama NDJSON chat chunks into the harness `StreamChunk` protocol.
547
+ * Ollama streams one JSON object per line: `message.content` and
548
+ * `message.thinking` are incremental deltas, while `message.tool_calls`
549
+ * carries the CUMULATIVE arguments object, so tool-call deltas are computed by
550
+ * longest-common-prefix diffing. Usage and the finish reason are deferred to
551
+ * the `done: true` chunk, guaranteeing no chunk follows `finish`.
552
+ * @module dsh-local-ai/translate
553
+ */
554
+ /**
555
+ * Map the Ollama `done_reason` vocabulary to the harness FinishReason.
556
+ * @param reason - the wire `done_reason` string.
557
+ * @returns the mapped reason; unrecognized values become `{kind: 'error'}`.
558
+ */
559
+ function mapFinishReason(reason) {
560
+ switch (reason) {
561
+ case "stop": return { kind: "stop" };
562
+ case "tool_calls": return { kind: "tool-calls" };
563
+ case "length": return { kind: "max-tokens" };
564
+ default: return {
565
+ kind: "error",
566
+ failure: {
567
+ message: `model stopped: ${reason}`,
568
+ code: reason.toUpperCase()
569
+ }
570
+ };
571
+ }
572
+ }
573
+ /** Assemble the final ContentBlock for one open block. */
574
+ function closeBlock(block) {
575
+ switch (block.kind) {
576
+ case "text": return {
577
+ type: "text",
578
+ text: block.text
579
+ };
580
+ case "reasoning": return {
581
+ type: "reasoning",
582
+ text: block.text
583
+ };
584
+ case "tool-call": return {
585
+ type: "tool-call",
586
+ id: CallId(block.callId ?? ""),
587
+ name: block.name ?? "",
588
+ arguments: block.text
589
+ };
590
+ }
591
+ }
592
+ /**
593
+ * Compute the append delta from a cumulative JSON string, so the harness's
594
+ * delta-concatenating assembler reconstructs the full arguments. Ollama grows
595
+ * the arguments object monotonically, so the delta is everything past the
596
+ * longest common prefix with the previously seen string.
597
+ * @param previous - the previously seen cumulative JSON (or `''`).
598
+ * @param next - the new cumulative JSON.
599
+ * @returns the fragment to append.
600
+ */
601
+ function argumentsDelta(previous, next) {
602
+ if (next.startsWith(previous)) return next.slice(previous.length);
603
+ let index = 0;
604
+ while (index < previous.length && index < next.length && previous[index] === next[index]) index += 1;
605
+ return next.slice(index);
606
+ }
607
+ /**
608
+ * Consume Ollama NDJSON lines and yield StreamChunks. Text and reasoning deltas
609
+ * stream as they arrive; tool-call deltas are diffed from the cumulative wire
610
+ * arguments; `block-end`, `usage`, and `finish` are deferred to the `done`
611
+ * chunk. A `stop` finish with no opened blocks maps to an `EMPTY_RESPONSE`
612
+ * error finish.
613
+ * @param lines - newline-delimited Ollama chat payloads.
614
+ * @returns deltas as they arrive; `block-end`s, `usage`, and `finish` deferred to `done`.
615
+ */
616
+ async function* translate(lines) {
617
+ let nextIndex = 0;
618
+ let textBlock;
619
+ let reasoningBlock;
620
+ const toolBlocks = /* @__PURE__ */ new Map();
621
+ const toolArguments = /* @__PURE__ */ new Map();
622
+ const order = [];
623
+ function open(kind) {
624
+ const block = {
625
+ index: nextIndex++,
626
+ kind,
627
+ text: ""
628
+ };
629
+ order.push(block);
630
+ return block;
631
+ }
632
+ for await (const line of lines) {
633
+ if (line.length === 0) continue;
634
+ let chunk;
635
+ try {
636
+ chunk = JSON.parse(line);
637
+ } catch {
638
+ throw new LlmError(`malformed Ollama NDJSON payload: ${line.slice(0, 120)}`, "MALFORMED_RESPONSE");
639
+ }
640
+ if (typeof chunk.error === "string" && chunk.error.length > 0) throw new LlmError(chunk.error, "PROVIDER");
641
+ const message = chunk.message;
642
+ if (message !== void 0) {
643
+ const reasoning = message.thinking;
644
+ if (typeof reasoning === "string" && reasoning.length > 0) {
645
+ if (!reasoningBlock) {
646
+ reasoningBlock = open("reasoning");
647
+ yield {
648
+ type: "block-start",
649
+ index: reasoningBlock.index,
650
+ blockType: "reasoning"
651
+ };
652
+ }
653
+ reasoningBlock.text += reasoning;
654
+ yield {
655
+ type: "reasoning-delta",
656
+ index: reasoningBlock.index,
657
+ text: reasoning
658
+ };
659
+ }
660
+ const content = message.content;
661
+ if (typeof content === "string" && content.length > 0) {
662
+ if (!textBlock) {
663
+ textBlock = open("text");
664
+ yield {
665
+ type: "block-start",
666
+ index: textBlock.index,
667
+ blockType: "text"
668
+ };
669
+ }
670
+ textBlock.text += content;
671
+ yield {
672
+ type: "text-delta",
673
+ index: textBlock.index,
674
+ text: content
675
+ };
676
+ }
677
+ const toolCalls = message.tool_calls ?? [];
678
+ for (let callIndex = 0; callIndex < toolCalls.length; callIndex++) {
679
+ const call = toolCalls[callIndex];
680
+ let block = toolBlocks.get(callIndex);
681
+ if (!block) {
682
+ block = open("tool-call");
683
+ toolBlocks.set(callIndex, block);
684
+ toolArguments.set(callIndex, "");
685
+ yield {
686
+ type: "block-start",
687
+ index: block.index,
688
+ blockType: "tool-call"
689
+ };
690
+ }
691
+ if (call?.function?.name !== void 0 && block.name === void 0) block.name = call.function.name;
692
+ const cumulative = JSON.stringify(call?.function?.arguments ?? {});
693
+ const previous = toolArguments.get(callIndex) ?? "";
694
+ if (cumulative !== previous) {
695
+ const fragment = argumentsDelta(previous, cumulative);
696
+ toolArguments.set(callIndex, cumulative);
697
+ block.text += fragment;
698
+ yield {
699
+ type: "tool-call-delta",
700
+ index: block.index,
701
+ id: CallId(block.callId ?? ""),
702
+ ...block.name !== void 0 ? { name: block.name } : {},
703
+ argumentsDelta: fragment
704
+ };
705
+ }
706
+ }
707
+ }
708
+ if (chunk.done === true) {
709
+ for (const block of order) yield {
710
+ type: "block-end",
711
+ index: block.index,
712
+ block: closeBlock(block)
713
+ };
714
+ const promptCount = chunk.prompt_eval_count;
715
+ const evalCount = chunk.eval_count;
716
+ if (promptCount !== void 0 || evalCount !== void 0) yield {
717
+ type: "usage",
718
+ usage: {
719
+ inputTokens: promptCount ?? 0,
720
+ outputTokens: evalCount ?? 0
721
+ }
722
+ };
723
+ const reason = mapFinishReason(chunk.done_reason ?? "stop");
724
+ yield {
725
+ type: "finish",
726
+ reason: reason.kind === "stop" && order.length === 0 ? {
727
+ kind: "error",
728
+ failure: {
729
+ message: "model returned a completed response with no content",
730
+ code: EMPTY_RESPONSE_CODE
731
+ }
732
+ } : reason
733
+ };
734
+ return;
735
+ }
736
+ }
737
+ throw new LlmError("Ollama NDJSON stream ended without a done chunk", "STREAM_CLOSED");
738
+ }
739
+ //#endregion
740
+ //#region \0@oxc-project+runtime@0.144.0/helpers/esm/usingCtx.js
741
+ function _usingCtx() {
742
+ var r = "function" == typeof SuppressedError ? SuppressedError : function(r, e) {
743
+ var n = Error();
744
+ return n.name = "SuppressedError", n.error = r, n.suppressed = e, n;
745
+ }, e = {}, n = [];
746
+ function using(r, e) {
747
+ if (null != e) {
748
+ if (Object(e) !== e) throw new TypeError("using declarations can only be used with objects, functions, null, or undefined.");
749
+ if (r) var o = e[Symbol.asyncDispose || Symbol["for"]("Symbol.asyncDispose")];
750
+ if (void 0 === o && (o = e[Symbol.dispose || Symbol["for"]("Symbol.dispose")], r)) var t = o;
751
+ if ("function" != typeof o) throw new TypeError("Object is not disposable.");
752
+ t && (o = function o() {
753
+ try {
754
+ t.call(e);
755
+ } catch (r) {
756
+ return Promise.reject(r);
757
+ }
758
+ }), n.push({
759
+ v: e,
760
+ d: o,
761
+ a: r
762
+ });
763
+ } else r && n.push({
764
+ d: e,
765
+ a: r
766
+ });
767
+ return e;
768
+ }
769
+ return {
770
+ e,
771
+ u: using.bind(null, !1),
772
+ a: using.bind(null, !0),
773
+ d: function d() {
774
+ var o, t = this.e, s = 0;
775
+ function next() {
776
+ for (; o = n.pop();) try {
777
+ if (!o.a && 1 === s) return s = 0, n.push(o), Promise.resolve().then(next);
778
+ if (o.d) {
779
+ var r = o.d.call(o.v);
780
+ if (o.a) return s |= 2, Promise.resolve(r).then(next, err);
781
+ } else s |= 1;
782
+ } catch (r) {
783
+ return err(r);
784
+ }
785
+ if (1 === s) return t !== e ? Promise.reject(t) : Promise.resolve();
786
+ if (t !== e) throw t;
787
+ }
788
+ function err(n) {
789
+ return t = t !== e ? new r(n, t) : n, next();
790
+ }
791
+ return next();
792
+ }
793
+ };
794
+ }
795
+ //#endregion
796
+ //#region src/adapter.ts
797
+ /**
798
+ * `OllamaAdapter`: the harness `LlmAdapter` over Ollama's native `/api/chat`
799
+ * streaming endpoint. Transport-only: the registering plugin owns config
800
+ * resolution (one `() => ResolvedConfig` thunk re-read per operation), so a
801
+ * changed base URL, model mapping, or timeout reaches the next request without
802
+ * re-registration, while an in-flight stream keeps the facts it started with.
803
+ * The adapter is text-only (`inputModalities: ['text']`); tool calls and tool
804
+ * results are translated by {@link serializeRequest}.
805
+ * @module dsh-local-ai/adapter
806
+ */
807
+ /** Idle-timeout abort code stamped onto a stalled stream's timeout reason. */
808
+ const STREAM_IDLE_TIMEOUT_CODE = "LLM_STREAM_IDLE_TIMEOUT";
809
+ /** The single provider route this adapter owns. */
810
+ const OLLAMA_PROVIDER = "ollama";
811
+ /** Reverse a model mapping: Ollama model id → harness-visible name. */
812
+ function harnessNameOf(resolved, ollamaName) {
813
+ return resolved.models.find((entry) => entry.model === ollamaName)?.name ?? ollamaName;
814
+ }
815
+ /** Advertise one configured or discovered local model as text-only. */
816
+ function modelInfo(provider, id, name) {
817
+ return {
818
+ provider,
819
+ id,
820
+ name,
821
+ inputModalities: ["text"]
822
+ };
823
+ }
824
+ /**
825
+ * The Ollama provider adapter. One instance serves every harness-visible local
826
+ * model name; the harness model name maps to the wire model id through the
827
+ * configured model mapping (identity when unmapped).
828
+ */
829
+ var OllamaAdapter = class extends LlmAdapter {
830
+ options;
831
+ constructor(options) {
832
+ super();
833
+ this.options = options;
834
+ }
835
+ fetchImpl() {
836
+ return this.options.fetchImpl ?? ((input, init) => globalThis.fetch(input, init));
837
+ }
838
+ providerInfo(provider) {
839
+ return {
840
+ id: provider,
841
+ name: "Ollama (local)"
842
+ };
843
+ }
844
+ listModels(provider) {
845
+ const resolved = this.options.config();
846
+ return listModels(resolved.baseURL, this.fetchImpl()).then((models) => models.map((model) => modelInfo(provider, harnessNameOf(resolved, model.name), harnessNameOf(resolved, model.name)))).catch(() => resolved.models.map((entry) => modelInfo(provider, entry.name, entry.name)));
847
+ }
848
+ resolveModel(provider, model, _signal) {
849
+ const resolved = this.options.config();
850
+ const mapping = resolved.models.find((entry) => entry.name === model);
851
+ return Promise.resolve({
852
+ provider,
853
+ id: model,
854
+ name: model,
855
+ inputModalities: ["text"],
856
+ context: { contextWindow: mapping?.contextWindow ?? resolved.defaultContextWindow },
857
+ defaultMaxTokens: mapping?.maxTokens ?? resolved.maxTokens
858
+ });
859
+ }
860
+ async *stream(options) {
861
+ try {
862
+ var _usingCtx$1 = _usingCtx();
863
+ const resolved = this.options.config();
864
+ const consumer = new AbortController();
865
+ const upstream = options.signal === void 0 ? consumer.signal : AbortSignal.any([options.signal, consumer.signal]);
866
+ const watchdog = _usingCtx$1.u(idleWatchdog(upstream, resolved.requestTimeoutMs, STREAM_IDLE_TIMEOUT_CODE));
867
+ const iterator = this.request(options, resolved, watchdog.signal)[Symbol.asyncIterator]();
868
+ let exhausted = false;
869
+ try {
870
+ while (true) {
871
+ const result = await watchdog.next(iterator);
872
+ if (result.done) {
873
+ exhausted = true;
874
+ return;
875
+ }
876
+ yield result.value;
877
+ }
878
+ } catch (error) {
879
+ if (timeoutOf(watchdog.signal, STREAM_IDLE_TIMEOUT_CODE) !== void 0) throw new LlmError(`Ollama stream idle timeout after ${resolved.requestTimeoutMs}ms`, "TIMEOUT", { cause: error });
880
+ if (options.signal?.aborted) throw new LlmError("Ollama request aborted by caller", "ABORTED", { cause: error });
881
+ if (error instanceof LlmError) throw error;
882
+ throw new LlmError(`Ollama API stream from ${sanitizeEndpoint(resolved.baseURL)} failed`, "TRANSPORT", { cause: error });
883
+ } finally {
884
+ consumer.abort("Ollama stream consumer stopped");
885
+ if (!exhausted && iterator.return !== void 0) try {
886
+ await iterator.return();
887
+ } catch (_abortedTransportTeardown) {}
888
+ }
889
+ } catch (_) {
890
+ _usingCtx$1.e = _;
891
+ } finally {
892
+ _usingCtx$1.d();
893
+ }
894
+ }
895
+ async *request(options, resolved, signal) {
896
+ const body = serializeRequest(options, resolved);
897
+ let response;
898
+ try {
899
+ response = await postStream(resolved.baseURL, "/api/chat", body, this.fetchImpl(), signal);
900
+ } catch (error) {
901
+ if (signal.aborted) throw error;
902
+ throw new LlmError(`Ollama API request to ${sanitizeEndpoint(resolved.baseURL)} failed`, "TRANSPORT", { cause: error });
903
+ }
904
+ if (!response.body) throw new LlmError("Ollama API returned no response body", "EMPTY_RESPONSE");
905
+ yield* translate(readNdjsonLines(response.body));
906
+ }
907
+ };
908
+ //#endregion
909
+ //#region src/route.ts
910
+ /**
911
+ * Local-model routing decision and the streaming fallback. The pure decision
912
+ * matches configured rules (task type via `purpose`, case-insensitive
913
+ * keywords, or a blanket `always`) against a request, in list order — first
914
+ * match wins. The streaming helper routes a matched request to the local
915
+ * Ollama adapter and, when the local route fails BEFORE producing any visible
916
+ * content, falls back to the cloud (`next()`) so a down Ollama never bricks a
917
+ * conversation. Once local content has started, it is streamed through — a
918
+ * mid-stream failure cannot be retracted.
919
+ * @module dsh-local-ai/route
920
+ */
921
+ /**
922
+ * Concatenate the request's model-visible text (system prompt + every text
923
+ * block) for keyword matching.
924
+ * @param options - the request.
925
+ * @returns the joined text.
926
+ */
927
+ function requestText(options) {
928
+ const parts = [];
929
+ if (options.system !== void 0) parts.push(options.system);
930
+ for (const message of options.messages) for (const block of message.content) if (block.type === "text") parts.push(block.text);
931
+ return parts.join("\n");
932
+ }
933
+ /** Whether a keyword appears case-insensitively in the text. */
934
+ function matchesKeyword(text, keyword) {
935
+ return text.toLowerCase().includes(keyword.toLowerCase());
936
+ }
937
+ /** Whether one resolved rule matches a request. */
938
+ function ruleMatches(rule, options) {
939
+ if (rule.always) return true;
940
+ if (rule.purpose !== void 0 && options.purpose === rule.purpose) return true;
941
+ if (rule.keywords.length > 0) {
942
+ const text = requestText(options);
943
+ return rule.keywords.some((keyword) => matchesKeyword(text, keyword));
944
+ }
945
+ return false;
946
+ }
947
+ /**
948
+ * Decide whether a request should route to a local model. A request already
949
+ * addressed to the `ollama` provider (explicit selection or a prior re-route)
950
+ * never re-routes.
951
+ * @param options - the request.
952
+ * @param resolved - the resolved config.
953
+ * @returns the local model to use, or `undefined` to stay on the cloud route.
954
+ */
955
+ function decideRoute(options, resolved) {
956
+ if (options.provider === "ollama") return void 0;
957
+ for (const rule of resolved.route) if (ruleMatches(rule, options)) return { model: rule.model };
958
+ }
959
+ /**
960
+ * Stream a locally-routed request with automatic cloud fallback. The local
961
+ * stream is produced through `streamLocal` (the full harness stream, so the
962
+ * local route keeps retry and failure normalization). If the local route
963
+ * finishes with an error or aborts before any token delta, `next()` (the
964
+ * cloud) is streamed instead; otherwise the local stream is forwarded.
965
+ * @param streamLocal - produces the local stream for a re-routed request.
966
+ * @param options - the original request.
967
+ * @param decision - the local model to route to.
968
+ * @param next - the cloud stream (the waterfall's `next()`).
969
+ * @returns the effective chunk stream.
970
+ */
971
+ async function* routeLocal(streamLocal, options, decision, next) {
972
+ const upstream = streamLocal({
973
+ ...options,
974
+ provider: OLLAMA_PROVIDER,
975
+ model: decision.model
976
+ });
977
+ let producedContent = false;
978
+ const pending = [];
979
+ try {
980
+ for await (const chunk of upstream) {
981
+ if (producedContent) {
982
+ yield chunk;
983
+ continue;
984
+ }
985
+ if (chunk.type === "finish") {
986
+ if (chunk.reason.kind === "error" || chunk.reason.kind === "aborted") {
987
+ yield* next();
988
+ return;
989
+ }
990
+ yield chunk;
991
+ return;
992
+ }
993
+ pending.push(chunk);
994
+ if (isTokenDelta(chunk)) {
995
+ producedContent = true;
996
+ for (const buffered of pending) yield buffered;
997
+ pending.length = 0;
998
+ }
999
+ }
1000
+ for (const buffered of pending) yield buffered;
1001
+ } catch (error) {
1002
+ if (!producedContent) {
1003
+ yield* next();
1004
+ return;
1005
+ }
1006
+ for (const buffered of pending) yield buffered;
1007
+ throw error;
1008
+ }
1009
+ }
1010
+ //#endregion
1011
+ //#region src/health.ts
1012
+ /** The default probe: `ollama list` reaches the server through the CLI's own channel. */
1013
+ const OLLAMA_PROCESS_PROBE = {
1014
+ command: "ollama",
1015
+ args: ["list"]
1016
+ };
1017
+ /**
1018
+ * Check whether the Ollama HTTP API responds to `/api/version` within a
1019
+ * deadline. A timeout, transport failure, or non-2xx response is `ok: false`
1020
+ * with a sanitized error.
1021
+ * @param baseURL - the Ollama base URL.
1022
+ * @param fetchImpl - the fetch implementation.
1023
+ * @param timeoutMs - deadline in milliseconds.
1024
+ * @returns the API health result.
1025
+ */
1026
+ async function checkApiHealth(baseURL, fetchImpl, timeoutMs) {
1027
+ const controller = new AbortController();
1028
+ const timer = setTimeout(() => controller.abort(/* @__PURE__ */ new Error("timeout")), timeoutMs);
1029
+ try {
1030
+ return {
1031
+ ok: true,
1032
+ version: sanitizeText(await apiVersion(baseURL, fetchImpl, controller.signal), 64)
1033
+ };
1034
+ } catch (error) {
1035
+ return {
1036
+ ok: false,
1037
+ error: sanitizeText(error instanceof Error ? error.message : String(error), 500)
1038
+ };
1039
+ } finally {
1040
+ clearTimeout(timer);
1041
+ }
1042
+ }
1043
+ /**
1044
+ * Check process liveness through the subprocess seam by spawning the probe
1045
+ * command in collect mode. Exit code 0 means the CLI (and, for the default
1046
+ * `ollama list` probe, the server it talks to) is alive; a missing executable
1047
+ * or non-zero exit is `present: false` with a sanitized error.
1048
+ * @param subprocess - the real subprocess runtime.
1049
+ * @param graceMs - terminate grace for the spawned probe.
1050
+ * @param probe - the command to probe with (defaults to `ollama list`).
1051
+ * @returns the process health result.
1052
+ */
1053
+ async function checkProcessHealth(subprocess, graceMs, probe = OLLAMA_PROCESS_PROBE) {
1054
+ let executable;
1055
+ try {
1056
+ executable = await subprocess.resolveExecutable(probe.command);
1057
+ } catch (error) {
1058
+ return {
1059
+ present: false,
1060
+ error: sanitizeText(`"${probe.command}" not found: ${error instanceof Error ? error.message : String(error)}`, 500)
1061
+ };
1062
+ }
1063
+ const handle = subprocess.spawn({
1064
+ argv: [executable, ...probe.args],
1065
+ cwd: tmpdir(),
1066
+ stdio: {
1067
+ stdin: "ignore",
1068
+ stdout: { maxBytes: 4096 },
1069
+ stderr: { maxBytes: 4096 }
1070
+ },
1071
+ graceMs
1072
+ });
1073
+ const outcome = await handle.done;
1074
+ if (outcome.exitCode === 0) return { present: true };
1075
+ const stderr = handle.collected.stderr?.readFrom(0).text ?? "";
1076
+ return {
1077
+ present: false,
1078
+ error: sanitizeText(stderr.trim().length > 0 ? stderr : `exit code ${String(outcome.exitCode)}`, 500)
1079
+ };
1080
+ }
1081
+ /**
1082
+ * Check both health signals.
1083
+ * @param baseURL - the Ollama base URL.
1084
+ * @param fetchImpl - the fetch implementation.
1085
+ * @param subprocess - the real subprocess runtime.
1086
+ * @param requestTimeoutMs - HTTP deadline in milliseconds.
1087
+ * @param graceMs - subprocess terminate grace in milliseconds.
1088
+ * @returns the combined health result.
1089
+ */
1090
+ async function checkHealth(baseURL, fetchImpl, subprocess, requestTimeoutMs, graceMs) {
1091
+ const [api, process] = await Promise.all([checkApiHealth(baseURL, fetchImpl, requestTimeoutMs), checkProcessHealth(subprocess, graceMs)]);
1092
+ return {
1093
+ api,
1094
+ process
1095
+ };
1096
+ }
1097
+ //#endregion
1098
+ //#region src/version.ts
1099
+ /**
1100
+ * The `dsh-local-ai` plugin version. The release script bumps this string
1101
+ * alongside `package.json`; `test/version.spec.ts` trips when the two drift.
1102
+ */
1103
+ const VERSION = "0.1.0";
1104
+ //#endregion
1105
+ //#region src/index.ts
1106
+ const name = "local-ai";
1107
+ const inject = [
1108
+ "llm",
1109
+ "tools",
1110
+ "subprocess",
1111
+ "commands"
1112
+ ];
1113
+ /** Format a byte count into a compact human-readable string. */
1114
+ function formatBytes(bytes) {
1115
+ if (!Number.isFinite(bytes) || bytes < 0) return "0 B";
1116
+ const units = [
1117
+ "B",
1118
+ "KB",
1119
+ "MB",
1120
+ "GB",
1121
+ "TB"
1122
+ ];
1123
+ let value = bytes;
1124
+ let unit = 0;
1125
+ while (value >= 1024 && unit < units.length - 1) {
1126
+ value /= 1024;
1127
+ unit += 1;
1128
+ }
1129
+ return `${unit === 0 ? String(Math.round(value)) : value.toFixed(1)} ${units[unit]}`;
1130
+ }
1131
+ /** Render the `ollama_list` canonical value as model-visible text. */
1132
+ function renderList(value) {
1133
+ const list = value;
1134
+ const models = list.models ?? [];
1135
+ const lines = [`${models.length} local model(s), ${formatBytes(list.totalBytes ?? 0)} on disk`];
1136
+ for (const model of models) {
1137
+ const detail = [model.parameterSize, model.quantization].filter((part) => part !== void 0).join(" ");
1138
+ lines.push(`- ${model.name}${detail.length > 0 ? ` (${detail})` : ""} — ${formatBytes(model.size)}${model.running ? " [running]" : ""}`);
1139
+ }
1140
+ if (list.running !== void 0 && list.running.length > 0) lines.push(`running: ${list.running.join(", ")}`);
1141
+ return lines.join("\n");
1142
+ }
1143
+ /** Render the `ollama_show` canonical value as model-visible text. */
1144
+ function renderShow(value) {
1145
+ const show = value;
1146
+ const parts = [
1147
+ [show.parameterSize, show.quantization].filter((part) => part !== void 0).join(" "),
1148
+ show.contextLength !== void 0 ? `context ${show.contextLength}` : void 0,
1149
+ show.family,
1150
+ show.format
1151
+ ].filter((part) => part !== void 0 && part.length > 0);
1152
+ return `${show.name}${parts.length > 0 ? ` — ${parts.join(", ")}` : ""}`;
1153
+ }
1154
+ /** Render a health canonical value as model-visible text. */
1155
+ function renderHealth(value) {
1156
+ const health = value;
1157
+ return [`API: ${health.api.ok ? `ok${health.api.version !== void 0 ? ` (v${health.api.version})` : ""}` : "down"}`, `process: ${health.process.present ? "alive" : "not detected"}`].join("\n");
1158
+ }
1159
+ /** Render a pull/remove canonical value as model-visible text. */
1160
+ function renderOperation(value) {
1161
+ const op = value;
1162
+ if (op.removed === true) return `removed ${op.name}`;
1163
+ return `${op.name}: ${op.status ?? "done"}`;
1164
+ }
1165
+ /**
1166
+ * Mount the plugin: resolve config (fail loud), register the Ollama adapter,
1167
+ * the `llm/stream` routing waterfall, the five management tools, and the
1168
+ * `/ollama` command. Every contribution goes through its registry's effect
1169
+ * (register/on), so stop and hot-reload withdraw all of it.
1170
+ * @param ctx - the plugin context (host).
1171
+ * @param config - raw plugin config.
1172
+ */
1173
+ function apply(ctx, config = {}) {
1174
+ const resolved = resolveConfig(config);
1175
+ const logger = ctx.logger("local-ai");
1176
+ const fetchImpl = (input, init) => globalThis.fetch(input, init);
1177
+ const adapter = new OllamaAdapter({
1178
+ config: () => resolved,
1179
+ fetchImpl
1180
+ });
1181
+ ctx.llm.registerAdapter([OLLAMA_PROVIDER], adapter);
1182
+ ctx.on("llm/stream", (options, next) => {
1183
+ const decision = decideRoute(options, resolved);
1184
+ if (decision === void 0) return next();
1185
+ return routeLocal((reRouted) => ctx.llm.stream(reRouted), options, decision, next);
1186
+ });
1187
+ ctx.tools.register(defineTool({
1188
+ name: "ollama_list",
1189
+ description: "List local Ollama models with disk usage and which are currently loaded (running).",
1190
+ parameters: {},
1191
+ output: {
1192
+ schema: { type: "json" },
1193
+ render: (_args, value) => [{
1194
+ type: "text",
1195
+ text: renderList(value)
1196
+ }]
1197
+ },
1198
+ async execute(_args, exec) {
1199
+ const [models, running] = await Promise.all([listModels(resolved.baseURL, fetchImpl, exec.signal), listRunning(resolved.baseURL, fetchImpl, exec.signal).catch(() => [])]);
1200
+ const runningNames = new Set(running.map((model) => model.name));
1201
+ return {
1202
+ models: models.map((model) => ({
1203
+ name: model.name,
1204
+ size: model.size,
1205
+ ...model.details?.parameter_size === void 0 ? {} : { parameterSize: model.details.parameter_size },
1206
+ ...model.details?.quantization_level === void 0 ? {} : { quantization: model.details.quantization_level },
1207
+ running: runningNames.has(model.name)
1208
+ })),
1209
+ running: running.map((model) => model.name),
1210
+ count: models.length,
1211
+ totalBytes: models.reduce((sum, model) => sum + model.size, 0)
1212
+ };
1213
+ }
1214
+ }));
1215
+ ctx.tools.register(defineTool({
1216
+ name: "ollama_show",
1217
+ description: "Show details for one local Ollama model: parameter size, quantization, and context length.",
1218
+ parameters: { name: {
1219
+ type: "string",
1220
+ required: true,
1221
+ description: "The Ollama model name to inspect."
1222
+ } },
1223
+ output: {
1224
+ schema: { type: "json" },
1225
+ render: (_args, value) => [{
1226
+ type: "text",
1227
+ text: renderShow(value)
1228
+ }]
1229
+ },
1230
+ async execute(args, exec) {
1231
+ const name = args.name;
1232
+ const show = await showModel(resolved.baseURL, name, fetchImpl, exec.signal);
1233
+ return {
1234
+ name,
1235
+ ...show.details?.parameter_size === void 0 ? {} : { parameterSize: show.details.parameter_size },
1236
+ ...show.details?.quantization_level === void 0 ? {} : { quantization: show.details.quantization_level },
1237
+ ...show.details?.family === void 0 ? {} : { family: show.details.family },
1238
+ ...show.details?.format === void 0 ? {} : { format: show.details.format },
1239
+ ...(() => {
1240
+ const contextLength = contextLengthOf(show);
1241
+ return contextLength === void 0 ? {} : { contextLength };
1242
+ })()
1243
+ };
1244
+ }
1245
+ }));
1246
+ ctx.tools.register(defineTool({
1247
+ name: "ollama_pull",
1248
+ description: "Pull (download) a model into the local Ollama server.",
1249
+ parameters: { name: {
1250
+ type: "string",
1251
+ required: true,
1252
+ description: "The Ollama model name to pull (e.g. llama3.2)."
1253
+ } },
1254
+ output: {
1255
+ schema: { type: "json" },
1256
+ render: (_args, value) => [{
1257
+ type: "text",
1258
+ text: renderOperation(value)
1259
+ }]
1260
+ },
1261
+ async execute(args, exec) {
1262
+ const name = args.name;
1263
+ return {
1264
+ name,
1265
+ status: (await pullModel(resolved.baseURL, name, fetchImpl, exec.signal)).status
1266
+ };
1267
+ }
1268
+ }));
1269
+ ctx.tools.register(defineTool({
1270
+ name: "ollama_remove",
1271
+ description: "Remove (delete) a model from the local Ollama server.",
1272
+ parameters: { name: {
1273
+ type: "string",
1274
+ required: true,
1275
+ description: "The Ollama model name to remove."
1276
+ } },
1277
+ output: {
1278
+ schema: { type: "json" },
1279
+ render: (_args, value) => [{
1280
+ type: "text",
1281
+ text: renderOperation(value)
1282
+ }]
1283
+ },
1284
+ async execute(args, exec) {
1285
+ const name = args.name;
1286
+ await removeModel(resolved.baseURL, name, fetchImpl, exec.signal);
1287
+ return {
1288
+ name,
1289
+ removed: true
1290
+ };
1291
+ }
1292
+ }));
1293
+ ctx.tools.register(defineTool({
1294
+ name: "ollama_health",
1295
+ description: "Check the local Ollama server: whether the process is alive and whether the API responds.",
1296
+ parameters: {},
1297
+ output: {
1298
+ schema: { type: "json" },
1299
+ render: (_args, value) => [{
1300
+ type: "text",
1301
+ text: renderHealth(value)
1302
+ }]
1303
+ },
1304
+ async execute(_args, exec) {
1305
+ return await checkHealth(resolved.baseURL, fetchImpl, ctx.subprocess, resolved.requestTimeoutMs, resolved.graceMs);
1306
+ }
1307
+ }));
1308
+ ctx.commands.register({
1309
+ name: "ollama",
1310
+ description: "One-shot status overview: local models, disk usage, health, and routing suggestions.",
1311
+ async handler() {
1312
+ const health = await checkHealth(resolved.baseURL, fetchImpl, ctx.subprocess, resolved.requestTimeoutMs, resolved.graceMs);
1313
+ const lines = ["Ollama status:"];
1314
+ lines.push(`- API: ${health.api.ok ? `ok${health.api.version !== void 0 ? ` (v${health.api.version})` : ""}` : "down"}`);
1315
+ lines.push(`- process: ${health.process.present ? "alive" : "not detected"}`);
1316
+ let models = [];
1317
+ try {
1318
+ models = await listModels(resolved.baseURL, fetchImpl);
1319
+ } catch {}
1320
+ const totalBytes = models.reduce((sum, model) => sum + model.size, 0);
1321
+ lines.push(`- models: ${models.length} installed (${formatBytes(totalBytes)})`);
1322
+ for (const model of models) lines.push(` - ${model.name} (${formatBytes(model.size)})`);
1323
+ if (!health.api.ok && !health.process.present) lines.push("suggestion: start the Ollama server (e.g. `ollama serve`)");
1324
+ else if (resolved.route.length === 0) lines.push("suggestion: configure `route` rules to route requests to local models");
1325
+ return {
1326
+ kind: "success",
1327
+ text: lines.join("\n")
1328
+ };
1329
+ }
1330
+ });
1331
+ logger.info(`ollama adapter registered at ${resolved.baseURL} (${resolved.models.length} mapping(s), ${resolved.route.length} route rule(s))`);
1332
+ }
1333
+ //#endregion
1334
+ export { Config, REDACTED, VERSION, apply, formatBytes, inject, name, redactSecrets, renderHealth, renderList, renderOperation, renderShow, resolveConfig, sanitizeEndpoint, sanitizePath, sanitizeText, truncate };