dsh-local-ai 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/LICENSE +201 -0
- package/README.es.md +211 -0
- package/README.hi.md +211 -0
- package/README.md +211 -0
- package/README.pt.md +211 -0
- package/README.zh.md +211 -0
- package/THIRD_PARTY_NOTICES.md +20 -0
- package/cordis.patch.yml +44 -0
- package/lib/index.js +1334 -0
- package/lib/types/adapter.d.ts +39 -0
- package/lib/types/adapter.d.ts.map +1 -0
- package/lib/types/adapter.js +190 -0
- package/lib/types/adapter.js.map +1 -0
- package/lib/types/config.d.ts +109 -0
- package/lib/types/config.d.ts.map +1 -0
- package/lib/types/config.js +161 -0
- package/lib/types/config.js.map +1 -0
- package/lib/types/health.d.ts +64 -0
- package/lib/types/health.d.ts.map +1 -0
- package/lib/types/health.js +92 -0
- package/lib/types/health.js.map +1 -0
- package/lib/types/index.d.ts +43 -0
- package/lib/types/index.d.ts.map +1 -0
- package/lib/types/index.js +232 -0
- package/lib/types/index.js.map +1 -0
- package/lib/types/ollama.d.ts +91 -0
- package/lib/types/ollama.d.ts.map +1 -0
- package/lib/types/ollama.js +184 -0
- package/lib/types/ollama.js.map +1 -0
- package/lib/types/route.d.ts +52 -0
- package/lib/types/route.d.ts.map +1 -0
- package/lib/types/route.js +119 -0
- package/lib/types/route.js.map +1 -0
- package/lib/types/sanitize.d.ts +58 -0
- package/lib/types/sanitize.d.ts.map +1 -0
- package/lib/types/sanitize.js +110 -0
- package/lib/types/sanitize.js.map +1 -0
- package/lib/types/serialize.d.ts +65 -0
- package/lib/types/serialize.d.ts.map +1 -0
- package/lib/types/serialize.js +149 -0
- package/lib/types/serialize.js.map +1 -0
- package/lib/types/translate.d.ts +57 -0
- package/lib/types/translate.d.ts.map +1 -0
- package/lib/types/translate.js +169 -0
- package/lib/types/translate.js.map +1 -0
- package/lib/types/version.d.ts +6 -0
- package/lib/types/version.d.ts.map +1 -0
- package/lib/types/version.js +6 -0
- package/lib/types/version.js.map +1 -0
- package/package.json +141 -0
- package/src/adapter.ts +158 -0
- package/src/config.ts +243 -0
- package/src/health.ts +133 -0
- package/src/index.ts +279 -0
- package/src/ollama.ts +274 -0
- package/src/route.ts +127 -0
- package/src/sanitize.ts +114 -0
- package/src/serialize.ts +178 -0
- package/src/translate.ts +205 -0
- package/src/version.ts +5 -0
package/lib/index.js
ADDED
|
@@ -0,0 +1,1334 @@
|
|
|
1
|
+
import { defineTool } from "@deepseek-ai/dsh-tools";
|
|
2
|
+
import z from "@deepseek-ai/schemastery";
|
|
3
|
+
import { CallId, EMPTY_RESPONSE_CODE, LlmAdapter, LlmError, attributionHeaders, contentHasImage, isTokenDelta } from "@deepseek-ai/dsh-llm";
|
|
4
|
+
import { idleWatchdog, timeoutOf } from "@deepseek-ai/dsh-timeout";
|
|
5
|
+
import { tmpdir } from "node:os";
|
|
6
|
+
//#region src/config.ts
|
|
7
|
+
/**
|
|
8
|
+
* Config schema and resolution for `dsh-local-ai`. Every tunable is a
|
|
9
|
+
* validated {@link Config} field changeable from cordis.yml; the resolution
|
|
10
|
+
* step validates URLs, numeric bounds, and route/model entries so
|
|
11
|
+
* misconfiguration fails loud at mount — never silently skips a rule or
|
|
12
|
+
* half-configures the adapter. The plugin is inert until at least one route
|
|
13
|
+
* rule exists or a caller selects the `ollama` provider explicitly: routing
|
|
14
|
+
* every request to a local model requires an explicit opt-in (privacy and
|
|
15
|
+
* cost default: no automatic re-routing).
|
|
16
|
+
* @module dsh-local-ai/config
|
|
17
|
+
*/
|
|
18
|
+
/** Schemastery schema: the loader validates and fills defaults before `apply`. */
|
|
19
|
+
const Config = z.object({
|
|
20
|
+
baseURL: z.string().default("http://127.0.0.1:11434"),
|
|
21
|
+
requestTimeoutMs: z.number().default(3e4),
|
|
22
|
+
graceMs: z.number().default(15e3),
|
|
23
|
+
defaultContextWindow: z.number().default(8192),
|
|
24
|
+
maxTokens: z.number().default(4096),
|
|
25
|
+
temperature: z.number(),
|
|
26
|
+
models: z.array(z.object({
|
|
27
|
+
name: z.string().required(),
|
|
28
|
+
model: z.string(),
|
|
29
|
+
contextWindow: z.number(),
|
|
30
|
+
maxTokens: z.number(),
|
|
31
|
+
temperature: z.number()
|
|
32
|
+
})).default([]),
|
|
33
|
+
route: z.array(z.object({
|
|
34
|
+
model: z.string().required(),
|
|
35
|
+
purpose: z.union(["compaction", "session-title"]),
|
|
36
|
+
keywords: z.array(z.string()).default([]),
|
|
37
|
+
always: z.boolean().default(false)
|
|
38
|
+
})).default([])
|
|
39
|
+
});
|
|
40
|
+
/** Throw unless `value` is a positive safe integer. */
|
|
41
|
+
function assertPositiveInt(name, value) {
|
|
42
|
+
if (!Number.isSafeInteger(value) || value <= 0) throw new TypeError(`${name} must be a positive safe integer, got ${String(value)}`);
|
|
43
|
+
}
|
|
44
|
+
/** Throw unless `value` is a finite number in `[min, max]`. */
|
|
45
|
+
function assertFiniteRange(name, value, min, max) {
|
|
46
|
+
if (typeof value !== "number" || !Number.isFinite(value) || value < min || value > max) throw new TypeError(`${name} must be a finite number in [${min}, ${max}], got ${String(value)}`);
|
|
47
|
+
}
|
|
48
|
+
/**
|
|
49
|
+
* Validate an http(s) URL string and normalize it to a clean base (no query,
|
|
50
|
+
* no fragment, no trailing slash).
|
|
51
|
+
* @param name - config key, for the error message.
|
|
52
|
+
* @param value - raw URL value.
|
|
53
|
+
* @returns the normalized base URL.
|
|
54
|
+
*/
|
|
55
|
+
function normalizeBaseUrl(name, value) {
|
|
56
|
+
let parsed;
|
|
57
|
+
try {
|
|
58
|
+
parsed = new URL(value);
|
|
59
|
+
} catch (error) {
|
|
60
|
+
throw new TypeError(`${name} must be a valid URL, got ${JSON.stringify(value)} (${error instanceof Error ? error.message : "invalid URL"})`);
|
|
61
|
+
}
|
|
62
|
+
if (parsed.protocol !== "http:" && parsed.protocol !== "https:") throw new TypeError(`${name} must use http(s), got ${JSON.stringify(parsed.protocol)}`);
|
|
63
|
+
parsed.search = "";
|
|
64
|
+
parsed.hash = "";
|
|
65
|
+
return parsed.href.replace(/\/+$/u, "");
|
|
66
|
+
}
|
|
67
|
+
/**
|
|
68
|
+
* Validate raw values and fill explicit defaults. Invalid URLs, numeric
|
|
69
|
+
* bounds, duplicate model names, or empty route/model names throw here —
|
|
70
|
+
* misconfiguration fails loud at mount even when the plugin is mounted
|
|
71
|
+
* without the Schemastery loader.
|
|
72
|
+
* @param config - raw (possibly partial) plugin config.
|
|
73
|
+
* @returns the fully resolved config.
|
|
74
|
+
*/
|
|
75
|
+
function resolveConfig(config = {}) {
|
|
76
|
+
const baseURL = normalizeBaseUrl("baseURL", config.baseURL ?? "http://127.0.0.1:11434");
|
|
77
|
+
const requestTimeoutMs = config.requestTimeoutMs ?? 3e4;
|
|
78
|
+
assertPositiveInt("requestTimeoutMs", requestTimeoutMs);
|
|
79
|
+
const graceMs = config.graceMs ?? 15e3;
|
|
80
|
+
assertPositiveInt("graceMs", graceMs);
|
|
81
|
+
const defaultContextWindow = config.defaultContextWindow ?? 8192;
|
|
82
|
+
assertPositiveInt("defaultContextWindow", defaultContextWindow);
|
|
83
|
+
const maxTokens = config.maxTokens ?? 4096;
|
|
84
|
+
assertPositiveInt("maxTokens", maxTokens);
|
|
85
|
+
const temperature = config.temperature;
|
|
86
|
+
if (temperature !== void 0) assertFiniteRange("temperature", temperature, 0, 2);
|
|
87
|
+
const seenNames = /* @__PURE__ */ new Set();
|
|
88
|
+
const models = (config.models ?? []).map((mapping, index) => {
|
|
89
|
+
if (typeof mapping.name !== "string" || mapping.name.trim().length === 0) throw new TypeError(`models[${index}].name must be a non-empty string`);
|
|
90
|
+
const name = mapping.name.trim();
|
|
91
|
+
if (seenNames.has(name)) throw new TypeError(`models[${index}]: duplicate model name ${JSON.stringify(name)}`);
|
|
92
|
+
seenNames.add(name);
|
|
93
|
+
const model = (mapping.model ?? name).trim();
|
|
94
|
+
if (model.length === 0) throw new TypeError(`models[${index}].model must be a non-empty string`);
|
|
95
|
+
const contextWindow = mapping.contextWindow;
|
|
96
|
+
if (contextWindow !== void 0) assertPositiveInt(`models[${index}].contextWindow`, contextWindow);
|
|
97
|
+
const modelMaxTokens = mapping.maxTokens;
|
|
98
|
+
if (modelMaxTokens !== void 0) assertPositiveInt(`models[${index}].maxTokens`, modelMaxTokens);
|
|
99
|
+
const modelTemperature = mapping.temperature;
|
|
100
|
+
if (modelTemperature !== void 0) assertFiniteRange(`models[${index}].temperature`, modelTemperature, 0, 2);
|
|
101
|
+
return {
|
|
102
|
+
name,
|
|
103
|
+
model,
|
|
104
|
+
...contextWindow === void 0 ? {} : { contextWindow },
|
|
105
|
+
...modelMaxTokens === void 0 ? {} : { maxTokens: modelMaxTokens },
|
|
106
|
+
...modelTemperature === void 0 ? {} : { temperature: modelTemperature }
|
|
107
|
+
};
|
|
108
|
+
});
|
|
109
|
+
const route = (config.route ?? []).map((rule, index) => {
|
|
110
|
+
if (typeof rule.model !== "string" || rule.model.trim().length === 0) throw new TypeError(`route[${index}].model must be a non-empty string`);
|
|
111
|
+
const keywords = (rule.keywords ?? []).map((keyword, keywordIndex) => {
|
|
112
|
+
if (typeof keyword !== "string" || keyword.trim().length === 0) throw new TypeError(`route[${index}].keywords[${keywordIndex}] must be a non-empty string`);
|
|
113
|
+
return keyword;
|
|
114
|
+
});
|
|
115
|
+
if (rule.always !== true && rule.purpose === void 0 && keywords.length === 0) throw new TypeError(`route[${index}] must declare a purpose, at least one keyword, or always: true`);
|
|
116
|
+
return {
|
|
117
|
+
model: rule.model.trim(),
|
|
118
|
+
...rule.purpose === void 0 ? {} : { purpose: rule.purpose },
|
|
119
|
+
keywords,
|
|
120
|
+
always: rule.always ?? false
|
|
121
|
+
};
|
|
122
|
+
});
|
|
123
|
+
return {
|
|
124
|
+
baseURL,
|
|
125
|
+
requestTimeoutMs,
|
|
126
|
+
graceMs,
|
|
127
|
+
defaultContextWindow,
|
|
128
|
+
maxTokens,
|
|
129
|
+
...temperature === void 0 ? {} : { temperature },
|
|
130
|
+
models,
|
|
131
|
+
route
|
|
132
|
+
};
|
|
133
|
+
}
|
|
134
|
+
//#endregion
|
|
135
|
+
//#region src/sanitize.ts
|
|
136
|
+
/**
|
|
137
|
+
* Display/log sanitization for `dsh-local-ai`. Every value shown to the model
|
|
138
|
+
* or to the user (tool results, the `/ollama` command, error messages) passes
|
|
139
|
+
* through one of these pure functions first, so an endpoint address or a local
|
|
140
|
+
* path can never leak credentials, secret query parameters, or unbounded text.
|
|
141
|
+
*
|
|
142
|
+
* All functions are pure: they depend only on their arguments, never on
|
|
143
|
+
* process state (the caller supplies a home directory for path redaction).
|
|
144
|
+
* @module dsh-local-ai/sanitize
|
|
145
|
+
*/
|
|
146
|
+
/** Placeholder substituted for a redacted secret or credential. */
|
|
147
|
+
const REDACTED = "[REDACTED]";
|
|
148
|
+
/** Control characters (C0 + DEL) stripped from every sanitized value. */
|
|
149
|
+
const CONTROL_CHARS = /[\u0000-\u001f\u007f]/gu;
|
|
150
|
+
/** Secret-shaped query-parameter keys removed from endpoint URLs. */
|
|
151
|
+
const SECRET_KEY_PATTERN = /key|token|secret|password|credential|auth/iu;
|
|
152
|
+
/** Built-in secret literal patterns redacted from arbitrary text. */
|
|
153
|
+
const BUILTIN_SECRET_PATTERNS = [
|
|
154
|
+
/\bsk-[A-Za-z0-9]{16,}\b/u,
|
|
155
|
+
/\bghp_[A-Za-z0-9]{20,}\b/u,
|
|
156
|
+
/\bgho_[A-Za-z0-9]{20,}\b/u,
|
|
157
|
+
/\bAKIA[0-9A-Z]{16}\b/u,
|
|
158
|
+
/\bBearer\s+[A-Za-z0-9._~+/=-]{8,}\b/u,
|
|
159
|
+
/-----BEGIN [A-Z ]*PRIVATE KEY-----[A-Za-z0-9+/=\s]*-----END [A-Z ]*PRIVATE KEY-----/u
|
|
160
|
+
];
|
|
161
|
+
/** Remove C0/DEL control characters from a string. */
|
|
162
|
+
function stripControl(value) {
|
|
163
|
+
return value.replace(CONTROL_CHARS, "");
|
|
164
|
+
}
|
|
165
|
+
/**
|
|
166
|
+
* Truncate a string to `maxChars`, appending an ellipsis when it was cut.
|
|
167
|
+
* A non-positive `maxChars` yields the empty string.
|
|
168
|
+
* @param value - the string to bound.
|
|
169
|
+
* @param maxChars - maximum returned length, including the ellipsis.
|
|
170
|
+
* @returns the bounded string.
|
|
171
|
+
*/
|
|
172
|
+
function truncate(value, maxChars) {
|
|
173
|
+
if (value.length <= maxChars) return value;
|
|
174
|
+
if (maxChars <= 1) return maxChars <= 0 ? "" : "…";
|
|
175
|
+
return `${value.slice(0, maxChars - 1)}…`;
|
|
176
|
+
}
|
|
177
|
+
/**
|
|
178
|
+
* Sanitize an endpoint address for display: strip the URL userinfo
|
|
179
|
+
* (`user:pass@`), drop query parameters whose key looks like a secret, strip
|
|
180
|
+
* control characters, and bound the length. Values that are not parseable as
|
|
181
|
+
* a URL are still stripped and truncated.
|
|
182
|
+
* @param value - the raw endpoint (URL string or anything stringifiable).
|
|
183
|
+
* @param maxChars - maximum returned length.
|
|
184
|
+
* @returns the sanitized endpoint text.
|
|
185
|
+
*/
|
|
186
|
+
function sanitizeEndpoint(value, maxChars = 2048) {
|
|
187
|
+
const text = stripControl(typeof value === "string" ? value : String(value));
|
|
188
|
+
let out;
|
|
189
|
+
try {
|
|
190
|
+
const url = new URL(text);
|
|
191
|
+
url.username = "";
|
|
192
|
+
url.password = "";
|
|
193
|
+
for (const key of [...url.searchParams.keys()]) if (SECRET_KEY_PATTERN.test(key)) url.searchParams.delete(key);
|
|
194
|
+
out = url.href;
|
|
195
|
+
} catch {
|
|
196
|
+
out = text;
|
|
197
|
+
}
|
|
198
|
+
return truncate(out, maxChars);
|
|
199
|
+
}
|
|
200
|
+
/**
|
|
201
|
+
* Sanitize a local path for display: strip control characters, redact a
|
|
202
|
+
* leading home directory to `~`, and bound the length.
|
|
203
|
+
* @param value - the raw path (string or anything stringifiable).
|
|
204
|
+
* @param home - the user's home directory to redact; omit to skip redaction.
|
|
205
|
+
* @param maxChars - maximum returned length.
|
|
206
|
+
* @returns the sanitized path text.
|
|
207
|
+
*/
|
|
208
|
+
function sanitizePath(value, home = "", maxChars = 1024) {
|
|
209
|
+
const text = stripControl(typeof value === "string" ? value : String(value));
|
|
210
|
+
return truncate(home.length > 0 && text.startsWith(home) ? `~${text.slice(home.length)}` : text, maxChars);
|
|
211
|
+
}
|
|
212
|
+
/**
|
|
213
|
+
* Redact built-in secret literals (API keys, GitHub tokens, AWS keys, bearer
|
|
214
|
+
* credentials, PEM private keys) from arbitrary text. Control characters are
|
|
215
|
+
* stripped first.
|
|
216
|
+
* @param value - the raw text (string or anything stringifiable).
|
|
217
|
+
* @returns the text with secret literals replaced by {@link REDACTED}.
|
|
218
|
+
*/
|
|
219
|
+
function redactSecrets(value) {
|
|
220
|
+
let out = stripControl(typeof value === "string" ? value : String(value));
|
|
221
|
+
for (const pattern of BUILTIN_SECRET_PATTERNS) out = out.replace(pattern, REDACTED);
|
|
222
|
+
return out;
|
|
223
|
+
}
|
|
224
|
+
/**
|
|
225
|
+
* Sanitize arbitrary display text: redact secrets, strip control characters,
|
|
226
|
+
* and bound the length.
|
|
227
|
+
* @param value - the raw text (string or anything stringifiable).
|
|
228
|
+
* @param maxChars - maximum returned length.
|
|
229
|
+
* @returns the sanitized text.
|
|
230
|
+
*/
|
|
231
|
+
function sanitizeText(value, maxChars = 4e3) {
|
|
232
|
+
return truncate(redactSecrets(value), maxChars);
|
|
233
|
+
}
|
|
234
|
+
//#endregion
|
|
235
|
+
//#region src/ollama.ts
|
|
236
|
+
/**
|
|
237
|
+
* Ollama HTTP API client (zero runtime dependencies — plain `fetch`). The
|
|
238
|
+
* harness `LlmAdapter` streams through `/api/chat`; discovery and management
|
|
239
|
+
* tools call `/api/tags`, `/api/show`, `/api/pull`, `/api/delete`, and
|
|
240
|
+
* `/api/version`. Every request carries the harness attribution headers and
|
|
241
|
+
* honors the caller's abort signal; non-2xx responses fail with a normalized
|
|
242
|
+
* `LlmError`. The fetch implementation is injectable for tests.
|
|
243
|
+
* @module dsh-local-ai/ollama
|
|
244
|
+
*/
|
|
245
|
+
/** Build an absolute API URL from a normalized base URL. */
|
|
246
|
+
function endpointUrl(baseURL, path) {
|
|
247
|
+
return `${baseURL}${path}`;
|
|
248
|
+
}
|
|
249
|
+
/** Map an HTTP status to a stable LlmError code. */
|
|
250
|
+
function httpErrorCode(status) {
|
|
251
|
+
if (status === 404) return "NOT_FOUND";
|
|
252
|
+
if (status === 400) return "INVALID_REQUEST";
|
|
253
|
+
if (status >= 500) return "SERVER";
|
|
254
|
+
return `HTTP_${status}`;
|
|
255
|
+
}
|
|
256
|
+
/** Throw a normalized LlmError from a non-2xx response, using the body's `error`. */
|
|
257
|
+
async function throwHttpError(response, context) {
|
|
258
|
+
let message = `Ollama API error (HTTP ${response.status}) from ${sanitizeEndpoint(context)}`;
|
|
259
|
+
try {
|
|
260
|
+
const body = await response.json();
|
|
261
|
+
if (typeof body.error === "string" && body.error.length > 0) message = body.error;
|
|
262
|
+
} catch {}
|
|
263
|
+
throw new LlmError(message, httpErrorCode(response.status), { status: response.status });
|
|
264
|
+
}
|
|
265
|
+
/** Send a GET request and parse the JSON response. */
|
|
266
|
+
async function requestJson(baseURL, path, fetchImpl, signal) {
|
|
267
|
+
const response = await fetchImpl(endpointUrl(baseURL, path), {
|
|
268
|
+
method: "GET",
|
|
269
|
+
headers: {
|
|
270
|
+
accept: "application/json",
|
|
271
|
+
...attributionHeaders()
|
|
272
|
+
},
|
|
273
|
+
...signal === void 0 ? {} : { signal }
|
|
274
|
+
});
|
|
275
|
+
if (!response.ok) await throwHttpError(response, endpointUrl(baseURL, path));
|
|
276
|
+
return response.json();
|
|
277
|
+
}
|
|
278
|
+
/** Send a POST request and parse the JSON response. */
|
|
279
|
+
async function postJson(baseURL, path, body, fetchImpl, signal) {
|
|
280
|
+
const response = await fetchImpl(endpointUrl(baseURL, path), {
|
|
281
|
+
method: "POST",
|
|
282
|
+
headers: {
|
|
283
|
+
"content-type": "application/json",
|
|
284
|
+
accept: "application/json",
|
|
285
|
+
...attributionHeaders()
|
|
286
|
+
},
|
|
287
|
+
body: JSON.stringify(body),
|
|
288
|
+
...signal === void 0 ? {} : { signal }
|
|
289
|
+
});
|
|
290
|
+
if (!response.ok) await throwHttpError(response, endpointUrl(baseURL, path));
|
|
291
|
+
return response.json();
|
|
292
|
+
}
|
|
293
|
+
/** Send a DELETE request and parse the JSON response. */
|
|
294
|
+
async function deleteJson(baseURL, path, body, fetchImpl, signal) {
|
|
295
|
+
const response = await fetchImpl(endpointUrl(baseURL, path), {
|
|
296
|
+
method: "DELETE",
|
|
297
|
+
headers: {
|
|
298
|
+
"content-type": "application/json",
|
|
299
|
+
accept: "application/json",
|
|
300
|
+
...attributionHeaders()
|
|
301
|
+
},
|
|
302
|
+
body: JSON.stringify(body),
|
|
303
|
+
...signal === void 0 ? {} : { signal }
|
|
304
|
+
});
|
|
305
|
+
if (!response.ok) await throwHttpError(response, endpointUrl(baseURL, path));
|
|
306
|
+
return response.json();
|
|
307
|
+
}
|
|
308
|
+
/**
|
|
309
|
+
* Send a POST and return the raw `Response` after validating 2xx. Used by the
|
|
310
|
+
* streaming adapter, which owns body decoding and the idle watchdog.
|
|
311
|
+
*/
|
|
312
|
+
async function postStream(baseURL, path, body, fetchImpl, signal) {
|
|
313
|
+
const response = await fetchImpl(endpointUrl(baseURL, path), {
|
|
314
|
+
method: "POST",
|
|
315
|
+
headers: {
|
|
316
|
+
"content-type": "application/json",
|
|
317
|
+
accept: "application/x-ndjson",
|
|
318
|
+
...attributionHeaders()
|
|
319
|
+
},
|
|
320
|
+
body: JSON.stringify(body),
|
|
321
|
+
...signal === void 0 ? {} : { signal }
|
|
322
|
+
});
|
|
323
|
+
if (!response.ok) await throwHttpError(response, endpointUrl(baseURL, path));
|
|
324
|
+
return response;
|
|
325
|
+
}
|
|
326
|
+
/** List installed models from `/api/tags`. */
|
|
327
|
+
async function listModels(baseURL, fetchImpl, signal) {
|
|
328
|
+
return (await requestJson(baseURL, "/api/tags", fetchImpl, signal)).models ?? [];
|
|
329
|
+
}
|
|
330
|
+
/** List currently-loaded (running) models from `/api/ps`. */
|
|
331
|
+
async function listRunning(baseURL, fetchImpl, signal) {
|
|
332
|
+
return (await requestJson(baseURL, "/api/ps", fetchImpl, signal)).models ?? [];
|
|
333
|
+
}
|
|
334
|
+
/** Inspect one model via `/api/show`. */
|
|
335
|
+
async function showModel(baseURL, name, fetchImpl, signal) {
|
|
336
|
+
return postJson(baseURL, "/api/show", { name }, fetchImpl, signal);
|
|
337
|
+
}
|
|
338
|
+
/** Remove one model via `/api/delete`. */
|
|
339
|
+
async function removeModel(baseURL, name, fetchImpl, signal) {
|
|
340
|
+
await deleteJson(baseURL, "/api/delete", { name }, fetchImpl, signal);
|
|
341
|
+
}
|
|
342
|
+
/** Query the Ollama server version via `/api/version`. */
|
|
343
|
+
async function apiVersion(baseURL, fetchImpl, signal) {
|
|
344
|
+
return (await requestJson(baseURL, "/api/version", fetchImpl, signal)).version;
|
|
345
|
+
}
|
|
346
|
+
/**
|
|
347
|
+
* Pull a model via `/api/pull`, consuming the progress stream and returning the
|
|
348
|
+
* final status. An intermediate error status or a non-2xx response fails loud.
|
|
349
|
+
*/
|
|
350
|
+
async function pullModel(baseURL, name, fetchImpl, signal) {
|
|
351
|
+
const response = await postStream(baseURL, "/api/pull", {
|
|
352
|
+
name,
|
|
353
|
+
stream: true
|
|
354
|
+
}, fetchImpl, signal);
|
|
355
|
+
if (!response.body) throw new LlmError("Ollama pull returned no response body", "EMPTY_RESPONSE");
|
|
356
|
+
let last = { status: "success" };
|
|
357
|
+
for await (const line of readNdjsonLines(response.body)) {
|
|
358
|
+
if (line.length === 0) continue;
|
|
359
|
+
const chunk = JSON.parse(line);
|
|
360
|
+
if (typeof chunk.error === "string" && chunk.error.length > 0) throw new LlmError(chunk.error, "PROVIDER");
|
|
361
|
+
if (typeof chunk.status === "string") last = { status: chunk.status };
|
|
362
|
+
}
|
|
363
|
+
return last;
|
|
364
|
+
}
|
|
365
|
+
/**
|
|
366
|
+
* Decode a `ReadableStream<Uint8Array>` into newline-delimited text lines.
|
|
367
|
+
* The final line is yielded even without a trailing newline; a missing body
|
|
368
|
+
* yields nothing.
|
|
369
|
+
* @param body - the response body stream.
|
|
370
|
+
* @returns text lines in delivery order.
|
|
371
|
+
*/
|
|
372
|
+
async function* readNdjsonLines(body) {
|
|
373
|
+
const reader = body.getReader();
|
|
374
|
+
const decoder = new TextDecoder();
|
|
375
|
+
let buffer = "";
|
|
376
|
+
try {
|
|
377
|
+
while (true) {
|
|
378
|
+
const { done, value } = await reader.read();
|
|
379
|
+
if (done) break;
|
|
380
|
+
buffer += decoder.decode(value, { stream: true });
|
|
381
|
+
let newline = buffer.indexOf("\n");
|
|
382
|
+
while (newline >= 0) {
|
|
383
|
+
const line = buffer.slice(0, newline).replace(/\r$/u, "");
|
|
384
|
+
buffer = buffer.slice(newline + 1);
|
|
385
|
+
newline = buffer.indexOf("\n");
|
|
386
|
+
yield line;
|
|
387
|
+
}
|
|
388
|
+
}
|
|
389
|
+
} finally {
|
|
390
|
+
reader.releaseLock();
|
|
391
|
+
}
|
|
392
|
+
buffer += decoder.decode();
|
|
393
|
+
if (buffer.length > 0) yield buffer;
|
|
394
|
+
}
|
|
395
|
+
/**
|
|
396
|
+
* Extract the context length from an `/api/show` result by scanning
|
|
397
|
+
* `model_info` for a `*.context_length` or bare `context_length` entry.
|
|
398
|
+
* @param show - the `/api/show` result.
|
|
399
|
+
* @returns the context length, or `undefined` when not reported.
|
|
400
|
+
*/
|
|
401
|
+
function contextLengthOf(show) {
|
|
402
|
+
const info = show.model_info;
|
|
403
|
+
if (info === void 0) return void 0;
|
|
404
|
+
for (const [key, value] of Object.entries(info)) if (key === "context_length" || key.endsWith(".context_length")) {
|
|
405
|
+
if (typeof value === "number" && Number.isInteger(value) && value > 0) return value;
|
|
406
|
+
}
|
|
407
|
+
}
|
|
408
|
+
//#endregion
|
|
409
|
+
//#region src/serialize.ts
|
|
410
|
+
/**
|
|
411
|
+
* Serialize harness messages and requests into the Ollama `/api/chat` wire
|
|
412
|
+
* vocabulary. User text is joined; assistant text becomes `content` and tool
|
|
413
|
+
* calls become `tool_calls` (with arguments parsed from the raw JSON string to
|
|
414
|
+
* the object Ollama expects); tool results become separate `tool` messages.
|
|
415
|
+
* Core image blocks are rejected explicitly because this route is text-only;
|
|
416
|
+
* unknown declaration-merged block types retain the documented extension
|
|
417
|
+
* fallback (ignored for content, retained as text where text is expected).
|
|
418
|
+
* @module dsh-local-ai/serialize
|
|
419
|
+
*/
|
|
420
|
+
/** Join the text blocks of a message (used for user/tool-result content). */
|
|
421
|
+
function flattenText(blocks) {
|
|
422
|
+
return blocks.filter((block) => block.type === "text").map((block) => block.text).join("");
|
|
423
|
+
}
|
|
424
|
+
/** Reject core image content before any text-flattening path can silently erase it. */
|
|
425
|
+
function assertTextOnly(blocks) {
|
|
426
|
+
if (contentHasImage(blocks)) throw new LlmError("The Ollama adapter does not support image content.", "UNSUPPORTED_CONTENT");
|
|
427
|
+
}
|
|
428
|
+
/**
|
|
429
|
+
* Parse a tool-call argument string into the object Ollama expects. The raw
|
|
430
|
+
* string is guaranteed by the harness contract to be JSON; a malformed value
|
|
431
|
+
* from a hand-built call degrades to a single `value` field rather than
|
|
432
|
+
* bricking the whole session.
|
|
433
|
+
* @param raw - the raw JSON string produced by the model.
|
|
434
|
+
* @returns the parsed object, or a `{ value }` fallback.
|
|
435
|
+
*/
|
|
436
|
+
function parseToolArguments(raw) {
|
|
437
|
+
try {
|
|
438
|
+
const parsed = JSON.parse(raw);
|
|
439
|
+
if (typeof parsed === "object" && parsed !== null && !Array.isArray(parsed)) return parsed;
|
|
440
|
+
return { value: raw };
|
|
441
|
+
} catch {
|
|
442
|
+
return { value: raw };
|
|
443
|
+
}
|
|
444
|
+
}
|
|
445
|
+
/** Serialize one assistant message (text + tool calls). */
|
|
446
|
+
function serializeAssistant(message) {
|
|
447
|
+
const text = flattenText(message.content);
|
|
448
|
+
const toolCalls = message.content.filter((block) => block.type === "tool-call").map((block) => ({ function: {
|
|
449
|
+
name: block.name,
|
|
450
|
+
arguments: parseToolArguments(block.arguments)
|
|
451
|
+
} }));
|
|
452
|
+
return {
|
|
453
|
+
role: "assistant",
|
|
454
|
+
content: text,
|
|
455
|
+
...toolCalls.length > 0 ? { tool_calls: toolCalls } : {}
|
|
456
|
+
};
|
|
457
|
+
}
|
|
458
|
+
/** Resolve a tool-result block's name from the assistant tool calls that precede it. */
|
|
459
|
+
function toolNameOf(callId, namesByCallId) {
|
|
460
|
+
return namesByCallId.get(callId);
|
|
461
|
+
}
|
|
462
|
+
/**
|
|
463
|
+
* Serialize the conversation. `tool-result` blocks become standalone
|
|
464
|
+
* `{role: 'tool'}` messages; the harness puts each tool result in its own
|
|
465
|
+
* user-role message, so a mixed user message contributes its text first and
|
|
466
|
+
* its tool results as separate wire messages after. Assistant tool calls are
|
|
467
|
+
* indexed first so their results can carry the tool name Ollama needs.
|
|
468
|
+
* @param messages - the harness conversation, in order.
|
|
469
|
+
* @returns the wire messages; order preserved, each tool result expanded into its own entry.
|
|
470
|
+
*/
|
|
471
|
+
function serializeMessages(messages) {
|
|
472
|
+
const namesByCallId = /* @__PURE__ */ new Map();
|
|
473
|
+
for (const message of messages) {
|
|
474
|
+
if (message.role !== "assistant") continue;
|
|
475
|
+
for (const block of message.content) if (block.type === "tool-call") namesByCallId.set(String(block.id), block.name);
|
|
476
|
+
}
|
|
477
|
+
const wire = [];
|
|
478
|
+
for (const message of messages) {
|
|
479
|
+
assertTextOnly(message.content);
|
|
480
|
+
if (message.role === "system") {
|
|
481
|
+
wire.push({
|
|
482
|
+
role: "system",
|
|
483
|
+
content: flattenText(message.content)
|
|
484
|
+
});
|
|
485
|
+
continue;
|
|
486
|
+
}
|
|
487
|
+
if (message.role === "assistant") {
|
|
488
|
+
wire.push(serializeAssistant(message));
|
|
489
|
+
continue;
|
|
490
|
+
}
|
|
491
|
+
const toolResults = message.content.filter((block) => block.type === "tool-result");
|
|
492
|
+
const text = flattenText(message.content);
|
|
493
|
+
if (text.length > 0 || toolResults.length === 0) wire.push({
|
|
494
|
+
role: "user",
|
|
495
|
+
content: text
|
|
496
|
+
});
|
|
497
|
+
for (const result of toolResults) {
|
|
498
|
+
const name = toolNameOf(String(result.toolCallId), namesByCallId);
|
|
499
|
+
wire.push({
|
|
500
|
+
role: "tool",
|
|
501
|
+
content: flattenText(result.content) || "(no output)",
|
|
502
|
+
...name === void 0 ? {} : { tool_name: name }
|
|
503
|
+
});
|
|
504
|
+
}
|
|
505
|
+
}
|
|
506
|
+
return wire;
|
|
507
|
+
}
|
|
508
|
+
/**
|
|
509
|
+
* Build the full wire request. Always streaming (`stream: true`); optional
|
|
510
|
+
* fields are omitted rather than sent as null, so Ollama defaults apply.
|
|
511
|
+
* `temperature` resolves request → model mapping → plugin default; `num_predict`
|
|
512
|
+
* is the harness-materialized `maxTokens`.
|
|
513
|
+
* @param options - the harness request (model, history, system, tools, sampling).
|
|
514
|
+
* @param resolved - the resolved plugin config.
|
|
515
|
+
* @returns the `/api/chat` request body.
|
|
516
|
+
*/
|
|
517
|
+
function serializeRequest(options, resolved) {
|
|
518
|
+
const mapping = resolved.models.find((entry) => entry.name === options.model);
|
|
519
|
+
const model = mapping?.model ?? options.model;
|
|
520
|
+
const messages = [];
|
|
521
|
+
if (options.system !== void 0) messages.push({
|
|
522
|
+
role: "system",
|
|
523
|
+
content: options.system
|
|
524
|
+
});
|
|
525
|
+
messages.push(...serializeMessages(options.messages));
|
|
526
|
+
const temperature = options.temperature ?? mapping?.temperature ?? resolved.temperature;
|
|
527
|
+
const ollamaOptions = {};
|
|
528
|
+
if (temperature !== void 0) ollamaOptions.temperature = temperature;
|
|
529
|
+
if (options.maxTokens !== void 0) ollamaOptions.num_predict = options.maxTokens;
|
|
530
|
+
if (options.stop !== void 0 && options.stop.length > 0) ollamaOptions.stop = options.stop;
|
|
531
|
+
const tools = options.tools?.map((tool) => ({
|
|
532
|
+
type: "function",
|
|
533
|
+
function: tool
|
|
534
|
+
}));
|
|
535
|
+
return {
|
|
536
|
+
model,
|
|
537
|
+
messages,
|
|
538
|
+
stream: true,
|
|
539
|
+
...Object.keys(ollamaOptions).length > 0 ? { options: ollamaOptions } : {},
|
|
540
|
+
...tools !== void 0 && tools.length > 0 ? { tools } : {}
|
|
541
|
+
};
|
|
542
|
+
}
|
|
543
|
+
//#endregion
|
|
544
|
+
//#region src/translate.ts
|
|
545
|
+
/**
|
|
546
|
+
* Translate Ollama NDJSON chat chunks into the harness `StreamChunk` protocol.
|
|
547
|
+
* Ollama streams one JSON object per line: `message.content` and
|
|
548
|
+
* `message.thinking` are incremental deltas, while `message.tool_calls`
|
|
549
|
+
* carries the CUMULATIVE arguments object, so tool-call deltas are computed by
|
|
550
|
+
* longest-common-prefix diffing. Usage and the finish reason are deferred to
|
|
551
|
+
* the `done: true` chunk, guaranteeing no chunk follows `finish`.
|
|
552
|
+
* @module dsh-local-ai/translate
|
|
553
|
+
*/
|
|
554
|
+
/**
|
|
555
|
+
* Map the Ollama `done_reason` vocabulary to the harness FinishReason.
|
|
556
|
+
* @param reason - the wire `done_reason` string.
|
|
557
|
+
* @returns the mapped reason; unrecognized values become `{kind: 'error'}`.
|
|
558
|
+
*/
|
|
559
|
+
function mapFinishReason(reason) {
|
|
560
|
+
switch (reason) {
|
|
561
|
+
case "stop": return { kind: "stop" };
|
|
562
|
+
case "tool_calls": return { kind: "tool-calls" };
|
|
563
|
+
case "length": return { kind: "max-tokens" };
|
|
564
|
+
default: return {
|
|
565
|
+
kind: "error",
|
|
566
|
+
failure: {
|
|
567
|
+
message: `model stopped: ${reason}`,
|
|
568
|
+
code: reason.toUpperCase()
|
|
569
|
+
}
|
|
570
|
+
};
|
|
571
|
+
}
|
|
572
|
+
}
|
|
573
|
+
/** Assemble the final ContentBlock for one open block. */
|
|
574
|
+
function closeBlock(block) {
|
|
575
|
+
switch (block.kind) {
|
|
576
|
+
case "text": return {
|
|
577
|
+
type: "text",
|
|
578
|
+
text: block.text
|
|
579
|
+
};
|
|
580
|
+
case "reasoning": return {
|
|
581
|
+
type: "reasoning",
|
|
582
|
+
text: block.text
|
|
583
|
+
};
|
|
584
|
+
case "tool-call": return {
|
|
585
|
+
type: "tool-call",
|
|
586
|
+
id: CallId(block.callId ?? ""),
|
|
587
|
+
name: block.name ?? "",
|
|
588
|
+
arguments: block.text
|
|
589
|
+
};
|
|
590
|
+
}
|
|
591
|
+
}
|
|
592
|
+
/**
|
|
593
|
+
* Compute the append delta from a cumulative JSON string, so the harness's
|
|
594
|
+
* delta-concatenating assembler reconstructs the full arguments. Ollama grows
|
|
595
|
+
* the arguments object monotonically, so the delta is everything past the
|
|
596
|
+
* longest common prefix with the previously seen string.
|
|
597
|
+
* @param previous - the previously seen cumulative JSON (or `''`).
|
|
598
|
+
* @param next - the new cumulative JSON.
|
|
599
|
+
* @returns the fragment to append.
|
|
600
|
+
*/
|
|
601
|
+
function argumentsDelta(previous, next) {
|
|
602
|
+
if (next.startsWith(previous)) return next.slice(previous.length);
|
|
603
|
+
let index = 0;
|
|
604
|
+
while (index < previous.length && index < next.length && previous[index] === next[index]) index += 1;
|
|
605
|
+
return next.slice(index);
|
|
606
|
+
}
|
|
607
|
+
/**
|
|
608
|
+
* Consume Ollama NDJSON lines and yield StreamChunks. Text and reasoning deltas
|
|
609
|
+
* stream as they arrive; tool-call deltas are diffed from the cumulative wire
|
|
610
|
+
* arguments; `block-end`, `usage`, and `finish` are deferred to the `done`
|
|
611
|
+
* chunk. A `stop` finish with no opened blocks maps to an `EMPTY_RESPONSE`
|
|
612
|
+
* error finish.
|
|
613
|
+
* @param lines - newline-delimited Ollama chat payloads.
|
|
614
|
+
* @returns deltas as they arrive; `block-end`s, `usage`, and `finish` deferred to `done`.
|
|
615
|
+
*/
|
|
616
|
+
async function* translate(lines) {
|
|
617
|
+
let nextIndex = 0;
|
|
618
|
+
let textBlock;
|
|
619
|
+
let reasoningBlock;
|
|
620
|
+
const toolBlocks = /* @__PURE__ */ new Map();
|
|
621
|
+
const toolArguments = /* @__PURE__ */ new Map();
|
|
622
|
+
const order = [];
|
|
623
|
+
function open(kind) {
|
|
624
|
+
const block = {
|
|
625
|
+
index: nextIndex++,
|
|
626
|
+
kind,
|
|
627
|
+
text: ""
|
|
628
|
+
};
|
|
629
|
+
order.push(block);
|
|
630
|
+
return block;
|
|
631
|
+
}
|
|
632
|
+
for await (const line of lines) {
|
|
633
|
+
if (line.length === 0) continue;
|
|
634
|
+
let chunk;
|
|
635
|
+
try {
|
|
636
|
+
chunk = JSON.parse(line);
|
|
637
|
+
} catch {
|
|
638
|
+
throw new LlmError(`malformed Ollama NDJSON payload: ${line.slice(0, 120)}`, "MALFORMED_RESPONSE");
|
|
639
|
+
}
|
|
640
|
+
if (typeof chunk.error === "string" && chunk.error.length > 0) throw new LlmError(chunk.error, "PROVIDER");
|
|
641
|
+
const message = chunk.message;
|
|
642
|
+
if (message !== void 0) {
|
|
643
|
+
const reasoning = message.thinking;
|
|
644
|
+
if (typeof reasoning === "string" && reasoning.length > 0) {
|
|
645
|
+
if (!reasoningBlock) {
|
|
646
|
+
reasoningBlock = open("reasoning");
|
|
647
|
+
yield {
|
|
648
|
+
type: "block-start",
|
|
649
|
+
index: reasoningBlock.index,
|
|
650
|
+
blockType: "reasoning"
|
|
651
|
+
};
|
|
652
|
+
}
|
|
653
|
+
reasoningBlock.text += reasoning;
|
|
654
|
+
yield {
|
|
655
|
+
type: "reasoning-delta",
|
|
656
|
+
index: reasoningBlock.index,
|
|
657
|
+
text: reasoning
|
|
658
|
+
};
|
|
659
|
+
}
|
|
660
|
+
const content = message.content;
|
|
661
|
+
if (typeof content === "string" && content.length > 0) {
|
|
662
|
+
if (!textBlock) {
|
|
663
|
+
textBlock = open("text");
|
|
664
|
+
yield {
|
|
665
|
+
type: "block-start",
|
|
666
|
+
index: textBlock.index,
|
|
667
|
+
blockType: "text"
|
|
668
|
+
};
|
|
669
|
+
}
|
|
670
|
+
textBlock.text += content;
|
|
671
|
+
yield {
|
|
672
|
+
type: "text-delta",
|
|
673
|
+
index: textBlock.index,
|
|
674
|
+
text: content
|
|
675
|
+
};
|
|
676
|
+
}
|
|
677
|
+
const toolCalls = message.tool_calls ?? [];
|
|
678
|
+
for (let callIndex = 0; callIndex < toolCalls.length; callIndex++) {
|
|
679
|
+
const call = toolCalls[callIndex];
|
|
680
|
+
let block = toolBlocks.get(callIndex);
|
|
681
|
+
if (!block) {
|
|
682
|
+
block = open("tool-call");
|
|
683
|
+
toolBlocks.set(callIndex, block);
|
|
684
|
+
toolArguments.set(callIndex, "");
|
|
685
|
+
yield {
|
|
686
|
+
type: "block-start",
|
|
687
|
+
index: block.index,
|
|
688
|
+
blockType: "tool-call"
|
|
689
|
+
};
|
|
690
|
+
}
|
|
691
|
+
if (call?.function?.name !== void 0 && block.name === void 0) block.name = call.function.name;
|
|
692
|
+
const cumulative = JSON.stringify(call?.function?.arguments ?? {});
|
|
693
|
+
const previous = toolArguments.get(callIndex) ?? "";
|
|
694
|
+
if (cumulative !== previous) {
|
|
695
|
+
const fragment = argumentsDelta(previous, cumulative);
|
|
696
|
+
toolArguments.set(callIndex, cumulative);
|
|
697
|
+
block.text += fragment;
|
|
698
|
+
yield {
|
|
699
|
+
type: "tool-call-delta",
|
|
700
|
+
index: block.index,
|
|
701
|
+
id: CallId(block.callId ?? ""),
|
|
702
|
+
...block.name !== void 0 ? { name: block.name } : {},
|
|
703
|
+
argumentsDelta: fragment
|
|
704
|
+
};
|
|
705
|
+
}
|
|
706
|
+
}
|
|
707
|
+
}
|
|
708
|
+
if (chunk.done === true) {
|
|
709
|
+
for (const block of order) yield {
|
|
710
|
+
type: "block-end",
|
|
711
|
+
index: block.index,
|
|
712
|
+
block: closeBlock(block)
|
|
713
|
+
};
|
|
714
|
+
const promptCount = chunk.prompt_eval_count;
|
|
715
|
+
const evalCount = chunk.eval_count;
|
|
716
|
+
if (promptCount !== void 0 || evalCount !== void 0) yield {
|
|
717
|
+
type: "usage",
|
|
718
|
+
usage: {
|
|
719
|
+
inputTokens: promptCount ?? 0,
|
|
720
|
+
outputTokens: evalCount ?? 0
|
|
721
|
+
}
|
|
722
|
+
};
|
|
723
|
+
const reason = mapFinishReason(chunk.done_reason ?? "stop");
|
|
724
|
+
yield {
|
|
725
|
+
type: "finish",
|
|
726
|
+
reason: reason.kind === "stop" && order.length === 0 ? {
|
|
727
|
+
kind: "error",
|
|
728
|
+
failure: {
|
|
729
|
+
message: "model returned a completed response with no content",
|
|
730
|
+
code: EMPTY_RESPONSE_CODE
|
|
731
|
+
}
|
|
732
|
+
} : reason
|
|
733
|
+
};
|
|
734
|
+
return;
|
|
735
|
+
}
|
|
736
|
+
}
|
|
737
|
+
throw new LlmError("Ollama NDJSON stream ended without a done chunk", "STREAM_CLOSED");
|
|
738
|
+
}
|
|
739
|
+
//#endregion
|
|
740
|
+
//#region \0@oxc-project+runtime@0.144.0/helpers/esm/usingCtx.js
|
|
741
|
+
function _usingCtx() {
|
|
742
|
+
var r = "function" == typeof SuppressedError ? SuppressedError : function(r, e) {
|
|
743
|
+
var n = Error();
|
|
744
|
+
return n.name = "SuppressedError", n.error = r, n.suppressed = e, n;
|
|
745
|
+
}, e = {}, n = [];
|
|
746
|
+
function using(r, e) {
|
|
747
|
+
if (null != e) {
|
|
748
|
+
if (Object(e) !== e) throw new TypeError("using declarations can only be used with objects, functions, null, or undefined.");
|
|
749
|
+
if (r) var o = e[Symbol.asyncDispose || Symbol["for"]("Symbol.asyncDispose")];
|
|
750
|
+
if (void 0 === o && (o = e[Symbol.dispose || Symbol["for"]("Symbol.dispose")], r)) var t = o;
|
|
751
|
+
if ("function" != typeof o) throw new TypeError("Object is not disposable.");
|
|
752
|
+
t && (o = function o() {
|
|
753
|
+
try {
|
|
754
|
+
t.call(e);
|
|
755
|
+
} catch (r) {
|
|
756
|
+
return Promise.reject(r);
|
|
757
|
+
}
|
|
758
|
+
}), n.push({
|
|
759
|
+
v: e,
|
|
760
|
+
d: o,
|
|
761
|
+
a: r
|
|
762
|
+
});
|
|
763
|
+
} else r && n.push({
|
|
764
|
+
d: e,
|
|
765
|
+
a: r
|
|
766
|
+
});
|
|
767
|
+
return e;
|
|
768
|
+
}
|
|
769
|
+
return {
|
|
770
|
+
e,
|
|
771
|
+
u: using.bind(null, !1),
|
|
772
|
+
a: using.bind(null, !0),
|
|
773
|
+
d: function d() {
|
|
774
|
+
var o, t = this.e, s = 0;
|
|
775
|
+
function next() {
|
|
776
|
+
for (; o = n.pop();) try {
|
|
777
|
+
if (!o.a && 1 === s) return s = 0, n.push(o), Promise.resolve().then(next);
|
|
778
|
+
if (o.d) {
|
|
779
|
+
var r = o.d.call(o.v);
|
|
780
|
+
if (o.a) return s |= 2, Promise.resolve(r).then(next, err);
|
|
781
|
+
} else s |= 1;
|
|
782
|
+
} catch (r) {
|
|
783
|
+
return err(r);
|
|
784
|
+
}
|
|
785
|
+
if (1 === s) return t !== e ? Promise.reject(t) : Promise.resolve();
|
|
786
|
+
if (t !== e) throw t;
|
|
787
|
+
}
|
|
788
|
+
function err(n) {
|
|
789
|
+
return t = t !== e ? new r(n, t) : n, next();
|
|
790
|
+
}
|
|
791
|
+
return next();
|
|
792
|
+
}
|
|
793
|
+
};
|
|
794
|
+
}
|
|
795
|
+
//#endregion
|
|
796
|
+
//#region src/adapter.ts
|
|
797
|
+
/**
|
|
798
|
+
* `OllamaAdapter`: the harness `LlmAdapter` over Ollama's native `/api/chat`
|
|
799
|
+
* streaming endpoint. Transport-only: the registering plugin owns config
|
|
800
|
+
* resolution (one `() => ResolvedConfig` thunk re-read per operation), so a
|
|
801
|
+
* changed base URL, model mapping, or timeout reaches the next request without
|
|
802
|
+
* re-registration, while an in-flight stream keeps the facts it started with.
|
|
803
|
+
* The adapter is text-only (`inputModalities: ['text']`); tool calls and tool
|
|
804
|
+
* results are translated by {@link serializeRequest}.
|
|
805
|
+
* @module dsh-local-ai/adapter
|
|
806
|
+
*/
|
|
807
|
+
/** Idle-timeout abort code stamped onto a stalled stream's timeout reason. */
|
|
808
|
+
const STREAM_IDLE_TIMEOUT_CODE = "LLM_STREAM_IDLE_TIMEOUT";
|
|
809
|
+
/** The single provider route this adapter owns. */
|
|
810
|
+
const OLLAMA_PROVIDER = "ollama";
|
|
811
|
+
/** Reverse a model mapping: Ollama model id → harness-visible name. */
|
|
812
|
+
function harnessNameOf(resolved, ollamaName) {
|
|
813
|
+
return resolved.models.find((entry) => entry.model === ollamaName)?.name ?? ollamaName;
|
|
814
|
+
}
|
|
815
|
+
/** Advertise one configured or discovered local model as text-only. */
|
|
816
|
+
function modelInfo(provider, id, name) {
|
|
817
|
+
return {
|
|
818
|
+
provider,
|
|
819
|
+
id,
|
|
820
|
+
name,
|
|
821
|
+
inputModalities: ["text"]
|
|
822
|
+
};
|
|
823
|
+
}
|
|
824
|
+
/**
|
|
825
|
+
* The Ollama provider adapter. One instance serves every harness-visible local
|
|
826
|
+
* model name; the harness model name maps to the wire model id through the
|
|
827
|
+
* configured model mapping (identity when unmapped).
|
|
828
|
+
*/
|
|
829
|
+
var OllamaAdapter = class extends LlmAdapter {
|
|
830
|
+
options;
|
|
831
|
+
constructor(options) {
|
|
832
|
+
super();
|
|
833
|
+
this.options = options;
|
|
834
|
+
}
|
|
835
|
+
fetchImpl() {
|
|
836
|
+
return this.options.fetchImpl ?? ((input, init) => globalThis.fetch(input, init));
|
|
837
|
+
}
|
|
838
|
+
providerInfo(provider) {
|
|
839
|
+
return {
|
|
840
|
+
id: provider,
|
|
841
|
+
name: "Ollama (local)"
|
|
842
|
+
};
|
|
843
|
+
}
|
|
844
|
+
listModels(provider) {
|
|
845
|
+
const resolved = this.options.config();
|
|
846
|
+
return listModels(resolved.baseURL, this.fetchImpl()).then((models) => models.map((model) => modelInfo(provider, harnessNameOf(resolved, model.name), harnessNameOf(resolved, model.name)))).catch(() => resolved.models.map((entry) => modelInfo(provider, entry.name, entry.name)));
|
|
847
|
+
}
|
|
848
|
+
resolveModel(provider, model, _signal) {
|
|
849
|
+
const resolved = this.options.config();
|
|
850
|
+
const mapping = resolved.models.find((entry) => entry.name === model);
|
|
851
|
+
return Promise.resolve({
|
|
852
|
+
provider,
|
|
853
|
+
id: model,
|
|
854
|
+
name: model,
|
|
855
|
+
inputModalities: ["text"],
|
|
856
|
+
context: { contextWindow: mapping?.contextWindow ?? resolved.defaultContextWindow },
|
|
857
|
+
defaultMaxTokens: mapping?.maxTokens ?? resolved.maxTokens
|
|
858
|
+
});
|
|
859
|
+
}
|
|
860
|
+
async *stream(options) {
|
|
861
|
+
try {
|
|
862
|
+
var _usingCtx$1 = _usingCtx();
|
|
863
|
+
const resolved = this.options.config();
|
|
864
|
+
const consumer = new AbortController();
|
|
865
|
+
const upstream = options.signal === void 0 ? consumer.signal : AbortSignal.any([options.signal, consumer.signal]);
|
|
866
|
+
const watchdog = _usingCtx$1.u(idleWatchdog(upstream, resolved.requestTimeoutMs, STREAM_IDLE_TIMEOUT_CODE));
|
|
867
|
+
const iterator = this.request(options, resolved, watchdog.signal)[Symbol.asyncIterator]();
|
|
868
|
+
let exhausted = false;
|
|
869
|
+
try {
|
|
870
|
+
while (true) {
|
|
871
|
+
const result = await watchdog.next(iterator);
|
|
872
|
+
if (result.done) {
|
|
873
|
+
exhausted = true;
|
|
874
|
+
return;
|
|
875
|
+
}
|
|
876
|
+
yield result.value;
|
|
877
|
+
}
|
|
878
|
+
} catch (error) {
|
|
879
|
+
if (timeoutOf(watchdog.signal, STREAM_IDLE_TIMEOUT_CODE) !== void 0) throw new LlmError(`Ollama stream idle timeout after ${resolved.requestTimeoutMs}ms`, "TIMEOUT", { cause: error });
|
|
880
|
+
if (options.signal?.aborted) throw new LlmError("Ollama request aborted by caller", "ABORTED", { cause: error });
|
|
881
|
+
if (error instanceof LlmError) throw error;
|
|
882
|
+
throw new LlmError(`Ollama API stream from ${sanitizeEndpoint(resolved.baseURL)} failed`, "TRANSPORT", { cause: error });
|
|
883
|
+
} finally {
|
|
884
|
+
consumer.abort("Ollama stream consumer stopped");
|
|
885
|
+
if (!exhausted && iterator.return !== void 0) try {
|
|
886
|
+
await iterator.return();
|
|
887
|
+
} catch (_abortedTransportTeardown) {}
|
|
888
|
+
}
|
|
889
|
+
} catch (_) {
|
|
890
|
+
_usingCtx$1.e = _;
|
|
891
|
+
} finally {
|
|
892
|
+
_usingCtx$1.d();
|
|
893
|
+
}
|
|
894
|
+
}
|
|
895
|
+
async *request(options, resolved, signal) {
|
|
896
|
+
const body = serializeRequest(options, resolved);
|
|
897
|
+
let response;
|
|
898
|
+
try {
|
|
899
|
+
response = await postStream(resolved.baseURL, "/api/chat", body, this.fetchImpl(), signal);
|
|
900
|
+
} catch (error) {
|
|
901
|
+
if (signal.aborted) throw error;
|
|
902
|
+
throw new LlmError(`Ollama API request to ${sanitizeEndpoint(resolved.baseURL)} failed`, "TRANSPORT", { cause: error });
|
|
903
|
+
}
|
|
904
|
+
if (!response.body) throw new LlmError("Ollama API returned no response body", "EMPTY_RESPONSE");
|
|
905
|
+
yield* translate(readNdjsonLines(response.body));
|
|
906
|
+
}
|
|
907
|
+
};
|
|
908
|
+
//#endregion
|
|
909
|
+
//#region src/route.ts
|
|
910
|
+
/**
|
|
911
|
+
* Local-model routing decision and the streaming fallback. The pure decision
|
|
912
|
+
* matches configured rules (task type via `purpose`, case-insensitive
|
|
913
|
+
* keywords, or a blanket `always`) against a request, in list order — first
|
|
914
|
+
* match wins. The streaming helper routes a matched request to the local
|
|
915
|
+
* Ollama adapter and, when the local route fails BEFORE producing any visible
|
|
916
|
+
* content, falls back to the cloud (`next()`) so a down Ollama never bricks a
|
|
917
|
+
* conversation. Once local content has started, it is streamed through — a
|
|
918
|
+
* mid-stream failure cannot be retracted.
|
|
919
|
+
* @module dsh-local-ai/route
|
|
920
|
+
*/
|
|
921
|
+
/**
|
|
922
|
+
* Concatenate the request's model-visible text (system prompt + every text
|
|
923
|
+
* block) for keyword matching.
|
|
924
|
+
* @param options - the request.
|
|
925
|
+
* @returns the joined text.
|
|
926
|
+
*/
|
|
927
|
+
function requestText(options) {
|
|
928
|
+
const parts = [];
|
|
929
|
+
if (options.system !== void 0) parts.push(options.system);
|
|
930
|
+
for (const message of options.messages) for (const block of message.content) if (block.type === "text") parts.push(block.text);
|
|
931
|
+
return parts.join("\n");
|
|
932
|
+
}
|
|
933
|
+
/** Whether a keyword appears case-insensitively in the text. */
|
|
934
|
+
function matchesKeyword(text, keyword) {
|
|
935
|
+
return text.toLowerCase().includes(keyword.toLowerCase());
|
|
936
|
+
}
|
|
937
|
+
/** Whether one resolved rule matches a request. */
|
|
938
|
+
function ruleMatches(rule, options) {
|
|
939
|
+
if (rule.always) return true;
|
|
940
|
+
if (rule.purpose !== void 0 && options.purpose === rule.purpose) return true;
|
|
941
|
+
if (rule.keywords.length > 0) {
|
|
942
|
+
const text = requestText(options);
|
|
943
|
+
return rule.keywords.some((keyword) => matchesKeyword(text, keyword));
|
|
944
|
+
}
|
|
945
|
+
return false;
|
|
946
|
+
}
|
|
947
|
+
/**
|
|
948
|
+
* Decide whether a request should route to a local model. A request already
|
|
949
|
+
* addressed to the `ollama` provider (explicit selection or a prior re-route)
|
|
950
|
+
* never re-routes.
|
|
951
|
+
* @param options - the request.
|
|
952
|
+
* @param resolved - the resolved config.
|
|
953
|
+
* @returns the local model to use, or `undefined` to stay on the cloud route.
|
|
954
|
+
*/
|
|
955
|
+
function decideRoute(options, resolved) {
|
|
956
|
+
if (options.provider === "ollama") return void 0;
|
|
957
|
+
for (const rule of resolved.route) if (ruleMatches(rule, options)) return { model: rule.model };
|
|
958
|
+
}
|
|
959
|
+
/**
|
|
960
|
+
* Stream a locally-routed request with automatic cloud fallback. The local
|
|
961
|
+
* stream is produced through `streamLocal` (the full harness stream, so the
|
|
962
|
+
* local route keeps retry and failure normalization). If the local route
|
|
963
|
+
* finishes with an error or aborts before any token delta, `next()` (the
|
|
964
|
+
* cloud) is streamed instead; otherwise the local stream is forwarded.
|
|
965
|
+
* @param streamLocal - produces the local stream for a re-routed request.
|
|
966
|
+
* @param options - the original request.
|
|
967
|
+
* @param decision - the local model to route to.
|
|
968
|
+
* @param next - the cloud stream (the waterfall's `next()`).
|
|
969
|
+
* @returns the effective chunk stream.
|
|
970
|
+
*/
|
|
971
|
+
async function* routeLocal(streamLocal, options, decision, next) {
|
|
972
|
+
const upstream = streamLocal({
|
|
973
|
+
...options,
|
|
974
|
+
provider: OLLAMA_PROVIDER,
|
|
975
|
+
model: decision.model
|
|
976
|
+
});
|
|
977
|
+
let producedContent = false;
|
|
978
|
+
const pending = [];
|
|
979
|
+
try {
|
|
980
|
+
for await (const chunk of upstream) {
|
|
981
|
+
if (producedContent) {
|
|
982
|
+
yield chunk;
|
|
983
|
+
continue;
|
|
984
|
+
}
|
|
985
|
+
if (chunk.type === "finish") {
|
|
986
|
+
if (chunk.reason.kind === "error" || chunk.reason.kind === "aborted") {
|
|
987
|
+
yield* next();
|
|
988
|
+
return;
|
|
989
|
+
}
|
|
990
|
+
yield chunk;
|
|
991
|
+
return;
|
|
992
|
+
}
|
|
993
|
+
pending.push(chunk);
|
|
994
|
+
if (isTokenDelta(chunk)) {
|
|
995
|
+
producedContent = true;
|
|
996
|
+
for (const buffered of pending) yield buffered;
|
|
997
|
+
pending.length = 0;
|
|
998
|
+
}
|
|
999
|
+
}
|
|
1000
|
+
for (const buffered of pending) yield buffered;
|
|
1001
|
+
} catch (error) {
|
|
1002
|
+
if (!producedContent) {
|
|
1003
|
+
yield* next();
|
|
1004
|
+
return;
|
|
1005
|
+
}
|
|
1006
|
+
for (const buffered of pending) yield buffered;
|
|
1007
|
+
throw error;
|
|
1008
|
+
}
|
|
1009
|
+
}
|
|
1010
|
+
//#endregion
|
|
1011
|
+
//#region src/health.ts
|
|
1012
|
+
/** The default probe: `ollama list` reaches the server through the CLI's own channel. */
|
|
1013
|
+
const OLLAMA_PROCESS_PROBE = {
|
|
1014
|
+
command: "ollama",
|
|
1015
|
+
args: ["list"]
|
|
1016
|
+
};
|
|
1017
|
+
/**
|
|
1018
|
+
* Check whether the Ollama HTTP API responds to `/api/version` within a
|
|
1019
|
+
* deadline. A timeout, transport failure, or non-2xx response is `ok: false`
|
|
1020
|
+
* with a sanitized error.
|
|
1021
|
+
* @param baseURL - the Ollama base URL.
|
|
1022
|
+
* @param fetchImpl - the fetch implementation.
|
|
1023
|
+
* @param timeoutMs - deadline in milliseconds.
|
|
1024
|
+
* @returns the API health result.
|
|
1025
|
+
*/
|
|
1026
|
+
async function checkApiHealth(baseURL, fetchImpl, timeoutMs) {
|
|
1027
|
+
const controller = new AbortController();
|
|
1028
|
+
const timer = setTimeout(() => controller.abort(/* @__PURE__ */ new Error("timeout")), timeoutMs);
|
|
1029
|
+
try {
|
|
1030
|
+
return {
|
|
1031
|
+
ok: true,
|
|
1032
|
+
version: sanitizeText(await apiVersion(baseURL, fetchImpl, controller.signal), 64)
|
|
1033
|
+
};
|
|
1034
|
+
} catch (error) {
|
|
1035
|
+
return {
|
|
1036
|
+
ok: false,
|
|
1037
|
+
error: sanitizeText(error instanceof Error ? error.message : String(error), 500)
|
|
1038
|
+
};
|
|
1039
|
+
} finally {
|
|
1040
|
+
clearTimeout(timer);
|
|
1041
|
+
}
|
|
1042
|
+
}
|
|
1043
|
+
/**
|
|
1044
|
+
* Check process liveness through the subprocess seam by spawning the probe
|
|
1045
|
+
* command in collect mode. Exit code 0 means the CLI (and, for the default
|
|
1046
|
+
* `ollama list` probe, the server it talks to) is alive; a missing executable
|
|
1047
|
+
* or non-zero exit is `present: false` with a sanitized error.
|
|
1048
|
+
* @param subprocess - the real subprocess runtime.
|
|
1049
|
+
* @param graceMs - terminate grace for the spawned probe.
|
|
1050
|
+
* @param probe - the command to probe with (defaults to `ollama list`).
|
|
1051
|
+
* @returns the process health result.
|
|
1052
|
+
*/
|
|
1053
|
+
async function checkProcessHealth(subprocess, graceMs, probe = OLLAMA_PROCESS_PROBE) {
|
|
1054
|
+
let executable;
|
|
1055
|
+
try {
|
|
1056
|
+
executable = await subprocess.resolveExecutable(probe.command);
|
|
1057
|
+
} catch (error) {
|
|
1058
|
+
return {
|
|
1059
|
+
present: false,
|
|
1060
|
+
error: sanitizeText(`"${probe.command}" not found: ${error instanceof Error ? error.message : String(error)}`, 500)
|
|
1061
|
+
};
|
|
1062
|
+
}
|
|
1063
|
+
const handle = subprocess.spawn({
|
|
1064
|
+
argv: [executable, ...probe.args],
|
|
1065
|
+
cwd: tmpdir(),
|
|
1066
|
+
stdio: {
|
|
1067
|
+
stdin: "ignore",
|
|
1068
|
+
stdout: { maxBytes: 4096 },
|
|
1069
|
+
stderr: { maxBytes: 4096 }
|
|
1070
|
+
},
|
|
1071
|
+
graceMs
|
|
1072
|
+
});
|
|
1073
|
+
const outcome = await handle.done;
|
|
1074
|
+
if (outcome.exitCode === 0) return { present: true };
|
|
1075
|
+
const stderr = handle.collected.stderr?.readFrom(0).text ?? "";
|
|
1076
|
+
return {
|
|
1077
|
+
present: false,
|
|
1078
|
+
error: sanitizeText(stderr.trim().length > 0 ? stderr : `exit code ${String(outcome.exitCode)}`, 500)
|
|
1079
|
+
};
|
|
1080
|
+
}
|
|
1081
|
+
/**
|
|
1082
|
+
* Check both health signals.
|
|
1083
|
+
* @param baseURL - the Ollama base URL.
|
|
1084
|
+
* @param fetchImpl - the fetch implementation.
|
|
1085
|
+
* @param subprocess - the real subprocess runtime.
|
|
1086
|
+
* @param requestTimeoutMs - HTTP deadline in milliseconds.
|
|
1087
|
+
* @param graceMs - subprocess terminate grace in milliseconds.
|
|
1088
|
+
* @returns the combined health result.
|
|
1089
|
+
*/
|
|
1090
|
+
async function checkHealth(baseURL, fetchImpl, subprocess, requestTimeoutMs, graceMs) {
|
|
1091
|
+
const [api, process] = await Promise.all([checkApiHealth(baseURL, fetchImpl, requestTimeoutMs), checkProcessHealth(subprocess, graceMs)]);
|
|
1092
|
+
return {
|
|
1093
|
+
api,
|
|
1094
|
+
process
|
|
1095
|
+
};
|
|
1096
|
+
}
|
|
1097
|
+
//#endregion
|
|
1098
|
+
//#region src/version.ts
|
|
1099
|
+
/**
|
|
1100
|
+
* The `dsh-local-ai` plugin version. The release script bumps this string
|
|
1101
|
+
* alongside `package.json`; `test/version.spec.ts` trips when the two drift.
|
|
1102
|
+
*/
|
|
1103
|
+
const VERSION = "0.1.0";
|
|
1104
|
+
//#endregion
|
|
1105
|
+
//#region src/index.ts
|
|
1106
|
+
const name = "local-ai";
|
|
1107
|
+
const inject = [
|
|
1108
|
+
"llm",
|
|
1109
|
+
"tools",
|
|
1110
|
+
"subprocess",
|
|
1111
|
+
"commands"
|
|
1112
|
+
];
|
|
1113
|
+
/** Format a byte count into a compact human-readable string. */
|
|
1114
|
+
function formatBytes(bytes) {
|
|
1115
|
+
if (!Number.isFinite(bytes) || bytes < 0) return "0 B";
|
|
1116
|
+
const units = [
|
|
1117
|
+
"B",
|
|
1118
|
+
"KB",
|
|
1119
|
+
"MB",
|
|
1120
|
+
"GB",
|
|
1121
|
+
"TB"
|
|
1122
|
+
];
|
|
1123
|
+
let value = bytes;
|
|
1124
|
+
let unit = 0;
|
|
1125
|
+
while (value >= 1024 && unit < units.length - 1) {
|
|
1126
|
+
value /= 1024;
|
|
1127
|
+
unit += 1;
|
|
1128
|
+
}
|
|
1129
|
+
return `${unit === 0 ? String(Math.round(value)) : value.toFixed(1)} ${units[unit]}`;
|
|
1130
|
+
}
|
|
1131
|
+
/** Render the `ollama_list` canonical value as model-visible text. */
|
|
1132
|
+
function renderList(value) {
|
|
1133
|
+
const list = value;
|
|
1134
|
+
const models = list.models ?? [];
|
|
1135
|
+
const lines = [`${models.length} local model(s), ${formatBytes(list.totalBytes ?? 0)} on disk`];
|
|
1136
|
+
for (const model of models) {
|
|
1137
|
+
const detail = [model.parameterSize, model.quantization].filter((part) => part !== void 0).join(" ");
|
|
1138
|
+
lines.push(`- ${model.name}${detail.length > 0 ? ` (${detail})` : ""} — ${formatBytes(model.size)}${model.running ? " [running]" : ""}`);
|
|
1139
|
+
}
|
|
1140
|
+
if (list.running !== void 0 && list.running.length > 0) lines.push(`running: ${list.running.join(", ")}`);
|
|
1141
|
+
return lines.join("\n");
|
|
1142
|
+
}
|
|
1143
|
+
/** Render the `ollama_show` canonical value as model-visible text. */
|
|
1144
|
+
function renderShow(value) {
|
|
1145
|
+
const show = value;
|
|
1146
|
+
const parts = [
|
|
1147
|
+
[show.parameterSize, show.quantization].filter((part) => part !== void 0).join(" "),
|
|
1148
|
+
show.contextLength !== void 0 ? `context ${show.contextLength}` : void 0,
|
|
1149
|
+
show.family,
|
|
1150
|
+
show.format
|
|
1151
|
+
].filter((part) => part !== void 0 && part.length > 0);
|
|
1152
|
+
return `${show.name}${parts.length > 0 ? ` — ${parts.join(", ")}` : ""}`;
|
|
1153
|
+
}
|
|
1154
|
+
/** Render a health canonical value as model-visible text. */
|
|
1155
|
+
function renderHealth(value) {
|
|
1156
|
+
const health = value;
|
|
1157
|
+
return [`API: ${health.api.ok ? `ok${health.api.version !== void 0 ? ` (v${health.api.version})` : ""}` : "down"}`, `process: ${health.process.present ? "alive" : "not detected"}`].join("\n");
|
|
1158
|
+
}
|
|
1159
|
+
/** Render a pull/remove canonical value as model-visible text. */
|
|
1160
|
+
function renderOperation(value) {
|
|
1161
|
+
const op = value;
|
|
1162
|
+
if (op.removed === true) return `removed ${op.name}`;
|
|
1163
|
+
return `${op.name}: ${op.status ?? "done"}`;
|
|
1164
|
+
}
|
|
1165
|
+
/**
|
|
1166
|
+
* Mount the plugin: resolve config (fail loud), register the Ollama adapter,
|
|
1167
|
+
* the `llm/stream` routing waterfall, the five management tools, and the
|
|
1168
|
+
* `/ollama` command. Every contribution goes through its registry's effect
|
|
1169
|
+
* (register/on), so stop and hot-reload withdraw all of it.
|
|
1170
|
+
* @param ctx - the plugin context (host).
|
|
1171
|
+
* @param config - raw plugin config.
|
|
1172
|
+
*/
|
|
1173
|
+
function apply(ctx, config = {}) {
|
|
1174
|
+
const resolved = resolveConfig(config);
|
|
1175
|
+
const logger = ctx.logger("local-ai");
|
|
1176
|
+
const fetchImpl = (input, init) => globalThis.fetch(input, init);
|
|
1177
|
+
const adapter = new OllamaAdapter({
|
|
1178
|
+
config: () => resolved,
|
|
1179
|
+
fetchImpl
|
|
1180
|
+
});
|
|
1181
|
+
ctx.llm.registerAdapter([OLLAMA_PROVIDER], adapter);
|
|
1182
|
+
ctx.on("llm/stream", (options, next) => {
|
|
1183
|
+
const decision = decideRoute(options, resolved);
|
|
1184
|
+
if (decision === void 0) return next();
|
|
1185
|
+
return routeLocal((reRouted) => ctx.llm.stream(reRouted), options, decision, next);
|
|
1186
|
+
});
|
|
1187
|
+
ctx.tools.register(defineTool({
|
|
1188
|
+
name: "ollama_list",
|
|
1189
|
+
description: "List local Ollama models with disk usage and which are currently loaded (running).",
|
|
1190
|
+
parameters: {},
|
|
1191
|
+
output: {
|
|
1192
|
+
schema: { type: "json" },
|
|
1193
|
+
render: (_args, value) => [{
|
|
1194
|
+
type: "text",
|
|
1195
|
+
text: renderList(value)
|
|
1196
|
+
}]
|
|
1197
|
+
},
|
|
1198
|
+
async execute(_args, exec) {
|
|
1199
|
+
const [models, running] = await Promise.all([listModels(resolved.baseURL, fetchImpl, exec.signal), listRunning(resolved.baseURL, fetchImpl, exec.signal).catch(() => [])]);
|
|
1200
|
+
const runningNames = new Set(running.map((model) => model.name));
|
|
1201
|
+
return {
|
|
1202
|
+
models: models.map((model) => ({
|
|
1203
|
+
name: model.name,
|
|
1204
|
+
size: model.size,
|
|
1205
|
+
...model.details?.parameter_size === void 0 ? {} : { parameterSize: model.details.parameter_size },
|
|
1206
|
+
...model.details?.quantization_level === void 0 ? {} : { quantization: model.details.quantization_level },
|
|
1207
|
+
running: runningNames.has(model.name)
|
|
1208
|
+
})),
|
|
1209
|
+
running: running.map((model) => model.name),
|
|
1210
|
+
count: models.length,
|
|
1211
|
+
totalBytes: models.reduce((sum, model) => sum + model.size, 0)
|
|
1212
|
+
};
|
|
1213
|
+
}
|
|
1214
|
+
}));
|
|
1215
|
+
ctx.tools.register(defineTool({
|
|
1216
|
+
name: "ollama_show",
|
|
1217
|
+
description: "Show details for one local Ollama model: parameter size, quantization, and context length.",
|
|
1218
|
+
parameters: { name: {
|
|
1219
|
+
type: "string",
|
|
1220
|
+
required: true,
|
|
1221
|
+
description: "The Ollama model name to inspect."
|
|
1222
|
+
} },
|
|
1223
|
+
output: {
|
|
1224
|
+
schema: { type: "json" },
|
|
1225
|
+
render: (_args, value) => [{
|
|
1226
|
+
type: "text",
|
|
1227
|
+
text: renderShow(value)
|
|
1228
|
+
}]
|
|
1229
|
+
},
|
|
1230
|
+
async execute(args, exec) {
|
|
1231
|
+
const name = args.name;
|
|
1232
|
+
const show = await showModel(resolved.baseURL, name, fetchImpl, exec.signal);
|
|
1233
|
+
return {
|
|
1234
|
+
name,
|
|
1235
|
+
...show.details?.parameter_size === void 0 ? {} : { parameterSize: show.details.parameter_size },
|
|
1236
|
+
...show.details?.quantization_level === void 0 ? {} : { quantization: show.details.quantization_level },
|
|
1237
|
+
...show.details?.family === void 0 ? {} : { family: show.details.family },
|
|
1238
|
+
...show.details?.format === void 0 ? {} : { format: show.details.format },
|
|
1239
|
+
...(() => {
|
|
1240
|
+
const contextLength = contextLengthOf(show);
|
|
1241
|
+
return contextLength === void 0 ? {} : { contextLength };
|
|
1242
|
+
})()
|
|
1243
|
+
};
|
|
1244
|
+
}
|
|
1245
|
+
}));
|
|
1246
|
+
ctx.tools.register(defineTool({
|
|
1247
|
+
name: "ollama_pull",
|
|
1248
|
+
description: "Pull (download) a model into the local Ollama server.",
|
|
1249
|
+
parameters: { name: {
|
|
1250
|
+
type: "string",
|
|
1251
|
+
required: true,
|
|
1252
|
+
description: "The Ollama model name to pull (e.g. llama3.2)."
|
|
1253
|
+
} },
|
|
1254
|
+
output: {
|
|
1255
|
+
schema: { type: "json" },
|
|
1256
|
+
render: (_args, value) => [{
|
|
1257
|
+
type: "text",
|
|
1258
|
+
text: renderOperation(value)
|
|
1259
|
+
}]
|
|
1260
|
+
},
|
|
1261
|
+
async execute(args, exec) {
|
|
1262
|
+
const name = args.name;
|
|
1263
|
+
return {
|
|
1264
|
+
name,
|
|
1265
|
+
status: (await pullModel(resolved.baseURL, name, fetchImpl, exec.signal)).status
|
|
1266
|
+
};
|
|
1267
|
+
}
|
|
1268
|
+
}));
|
|
1269
|
+
ctx.tools.register(defineTool({
|
|
1270
|
+
name: "ollama_remove",
|
|
1271
|
+
description: "Remove (delete) a model from the local Ollama server.",
|
|
1272
|
+
parameters: { name: {
|
|
1273
|
+
type: "string",
|
|
1274
|
+
required: true,
|
|
1275
|
+
description: "The Ollama model name to remove."
|
|
1276
|
+
} },
|
|
1277
|
+
output: {
|
|
1278
|
+
schema: { type: "json" },
|
|
1279
|
+
render: (_args, value) => [{
|
|
1280
|
+
type: "text",
|
|
1281
|
+
text: renderOperation(value)
|
|
1282
|
+
}]
|
|
1283
|
+
},
|
|
1284
|
+
async execute(args, exec) {
|
|
1285
|
+
const name = args.name;
|
|
1286
|
+
await removeModel(resolved.baseURL, name, fetchImpl, exec.signal);
|
|
1287
|
+
return {
|
|
1288
|
+
name,
|
|
1289
|
+
removed: true
|
|
1290
|
+
};
|
|
1291
|
+
}
|
|
1292
|
+
}));
|
|
1293
|
+
ctx.tools.register(defineTool({
|
|
1294
|
+
name: "ollama_health",
|
|
1295
|
+
description: "Check the local Ollama server: whether the process is alive and whether the API responds.",
|
|
1296
|
+
parameters: {},
|
|
1297
|
+
output: {
|
|
1298
|
+
schema: { type: "json" },
|
|
1299
|
+
render: (_args, value) => [{
|
|
1300
|
+
type: "text",
|
|
1301
|
+
text: renderHealth(value)
|
|
1302
|
+
}]
|
|
1303
|
+
},
|
|
1304
|
+
async execute(_args, exec) {
|
|
1305
|
+
return await checkHealth(resolved.baseURL, fetchImpl, ctx.subprocess, resolved.requestTimeoutMs, resolved.graceMs);
|
|
1306
|
+
}
|
|
1307
|
+
}));
|
|
1308
|
+
ctx.commands.register({
|
|
1309
|
+
name: "ollama",
|
|
1310
|
+
description: "One-shot status overview: local models, disk usage, health, and routing suggestions.",
|
|
1311
|
+
async handler() {
|
|
1312
|
+
const health = await checkHealth(resolved.baseURL, fetchImpl, ctx.subprocess, resolved.requestTimeoutMs, resolved.graceMs);
|
|
1313
|
+
const lines = ["Ollama status:"];
|
|
1314
|
+
lines.push(`- API: ${health.api.ok ? `ok${health.api.version !== void 0 ? ` (v${health.api.version})` : ""}` : "down"}`);
|
|
1315
|
+
lines.push(`- process: ${health.process.present ? "alive" : "not detected"}`);
|
|
1316
|
+
let models = [];
|
|
1317
|
+
try {
|
|
1318
|
+
models = await listModels(resolved.baseURL, fetchImpl);
|
|
1319
|
+
} catch {}
|
|
1320
|
+
const totalBytes = models.reduce((sum, model) => sum + model.size, 0);
|
|
1321
|
+
lines.push(`- models: ${models.length} installed (${formatBytes(totalBytes)})`);
|
|
1322
|
+
for (const model of models) lines.push(` - ${model.name} (${formatBytes(model.size)})`);
|
|
1323
|
+
if (!health.api.ok && !health.process.present) lines.push("suggestion: start the Ollama server (e.g. `ollama serve`)");
|
|
1324
|
+
else if (resolved.route.length === 0) lines.push("suggestion: configure `route` rules to route requests to local models");
|
|
1325
|
+
return {
|
|
1326
|
+
kind: "success",
|
|
1327
|
+
text: lines.join("\n")
|
|
1328
|
+
};
|
|
1329
|
+
}
|
|
1330
|
+
});
|
|
1331
|
+
logger.info(`ollama adapter registered at ${resolved.baseURL} (${resolved.models.length} mapping(s), ${resolved.route.length} route rule(s))`);
|
|
1332
|
+
}
|
|
1333
|
+
//#endregion
|
|
1334
|
+
export { Config, REDACTED, VERSION, apply, formatBytes, inject, name, redactSecrets, renderHealth, renderList, renderOperation, renderShow, resolveConfig, sanitizeEndpoint, sanitizePath, sanitizeText, truncate };
|