dsh-local-ai 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/LICENSE +201 -0
- package/README.es.md +211 -0
- package/README.hi.md +211 -0
- package/README.md +211 -0
- package/README.pt.md +211 -0
- package/README.zh.md +211 -0
- package/THIRD_PARTY_NOTICES.md +20 -0
- package/cordis.patch.yml +44 -0
- package/lib/index.js +1334 -0
- package/lib/types/adapter.d.ts +39 -0
- package/lib/types/adapter.d.ts.map +1 -0
- package/lib/types/adapter.js +190 -0
- package/lib/types/adapter.js.map +1 -0
- package/lib/types/config.d.ts +109 -0
- package/lib/types/config.d.ts.map +1 -0
- package/lib/types/config.js +161 -0
- package/lib/types/config.js.map +1 -0
- package/lib/types/health.d.ts +64 -0
- package/lib/types/health.d.ts.map +1 -0
- package/lib/types/health.js +92 -0
- package/lib/types/health.js.map +1 -0
- package/lib/types/index.d.ts +43 -0
- package/lib/types/index.d.ts.map +1 -0
- package/lib/types/index.js +232 -0
- package/lib/types/index.js.map +1 -0
- package/lib/types/ollama.d.ts +91 -0
- package/lib/types/ollama.d.ts.map +1 -0
- package/lib/types/ollama.js +184 -0
- package/lib/types/ollama.js.map +1 -0
- package/lib/types/route.d.ts +52 -0
- package/lib/types/route.d.ts.map +1 -0
- package/lib/types/route.js +119 -0
- package/lib/types/route.js.map +1 -0
- package/lib/types/sanitize.d.ts +58 -0
- package/lib/types/sanitize.d.ts.map +1 -0
- package/lib/types/sanitize.js +110 -0
- package/lib/types/sanitize.js.map +1 -0
- package/lib/types/serialize.d.ts +65 -0
- package/lib/types/serialize.d.ts.map +1 -0
- package/lib/types/serialize.js +149 -0
- package/lib/types/serialize.js.map +1 -0
- package/lib/types/translate.d.ts +57 -0
- package/lib/types/translate.d.ts.map +1 -0
- package/lib/types/translate.js +169 -0
- package/lib/types/translate.js.map +1 -0
- package/lib/types/version.d.ts +6 -0
- package/lib/types/version.d.ts.map +1 -0
- package/lib/types/version.js +6 -0
- package/lib/types/version.js.map +1 -0
- package/package.json +141 -0
- package/src/adapter.ts +158 -0
- package/src/config.ts +243 -0
- package/src/health.ts +133 -0
- package/src/index.ts +279 -0
- package/src/ollama.ts +274 -0
- package/src/route.ts +127 -0
- package/src/sanitize.ts +114 -0
- package/src/serialize.ts +178 -0
- package/src/translate.ts +205 -0
- package/src/version.ts +5 -0
package/src/index.ts
ADDED
|
@@ -0,0 +1,279 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `dsh-local-ai` — local-model (Ollama) integration for DeepSeek Harness.
|
|
3
|
+
* Registers the `ollama` `LlmAdapter` route, exposes discovery/management
|
|
4
|
+
* tools (`ollama_list`, `ollama_show`, `ollama_pull`, `ollama_remove`) plus a
|
|
5
|
+
* health check, routes requests to local models by task type or keyword with
|
|
6
|
+
* automatic fallback to the cloud, and provides the `/ollama` one-shot status
|
|
7
|
+
* command. Zero runtime dependencies beyond the harness peers: everything
|
|
8
|
+
* talks to Ollama over its HTTP API (or, for process liveness, the CLI).
|
|
9
|
+
*
|
|
10
|
+
* Function plugin — no default export (the Loader unwraps
|
|
11
|
+
* `exports.default ?? exports`, and a stray default would discard
|
|
12
|
+
* `name`/`inject`/`Config`/`apply`).
|
|
13
|
+
* @module dsh-local-ai
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
import type { Context } from '@deepseek-ai/cordis'
|
|
17
|
+
import { defineTool } from '@deepseek-ai/dsh-tools'
|
|
18
|
+
import type { JsonValue } from '@deepseek-ai/dsh-tools'
|
|
19
|
+
import type { CommandResult } from '@deepseek-ai/dsh-commands'
|
|
20
|
+
import type { GenerateOptions, StreamChunk } from '@deepseek-ai/dsh-llm'
|
|
21
|
+
import { Config, resolveConfig } from './config.ts'
|
|
22
|
+
import { OllamaAdapter, OLLAMA_PROVIDER } from './adapter.ts'
|
|
23
|
+
import { decideRoute, routeLocal } from './route.ts'
|
|
24
|
+
import { checkHealth } from './health.ts'
|
|
25
|
+
import { contextLengthOf, listModels, listRunning, pullModel, removeModel, showModel } from './ollama.ts'
|
|
26
|
+
import type { FetchLike } from './ollama.ts'
|
|
27
|
+
|
|
28
|
+
export const name = 'local-ai'
|
|
29
|
+
export const inject = ['llm', 'tools', 'subprocess', 'commands']
|
|
30
|
+
|
|
31
|
+
export { Config, resolveConfig } from './config.ts'
|
|
32
|
+
export type { Config as LocalAiConfig, ModelMapping, ResolvedConfig, ResolvedModelMapping, ResolvedRouteRule, RouteRule } from './config.ts'
|
|
33
|
+
export { VERSION } from './version.ts'
|
|
34
|
+
export { REDACTED, redactSecrets, sanitizeEndpoint, sanitizePath, sanitizeText, truncate } from './sanitize.ts'
|
|
35
|
+
|
|
36
|
+
/** Format a byte count into a compact human-readable string. */
|
|
37
|
+
export function formatBytes(bytes: number): string {
|
|
38
|
+
if (!Number.isFinite(bytes) || bytes < 0) return '0 B'
|
|
39
|
+
const units = ['B', 'KB', 'MB', 'GB', 'TB']
|
|
40
|
+
let value = bytes
|
|
41
|
+
let unit = 0
|
|
42
|
+
while (value >= 1024 && unit < units.length - 1) {
|
|
43
|
+
value /= 1024
|
|
44
|
+
unit += 1
|
|
45
|
+
}
|
|
46
|
+
const rounded = unit === 0 ? String(Math.round(value)) : value.toFixed(1)
|
|
47
|
+
return `${rounded} ${units[unit]}`
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/** One installed-model row reported by `ollama_list`. */
|
|
51
|
+
interface ListedModel {
|
|
52
|
+
name: string
|
|
53
|
+
size: number
|
|
54
|
+
parameterSize?: string
|
|
55
|
+
quantization?: string
|
|
56
|
+
running: boolean
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/** The `ollama_list` canonical value. */
|
|
60
|
+
interface ListValue {
|
|
61
|
+
models: ListedModel[]
|
|
62
|
+
running: string[]
|
|
63
|
+
count: number
|
|
64
|
+
totalBytes: number
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/** The `ollama_show` canonical value. */
|
|
68
|
+
interface ShowValue {
|
|
69
|
+
name: string
|
|
70
|
+
parameterSize?: string
|
|
71
|
+
quantization?: string
|
|
72
|
+
contextLength?: number
|
|
73
|
+
family?: string
|
|
74
|
+
format?: string
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/** Render the `ollama_list` canonical value as model-visible text. */
|
|
78
|
+
export function renderList(value: JsonValue): string {
|
|
79
|
+
const list = value as unknown as ListValue
|
|
80
|
+
const models = list.models ?? []
|
|
81
|
+
const lines = [`${models.length} local model(s), ${formatBytes(list.totalBytes ?? 0)} on disk`]
|
|
82
|
+
for (const model of models) {
|
|
83
|
+
const detail = [model.parameterSize, model.quantization].filter(part => part !== undefined).join(' ')
|
|
84
|
+
lines.push(`- ${model.name}${detail.length > 0 ? ` (${detail})` : ''} — ${formatBytes(model.size)}${model.running ? ' [running]' : ''}`)
|
|
85
|
+
}
|
|
86
|
+
if (list.running !== undefined && list.running.length > 0) {
|
|
87
|
+
lines.push(`running: ${list.running.join(', ')}`)
|
|
88
|
+
}
|
|
89
|
+
return lines.join('\n')
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/** Render the `ollama_show` canonical value as model-visible text. */
|
|
93
|
+
export function renderShow(value: JsonValue): string {
|
|
94
|
+
const show = value as unknown as ShowValue
|
|
95
|
+
const detail = [show.parameterSize, show.quantization].filter(part => part !== undefined).join(' ')
|
|
96
|
+
const context = show.contextLength !== undefined ? `context ${show.contextLength}` : undefined
|
|
97
|
+
const parts = [detail, context, show.family, show.format].filter(part => part !== undefined && part.length > 0)
|
|
98
|
+
return `${show.name}${parts.length > 0 ? ` — ${parts.join(', ')}` : ''}`
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
/** Render a health canonical value as model-visible text. */
|
|
102
|
+
export function renderHealth(value: JsonValue): string {
|
|
103
|
+
const health = value as unknown as {
|
|
104
|
+
api: { ok: boolean; version?: string }
|
|
105
|
+
process: { present: boolean; error?: string }
|
|
106
|
+
}
|
|
107
|
+
return [
|
|
108
|
+
`API: ${health.api.ok ? `ok${health.api.version !== undefined ? ` (v${health.api.version})` : ''}` : 'down'}`,
|
|
109
|
+
`process: ${health.process.present ? 'alive' : 'not detected'}`,
|
|
110
|
+
].join('\n')
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/** Render a pull/remove canonical value as model-visible text. */
|
|
114
|
+
export function renderOperation(value: JsonValue): string {
|
|
115
|
+
const op = value as unknown as { name: string; status?: string; removed?: boolean }
|
|
116
|
+
if (op.removed === true) return `removed ${op.name}`
|
|
117
|
+
return `${op.name}: ${op.status ?? 'done'}`
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* Mount the plugin: resolve config (fail loud), register the Ollama adapter,
|
|
122
|
+
* the `llm/stream` routing waterfall, the five management tools, and the
|
|
123
|
+
* `/ollama` command. Every contribution goes through its registry's effect
|
|
124
|
+
* (register/on), so stop and hot-reload withdraw all of it.
|
|
125
|
+
* @param ctx - the plugin context (host).
|
|
126
|
+
* @param config - raw plugin config.
|
|
127
|
+
*/
|
|
128
|
+
export function apply(ctx: Context, config: Config = {}): void {
|
|
129
|
+
const resolved = resolveConfig(config)
|
|
130
|
+
const logger = ctx.logger('local-ai')
|
|
131
|
+
// Lazily-bound so tests can stub `globalThis.fetch` before a call.
|
|
132
|
+
const fetchImpl: FetchLike = (input, init) => globalThis.fetch(input, init)
|
|
133
|
+
|
|
134
|
+
const adapter = new OllamaAdapter({ config: () => resolved, fetchImpl })
|
|
135
|
+
ctx.llm.registerAdapter([OLLAMA_PROVIDER], adapter)
|
|
136
|
+
|
|
137
|
+
// Routing waterfall: passthrough by default; a matched rule re-routes to the
|
|
138
|
+
// local model and falls back to `next()` (the cloud) when local fails first.
|
|
139
|
+
ctx.on('llm/stream', (options: GenerateOptions, next: () => AsyncIterable<StreamChunk>): AsyncIterable<StreamChunk> => {
|
|
140
|
+
const decision = decideRoute(options, resolved)
|
|
141
|
+
if (decision === undefined) return next()
|
|
142
|
+
return routeLocal((reRouted: GenerateOptions) => ctx.llm.stream(reRouted), options, decision, next)
|
|
143
|
+
})
|
|
144
|
+
|
|
145
|
+
ctx.tools.register(defineTool({
|
|
146
|
+
name: 'ollama_list',
|
|
147
|
+
description: 'List local Ollama models with disk usage and which are currently loaded (running).',
|
|
148
|
+
parameters: {},
|
|
149
|
+
output: {
|
|
150
|
+
schema: { type: 'json' },
|
|
151
|
+
render: (_args, value) => [{ type: 'text', text: renderList(value) }],
|
|
152
|
+
},
|
|
153
|
+
async execute(_args, exec): Promise<JsonValue> {
|
|
154
|
+
const [models, running] = await Promise.all([
|
|
155
|
+
listModels(resolved.baseURL, fetchImpl, exec.signal),
|
|
156
|
+
listRunning(resolved.baseURL, fetchImpl, exec.signal).catch(() => []),
|
|
157
|
+
])
|
|
158
|
+
const runningNames = new Set(running.map(model => model.name))
|
|
159
|
+
const value: ListValue = {
|
|
160
|
+
models: models.map(model => ({
|
|
161
|
+
name: model.name,
|
|
162
|
+
size: model.size,
|
|
163
|
+
...model.details?.parameter_size === undefined ? {} : { parameterSize: model.details.parameter_size },
|
|
164
|
+
...model.details?.quantization_level === undefined ? {} : { quantization: model.details.quantization_level },
|
|
165
|
+
running: runningNames.has(model.name),
|
|
166
|
+
})),
|
|
167
|
+
running: running.map(model => model.name),
|
|
168
|
+
count: models.length,
|
|
169
|
+
totalBytes: models.reduce((sum, model) => sum + model.size, 0),
|
|
170
|
+
}
|
|
171
|
+
return value as unknown as JsonValue
|
|
172
|
+
},
|
|
173
|
+
}))
|
|
174
|
+
|
|
175
|
+
ctx.tools.register(defineTool({
|
|
176
|
+
name: 'ollama_show',
|
|
177
|
+
description: 'Show details for one local Ollama model: parameter size, quantization, and context length.',
|
|
178
|
+
parameters: {
|
|
179
|
+
name: { type: 'string', required: true, description: 'The Ollama model name to inspect.' },
|
|
180
|
+
},
|
|
181
|
+
output: {
|
|
182
|
+
schema: { type: 'json' },
|
|
183
|
+
render: (_args, value) => [{ type: 'text', text: renderShow(value) }],
|
|
184
|
+
},
|
|
185
|
+
async execute(args, exec): Promise<JsonValue> {
|
|
186
|
+
const name = (args as { name: string }).name
|
|
187
|
+
const show = await showModel(resolved.baseURL, name, fetchImpl, exec.signal)
|
|
188
|
+
const value: ShowValue = {
|
|
189
|
+
name,
|
|
190
|
+
...show.details?.parameter_size === undefined ? {} : { parameterSize: show.details.parameter_size },
|
|
191
|
+
...show.details?.quantization_level === undefined ? {} : { quantization: show.details.quantization_level },
|
|
192
|
+
...show.details?.family === undefined ? {} : { family: show.details.family },
|
|
193
|
+
...show.details?.format === undefined ? {} : { format: show.details.format },
|
|
194
|
+
...((): { contextLength?: number } => {
|
|
195
|
+
const contextLength = contextLengthOf(show)
|
|
196
|
+
return contextLength === undefined ? {} : { contextLength }
|
|
197
|
+
})(),
|
|
198
|
+
}
|
|
199
|
+
return value as unknown as JsonValue
|
|
200
|
+
},
|
|
201
|
+
}))
|
|
202
|
+
|
|
203
|
+
ctx.tools.register(defineTool({
|
|
204
|
+
name: 'ollama_pull',
|
|
205
|
+
description: 'Pull (download) a model into the local Ollama server.',
|
|
206
|
+
parameters: {
|
|
207
|
+
name: { type: 'string', required: true, description: 'The Ollama model name to pull (e.g. llama3.2).' },
|
|
208
|
+
},
|
|
209
|
+
output: {
|
|
210
|
+
schema: { type: 'json' },
|
|
211
|
+
render: (_args, value) => [{ type: 'text', text: renderOperation(value) }],
|
|
212
|
+
},
|
|
213
|
+
async execute(args, exec): Promise<JsonValue> {
|
|
214
|
+
const name = (args as { name: string }).name
|
|
215
|
+
const result = await pullModel(resolved.baseURL, name, fetchImpl, exec.signal)
|
|
216
|
+
return { name, status: result.status } as unknown as JsonValue
|
|
217
|
+
},
|
|
218
|
+
}))
|
|
219
|
+
|
|
220
|
+
ctx.tools.register(defineTool({
|
|
221
|
+
name: 'ollama_remove',
|
|
222
|
+
description: 'Remove (delete) a model from the local Ollama server.',
|
|
223
|
+
parameters: {
|
|
224
|
+
name: { type: 'string', required: true, description: 'The Ollama model name to remove.' },
|
|
225
|
+
},
|
|
226
|
+
output: {
|
|
227
|
+
schema: { type: 'json' },
|
|
228
|
+
render: (_args, value) => [{ type: 'text', text: renderOperation(value) }],
|
|
229
|
+
},
|
|
230
|
+
async execute(args, exec): Promise<JsonValue> {
|
|
231
|
+
const name = (args as { name: string }).name
|
|
232
|
+
await removeModel(resolved.baseURL, name, fetchImpl, exec.signal)
|
|
233
|
+
return { name, removed: true } as unknown as JsonValue
|
|
234
|
+
},
|
|
235
|
+
}))
|
|
236
|
+
|
|
237
|
+
ctx.tools.register(defineTool({
|
|
238
|
+
name: 'ollama_health',
|
|
239
|
+
description: 'Check the local Ollama server: whether the process is alive and whether the API responds.',
|
|
240
|
+
parameters: {},
|
|
241
|
+
output: {
|
|
242
|
+
schema: { type: 'json' },
|
|
243
|
+
render: (_args, value) => [{ type: 'text', text: renderHealth(value) }],
|
|
244
|
+
},
|
|
245
|
+
async execute(_args, exec): Promise<JsonValue> {
|
|
246
|
+
const health = await checkHealth(resolved.baseURL, fetchImpl, ctx.subprocess, resolved.requestTimeoutMs, resolved.graceMs)
|
|
247
|
+
void exec
|
|
248
|
+
return health as unknown as JsonValue
|
|
249
|
+
},
|
|
250
|
+
}))
|
|
251
|
+
|
|
252
|
+
ctx.commands.register({
|
|
253
|
+
name: 'ollama',
|
|
254
|
+
description: 'One-shot status overview: local models, disk usage, health, and routing suggestions.',
|
|
255
|
+
async handler(): Promise<CommandResult> {
|
|
256
|
+
const health = await checkHealth(resolved.baseURL, fetchImpl, ctx.subprocess, resolved.requestTimeoutMs, resolved.graceMs)
|
|
257
|
+
const lines = ['Ollama status:']
|
|
258
|
+
lines.push(`- API: ${health.api.ok ? `ok${health.api.version !== undefined ? ` (v${health.api.version})` : ''}` : 'down'}`)
|
|
259
|
+
lines.push(`- process: ${health.process.present ? 'alive' : 'not detected'}`)
|
|
260
|
+
let models = [] as Array<{ name: string; size: number }>
|
|
261
|
+
try {
|
|
262
|
+
models = await listModels(resolved.baseURL, fetchImpl)
|
|
263
|
+
} catch {
|
|
264
|
+
// Model listing is best-effort in the overview; health already reported the failure.
|
|
265
|
+
}
|
|
266
|
+
const totalBytes = models.reduce((sum, model) => sum + model.size, 0)
|
|
267
|
+
lines.push(`- models: ${models.length} installed (${formatBytes(totalBytes)})`)
|
|
268
|
+
for (const model of models) lines.push(` - ${model.name} (${formatBytes(model.size)})`)
|
|
269
|
+
if (!health.api.ok && !health.process.present) {
|
|
270
|
+
lines.push('suggestion: start the Ollama server (e.g. `ollama serve`)')
|
|
271
|
+
} else if (resolved.route.length === 0) {
|
|
272
|
+
lines.push('suggestion: configure `route` rules to route requests to local models')
|
|
273
|
+
}
|
|
274
|
+
return { kind: 'success', text: lines.join('\n') }
|
|
275
|
+
},
|
|
276
|
+
})
|
|
277
|
+
|
|
278
|
+
logger.info(`ollama adapter registered at ${resolved.baseURL} (${resolved.models.length} mapping(s), ${resolved.route.length} route rule(s))`)
|
|
279
|
+
}
|
package/src/ollama.ts
ADDED
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Ollama HTTP API client (zero runtime dependencies — plain `fetch`). The
|
|
3
|
+
* harness `LlmAdapter` streams through `/api/chat`; discovery and management
|
|
4
|
+
* tools call `/api/tags`, `/api/show`, `/api/pull`, `/api/delete`, and
|
|
5
|
+
* `/api/version`. Every request carries the harness attribution headers and
|
|
6
|
+
* honors the caller's abort signal; non-2xx responses fail with a normalized
|
|
7
|
+
* `LlmError`. The fetch implementation is injectable for tests.
|
|
8
|
+
* @module dsh-local-ai/ollama
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import { attributionHeaders, LlmError } from '@deepseek-ai/dsh-llm'
|
|
12
|
+
import { sanitizeEndpoint } from './sanitize.ts'
|
|
13
|
+
|
|
14
|
+
/** A `fetch`-compatible function, injectable for tests. */
|
|
15
|
+
export type FetchLike = (input: string, init?: RequestInit) => Promise<Response>
|
|
16
|
+
|
|
17
|
+
/** One installed model as reported by `/api/tags`. */
|
|
18
|
+
export interface OllamaModel {
|
|
19
|
+
name: string
|
|
20
|
+
model: string
|
|
21
|
+
size: number
|
|
22
|
+
digest: string
|
|
23
|
+
modified_at?: string
|
|
24
|
+
details?: OllamaModelDetails
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/** Structured model details reported by `/api/show` and `/api/tags`. */
|
|
28
|
+
export interface OllamaModelDetails {
|
|
29
|
+
family?: string
|
|
30
|
+
parameter_size?: string
|
|
31
|
+
quantization_level?: string
|
|
32
|
+
format?: string
|
|
33
|
+
parent_model?: string
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/** The `/api/show` response body. */
|
|
37
|
+
export interface OllamaShowResult {
|
|
38
|
+
license?: string
|
|
39
|
+
modelfile?: string
|
|
40
|
+
parameters?: string
|
|
41
|
+
template?: string
|
|
42
|
+
details?: OllamaModelDetails
|
|
43
|
+
model_info?: Record<string, unknown>
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/** The `/api/version` response body. */
|
|
47
|
+
export interface OllamaVersionResult {
|
|
48
|
+
version: string
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/** The `/api/pull` final success status. */
|
|
52
|
+
export interface OllamaPullResult {
|
|
53
|
+
status: string
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/** Build an absolute API URL from a normalized base URL. */
|
|
57
|
+
export function endpointUrl(baseURL: string, path: string): string {
|
|
58
|
+
return `${baseURL}${path}`
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/** Map an HTTP status to a stable LlmError code. */
|
|
62
|
+
export function httpErrorCode(status: number): string {
|
|
63
|
+
if (status === 404) return 'NOT_FOUND'
|
|
64
|
+
if (status === 400) return 'INVALID_REQUEST'
|
|
65
|
+
if (status >= 500) return 'SERVER'
|
|
66
|
+
return `HTTP_${status}`
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/** Throw a normalized LlmError from a non-2xx response, using the body's `error`. */
|
|
70
|
+
async function throwHttpError(response: Response, context: string): Promise<never> {
|
|
71
|
+
let message = `Ollama API error (HTTP ${response.status}) from ${sanitizeEndpoint(context)}`
|
|
72
|
+
try {
|
|
73
|
+
const body = await response.json() as { error?: unknown }
|
|
74
|
+
if (typeof body.error === 'string' && body.error.length > 0) message = body.error
|
|
75
|
+
} catch {
|
|
76
|
+
// Only swallow error-body parsing: the HTTP status still identifies the failure.
|
|
77
|
+
}
|
|
78
|
+
throw new LlmError(message, httpErrorCode(response.status), { status: response.status })
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/** Send a GET request and parse the JSON response. */
|
|
82
|
+
export async function requestJson<T>(
|
|
83
|
+
baseURL: string,
|
|
84
|
+
path: string,
|
|
85
|
+
fetchImpl: FetchLike,
|
|
86
|
+
signal?: AbortSignal,
|
|
87
|
+
): Promise<T> {
|
|
88
|
+
const response = await fetchImpl(endpointUrl(baseURL, path), {
|
|
89
|
+
method: 'GET',
|
|
90
|
+
headers: { accept: 'application/json', ...attributionHeaders() },
|
|
91
|
+
...signal === undefined ? {} : { signal },
|
|
92
|
+
})
|
|
93
|
+
if (!response.ok) await throwHttpError(response, endpointUrl(baseURL, path))
|
|
94
|
+
return response.json() as Promise<T>
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
/** Send a POST request and parse the JSON response. */
|
|
98
|
+
export async function postJson<T>(
|
|
99
|
+
baseURL: string,
|
|
100
|
+
path: string,
|
|
101
|
+
body: unknown,
|
|
102
|
+
fetchImpl: FetchLike,
|
|
103
|
+
signal?: AbortSignal,
|
|
104
|
+
): Promise<T> {
|
|
105
|
+
const response = await fetchImpl(endpointUrl(baseURL, path), {
|
|
106
|
+
method: 'POST',
|
|
107
|
+
headers: { 'content-type': 'application/json', accept: 'application/json', ...attributionHeaders() },
|
|
108
|
+
body: JSON.stringify(body),
|
|
109
|
+
...signal === undefined ? {} : { signal },
|
|
110
|
+
})
|
|
111
|
+
if (!response.ok) await throwHttpError(response, endpointUrl(baseURL, path))
|
|
112
|
+
return response.json() as Promise<T>
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/** Send a DELETE request and parse the JSON response. */
|
|
116
|
+
export async function deleteJson<T>(
|
|
117
|
+
baseURL: string,
|
|
118
|
+
path: string,
|
|
119
|
+
body: unknown,
|
|
120
|
+
fetchImpl: FetchLike,
|
|
121
|
+
signal?: AbortSignal,
|
|
122
|
+
): Promise<T> {
|
|
123
|
+
const response = await fetchImpl(endpointUrl(baseURL, path), {
|
|
124
|
+
method: 'DELETE',
|
|
125
|
+
headers: { 'content-type': 'application/json', accept: 'application/json', ...attributionHeaders() },
|
|
126
|
+
body: JSON.stringify(body),
|
|
127
|
+
...signal === undefined ? {} : { signal },
|
|
128
|
+
})
|
|
129
|
+
if (!response.ok) await throwHttpError(response, endpointUrl(baseURL, path))
|
|
130
|
+
return response.json() as Promise<T>
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* Send a POST and return the raw `Response` after validating 2xx. Used by the
|
|
135
|
+
* streaming adapter, which owns body decoding and the idle watchdog.
|
|
136
|
+
*/
|
|
137
|
+
export async function postStream(
|
|
138
|
+
baseURL: string,
|
|
139
|
+
path: string,
|
|
140
|
+
body: unknown,
|
|
141
|
+
fetchImpl: FetchLike,
|
|
142
|
+
signal?: AbortSignal,
|
|
143
|
+
): Promise<Response> {
|
|
144
|
+
const response = await fetchImpl(endpointUrl(baseURL, path), {
|
|
145
|
+
method: 'POST',
|
|
146
|
+
headers: { 'content-type': 'application/json', accept: 'application/x-ndjson', ...attributionHeaders() },
|
|
147
|
+
body: JSON.stringify(body),
|
|
148
|
+
...signal === undefined ? {} : { signal },
|
|
149
|
+
})
|
|
150
|
+
if (!response.ok) await throwHttpError(response, endpointUrl(baseURL, path))
|
|
151
|
+
return response
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/** List installed models from `/api/tags`. */
|
|
155
|
+
export async function listModels(
|
|
156
|
+
baseURL: string,
|
|
157
|
+
fetchImpl: FetchLike,
|
|
158
|
+
signal?: AbortSignal,
|
|
159
|
+
): Promise<OllamaModel[]> {
|
|
160
|
+
const result = await requestJson<{ models?: OllamaModel[] }>(baseURL, '/api/tags', fetchImpl, signal)
|
|
161
|
+
return result.models ?? []
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
/** List currently-loaded (running) models from `/api/ps`. */
|
|
165
|
+
export async function listRunning(
|
|
166
|
+
baseURL: string,
|
|
167
|
+
fetchImpl: FetchLike,
|
|
168
|
+
signal?: AbortSignal,
|
|
169
|
+
): Promise<OllamaModel[]> {
|
|
170
|
+
const result = await requestJson<{ models?: OllamaModel[] }>(baseURL, '/api/ps', fetchImpl, signal)
|
|
171
|
+
return result.models ?? []
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/** Inspect one model via `/api/show`. */
|
|
175
|
+
export async function showModel(
|
|
176
|
+
baseURL: string,
|
|
177
|
+
name: string,
|
|
178
|
+
fetchImpl: FetchLike,
|
|
179
|
+
signal?: AbortSignal,
|
|
180
|
+
): Promise<OllamaShowResult> {
|
|
181
|
+
return postJson<OllamaShowResult>(baseURL, '/api/show', { name }, fetchImpl, signal)
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
/** Remove one model via `/api/delete`. */
|
|
185
|
+
export async function removeModel(
|
|
186
|
+
baseURL: string,
|
|
187
|
+
name: string,
|
|
188
|
+
fetchImpl: FetchLike,
|
|
189
|
+
signal?: AbortSignal,
|
|
190
|
+
): Promise<void> {
|
|
191
|
+
await deleteJson<{ status?: string }>(baseURL, '/api/delete', { name }, fetchImpl, signal)
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
/** Query the Ollama server version via `/api/version`. */
|
|
195
|
+
export async function apiVersion(
|
|
196
|
+
baseURL: string,
|
|
197
|
+
fetchImpl: FetchLike,
|
|
198
|
+
signal?: AbortSignal,
|
|
199
|
+
): Promise<string> {
|
|
200
|
+
const result = await requestJson<OllamaVersionResult>(baseURL, '/api/version', fetchImpl, signal)
|
|
201
|
+
return result.version
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
/**
|
|
205
|
+
* Pull a model via `/api/pull`, consuming the progress stream and returning the
|
|
206
|
+
* final status. An intermediate error status or a non-2xx response fails loud.
|
|
207
|
+
*/
|
|
208
|
+
export async function pullModel(
|
|
209
|
+
baseURL: string,
|
|
210
|
+
name: string,
|
|
211
|
+
fetchImpl: FetchLike,
|
|
212
|
+
signal?: AbortSignal,
|
|
213
|
+
): Promise<OllamaPullResult> {
|
|
214
|
+
const response = await postStream(baseURL, '/api/pull', { name, stream: true }, fetchImpl, signal)
|
|
215
|
+
if (!response.body) throw new LlmError('Ollama pull returned no response body', 'EMPTY_RESPONSE')
|
|
216
|
+
let last: OllamaPullResult = { status: 'success' }
|
|
217
|
+
for await (const line of readNdjsonLines(response.body)) {
|
|
218
|
+
if (line.length === 0) continue
|
|
219
|
+
const chunk = JSON.parse(line) as { status?: string; error?: string }
|
|
220
|
+
if (typeof chunk.error === 'string' && chunk.error.length > 0) {
|
|
221
|
+
throw new LlmError(chunk.error, 'PROVIDER')
|
|
222
|
+
}
|
|
223
|
+
if (typeof chunk.status === 'string') last = { status: chunk.status }
|
|
224
|
+
}
|
|
225
|
+
return last
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
/**
|
|
229
|
+
* Decode a `ReadableStream<Uint8Array>` into newline-delimited text lines.
|
|
230
|
+
* The final line is yielded even without a trailing newline; a missing body
|
|
231
|
+
* yields nothing.
|
|
232
|
+
* @param body - the response body stream.
|
|
233
|
+
* @returns text lines in delivery order.
|
|
234
|
+
*/
|
|
235
|
+
export async function* readNdjsonLines(body: ReadableStream<Uint8Array>): AsyncGenerator<string> {
|
|
236
|
+
const reader = body.getReader()
|
|
237
|
+
const decoder = new TextDecoder()
|
|
238
|
+
let buffer = ''
|
|
239
|
+
try {
|
|
240
|
+
while (true) {
|
|
241
|
+
const { done, value } = await reader.read()
|
|
242
|
+
if (done) break
|
|
243
|
+
buffer += decoder.decode(value, { stream: true })
|
|
244
|
+
let newline = buffer.indexOf('\n')
|
|
245
|
+
while (newline >= 0) {
|
|
246
|
+
const line = buffer.slice(0, newline).replace(/\r$/u, '')
|
|
247
|
+
buffer = buffer.slice(newline + 1)
|
|
248
|
+
newline = buffer.indexOf('\n')
|
|
249
|
+
yield line
|
|
250
|
+
}
|
|
251
|
+
}
|
|
252
|
+
} finally {
|
|
253
|
+
reader.releaseLock()
|
|
254
|
+
}
|
|
255
|
+
buffer += decoder.decode()
|
|
256
|
+
if (buffer.length > 0) yield buffer
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
/**
|
|
260
|
+
* Extract the context length from an `/api/show` result by scanning
|
|
261
|
+
* `model_info` for a `*.context_length` or bare `context_length` entry.
|
|
262
|
+
* @param show - the `/api/show` result.
|
|
263
|
+
* @returns the context length, or `undefined` when not reported.
|
|
264
|
+
*/
|
|
265
|
+
export function contextLengthOf(show: OllamaShowResult): number | undefined {
|
|
266
|
+
const info = show.model_info
|
|
267
|
+
if (info === undefined) return undefined
|
|
268
|
+
for (const [key, value] of Object.entries(info)) {
|
|
269
|
+
if (key === 'context_length' || key.endsWith('.context_length')) {
|
|
270
|
+
if (typeof value === 'number' && Number.isInteger(value) && value > 0) return value
|
|
271
|
+
}
|
|
272
|
+
}
|
|
273
|
+
return undefined
|
|
274
|
+
}
|
package/src/route.ts
ADDED
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Local-model routing decision and the streaming fallback. The pure decision
|
|
3
|
+
* matches configured rules (task type via `purpose`, case-insensitive
|
|
4
|
+
* keywords, or a blanket `always`) against a request, in list order — first
|
|
5
|
+
* match wins. The streaming helper routes a matched request to the local
|
|
6
|
+
* Ollama adapter and, when the local route fails BEFORE producing any visible
|
|
7
|
+
* content, falls back to the cloud (`next()`) so a down Ollama never bricks a
|
|
8
|
+
* conversation. Once local content has started, it is streamed through — a
|
|
9
|
+
* mid-stream failure cannot be retracted.
|
|
10
|
+
* @module dsh-local-ai/route
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { isTokenDelta } from '@deepseek-ai/dsh-llm'
|
|
14
|
+
import type { GenerateOptions, StreamChunk } from '@deepseek-ai/dsh-llm'
|
|
15
|
+
import { OLLAMA_PROVIDER } from './adapter.ts'
|
|
16
|
+
import type { ResolvedConfig, ResolvedRouteRule } from './config.ts'
|
|
17
|
+
|
|
18
|
+
/** The chosen local model for one matched request. */
|
|
19
|
+
export interface RouteDecision {
|
|
20
|
+
/** Harness-visible local model name. */
|
|
21
|
+
readonly model: string
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Concatenate the request's model-visible text (system prompt + every text
|
|
26
|
+
* block) for keyword matching.
|
|
27
|
+
* @param options - the request.
|
|
28
|
+
* @returns the joined text.
|
|
29
|
+
*/
|
|
30
|
+
export function requestText(options: GenerateOptions): string {
|
|
31
|
+
const parts: string[] = []
|
|
32
|
+
if (options.system !== undefined) parts.push(options.system)
|
|
33
|
+
for (const message of options.messages) {
|
|
34
|
+
for (const block of message.content) {
|
|
35
|
+
if (block.type === 'text') parts.push(block.text)
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
return parts.join('\n')
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** Whether a keyword appears case-insensitively in the text. */
|
|
42
|
+
export function matchesKeyword(text: string, keyword: string): boolean {
|
|
43
|
+
return text.toLowerCase().includes(keyword.toLowerCase())
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/** Whether one resolved rule matches a request. */
|
|
47
|
+
export function ruleMatches(rule: ResolvedRouteRule, options: GenerateOptions): boolean {
|
|
48
|
+
if (rule.always) return true
|
|
49
|
+
if (rule.purpose !== undefined && options.purpose === rule.purpose) return true
|
|
50
|
+
if (rule.keywords.length > 0) {
|
|
51
|
+
const text = requestText(options)
|
|
52
|
+
return rule.keywords.some(keyword => matchesKeyword(text, keyword))
|
|
53
|
+
}
|
|
54
|
+
return false
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Decide whether a request should route to a local model. A request already
|
|
59
|
+
* addressed to the `ollama` provider (explicit selection or a prior re-route)
|
|
60
|
+
* never re-routes.
|
|
61
|
+
* @param options - the request.
|
|
62
|
+
* @param resolved - the resolved config.
|
|
63
|
+
* @returns the local model to use, or `undefined` to stay on the cloud route.
|
|
64
|
+
*/
|
|
65
|
+
export function decideRoute(options: GenerateOptions, resolved: ResolvedConfig): RouteDecision | undefined {
|
|
66
|
+
if (options.provider === OLLAMA_PROVIDER) return undefined
|
|
67
|
+
for (const rule of resolved.route) {
|
|
68
|
+
if (ruleMatches(rule, options)) return { model: rule.model }
|
|
69
|
+
}
|
|
70
|
+
return undefined
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/**
|
|
74
|
+
* Stream a locally-routed request with automatic cloud fallback. The local
|
|
75
|
+
* stream is produced through `streamLocal` (the full harness stream, so the
|
|
76
|
+
* local route keeps retry and failure normalization). If the local route
|
|
77
|
+
* finishes with an error or aborts before any token delta, `next()` (the
|
|
78
|
+
* cloud) is streamed instead; otherwise the local stream is forwarded.
|
|
79
|
+
* @param streamLocal - produces the local stream for a re-routed request.
|
|
80
|
+
* @param options - the original request.
|
|
81
|
+
* @param decision - the local model to route to.
|
|
82
|
+
* @param next - the cloud stream (the waterfall's `next()`).
|
|
83
|
+
* @returns the effective chunk stream.
|
|
84
|
+
*/
|
|
85
|
+
export async function* routeLocal(
|
|
86
|
+
streamLocal: (options: GenerateOptions) => AsyncIterable<StreamChunk>,
|
|
87
|
+
options: GenerateOptions,
|
|
88
|
+
decision: RouteDecision,
|
|
89
|
+
next: () => AsyncIterable<StreamChunk>,
|
|
90
|
+
): AsyncGenerator<StreamChunk> {
|
|
91
|
+
const localOptions: GenerateOptions = { ...options, provider: OLLAMA_PROVIDER, model: decision.model }
|
|
92
|
+
const upstream = streamLocal(localOptions)
|
|
93
|
+
let producedContent = false
|
|
94
|
+
const pending: StreamChunk[] = []
|
|
95
|
+
try {
|
|
96
|
+
for await (const chunk of upstream) {
|
|
97
|
+
if (producedContent) {
|
|
98
|
+
yield chunk
|
|
99
|
+
continue
|
|
100
|
+
}
|
|
101
|
+
if (chunk.type === 'finish') {
|
|
102
|
+
if (chunk.reason.kind === 'error' || chunk.reason.kind === 'aborted') {
|
|
103
|
+
// Local failed before producing content — fall back to the cloud.
|
|
104
|
+
yield* next()
|
|
105
|
+
return
|
|
106
|
+
}
|
|
107
|
+
yield chunk
|
|
108
|
+
return
|
|
109
|
+
}
|
|
110
|
+
pending.push(chunk)
|
|
111
|
+
if (isTokenDelta(chunk)) {
|
|
112
|
+
producedContent = true
|
|
113
|
+
for (const buffered of pending) yield buffered
|
|
114
|
+
pending.length = 0
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
// Stream ended without a finish chunk — flush whatever was buffered.
|
|
118
|
+
for (const buffered of pending) yield buffered
|
|
119
|
+
} catch (error) {
|
|
120
|
+
if (!producedContent) {
|
|
121
|
+
yield* next()
|
|
122
|
+
return
|
|
123
|
+
}
|
|
124
|
+
for (const buffered of pending) yield buffered
|
|
125
|
+
throw error
|
|
126
|
+
}
|
|
127
|
+
}
|