dsh-local-ai 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/LICENSE +201 -0
- package/README.es.md +211 -0
- package/README.hi.md +211 -0
- package/README.md +211 -0
- package/README.pt.md +211 -0
- package/README.zh.md +211 -0
- package/THIRD_PARTY_NOTICES.md +20 -0
- package/cordis.patch.yml +44 -0
- package/lib/index.js +1334 -0
- package/lib/types/adapter.d.ts +39 -0
- package/lib/types/adapter.d.ts.map +1 -0
- package/lib/types/adapter.js +190 -0
- package/lib/types/adapter.js.map +1 -0
- package/lib/types/config.d.ts +109 -0
- package/lib/types/config.d.ts.map +1 -0
- package/lib/types/config.js +161 -0
- package/lib/types/config.js.map +1 -0
- package/lib/types/health.d.ts +64 -0
- package/lib/types/health.d.ts.map +1 -0
- package/lib/types/health.js +92 -0
- package/lib/types/health.js.map +1 -0
- package/lib/types/index.d.ts +43 -0
- package/lib/types/index.d.ts.map +1 -0
- package/lib/types/index.js +232 -0
- package/lib/types/index.js.map +1 -0
- package/lib/types/ollama.d.ts +91 -0
- package/lib/types/ollama.d.ts.map +1 -0
- package/lib/types/ollama.js +184 -0
- package/lib/types/ollama.js.map +1 -0
- package/lib/types/route.d.ts +52 -0
- package/lib/types/route.d.ts.map +1 -0
- package/lib/types/route.js +119 -0
- package/lib/types/route.js.map +1 -0
- package/lib/types/sanitize.d.ts +58 -0
- package/lib/types/sanitize.d.ts.map +1 -0
- package/lib/types/sanitize.js +110 -0
- package/lib/types/sanitize.js.map +1 -0
- package/lib/types/serialize.d.ts +65 -0
- package/lib/types/serialize.d.ts.map +1 -0
- package/lib/types/serialize.js +149 -0
- package/lib/types/serialize.js.map +1 -0
- package/lib/types/translate.d.ts +57 -0
- package/lib/types/translate.d.ts.map +1 -0
- package/lib/types/translate.js +169 -0
- package/lib/types/translate.js.map +1 -0
- package/lib/types/version.d.ts +6 -0
- package/lib/types/version.d.ts.map +1 -0
- package/lib/types/version.js +6 -0
- package/lib/types/version.js.map +1 -0
- package/package.json +141 -0
- package/src/adapter.ts +158 -0
- package/src/config.ts +243 -0
- package/src/health.ts +133 -0
- package/src/index.ts +279 -0
- package/src/ollama.ts +274 -0
- package/src/route.ts +127 -0
- package/src/sanitize.ts +114 -0
- package/src/serialize.ts +178 -0
- package/src/translate.ts +205 -0
- package/src/version.ts +5 -0
package/src/adapter.ts
ADDED
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `OllamaAdapter`: the harness `LlmAdapter` over Ollama's native `/api/chat`
|
|
3
|
+
* streaming endpoint. Transport-only: the registering plugin owns config
|
|
4
|
+
* resolution (one `() => ResolvedConfig` thunk re-read per operation), so a
|
|
5
|
+
* changed base URL, model mapping, or timeout reaches the next request without
|
|
6
|
+
* re-registration, while an in-flight stream keeps the facts it started with.
|
|
7
|
+
* The adapter is text-only (`inputModalities: ['text']`); tool calls and tool
|
|
8
|
+
* results are translated by {@link serializeRequest}.
|
|
9
|
+
* @module dsh-local-ai/adapter
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import { LlmAdapter, LlmError } from '@deepseek-ai/dsh-llm'
|
|
13
|
+
import type { GenerateOptions, LlmModelInfo, LlmProviderInfo, LlmResolvedModelInfo, StreamChunk } from '@deepseek-ai/dsh-llm'
|
|
14
|
+
import { idleWatchdog, timeoutOf } from '@deepseek-ai/dsh-timeout'
|
|
15
|
+
import { sanitizeEndpoint } from './sanitize.ts'
|
|
16
|
+
import { listModels as listOllamaModels, postStream, readNdjsonLines } from './ollama.ts'
|
|
17
|
+
import type { FetchLike } from './ollama.ts'
|
|
18
|
+
import { serializeRequest } from './serialize.ts'
|
|
19
|
+
import { translate } from './translate.ts'
|
|
20
|
+
import type { ResolvedConfig } from './config.ts'
|
|
21
|
+
|
|
22
|
+
/** Idle-timeout abort code stamped onto a stalled stream's timeout reason. */
|
|
23
|
+
const STREAM_IDLE_TIMEOUT_CODE = 'LLM_STREAM_IDLE_TIMEOUT'
|
|
24
|
+
|
|
25
|
+
/** Constructor options for {@link OllamaAdapter}. */
|
|
26
|
+
export interface OllamaAdapterOptions {
|
|
27
|
+
/** Current validated config; called once per operation. */
|
|
28
|
+
config: () => ResolvedConfig
|
|
29
|
+
/** Fetch implementation, injectable for tests; defaults to `globalThis.fetch`. */
|
|
30
|
+
fetchImpl?: FetchLike
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/** The single provider route this adapter owns. */
|
|
34
|
+
export const OLLAMA_PROVIDER = 'ollama'
|
|
35
|
+
|
|
36
|
+
/** Reverse a model mapping: Ollama model id → harness-visible name. */
|
|
37
|
+
function harnessNameOf(resolved: ResolvedConfig, ollamaName: string): string {
|
|
38
|
+
const mapping = resolved.models.find(entry => entry.model === ollamaName)
|
|
39
|
+
return mapping?.name ?? ollamaName
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
/** Advertise one configured or discovered local model as text-only. */
|
|
43
|
+
function modelInfo(provider: string, id: string, name: string): LlmModelInfo {
|
|
44
|
+
return { provider, id, name, inputModalities: ['text'] }
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* The Ollama provider adapter. One instance serves every harness-visible local
|
|
49
|
+
* model name; the harness model name maps to the wire model id through the
|
|
50
|
+
* configured model mapping (identity when unmapped).
|
|
51
|
+
*/
|
|
52
|
+
export class OllamaAdapter extends LlmAdapter {
|
|
53
|
+
constructor(private readonly options: OllamaAdapterOptions) {
|
|
54
|
+
super()
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
private fetchImpl(): FetchLike {
|
|
58
|
+
return this.options.fetchImpl ?? ((input: string, init?: RequestInit) => globalThis.fetch(input, init))
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
override providerInfo(provider: string): LlmProviderInfo {
|
|
62
|
+
return { id: provider, name: 'Ollama (local)' }
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
override listModels(provider: string): Promise<readonly LlmModelInfo[]> {
|
|
66
|
+
const resolved = this.options.config()
|
|
67
|
+
return listOllamaModels(resolved.baseURL, this.fetchImpl())
|
|
68
|
+
.then(models => models.map(model => modelInfo(provider, harnessNameOf(resolved, model.name), harnessNameOf(resolved, model.name))))
|
|
69
|
+
.catch(() => resolved.models.map(entry => modelInfo(provider, entry.name, entry.name)))
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
override resolveModel(
|
|
73
|
+
provider: string,
|
|
74
|
+
model: string,
|
|
75
|
+
_signal?: AbortSignal,
|
|
76
|
+
): Promise<LlmResolvedModelInfo> {
|
|
77
|
+
const resolved = this.options.config()
|
|
78
|
+
const mapping = resolved.models.find(entry => entry.name === model)
|
|
79
|
+
return Promise.resolve({
|
|
80
|
+
provider,
|
|
81
|
+
id: model,
|
|
82
|
+
name: model,
|
|
83
|
+
inputModalities: ['text'],
|
|
84
|
+
context: { contextWindow: mapping?.contextWindow ?? resolved.defaultContextWindow },
|
|
85
|
+
defaultMaxTokens: mapping?.maxTokens ?? resolved.maxTokens,
|
|
86
|
+
})
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
async * stream(options: GenerateOptions): AsyncIterable<StreamChunk> {
|
|
90
|
+
// One resolution per stream call: connection facts freeze here and hold for
|
|
91
|
+
// this whole request, so an in-flight stream never observes a config change
|
|
92
|
+
// and the next call re-resolves.
|
|
93
|
+
const resolved = this.options.config()
|
|
94
|
+
const consumer = new AbortController()
|
|
95
|
+
const upstream = options.signal === undefined
|
|
96
|
+
? consumer.signal
|
|
97
|
+
: AbortSignal.any([options.signal, consumer.signal])
|
|
98
|
+
using watchdog = idleWatchdog(upstream, resolved.requestTimeoutMs, STREAM_IDLE_TIMEOUT_CODE)
|
|
99
|
+
const iterator = this.request(options, resolved, watchdog.signal)[Symbol.asyncIterator]()
|
|
100
|
+
let exhausted = false
|
|
101
|
+
try {
|
|
102
|
+
while (true) {
|
|
103
|
+
const result = await watchdog.next(iterator)
|
|
104
|
+
if (result.done) {
|
|
105
|
+
exhausted = true
|
|
106
|
+
return
|
|
107
|
+
}
|
|
108
|
+
yield result.value
|
|
109
|
+
}
|
|
110
|
+
} catch (error: unknown) {
|
|
111
|
+
if (timeoutOf(watchdog.signal, STREAM_IDLE_TIMEOUT_CODE) !== undefined) {
|
|
112
|
+
throw new LlmError(
|
|
113
|
+
`Ollama stream idle timeout after ${resolved.requestTimeoutMs}ms`,
|
|
114
|
+
'TIMEOUT',
|
|
115
|
+
{ cause: error },
|
|
116
|
+
)
|
|
117
|
+
}
|
|
118
|
+
if (options.signal?.aborted) {
|
|
119
|
+
throw new LlmError('Ollama request aborted by caller', 'ABORTED', { cause: error })
|
|
120
|
+
}
|
|
121
|
+
if (error instanceof LlmError) throw error
|
|
122
|
+
throw new LlmError(`Ollama API stream from ${sanitizeEndpoint(resolved.baseURL)} failed`, 'TRANSPORT', { cause: error })
|
|
123
|
+
} finally {
|
|
124
|
+
consumer.abort('Ollama stream consumer stopped')
|
|
125
|
+
if (!exhausted && iterator.return !== undefined) {
|
|
126
|
+
try {
|
|
127
|
+
await iterator.return()
|
|
128
|
+
} catch (_abortedTransportTeardown) {
|
|
129
|
+
// The consumer controller already owns termination; a return-time abort cannot add a second outcome.
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
private async * request(
|
|
136
|
+
options: GenerateOptions,
|
|
137
|
+
resolved: ResolvedConfig,
|
|
138
|
+
signal: AbortSignal,
|
|
139
|
+
): AsyncIterable<StreamChunk> {
|
|
140
|
+
const body = serializeRequest(options, resolved)
|
|
141
|
+
let response: Response
|
|
142
|
+
try {
|
|
143
|
+
response = await postStream(resolved.baseURL, '/api/chat', body, this.fetchImpl(), signal)
|
|
144
|
+
} catch (error: unknown) {
|
|
145
|
+
// The outer stream distinguishes caller cancellation and watchdog expiry.
|
|
146
|
+
if (signal.aborted) throw error
|
|
147
|
+
throw new LlmError(
|
|
148
|
+
`Ollama API request to ${sanitizeEndpoint(resolved.baseURL)} failed`,
|
|
149
|
+
'TRANSPORT',
|
|
150
|
+
{ cause: error },
|
|
151
|
+
)
|
|
152
|
+
}
|
|
153
|
+
if (!response.body) {
|
|
154
|
+
throw new LlmError('Ollama API returned no response body', 'EMPTY_RESPONSE')
|
|
155
|
+
}
|
|
156
|
+
yield* translate(readNdjsonLines(response.body))
|
|
157
|
+
}
|
|
158
|
+
}
|
package/src/config.ts
ADDED
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Config schema and resolution for `dsh-local-ai`. Every tunable is a
|
|
3
|
+
* validated {@link Config} field changeable from cordis.yml; the resolution
|
|
4
|
+
* step validates URLs, numeric bounds, and route/model entries so
|
|
5
|
+
* misconfiguration fails loud at mount — never silently skips a rule or
|
|
6
|
+
* half-configures the adapter. The plugin is inert until at least one route
|
|
7
|
+
* rule exists or a caller selects the `ollama` provider explicitly: routing
|
|
8
|
+
* every request to a local model requires an explicit opt-in (privacy and
|
|
9
|
+
* cost default: no automatic re-routing).
|
|
10
|
+
* @module dsh-local-ai/config
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import z from '@deepseek-ai/schemastery'
|
|
14
|
+
|
|
15
|
+
/** One harness-visible model mapping onto an Ollama model id. */
|
|
16
|
+
export interface ModelMapping {
|
|
17
|
+
/** Harness-visible model name (what `GenerateOptions.model` uses). */
|
|
18
|
+
name: string
|
|
19
|
+
/** Ollama model id; defaults to {@link name}. */
|
|
20
|
+
model?: string
|
|
21
|
+
/** Combined request/response context capacity in tokens. */
|
|
22
|
+
contextWindow?: number
|
|
23
|
+
/** Per-request output cap in tokens. */
|
|
24
|
+
maxTokens?: number
|
|
25
|
+
/** Default sampling temperature for this model (0..2). */
|
|
26
|
+
temperature?: number
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/** One local-model routing rule, matched in list order (first match wins). */
|
|
30
|
+
export interface RouteRule {
|
|
31
|
+
/** Target harness-visible model name, resolved through the model mapping. */
|
|
32
|
+
model: string
|
|
33
|
+
/** Route when the request purpose matches this task type. */
|
|
34
|
+
purpose?: 'compaction' | 'session-title'
|
|
35
|
+
/** Route when any keyword appears (case-insensitive) in the request text. */
|
|
36
|
+
keywords?: string[]
|
|
37
|
+
/** Route every eligible request to this local model (offline-first). */
|
|
38
|
+
always?: boolean
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** Raw plugin config — every field optional; {@link resolveConfig} supplies the defaults. */
|
|
42
|
+
export interface Config {
|
|
43
|
+
/** Ollama HTTP API base URL; `/api/*` paths are appended. */
|
|
44
|
+
baseURL?: string
|
|
45
|
+
/** Per-request HTTP timeout in milliseconds. */
|
|
46
|
+
requestTimeoutMs?: number
|
|
47
|
+
/** Subprocess terminate grace in milliseconds (health-check CLI). */
|
|
48
|
+
graceMs?: number
|
|
49
|
+
/** Context capacity used when a model has no exact value. */
|
|
50
|
+
defaultContextWindow?: number
|
|
51
|
+
/** Per-request output cap used when a model has no exact value. */
|
|
52
|
+
maxTokens?: number
|
|
53
|
+
/** Default sampling temperature; omitted leaves the provider default. */
|
|
54
|
+
temperature?: number
|
|
55
|
+
/** Harness-visible → Ollama model mappings. */
|
|
56
|
+
models?: ModelMapping[]
|
|
57
|
+
/** Local-model routing rules (offline-first / long-text / privacy tasks). */
|
|
58
|
+
route?: RouteRule[]
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/** Fully resolved model mapping. */
|
|
62
|
+
export interface ResolvedModelMapping {
|
|
63
|
+
readonly name: string
|
|
64
|
+
readonly model: string
|
|
65
|
+
readonly contextWindow?: number
|
|
66
|
+
readonly maxTokens?: number
|
|
67
|
+
readonly temperature?: number
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/** Fully resolved routing rule. */
|
|
71
|
+
export interface ResolvedRouteRule {
|
|
72
|
+
readonly model: string
|
|
73
|
+
readonly purpose?: 'compaction' | 'session-title'
|
|
74
|
+
readonly keywords: readonly string[]
|
|
75
|
+
readonly always: boolean
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/** The complete resolved config handed to the runtime. */
|
|
79
|
+
export interface ResolvedConfig {
|
|
80
|
+
readonly baseURL: string
|
|
81
|
+
readonly requestTimeoutMs: number
|
|
82
|
+
readonly graceMs: number
|
|
83
|
+
readonly defaultContextWindow: number
|
|
84
|
+
readonly maxTokens: number
|
|
85
|
+
readonly temperature?: number
|
|
86
|
+
readonly models: readonly ResolvedModelMapping[]
|
|
87
|
+
readonly route: readonly ResolvedRouteRule[]
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/** Schemastery schema: the loader validates and fills defaults before `apply`. */
|
|
91
|
+
export const Config: z<Config> = z.object({
|
|
92
|
+
baseURL: z.string().default('http://127.0.0.1:11434'),
|
|
93
|
+
requestTimeoutMs: z.number().default(30_000),
|
|
94
|
+
graceMs: z.number().default(15_000),
|
|
95
|
+
defaultContextWindow: z.number().default(8192),
|
|
96
|
+
maxTokens: z.number().default(4096),
|
|
97
|
+
temperature: z.number(),
|
|
98
|
+
models: z.array(z.object({
|
|
99
|
+
name: z.string().required(),
|
|
100
|
+
model: z.string(),
|
|
101
|
+
contextWindow: z.number(),
|
|
102
|
+
maxTokens: z.number(),
|
|
103
|
+
temperature: z.number(),
|
|
104
|
+
})).default([]),
|
|
105
|
+
route: z.array(z.object({
|
|
106
|
+
model: z.string().required(),
|
|
107
|
+
purpose: z.union(['compaction', 'session-title'] as const),
|
|
108
|
+
keywords: z.array(z.string()).default([]),
|
|
109
|
+
always: z.boolean().default(false),
|
|
110
|
+
})).default([]),
|
|
111
|
+
})
|
|
112
|
+
|
|
113
|
+
/** Throw unless `value` is a positive safe integer. */
|
|
114
|
+
function assertPositiveInt(name: string, value: number): void {
|
|
115
|
+
if (!Number.isSafeInteger(value) || value <= 0) {
|
|
116
|
+
throw new TypeError(`${name} must be a positive safe integer, got ${String(value)}`)
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/** Throw unless `value` is a finite number in `[min, max]`. */
|
|
121
|
+
function assertFiniteRange(name: string, value: number, min: number, max: number): void {
|
|
122
|
+
if (typeof value !== 'number' || !Number.isFinite(value) || value < min || value > max) {
|
|
123
|
+
throw new TypeError(`${name} must be a finite number in [${min}, ${max}], got ${String(value)}`)
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* Validate an http(s) URL string and normalize it to a clean base (no query,
|
|
129
|
+
* no fragment, no trailing slash).
|
|
130
|
+
* @param name - config key, for the error message.
|
|
131
|
+
* @param value - raw URL value.
|
|
132
|
+
* @returns the normalized base URL.
|
|
133
|
+
*/
|
|
134
|
+
export function normalizeBaseUrl(name: string, value: string): string {
|
|
135
|
+
let parsed: URL
|
|
136
|
+
try {
|
|
137
|
+
parsed = new URL(value)
|
|
138
|
+
} catch (error) {
|
|
139
|
+
throw new TypeError(`${name} must be a valid URL, got ${JSON.stringify(value)} (${error instanceof Error ? error.message : 'invalid URL'})`)
|
|
140
|
+
}
|
|
141
|
+
if (parsed.protocol !== 'http:' && parsed.protocol !== 'https:') {
|
|
142
|
+
throw new TypeError(`${name} must use http(s), got ${JSON.stringify(parsed.protocol)}`)
|
|
143
|
+
}
|
|
144
|
+
parsed.search = ''
|
|
145
|
+
parsed.hash = ''
|
|
146
|
+
return parsed.href.replace(/\/+$/u, '')
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Validate raw values and fill explicit defaults. Invalid URLs, numeric
|
|
151
|
+
* bounds, duplicate model names, or empty route/model names throw here —
|
|
152
|
+
* misconfiguration fails loud at mount even when the plugin is mounted
|
|
153
|
+
* without the Schemastery loader.
|
|
154
|
+
* @param config - raw (possibly partial) plugin config.
|
|
155
|
+
* @returns the fully resolved config.
|
|
156
|
+
*/
|
|
157
|
+
export function resolveConfig(config: Config = {}): ResolvedConfig {
|
|
158
|
+
const baseURL = normalizeBaseUrl('baseURL', config.baseURL ?? 'http://127.0.0.1:11434')
|
|
159
|
+
|
|
160
|
+
const requestTimeoutMs = config.requestTimeoutMs ?? 30_000
|
|
161
|
+
assertPositiveInt('requestTimeoutMs', requestTimeoutMs)
|
|
162
|
+
|
|
163
|
+
const graceMs = config.graceMs ?? 15_000
|
|
164
|
+
assertPositiveInt('graceMs', graceMs)
|
|
165
|
+
|
|
166
|
+
const defaultContextWindow = config.defaultContextWindow ?? 8192
|
|
167
|
+
assertPositiveInt('defaultContextWindow', defaultContextWindow)
|
|
168
|
+
|
|
169
|
+
const maxTokens = config.maxTokens ?? 4096
|
|
170
|
+
assertPositiveInt('maxTokens', maxTokens)
|
|
171
|
+
|
|
172
|
+
const temperature = config.temperature
|
|
173
|
+
if (temperature !== undefined) assertFiniteRange('temperature', temperature, 0, 2)
|
|
174
|
+
|
|
175
|
+
const seenNames = new Set<string>()
|
|
176
|
+
const models = (config.models ?? []).map((mapping, index) => {
|
|
177
|
+
if (typeof mapping.name !== 'string' || mapping.name.trim().length === 0) {
|
|
178
|
+
throw new TypeError(`models[${index}].name must be a non-empty string`)
|
|
179
|
+
}
|
|
180
|
+
const name = mapping.name.trim()
|
|
181
|
+
if (seenNames.has(name)) throw new TypeError(`models[${index}]: duplicate model name ${JSON.stringify(name)}`)
|
|
182
|
+
seenNames.add(name)
|
|
183
|
+
const model = (mapping.model ?? name).trim()
|
|
184
|
+
if (model.length === 0) throw new TypeError(`models[${index}].model must be a non-empty string`)
|
|
185
|
+
const contextWindow = mapping.contextWindow
|
|
186
|
+
if (contextWindow !== undefined) assertPositiveInt(`models[${index}].contextWindow`, contextWindow)
|
|
187
|
+
const modelMaxTokens = mapping.maxTokens
|
|
188
|
+
if (modelMaxTokens !== undefined) assertPositiveInt(`models[${index}].maxTokens`, modelMaxTokens)
|
|
189
|
+
const modelTemperature = mapping.temperature
|
|
190
|
+
if (modelTemperature !== undefined) assertFiniteRange(`models[${index}].temperature`, modelTemperature, 0, 2)
|
|
191
|
+
return {
|
|
192
|
+
name,
|
|
193
|
+
model,
|
|
194
|
+
...contextWindow === undefined ? {} : { contextWindow },
|
|
195
|
+
...modelMaxTokens === undefined ? {} : { maxTokens: modelMaxTokens },
|
|
196
|
+
...modelTemperature === undefined ? {} : { temperature: modelTemperature },
|
|
197
|
+
}
|
|
198
|
+
})
|
|
199
|
+
|
|
200
|
+
const route = (config.route ?? []).map((rule, index) => {
|
|
201
|
+
if (typeof rule.model !== 'string' || rule.model.trim().length === 0) {
|
|
202
|
+
throw new TypeError(`route[${index}].model must be a non-empty string`)
|
|
203
|
+
}
|
|
204
|
+
const keywords = (rule.keywords ?? []).map((keyword, keywordIndex) => {
|
|
205
|
+
if (typeof keyword !== 'string' || keyword.trim().length === 0) {
|
|
206
|
+
throw new TypeError(`route[${index}].keywords[${keywordIndex}] must be a non-empty string`)
|
|
207
|
+
}
|
|
208
|
+
return keyword
|
|
209
|
+
})
|
|
210
|
+
if (rule.always !== true && rule.purpose === undefined && keywords.length === 0) {
|
|
211
|
+
throw new TypeError(`route[${index}] must declare a purpose, at least one keyword, or always: true`)
|
|
212
|
+
}
|
|
213
|
+
return {
|
|
214
|
+
model: rule.model.trim(),
|
|
215
|
+
...rule.purpose === undefined ? {} : { purpose: rule.purpose },
|
|
216
|
+
keywords,
|
|
217
|
+
always: rule.always ?? false,
|
|
218
|
+
}
|
|
219
|
+
})
|
|
220
|
+
|
|
221
|
+
return {
|
|
222
|
+
baseURL,
|
|
223
|
+
requestTimeoutMs,
|
|
224
|
+
graceMs,
|
|
225
|
+
defaultContextWindow,
|
|
226
|
+
maxTokens,
|
|
227
|
+
...temperature === undefined ? {} : { temperature },
|
|
228
|
+
models,
|
|
229
|
+
route,
|
|
230
|
+
}
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
/**
|
|
234
|
+
* Resolve one harness-visible model name to its Ollama model id, or return
|
|
235
|
+
* `undefined` when no mapping matches (the identity mapping applies).
|
|
236
|
+
* @param resolved - resolved config.
|
|
237
|
+
* @param name - harness-visible model name.
|
|
238
|
+
* @returns the mapped Ollama model id, or `undefined` when unmapped.
|
|
239
|
+
*/
|
|
240
|
+
export function ollamaModelOf(resolved: ResolvedConfig, name: string): string | undefined {
|
|
241
|
+
const mapping = resolved.models.find(entry => entry.name === name)
|
|
242
|
+
return mapping?.model
|
|
243
|
+
}
|
package/src/health.ts
ADDED
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Health checks for the local Ollama server: API responsiveness over HTTP and
|
|
3
|
+
* process liveness through the real subprocess seam (the `ollama` CLI's own
|
|
4
|
+
* channel). Both signals are independent — a server that answers HTTP but has
|
|
5
|
+
* no reachable CLI, or vice versa, is reported as two facts, never conflated.
|
|
6
|
+
* @module dsh-local-ai/health
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import type { SubprocessRuntime } from '@deepseek-ai/dsh-subprocess'
|
|
10
|
+
import { apiVersion } from './ollama.ts'
|
|
11
|
+
import type { FetchLike } from './ollama.ts'
|
|
12
|
+
import { sanitizeText } from './sanitize.ts'
|
|
13
|
+
import { tmpdir } from 'node:os'
|
|
14
|
+
|
|
15
|
+
/** The HTTP responsiveness result. */
|
|
16
|
+
export interface ApiHealth {
|
|
17
|
+
readonly ok: boolean
|
|
18
|
+
readonly version?: string
|
|
19
|
+
readonly error?: string
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/** The process liveness result. */
|
|
23
|
+
export interface ProcessHealth {
|
|
24
|
+
readonly present: boolean
|
|
25
|
+
readonly error?: string
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/** The combined health result. */
|
|
29
|
+
export interface HealthResult {
|
|
30
|
+
readonly api: ApiHealth
|
|
31
|
+
readonly process: ProcessHealth
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/** The subprocess probe used to test process liveness. */
|
|
35
|
+
export interface ProcessProbe {
|
|
36
|
+
readonly command: string
|
|
37
|
+
readonly args: readonly string[]
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/** The default probe: `ollama list` reaches the server through the CLI's own channel. */
|
|
41
|
+
export const OLLAMA_PROCESS_PROBE: ProcessProbe = { command: 'ollama', args: ['list'] }
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Check whether the Ollama HTTP API responds to `/api/version` within a
|
|
45
|
+
* deadline. A timeout, transport failure, or non-2xx response is `ok: false`
|
|
46
|
+
* with a sanitized error.
|
|
47
|
+
* @param baseURL - the Ollama base URL.
|
|
48
|
+
* @param fetchImpl - the fetch implementation.
|
|
49
|
+
* @param timeoutMs - deadline in milliseconds.
|
|
50
|
+
* @returns the API health result.
|
|
51
|
+
*/
|
|
52
|
+
export async function checkApiHealth(
|
|
53
|
+
baseURL: string,
|
|
54
|
+
fetchImpl: FetchLike,
|
|
55
|
+
timeoutMs: number,
|
|
56
|
+
): Promise<ApiHealth> {
|
|
57
|
+
const controller = new AbortController()
|
|
58
|
+
const timer = setTimeout(() => controller.abort(new Error('timeout')), timeoutMs)
|
|
59
|
+
try {
|
|
60
|
+
const version = await apiVersion(baseURL, fetchImpl, controller.signal)
|
|
61
|
+
return { ok: true, version: sanitizeText(version, 64) }
|
|
62
|
+
} catch (error: unknown) {
|
|
63
|
+
return { ok: false, error: sanitizeText(error instanceof Error ? error.message : String(error), 500) }
|
|
64
|
+
} finally {
|
|
65
|
+
clearTimeout(timer)
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* Check process liveness through the subprocess seam by spawning the probe
|
|
71
|
+
* command in collect mode. Exit code 0 means the CLI (and, for the default
|
|
72
|
+
* `ollama list` probe, the server it talks to) is alive; a missing executable
|
|
73
|
+
* or non-zero exit is `present: false` with a sanitized error.
|
|
74
|
+
* @param subprocess - the real subprocess runtime.
|
|
75
|
+
* @param graceMs - terminate grace for the spawned probe.
|
|
76
|
+
* @param probe - the command to probe with (defaults to `ollama list`).
|
|
77
|
+
* @returns the process health result.
|
|
78
|
+
*/
|
|
79
|
+
export async function checkProcessHealth(
|
|
80
|
+
subprocess: SubprocessRuntime,
|
|
81
|
+
graceMs: number,
|
|
82
|
+
probe: ProcessProbe = OLLAMA_PROCESS_PROBE,
|
|
83
|
+
): Promise<ProcessHealth> {
|
|
84
|
+
let executable: string
|
|
85
|
+
try {
|
|
86
|
+
executable = await subprocess.resolveExecutable(probe.command)
|
|
87
|
+
} catch (error: unknown) {
|
|
88
|
+
return {
|
|
89
|
+
present: false,
|
|
90
|
+
error: sanitizeText(`"${probe.command}" not found: ${error instanceof Error ? error.message : String(error)}`, 500),
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
const handle = subprocess.spawn({
|
|
94
|
+
argv: [executable, ...probe.args],
|
|
95
|
+
cwd: tmpdir(),
|
|
96
|
+
stdio: {
|
|
97
|
+
stdin: 'ignore',
|
|
98
|
+
stdout: { maxBytes: 4096 },
|
|
99
|
+
stderr: { maxBytes: 4096 },
|
|
100
|
+
},
|
|
101
|
+
graceMs,
|
|
102
|
+
})
|
|
103
|
+
const outcome = await handle.done
|
|
104
|
+
if (outcome.exitCode === 0) return { present: true }
|
|
105
|
+
const stderr = handle.collected.stderr?.readFrom(0).text ?? ''
|
|
106
|
+
return {
|
|
107
|
+
present: false,
|
|
108
|
+
error: sanitizeText(stderr.trim().length > 0 ? stderr : `exit code ${String(outcome.exitCode)}`, 500),
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/**
|
|
113
|
+
* Check both health signals.
|
|
114
|
+
* @param baseURL - the Ollama base URL.
|
|
115
|
+
* @param fetchImpl - the fetch implementation.
|
|
116
|
+
* @param subprocess - the real subprocess runtime.
|
|
117
|
+
* @param requestTimeoutMs - HTTP deadline in milliseconds.
|
|
118
|
+
* @param graceMs - subprocess terminate grace in milliseconds.
|
|
119
|
+
* @returns the combined health result.
|
|
120
|
+
*/
|
|
121
|
+
export async function checkHealth(
|
|
122
|
+
baseURL: string,
|
|
123
|
+
fetchImpl: FetchLike,
|
|
124
|
+
subprocess: SubprocessRuntime,
|
|
125
|
+
requestTimeoutMs: number,
|
|
126
|
+
graceMs: number,
|
|
127
|
+
): Promise<HealthResult> {
|
|
128
|
+
const [api, process] = await Promise.all([
|
|
129
|
+
checkApiHealth(baseURL, fetchImpl, requestTimeoutMs),
|
|
130
|
+
checkProcessHealth(subprocess, graceMs),
|
|
131
|
+
])
|
|
132
|
+
return { api, process }
|
|
133
|
+
}
|