@kindgi/adapter-model-in-process 0.0.0-bootstrap.0 → 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +69 -2
- package/dist/index.d.ts +6 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +5 -0
- package/dist/index.js.map +1 -0
- package/dist/models.d.ts +44 -0
- package/dist/models.d.ts.map +1 -0
- package/dist/models.js +41 -0
- package/dist/models.js.map +1 -0
- package/dist/prepare.d.ts +44 -0
- package/dist/prepare.d.ts.map +1 -0
- package/dist/prepare.js +99 -0
- package/dist/prepare.js.map +1 -0
- package/dist/provider.d.ts +43 -0
- package/dist/provider.d.ts.map +1 -0
- package/dist/provider.js +165 -0
- package/dist/provider.js.map +1 -0
- package/package.json +48 -4
- package/src/index.ts +12 -0
- package/src/models.ts +84 -0
- package/src/prepare.ts +166 -0
- package/src/provider.ts +236 -0
package/src/provider.ts
ADDED
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
// Copyright (C) 2026 Kindgi Inc.
|
|
3
|
+
|
|
4
|
+
import type {
|
|
5
|
+
Feature,
|
|
6
|
+
ModelCallInput,
|
|
7
|
+
ModelCallResult,
|
|
8
|
+
ModelInfo,
|
|
9
|
+
ModelMessage,
|
|
10
|
+
ModelProvider,
|
|
11
|
+
ProviderMetadata,
|
|
12
|
+
} from '@kindgi/capabilities';
|
|
13
|
+
|
|
14
|
+
import { DEFAULT_LOCAL_MODEL, type LocalModel, MODEL_SPECS, type ModelSpec } from './models.js';
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* Structural type for the transformers.js text-generation pipeline —
|
|
18
|
+
* matches only the surface we actually depend on. The real module is
|
|
19
|
+
* imported dynamically on first invoke so shape tests never touch the
|
|
20
|
+
* ONNX runtime (which has ESM/CJS interop quirks with `onnxruntime-common`
|
|
21
|
+
* under some Node versions).
|
|
22
|
+
*/
|
|
23
|
+
type TextGenerationPipeline = {
|
|
24
|
+
tokenizer: {
|
|
25
|
+
apply_chat_template?: (input: unknown, options: unknown) => unknown;
|
|
26
|
+
encode?: (t: string) => number[];
|
|
27
|
+
};
|
|
28
|
+
(
|
|
29
|
+
messages: unknown,
|
|
30
|
+
options: Record<string, unknown>,
|
|
31
|
+
): Promise<Array<{ generated_text: Array<{ role: string; content: string }> }>>;
|
|
32
|
+
};
|
|
33
|
+
|
|
34
|
+
export interface InProcessProviderOptions {
|
|
35
|
+
/**
|
|
36
|
+
* Local models this provider exposes. Each entry becomes a
|
|
37
|
+
* `ModelInfo` in the returned `ProviderMetadata.models[]`. Pipelines
|
|
38
|
+
* are loaded lazily per model on the first invocation that names
|
|
39
|
+
* them. Defaults to `[DEFAULT_LOCAL_MODEL]` when omitted.
|
|
40
|
+
*/
|
|
41
|
+
readonly models?: readonly LocalModel[];
|
|
42
|
+
/**
|
|
43
|
+
* Override the provider id embedded in `ProviderMetadata.id` — useful
|
|
44
|
+
* when you want two registrations of the same model set with
|
|
45
|
+
* different capability declarations (e.g. one for routing, one for
|
|
46
|
+
* extraction).
|
|
47
|
+
*/
|
|
48
|
+
readonly providerId?: string;
|
|
49
|
+
/**
|
|
50
|
+
* Directory for downloaded model files, passed to transformers.js as
|
|
51
|
+
* `cache_dir`. Defaults to transformers.js's own cache (`env.cacheDir`;
|
|
52
|
+
* see `models.ts`).
|
|
53
|
+
*/
|
|
54
|
+
readonly cacheDir?: string;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Create an in-process `ModelProvider` backed by `@huggingface/transformers`.
|
|
59
|
+
*
|
|
60
|
+
* Pipelines are loaded lazily per model on the first `invoke()` call
|
|
61
|
+
* that names them — construction of the provider is cheap. Model
|
|
62
|
+
* weights download into the transformers.js cache (or `cacheDir`) on
|
|
63
|
+
* first use; later loads read from the cache.
|
|
64
|
+
*
|
|
65
|
+
* Cost is always `0` USD (no external service). Resource-usage recording
|
|
66
|
+
* still tracks token counts + duration so operators can see the local
|
|
67
|
+
* model's real load in aggregate reports.
|
|
68
|
+
*
|
|
69
|
+
* The returned provider is safe to share process-wide; each underlying
|
|
70
|
+
* pipeline serialises its own requests inside the ONNX runtime.
|
|
71
|
+
*/
|
|
72
|
+
export function createInProcessModelProvider(
|
|
73
|
+
options: InProcessProviderOptions = {},
|
|
74
|
+
): ModelProvider {
|
|
75
|
+
const modelKeys = options.models ?? [DEFAULT_LOCAL_MODEL];
|
|
76
|
+
if (modelKeys.length === 0) {
|
|
77
|
+
throw new Error('createInProcessModelProvider: `models` must not be empty.');
|
|
78
|
+
}
|
|
79
|
+
const providerId = options.providerId ?? `in-process/${modelKeys.join('+')}`;
|
|
80
|
+
const metadata: ProviderMetadata = buildProviderMetadata(providerId, modelKeys);
|
|
81
|
+
// Cache loaded pipelines per model key. Uses a Promise to dedupe
|
|
82
|
+
// concurrent first-invocations on the same model.
|
|
83
|
+
const pipelines = new Map<LocalModel, Promise<TextGenerationPipeline>>();
|
|
84
|
+
|
|
85
|
+
function loadPipeline(modelKey: LocalModel): Promise<TextGenerationPipeline> {
|
|
86
|
+
const cached = pipelines.get(modelKey);
|
|
87
|
+
if (cached !== undefined) return cached;
|
|
88
|
+
const spec = MODEL_SPECS[modelKey];
|
|
89
|
+
const loading = (async (): Promise<TextGenerationPipeline> => {
|
|
90
|
+
const mod = (await import('@huggingface/transformers')) as unknown as {
|
|
91
|
+
pipeline: (task: string, model: string, opts: Record<string, unknown>) => Promise<unknown>;
|
|
92
|
+
};
|
|
93
|
+
return (await mod.pipeline('text-generation', spec.hfName, {
|
|
94
|
+
dtype: spec.dtype,
|
|
95
|
+
...(options.cacheDir !== undefined && { cache_dir: options.cacheDir }),
|
|
96
|
+
})) as TextGenerationPipeline;
|
|
97
|
+
})();
|
|
98
|
+
pipelines.set(modelKey, loading);
|
|
99
|
+
return loading;
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
return {
|
|
103
|
+
metadata,
|
|
104
|
+
async invoke(input: ModelCallInput): Promise<ModelCallResult> {
|
|
105
|
+
const modelKey = input.model as LocalModel;
|
|
106
|
+
if (!modelKeys.includes(modelKey)) {
|
|
107
|
+
throw new Error(
|
|
108
|
+
`@kindgi/adapter-model-in-process: provider "${providerId}" does not expose model "${input.model}". ` +
|
|
109
|
+
`Available: ${modelKeys.join(', ') || '<none>'}.`,
|
|
110
|
+
);
|
|
111
|
+
}
|
|
112
|
+
const spec = MODEL_SPECS[modelKey];
|
|
113
|
+
const p = await loadPipeline(modelKey);
|
|
114
|
+
const messages = input.messages.map(toChatTemplateMessage);
|
|
115
|
+
const startedAt = Date.now();
|
|
116
|
+
|
|
117
|
+
const promptTokens = countPromptTokens(p, messages);
|
|
118
|
+
|
|
119
|
+
const output = await p(messages, {
|
|
120
|
+
max_new_tokens: input.maxOutputTokens ?? 512,
|
|
121
|
+
do_sample: input.temperature !== undefined && input.temperature > 0,
|
|
122
|
+
...(input.temperature !== undefined &&
|
|
123
|
+
input.temperature > 0 && {
|
|
124
|
+
temperature: input.temperature,
|
|
125
|
+
}),
|
|
126
|
+
// No TextStreamer: `invoke` waits for the complete generation and
|
|
127
|
+
// returns it in one result.
|
|
128
|
+
});
|
|
129
|
+
|
|
130
|
+
const durationMs = Date.now() - startedAt;
|
|
131
|
+
const firstResult = output[0];
|
|
132
|
+
const chat = firstResult?.generated_text ?? [];
|
|
133
|
+
const assistantTurn = chat.at(-1);
|
|
134
|
+
const responseText = assistantTurn?.content ?? '';
|
|
135
|
+
const completionTokens = countTextTokens(p, responseText);
|
|
136
|
+
void spec;
|
|
137
|
+
|
|
138
|
+
return {
|
|
139
|
+
message: { role: 'assistant', content: responseText },
|
|
140
|
+
finishReason: 'stop',
|
|
141
|
+
usage: { promptTokens, completionTokens },
|
|
142
|
+
// In-process = zero direct USD cost. Ledger still records tokens.
|
|
143
|
+
costUsd: 0,
|
|
144
|
+
durationMs,
|
|
145
|
+
provider: { id: providerId, model: modelKey },
|
|
146
|
+
};
|
|
147
|
+
},
|
|
148
|
+
};
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
/**
|
|
152
|
+
* Build `ProviderMetadata` from the set of local models this provider
|
|
153
|
+
* exposes. Each `LocalModel` key maps to a `ModelInfo` entry —
|
|
154
|
+
* `ModelInfo.name` is the key itself (short label), not the fully-
|
|
155
|
+
* qualified Hugging Face model id, so provider listings (such as
|
|
156
|
+
* `GET /v1/providers`) show `smollm2-360m` alongside vendor model ids.
|
|
157
|
+
* Provider-level `attributes` are the union of every model's
|
|
158
|
+
* `suitableFor` + `tier`, deduped, plus a marker `in-process`.
|
|
159
|
+
*/
|
|
160
|
+
function buildProviderMetadata(
|
|
161
|
+
providerId: string,
|
|
162
|
+
modelKeys: readonly LocalModel[],
|
|
163
|
+
): ProviderMetadata {
|
|
164
|
+
const attributes = new Set<string>(['in-process']);
|
|
165
|
+
for (const key of modelKeys) {
|
|
166
|
+
const spec = MODEL_SPECS[key];
|
|
167
|
+
for (const attr of spec.suitableFor) attributes.add(attr);
|
|
168
|
+
attributes.add(spec.tier);
|
|
169
|
+
}
|
|
170
|
+
return {
|
|
171
|
+
id: providerId,
|
|
172
|
+
region: 'in-process',
|
|
173
|
+
models: modelKeys.map((key) => buildModelInfo(key, MODEL_SPECS[key])),
|
|
174
|
+
attributes: [...attributes],
|
|
175
|
+
description:
|
|
176
|
+
'Local models via @huggingface/transformers — routing / classification / smoke-test tier.',
|
|
177
|
+
};
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
function buildModelInfo(key: LocalModel, spec: ModelSpec): ModelInfo {
|
|
181
|
+
const features: Feature[] = ['streaming'];
|
|
182
|
+
if (spec.toolUse) features.push('tool-use', 'structured-output');
|
|
183
|
+
if (spec.contextWindow >= 32_000) features.push('long-context');
|
|
184
|
+
return {
|
|
185
|
+
name: key,
|
|
186
|
+
contextWindow: spec.contextWindow,
|
|
187
|
+
features,
|
|
188
|
+
cost: { promptUsdPer1kTokens: 0, completionUsdPer1kTokens: 0 },
|
|
189
|
+
description: `Local ${spec.tier}-tier model (${spec.approxDownloadMb} MB, ${spec.contextWindow}-token context, HF id: ${spec.hfName}).`,
|
|
190
|
+
};
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/** Translate Kindgi ModelMessage → transformers.js Chat template message. */
|
|
194
|
+
function toChatTemplateMessage(m: ModelMessage): { role: string; content: string } {
|
|
195
|
+
return { role: m.role, content: m.content };
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
/**
|
|
199
|
+
* Token counting via the pipeline's tokenizer. Called before invocation
|
|
200
|
+
* for prompt tokens and after for completion tokens. The tokenizer is
|
|
201
|
+
* async-loaded with the pipeline; this must run after `loadPipeline()`.
|
|
202
|
+
*/
|
|
203
|
+
function countPromptTokens(
|
|
204
|
+
p: TextGenerationPipeline,
|
|
205
|
+
messages: readonly { role: string; content: string }[],
|
|
206
|
+
): number {
|
|
207
|
+
const applyTemplate = p.tokenizer.apply_chat_template;
|
|
208
|
+
if (typeof applyTemplate !== 'function') {
|
|
209
|
+
return messages.reduce((sum, m) => sum + countTextTokens(p, m.content), 0);
|
|
210
|
+
}
|
|
211
|
+
try {
|
|
212
|
+
const ids = applyTemplate(messages, {
|
|
213
|
+
tokenize: true,
|
|
214
|
+
add_generation_prompt: true,
|
|
215
|
+
});
|
|
216
|
+
return Array.isArray(ids) ? ids.length : 0;
|
|
217
|
+
} catch {
|
|
218
|
+
return messages.reduce((sum, m) => sum + countTextTokens(p, m.content), 0);
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
function countTextTokens(p: TextGenerationPipeline, text: string): number {
|
|
223
|
+
if (text.length === 0) return 0;
|
|
224
|
+
const encode = p.tokenizer.encode;
|
|
225
|
+
if (typeof encode !== 'function') {
|
|
226
|
+
return Math.ceil(text.length / 4);
|
|
227
|
+
}
|
|
228
|
+
try {
|
|
229
|
+
return encode(text).length;
|
|
230
|
+
} catch {
|
|
231
|
+
return Math.ceil(text.length / 4);
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
export type { LocalModel, ModelSpec } from './models.js';
|
|
236
|
+
export { MODEL_SPECS, DEFAULT_LOCAL_MODEL } from './models.js';
|