@kindgi/adapter-model-in-process 0.0.0-bootstrap.0 → 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,236 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ // Copyright (C) 2026 Kindgi Inc.
3
+
4
+ import type {
5
+ Feature,
6
+ ModelCallInput,
7
+ ModelCallResult,
8
+ ModelInfo,
9
+ ModelMessage,
10
+ ModelProvider,
11
+ ProviderMetadata,
12
+ } from '@kindgi/capabilities';
13
+
14
+ import { DEFAULT_LOCAL_MODEL, type LocalModel, MODEL_SPECS, type ModelSpec } from './models.js';
15
+
16
+ /**
17
+ * Structural type for the transformers.js text-generation pipeline —
18
+ * matches only the surface we actually depend on. The real module is
19
+ * imported dynamically on first invoke so shape tests never touch the
20
+ * ONNX runtime (which has ESM/CJS interop quirks with `onnxruntime-common`
21
+ * under some Node versions).
22
+ */
23
+ type TextGenerationPipeline = {
24
+ tokenizer: {
25
+ apply_chat_template?: (input: unknown, options: unknown) => unknown;
26
+ encode?: (t: string) => number[];
27
+ };
28
+ (
29
+ messages: unknown,
30
+ options: Record<string, unknown>,
31
+ ): Promise<Array<{ generated_text: Array<{ role: string; content: string }> }>>;
32
+ };
33
+
34
+ export interface InProcessProviderOptions {
35
+ /**
36
+ * Local models this provider exposes. Each entry becomes a
37
+ * `ModelInfo` in the returned `ProviderMetadata.models[]`. Pipelines
38
+ * are loaded lazily per model on the first invocation that names
39
+ * them. Defaults to `[DEFAULT_LOCAL_MODEL]` when omitted.
40
+ */
41
+ readonly models?: readonly LocalModel[];
42
+ /**
43
+ * Override the provider id embedded in `ProviderMetadata.id` — useful
44
+ * when you want two registrations of the same model set with
45
+ * different capability declarations (e.g. one for routing, one for
46
+ * extraction).
47
+ */
48
+ readonly providerId?: string;
49
+ /**
50
+ * Directory for downloaded model files, passed to transformers.js as
51
+ * `cache_dir`. Defaults to transformers.js's own cache (`env.cacheDir`;
52
+ * see `models.ts`).
53
+ */
54
+ readonly cacheDir?: string;
55
+ }
56
+
57
+ /**
58
+ * Create an in-process `ModelProvider` backed by `@huggingface/transformers`.
59
+ *
60
+ * Pipelines are loaded lazily per model on the first `invoke()` call
61
+ * that names them — construction of the provider is cheap. Model
62
+ * weights download into the transformers.js cache (or `cacheDir`) on
63
+ * first use; later loads read from the cache.
64
+ *
65
+ * Cost is always `0` USD (no external service). Resource-usage recording
66
+ * still tracks token counts + duration so operators can see the local
67
+ * model's real load in aggregate reports.
68
+ *
69
+ * The returned provider is safe to share process-wide; each underlying
70
+ * pipeline serialises its own requests inside the ONNX runtime.
71
+ */
72
+ export function createInProcessModelProvider(
73
+ options: InProcessProviderOptions = {},
74
+ ): ModelProvider {
75
+ const modelKeys = options.models ?? [DEFAULT_LOCAL_MODEL];
76
+ if (modelKeys.length === 0) {
77
+ throw new Error('createInProcessModelProvider: `models` must not be empty.');
78
+ }
79
+ const providerId = options.providerId ?? `in-process/${modelKeys.join('+')}`;
80
+ const metadata: ProviderMetadata = buildProviderMetadata(providerId, modelKeys);
81
+ // Cache loaded pipelines per model key. Uses a Promise to dedupe
82
+ // concurrent first-invocations on the same model.
83
+ const pipelines = new Map<LocalModel, Promise<TextGenerationPipeline>>();
84
+
85
+ function loadPipeline(modelKey: LocalModel): Promise<TextGenerationPipeline> {
86
+ const cached = pipelines.get(modelKey);
87
+ if (cached !== undefined) return cached;
88
+ const spec = MODEL_SPECS[modelKey];
89
+ const loading = (async (): Promise<TextGenerationPipeline> => {
90
+ const mod = (await import('@huggingface/transformers')) as unknown as {
91
+ pipeline: (task: string, model: string, opts: Record<string, unknown>) => Promise<unknown>;
92
+ };
93
+ return (await mod.pipeline('text-generation', spec.hfName, {
94
+ dtype: spec.dtype,
95
+ ...(options.cacheDir !== undefined && { cache_dir: options.cacheDir }),
96
+ })) as TextGenerationPipeline;
97
+ })();
98
+ pipelines.set(modelKey, loading);
99
+ return loading;
100
+ }
101
+
102
+ return {
103
+ metadata,
104
+ async invoke(input: ModelCallInput): Promise<ModelCallResult> {
105
+ const modelKey = input.model as LocalModel;
106
+ if (!modelKeys.includes(modelKey)) {
107
+ throw new Error(
108
+ `@kindgi/adapter-model-in-process: provider "${providerId}" does not expose model "${input.model}". ` +
109
+ `Available: ${modelKeys.join(', ') || '<none>'}.`,
110
+ );
111
+ }
112
+ const spec = MODEL_SPECS[modelKey];
113
+ const p = await loadPipeline(modelKey);
114
+ const messages = input.messages.map(toChatTemplateMessage);
115
+ const startedAt = Date.now();
116
+
117
+ const promptTokens = countPromptTokens(p, messages);
118
+
119
+ const output = await p(messages, {
120
+ max_new_tokens: input.maxOutputTokens ?? 512,
121
+ do_sample: input.temperature !== undefined && input.temperature > 0,
122
+ ...(input.temperature !== undefined &&
123
+ input.temperature > 0 && {
124
+ temperature: input.temperature,
125
+ }),
126
+ // No TextStreamer: `invoke` waits for the complete generation and
127
+ // returns it in one result.
128
+ });
129
+
130
+ const durationMs = Date.now() - startedAt;
131
+ const firstResult = output[0];
132
+ const chat = firstResult?.generated_text ?? [];
133
+ const assistantTurn = chat.at(-1);
134
+ const responseText = assistantTurn?.content ?? '';
135
+ const completionTokens = countTextTokens(p, responseText);
136
+ void spec;
137
+
138
+ return {
139
+ message: { role: 'assistant', content: responseText },
140
+ finishReason: 'stop',
141
+ usage: { promptTokens, completionTokens },
142
+ // In-process = zero direct USD cost. Ledger still records tokens.
143
+ costUsd: 0,
144
+ durationMs,
145
+ provider: { id: providerId, model: modelKey },
146
+ };
147
+ },
148
+ };
149
+ }
150
+
151
+ /**
152
+ * Build `ProviderMetadata` from the set of local models this provider
153
+ * exposes. Each `LocalModel` key maps to a `ModelInfo` entry —
154
+ * `ModelInfo.name` is the key itself (short label), not the fully-
155
+ * qualified Hugging Face model id, so provider listings (such as
156
+ * `GET /v1/providers`) show `smollm2-360m` alongside vendor model ids.
157
+ * Provider-level `attributes` are the union of every model's
158
+ * `suitableFor` + `tier`, deduped, plus a marker `in-process`.
159
+ */
160
+ function buildProviderMetadata(
161
+ providerId: string,
162
+ modelKeys: readonly LocalModel[],
163
+ ): ProviderMetadata {
164
+ const attributes = new Set<string>(['in-process']);
165
+ for (const key of modelKeys) {
166
+ const spec = MODEL_SPECS[key];
167
+ for (const attr of spec.suitableFor) attributes.add(attr);
168
+ attributes.add(spec.tier);
169
+ }
170
+ return {
171
+ id: providerId,
172
+ region: 'in-process',
173
+ models: modelKeys.map((key) => buildModelInfo(key, MODEL_SPECS[key])),
174
+ attributes: [...attributes],
175
+ description:
176
+ 'Local models via @huggingface/transformers — routing / classification / smoke-test tier.',
177
+ };
178
+ }
179
+
180
+ function buildModelInfo(key: LocalModel, spec: ModelSpec): ModelInfo {
181
+ const features: Feature[] = ['streaming'];
182
+ if (spec.toolUse) features.push('tool-use', 'structured-output');
183
+ if (spec.contextWindow >= 32_000) features.push('long-context');
184
+ return {
185
+ name: key,
186
+ contextWindow: spec.contextWindow,
187
+ features,
188
+ cost: { promptUsdPer1kTokens: 0, completionUsdPer1kTokens: 0 },
189
+ description: `Local ${spec.tier}-tier model (${spec.approxDownloadMb} MB, ${spec.contextWindow}-token context, HF id: ${spec.hfName}).`,
190
+ };
191
+ }
192
+
193
+ /** Translate Kindgi ModelMessage → transformers.js Chat template message. */
194
+ function toChatTemplateMessage(m: ModelMessage): { role: string; content: string } {
195
+ return { role: m.role, content: m.content };
196
+ }
197
+
198
+ /**
199
+ * Token counting via the pipeline's tokenizer. Called before invocation
200
+ * for prompt tokens and after for completion tokens. The tokenizer is
201
+ * async-loaded with the pipeline; this must run after `loadPipeline()`.
202
+ */
203
+ function countPromptTokens(
204
+ p: TextGenerationPipeline,
205
+ messages: readonly { role: string; content: string }[],
206
+ ): number {
207
+ const applyTemplate = p.tokenizer.apply_chat_template;
208
+ if (typeof applyTemplate !== 'function') {
209
+ return messages.reduce((sum, m) => sum + countTextTokens(p, m.content), 0);
210
+ }
211
+ try {
212
+ const ids = applyTemplate(messages, {
213
+ tokenize: true,
214
+ add_generation_prompt: true,
215
+ });
216
+ return Array.isArray(ids) ? ids.length : 0;
217
+ } catch {
218
+ return messages.reduce((sum, m) => sum + countTextTokens(p, m.content), 0);
219
+ }
220
+ }
221
+
222
+ function countTextTokens(p: TextGenerationPipeline, text: string): number {
223
+ if (text.length === 0) return 0;
224
+ const encode = p.tokenizer.encode;
225
+ if (typeof encode !== 'function') {
226
+ return Math.ceil(text.length / 4);
227
+ }
228
+ try {
229
+ return encode(text).length;
230
+ } catch {
231
+ return Math.ceil(text.length / 4);
232
+ }
233
+ }
234
+
235
+ export type { LocalModel, ModelSpec } from './models.js';
236
+ export { MODEL_SPECS, DEFAULT_LOCAL_MODEL } from './models.js';