@kindgi/adapter-model-in-process 0.0.0-bootstrap.0 → 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +69 -2
- package/dist/index.d.ts +6 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +5 -0
- package/dist/index.js.map +1 -0
- package/dist/models.d.ts +44 -0
- package/dist/models.d.ts.map +1 -0
- package/dist/models.js +41 -0
- package/dist/models.js.map +1 -0
- package/dist/prepare.d.ts +44 -0
- package/dist/prepare.d.ts.map +1 -0
- package/dist/prepare.js +99 -0
- package/dist/prepare.js.map +1 -0
- package/dist/provider.d.ts +43 -0
- package/dist/provider.d.ts.map +1 -0
- package/dist/provider.js +165 -0
- package/dist/provider.js.map +1 -0
- package/package.json +48 -4
- package/src/index.ts +12 -0
- package/src/models.ts +84 -0
- package/src/prepare.ts +166 -0
- package/src/provider.ts +236 -0
package/dist/provider.js
ADDED
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
// Copyright (C) 2026 Kindgi Inc.
|
|
3
|
+
import { DEFAULT_LOCAL_MODEL, MODEL_SPECS } from './models.js';
|
|
4
|
+
/**
|
|
5
|
+
* Create an in-process `ModelProvider` backed by `@huggingface/transformers`.
|
|
6
|
+
*
|
|
7
|
+
* Pipelines are loaded lazily per model on the first `invoke()` call
|
|
8
|
+
* that names them — construction of the provider is cheap. Model
|
|
9
|
+
* weights download into the transformers.js cache (or `cacheDir`) on
|
|
10
|
+
* first use; later loads read from the cache.
|
|
11
|
+
*
|
|
12
|
+
* Cost is always `0` USD (no external service). Resource-usage recording
|
|
13
|
+
* still tracks token counts + duration so operators can see the local
|
|
14
|
+
* model's real load in aggregate reports.
|
|
15
|
+
*
|
|
16
|
+
* The returned provider is safe to share process-wide; each underlying
|
|
17
|
+
* pipeline serialises its own requests inside the ONNX runtime.
|
|
18
|
+
*/
|
|
19
|
+
export function createInProcessModelProvider(options = {}) {
|
|
20
|
+
const modelKeys = options.models ?? [DEFAULT_LOCAL_MODEL];
|
|
21
|
+
if (modelKeys.length === 0) {
|
|
22
|
+
throw new Error('createInProcessModelProvider: `models` must not be empty.');
|
|
23
|
+
}
|
|
24
|
+
const providerId = options.providerId ?? `in-process/${modelKeys.join('+')}`;
|
|
25
|
+
const metadata = buildProviderMetadata(providerId, modelKeys);
|
|
26
|
+
// Cache loaded pipelines per model key. Uses a Promise to dedupe
|
|
27
|
+
// concurrent first-invocations on the same model.
|
|
28
|
+
const pipelines = new Map();
|
|
29
|
+
function loadPipeline(modelKey) {
|
|
30
|
+
const cached = pipelines.get(modelKey);
|
|
31
|
+
if (cached !== undefined)
|
|
32
|
+
return cached;
|
|
33
|
+
const spec = MODEL_SPECS[modelKey];
|
|
34
|
+
const loading = (async () => {
|
|
35
|
+
const mod = (await import('@huggingface/transformers'));
|
|
36
|
+
return (await mod.pipeline('text-generation', spec.hfName, {
|
|
37
|
+
dtype: spec.dtype,
|
|
38
|
+
...(options.cacheDir !== undefined && { cache_dir: options.cacheDir }),
|
|
39
|
+
}));
|
|
40
|
+
})();
|
|
41
|
+
pipelines.set(modelKey, loading);
|
|
42
|
+
return loading;
|
|
43
|
+
}
|
|
44
|
+
return {
|
|
45
|
+
metadata,
|
|
46
|
+
async invoke(input) {
|
|
47
|
+
const modelKey = input.model;
|
|
48
|
+
if (!modelKeys.includes(modelKey)) {
|
|
49
|
+
throw new Error(`@kindgi/adapter-model-in-process: provider "${providerId}" does not expose model "${input.model}". ` +
|
|
50
|
+
`Available: ${modelKeys.join(', ') || '<none>'}.`);
|
|
51
|
+
}
|
|
52
|
+
const spec = MODEL_SPECS[modelKey];
|
|
53
|
+
const p = await loadPipeline(modelKey);
|
|
54
|
+
const messages = input.messages.map(toChatTemplateMessage);
|
|
55
|
+
const startedAt = Date.now();
|
|
56
|
+
const promptTokens = countPromptTokens(p, messages);
|
|
57
|
+
const output = await p(messages, {
|
|
58
|
+
max_new_tokens: input.maxOutputTokens ?? 512,
|
|
59
|
+
do_sample: input.temperature !== undefined && input.temperature > 0,
|
|
60
|
+
...(input.temperature !== undefined &&
|
|
61
|
+
input.temperature > 0 && {
|
|
62
|
+
temperature: input.temperature,
|
|
63
|
+
}),
|
|
64
|
+
// No TextStreamer: `invoke` waits for the complete generation and
|
|
65
|
+
// returns it in one result.
|
|
66
|
+
});
|
|
67
|
+
const durationMs = Date.now() - startedAt;
|
|
68
|
+
const firstResult = output[0];
|
|
69
|
+
const chat = firstResult?.generated_text ?? [];
|
|
70
|
+
const assistantTurn = chat.at(-1);
|
|
71
|
+
const responseText = assistantTurn?.content ?? '';
|
|
72
|
+
const completionTokens = countTextTokens(p, responseText);
|
|
73
|
+
void spec;
|
|
74
|
+
return {
|
|
75
|
+
message: { role: 'assistant', content: responseText },
|
|
76
|
+
finishReason: 'stop',
|
|
77
|
+
usage: { promptTokens, completionTokens },
|
|
78
|
+
// In-process = zero direct USD cost. Ledger still records tokens.
|
|
79
|
+
costUsd: 0,
|
|
80
|
+
durationMs,
|
|
81
|
+
provider: { id: providerId, model: modelKey },
|
|
82
|
+
};
|
|
83
|
+
},
|
|
84
|
+
};
|
|
85
|
+
}
|
|
86
|
+
/**
|
|
87
|
+
* Build `ProviderMetadata` from the set of local models this provider
|
|
88
|
+
* exposes. Each `LocalModel` key maps to a `ModelInfo` entry —
|
|
89
|
+
* `ModelInfo.name` is the key itself (short label), not the fully-
|
|
90
|
+
* qualified Hugging Face model id, so provider listings (such as
|
|
91
|
+
* `GET /v1/providers`) show `smollm2-360m` alongside vendor model ids.
|
|
92
|
+
* Provider-level `attributes` are the union of every model's
|
|
93
|
+
* `suitableFor` + `tier`, deduped, plus a marker `in-process`.
|
|
94
|
+
*/
|
|
95
|
+
function buildProviderMetadata(providerId, modelKeys) {
|
|
96
|
+
const attributes = new Set(['in-process']);
|
|
97
|
+
for (const key of modelKeys) {
|
|
98
|
+
const spec = MODEL_SPECS[key];
|
|
99
|
+
for (const attr of spec.suitableFor)
|
|
100
|
+
attributes.add(attr);
|
|
101
|
+
attributes.add(spec.tier);
|
|
102
|
+
}
|
|
103
|
+
return {
|
|
104
|
+
id: providerId,
|
|
105
|
+
region: 'in-process',
|
|
106
|
+
models: modelKeys.map((key) => buildModelInfo(key, MODEL_SPECS[key])),
|
|
107
|
+
attributes: [...attributes],
|
|
108
|
+
description: 'Local models via @huggingface/transformers — routing / classification / smoke-test tier.',
|
|
109
|
+
};
|
|
110
|
+
}
|
|
111
|
+
function buildModelInfo(key, spec) {
|
|
112
|
+
const features = ['streaming'];
|
|
113
|
+
if (spec.toolUse)
|
|
114
|
+
features.push('tool-use', 'structured-output');
|
|
115
|
+
if (spec.contextWindow >= 32_000)
|
|
116
|
+
features.push('long-context');
|
|
117
|
+
return {
|
|
118
|
+
name: key,
|
|
119
|
+
contextWindow: spec.contextWindow,
|
|
120
|
+
features,
|
|
121
|
+
cost: { promptUsdPer1kTokens: 0, completionUsdPer1kTokens: 0 },
|
|
122
|
+
description: `Local ${spec.tier}-tier model (${spec.approxDownloadMb} MB, ${spec.contextWindow}-token context, HF id: ${spec.hfName}).`,
|
|
123
|
+
};
|
|
124
|
+
}
|
|
125
|
+
/** Translate Kindgi ModelMessage → transformers.js Chat template message. */
|
|
126
|
+
function toChatTemplateMessage(m) {
|
|
127
|
+
return { role: m.role, content: m.content };
|
|
128
|
+
}
|
|
129
|
+
/**
|
|
130
|
+
* Token counting via the pipeline's tokenizer. Called before invocation
|
|
131
|
+
* for prompt tokens and after for completion tokens. The tokenizer is
|
|
132
|
+
* async-loaded with the pipeline; this must run after `loadPipeline()`.
|
|
133
|
+
*/
|
|
134
|
+
function countPromptTokens(p, messages) {
|
|
135
|
+
const applyTemplate = p.tokenizer.apply_chat_template;
|
|
136
|
+
if (typeof applyTemplate !== 'function') {
|
|
137
|
+
return messages.reduce((sum, m) => sum + countTextTokens(p, m.content), 0);
|
|
138
|
+
}
|
|
139
|
+
try {
|
|
140
|
+
const ids = applyTemplate(messages, {
|
|
141
|
+
tokenize: true,
|
|
142
|
+
add_generation_prompt: true,
|
|
143
|
+
});
|
|
144
|
+
return Array.isArray(ids) ? ids.length : 0;
|
|
145
|
+
}
|
|
146
|
+
catch {
|
|
147
|
+
return messages.reduce((sum, m) => sum + countTextTokens(p, m.content), 0);
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
function countTextTokens(p, text) {
|
|
151
|
+
if (text.length === 0)
|
|
152
|
+
return 0;
|
|
153
|
+
const encode = p.tokenizer.encode;
|
|
154
|
+
if (typeof encode !== 'function') {
|
|
155
|
+
return Math.ceil(text.length / 4);
|
|
156
|
+
}
|
|
157
|
+
try {
|
|
158
|
+
return encode(text).length;
|
|
159
|
+
}
|
|
160
|
+
catch {
|
|
161
|
+
return Math.ceil(text.length / 4);
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
export { MODEL_SPECS, DEFAULT_LOCAL_MODEL } from './models.js';
|
|
165
|
+
//# sourceMappingURL=provider.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"provider.js","sourceRoot":"","sources":["../src/provider.ts"],"names":[],"mappings":"AAAA,sCAAsC;AACtC,iCAAiC;AAYjC,OAAO,EAAE,mBAAmB,EAAmB,WAAW,EAAkB,MAAM,aAAa,CAAC;AA2ChG;;;;;;;;;;;;;;GAcG;AACH,MAAM,UAAU,4BAA4B,CAC1C,UAAoC,EAAE;IAEtC,MAAM,SAAS,GAAG,OAAO,CAAC,MAAM,IAAI,CAAC,mBAAmB,CAAC,CAAC;IAC1D,IAAI,SAAS,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QAC3B,MAAM,IAAI,KAAK,CAAC,2DAA2D,CAAC,CAAC;IAC/E,CAAC;IACD,MAAM,UAAU,GAAG,OAAO,CAAC,UAAU,IAAI,cAAc,SAAS,CAAC,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC;IAC7E,MAAM,QAAQ,GAAqB,qBAAqB,CAAC,UAAU,EAAE,SAAS,CAAC,CAAC;IAChF,iEAAiE;IACjE,kDAAkD;IAClD,MAAM,SAAS,GAAG,IAAI,GAAG,EAA+C,CAAC;IAEzE,SAAS,YAAY,CAAC,QAAoB;QACxC,MAAM,MAAM,GAAG,SAAS,CAAC,GAAG,CAAC,QAAQ,CAAC,CAAC;QACvC,IAAI,MAAM,KAAK,SAAS;YAAE,OAAO,MAAM,CAAC;QACxC,MAAM,IAAI,GAAG,WAAW,CAAC,QAAQ,CAAC,CAAC;QACnC,MAAM,OAAO,GAAG,CAAC,KAAK,IAAqC,EAAE;YAC3D,MAAM,GAAG,GAAG,CAAC,MAAM,MAAM,CAAC,2BAA2B,CAAC,CAErD,CAAC;YACF,OAAO,CAAC,MAAM,GAAG,CAAC,QAAQ,CAAC,iBAAiB,EAAE,IAAI,CAAC,MAAM,EAAE;gBACzD,KAAK,EAAE,IAAI,CAAC,KAAK;gBACjB,GAAG,CAAC,OAAO,CAAC,QAAQ,KAAK,SAAS,IAAI,EAAE,SAAS,EAAE,OAAO,CAAC,QAAQ,EAAE,CAAC;aACvE,CAAC,CAA2B,CAAC;QAChC,CAAC,CAAC,EAAE,CAAC;QACL,SAAS,CAAC,GAAG,CAAC,QAAQ,EAAE,OAAO,CAAC,CAAC;QACjC,OAAO,OAAO,CAAC;IACjB,CAAC;IAED,OAAO;QACL,QAAQ;QACR,KAAK,CAAC,MAAM,CAAC,KAAqB;YAChC,MAAM,QAAQ,GAAG,KAAK,CAAC,KAAmB,CAAC;YAC3C,IAAI,CAAC,SAAS,CAAC,QAAQ,CAAC,QAAQ,CAAC,EAAE,CAAC;gBAClC,MAAM,IAAI,KAAK,CACb,+CAA+C,UAAU,4BAA4B,KAAK,CAAC,KAAK,KAAK;oBACnG,cAAc,SAAS,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,QAAQ,GAAG,CACpD,CAAC;YACJ,CAAC;YACD,MAAM,IAAI,GAAG,WAAW,CAAC,QAAQ,CAAC,CAAC;YACnC,MAAM,CAAC,GAAG,MAAM,YAAY,CAAC,QAAQ,CAAC,CAAC;YACvC,MAAM,QAAQ,GAAG,KAAK,CAAC,QAAQ,CAAC,GAAG,CAAC,qBAAqB,CAAC,CAAC;YAC3D,MAAM,SAAS,GAAG,IAAI,CAAC,GAAG,EAAE,CAAC;YAE7B,MAAM,YAAY,GAAG,iBAAiB,CAAC,CAAC,EAAE,QAAQ,CAAC,CAAC;YAEpD,MAAM,MAAM,GAAG,MAAM,CAAC,CAAC,QAAQ,EAAE;gBAC/B,cAAc,EAAE,KAAK,CAAC,eAAe,IAAI,GAAG;gBAC5C,SAAS,EAAE,KAAK,CAAC,WAAW,KAAK,SAAS,IAAI,KAAK,CAAC,WAAW,GAAG,CAAC;gBACnE,GAAG,CAAC,KAAK,CAAC,WAAW,KAAK,SAAS;oBACjC,KAAK,CAAC,WAAW,GAAG,CAAC,IAAI;oBACvB,WAAW,EAAE,KAAK,CAAC,WAAW;iBAC/B,CAAC;gBACJ,kEAAkE;gBAClE,4BAA4B;aAC7B,CAAC,CAAC;YAEH,MAAM,UAAU,GAAG,IAAI,CAAC,GAAG,EAAE,GAAG,SAAS,CAAC;YAC1C,MAAM,WAAW,GAAG,MAAM,CAAC,CAAC,CAAC,CAAC;YAC9B,MAAM,IAAI,GAAG,WAAW,EAAE,cAAc,IAAI,EAAE,CAAC;YAC/C,MAAM,aAAa,GAAG,IAAI,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC;YAClC,MAAM,YAAY,GAAG,aAAa,EAAE,OAAO,IAAI,EAAE,CAAC;YAClD,MAAM,gBAAgB,GAAG,eAAe,CAAC,CAAC,EAAE,YAAY,CAAC,CAAC;YAC1D,KAAK,IAAI,CAAC;YAEV,OAAO;gBACL,OAAO,EAAE,EAAE,IAAI,EAAE,WAAW,EAAE,OAAO,EAAE,YAAY,EAAE;gBACrD,YAAY,EAAE,MAAM;gBACpB,KAAK,EAAE,EAAE,YAAY,EAAE,gBAAgB,EAAE;gBACzC,kEAAkE;gBAClE,OAAO,EAAE,CAAC;gBACV,UAAU;gBACV,QAAQ,EAAE,EAAE,EAAE,EAAE,UAAU,EAAE,KAAK,EAAE,QAAQ,EAAE;aAC9C,CAAC;QACJ,CAAC;KACF,CAAC;AACJ,CAAC;AAED;;;;;;;;GAQG;AACH,SAAS,qBAAqB,CAC5B,UAAkB,EAClB,SAAgC;IAEhC,MAAM,UAAU,GAAG,IAAI,GAAG,CAAS,CAAC,YAAY,CAAC,CAAC,CAAC;IACnD,KAAK,MAAM,GAAG,IAAI,SAAS,EAAE,CAAC;QAC5B,MAAM,IAAI,GAAG,WAAW,CAAC,GAAG,CAAC,CAAC;QAC9B,KAAK,MAAM,IAAI,IAAI,IAAI,CAAC,WAAW;YAAE,UAAU,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC;QAC1D,UAAU,CAAC,GAAG,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;IAC5B,CAAC;IACD,OAAO;QACL,EAAE,EAAE,UAAU;QACd,MAAM,EAAE,YAAY;QACpB,MAAM,EAAE,SAAS,CAAC,GAAG,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,cAAc,CAAC,GAAG,EAAE,WAAW,CAAC,GAAG,CAAC,CAAC,CAAC;QACrE,UAAU,EAAE,CAAC,GAAG,UAAU,CAAC;QAC3B,WAAW,EACT,0FAA0F;KAC7F,CAAC;AACJ,CAAC;AAED,SAAS,cAAc,CAAC,GAAe,EAAE,IAAe;IACtD,MAAM,QAAQ,GAAc,CAAC,WAAW,CAAC,CAAC;IAC1C,IAAI,IAAI,CAAC,OAAO;QAAE,QAAQ,CAAC,IAAI,CAAC,UAAU,EAAE,mBAAmB,CAAC,CAAC;IACjE,IAAI,IAAI,CAAC,aAAa,IAAI,MAAM;QAAE,QAAQ,CAAC,IAAI,CAAC,cAAc,CAAC,CAAC;IAChE,OAAO;QACL,IAAI,EAAE,GAAG;QACT,aAAa,EAAE,IAAI,CAAC,aAAa;QACjC,QAAQ;QACR,IAAI,EAAE,EAAE,oBAAoB,EAAE,CAAC,EAAE,wBAAwB,EAAE,CAAC,EAAE;QAC9D,WAAW,EAAE,SAAS,IAAI,CAAC,IAAI,gBAAgB,IAAI,CAAC,gBAAgB,QAAQ,IAAI,CAAC,aAAa,0BAA0B,IAAI,CAAC,MAAM,IAAI;KACxI,CAAC;AACJ,CAAC;AAED,6EAA6E;AAC7E,SAAS,qBAAqB,CAAC,CAAe;IAC5C,OAAO,EAAE,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,OAAO,EAAE,CAAC,CAAC,OAAO,EAAE,CAAC;AAC9C,CAAC;AAED;;;;GAIG;AACH,SAAS,iBAAiB,CACxB,CAAyB,EACzB,QAAsD;IAEtD,MAAM,aAAa,GAAG,CAAC,CAAC,SAAS,CAAC,mBAAmB,CAAC;IACtD,IAAI,OAAO,aAAa,KAAK,UAAU,EAAE,CAAC;QACxC,OAAO,QAAQ,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,eAAe,CAAC,CAAC,EAAE,CAAC,CAAC,OAAO,CAAC,EAAE,CAAC,CAAC,CAAC;IAC7E,CAAC;IACD,IAAI,CAAC;QACH,MAAM,GAAG,GAAG,aAAa,CAAC,QAAQ,EAAE;YAClC,QAAQ,EAAE,IAAI;YACd,qBAAqB,EAAE,IAAI;SAC5B,CAAC,CAAC;QACH,OAAO,KAAK,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC;IAC7C,CAAC;IAAC,MAAM,CAAC;QACP,OAAO,QAAQ,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,eAAe,CAAC,CAAC,EAAE,CAAC,CAAC,OAAO,CAAC,EAAE,CAAC,CAAC,CAAC;IAC7E,CAAC;AACH,CAAC;AAED,SAAS,eAAe,CAAC,CAAyB,EAAE,IAAY;IAC9D,IAAI,IAAI,CAAC,MAAM,KAAK,CAAC;QAAE,OAAO,CAAC,CAAC;IAChC,MAAM,MAAM,GAAG,CAAC,CAAC,SAAS,CAAC,MAAM,CAAC;IAClC,IAAI,OAAO,MAAM,KAAK,UAAU,EAAE,CAAC;QACjC,OAAO,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;IACpC,CAAC;IACD,IAAI,CAAC;QACH,OAAO,MAAM,CAAC,IAAI,CAAC,CAAC,MAAM,CAAC;IAC7B,CAAC;IAAC,MAAM,CAAC;QACP,OAAO,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;IACpC,CAAC;AACH,CAAC;AAGD,OAAO,EAAE,WAAW,EAAE,mBAAmB,EAAE,MAAM,aAAa,CAAC"}
|
package/package.json
CHANGED
|
@@ -1,7 +1,51 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@kindgi/adapter-model-in-process",
|
|
3
|
-
"version": "0.
|
|
4
|
-
"description": "
|
|
3
|
+
"version": "0.1.1",
|
|
4
|
+
"description": "In-process ModelProvider for @kindgi/capabilities. Runs small instruction-tuned LLMs inside the Node.js process via @huggingface/transformers (ONNX runtime; no external processes, no API keys). SmolLM2-360M by default (~273MB q4f16, Apache 2.0, tuned for function calling); SmolLM2-135M and Qwen3-0.6B selectable via the `models` option. prepareInProcessModel downloads a model ahead of first use and streams progress. Positioned as the cheapest tier for routing/classification/smoke-test workloads — not a replacement for hosted providers on real reasoning tasks.",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
|
-
"repository": {
|
|
7
|
-
|
|
6
|
+
"repository": {
|
|
7
|
+
"type": "git",
|
|
8
|
+
"url": "git+https://github.com/kindgi/kindgi-sdk.git",
|
|
9
|
+
"directory": "packages/adapters/model-in-process"
|
|
10
|
+
},
|
|
11
|
+
"homepage": "https://github.com/kindgi/kindgi-sdk/tree/main/packages/adapters/model-in-process#readme",
|
|
12
|
+
"bugs": {
|
|
13
|
+
"url": "https://github.com/kindgi/kindgi-sdk/issues"
|
|
14
|
+
},
|
|
15
|
+
"type": "module",
|
|
16
|
+
"main": "./dist/index.js",
|
|
17
|
+
"types": "./dist/index.d.ts",
|
|
18
|
+
"exports": {
|
|
19
|
+
".": {
|
|
20
|
+
"types": "./dist/index.d.ts",
|
|
21
|
+
"import": "./dist/index.js"
|
|
22
|
+
}
|
|
23
|
+
},
|
|
24
|
+
"files": [
|
|
25
|
+
"dist",
|
|
26
|
+
"src",
|
|
27
|
+
"README.md"
|
|
28
|
+
],
|
|
29
|
+
"dependencies": {
|
|
30
|
+
"@huggingface/transformers": "^4.3.0",
|
|
31
|
+
"@kindgi/capabilities": "0.1.1"
|
|
32
|
+
},
|
|
33
|
+
"engines": {
|
|
34
|
+
"node": ">=22.0.0"
|
|
35
|
+
},
|
|
36
|
+
"publishConfig": {
|
|
37
|
+
"access": "public",
|
|
38
|
+
"provenance": true
|
|
39
|
+
},
|
|
40
|
+
"devDependencies": {
|
|
41
|
+
"@types/node": "^22.10.5",
|
|
42
|
+
"typescript": "^5.7.3",
|
|
43
|
+
"vitest": "^2.1.8"
|
|
44
|
+
},
|
|
45
|
+
"scripts": {
|
|
46
|
+
"build": "tsc -p tsconfig.build.json",
|
|
47
|
+
"typecheck": "tsc --noEmit",
|
|
48
|
+
"test": "vitest run",
|
|
49
|
+
"clean": "rm -rf dist *.tsbuildinfo"
|
|
50
|
+
}
|
|
51
|
+
}
|
package/src/index.ts
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
// Copyright (C) 2026 Kindgi Inc.
|
|
3
|
+
|
|
4
|
+
export {
|
|
5
|
+
DEFAULT_LOCAL_MODEL,
|
|
6
|
+
MODEL_SPECS,
|
|
7
|
+
createInProcessModelProvider,
|
|
8
|
+
} from './provider.js';
|
|
9
|
+
export type { InProcessProviderOptions, LocalModel, ModelSpec } from './provider.js';
|
|
10
|
+
export { prepareInProcessModel } from './prepare.js';
|
|
11
|
+
export type { PrepareInProcessParams } from './prepare.js';
|
|
12
|
+
export type { ModelProvider, PrepareEvent } from '@kindgi/capabilities';
|
package/src/models.ts
ADDED
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
// Copyright (C) 2026 Kindgi Inc.
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Model registry — three selectable tiers, all Apache 2.0.
|
|
6
|
+
*
|
|
7
|
+
* Sizes are approximate q4f16 (4-bit quantization, float16 activations)
|
|
8
|
+
* download sizes. The first `invoke()` naming a model (or a
|
|
9
|
+
* `prepareInProcessModel` run for it) downloads the files into the
|
|
10
|
+
* transformers.js cache; later loads read from the cache. The default
|
|
11
|
+
* cache is transformers.js's own `env.cacheDir` — with
|
|
12
|
+
* `@huggingface/transformers` 4.x, a `.cache` directory inside that
|
|
13
|
+
* package's install location. The `cacheDir` option (provider and
|
|
14
|
+
* prepare) is passed to the pipeline as `cache_dir` and redirects the
|
|
15
|
+
* model files; transformers.js 4.3 can still write a model's
|
|
16
|
+
* `config.json` to its default cache.
|
|
17
|
+
*
|
|
18
|
+
* Positioning per tier:
|
|
19
|
+
* - `smollm2-135m` — ultra-light. Routing, tag extraction, short
|
|
20
|
+
* classification, simple JSON. Do not use for reasoning or math.
|
|
21
|
+
* - `smollm2-360m` — default. Instruction-tuned with function-calling,
|
|
22
|
+
* summarisation and rewriting data. Best quality-per-megabyte in the
|
|
23
|
+
* tier.
|
|
24
|
+
* - `qwen3-0.6b` — highest local quality. Larger download; better on
|
|
25
|
+
* reasoning / longer contexts.
|
|
26
|
+
*/
|
|
27
|
+
export type LocalModel = 'smollm2-135m' | 'smollm2-360m' | 'qwen3-0.6b';
|
|
28
|
+
|
|
29
|
+
export interface ModelSpec {
|
|
30
|
+
/** Hugging Face hub repo id — must have transformers.js-compatible ONNX assets. */
|
|
31
|
+
readonly hfName: string;
|
|
32
|
+
/** Approximate q4f16 download size in MB. */
|
|
33
|
+
readonly approxDownloadMb: number;
|
|
34
|
+
/** Model context window in tokens. */
|
|
35
|
+
readonly contextWindow: number;
|
|
36
|
+
/** Whether the model was tuned for tool / function-calling use. */
|
|
37
|
+
readonly toolUse: boolean;
|
|
38
|
+
/** Rough size tier — added to the provider's `attributes` and each model's `description`. */
|
|
39
|
+
readonly tier: 'ultra-light' | 'small' | 'medium';
|
|
40
|
+
/** Suitable-for hints for capability router preference matching. */
|
|
41
|
+
readonly suitableFor: readonly string[];
|
|
42
|
+
/** dtype passed to transformers.js pipeline. Balances quality vs size. */
|
|
43
|
+
readonly dtype: 'q4f16' | 'q8' | 'fp16' | 'fp32';
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
export const MODEL_SPECS: Record<LocalModel, ModelSpec> = {
|
|
47
|
+
'smollm2-135m': {
|
|
48
|
+
hfName: 'HuggingFaceTB/SmolLM2-135M-Instruct',
|
|
49
|
+
approxDownloadMb: 118,
|
|
50
|
+
contextWindow: 8192,
|
|
51
|
+
toolUse: false,
|
|
52
|
+
tier: 'ultra-light',
|
|
53
|
+
suitableFor: ['routing', 'classification', 'smoke-test', 'local', 'lower-cost'],
|
|
54
|
+
dtype: 'q4f16',
|
|
55
|
+
},
|
|
56
|
+
'smollm2-360m': {
|
|
57
|
+
hfName: 'HuggingFaceTB/SmolLM2-360M-Instruct',
|
|
58
|
+
approxDownloadMb: 273,
|
|
59
|
+
contextWindow: 8192,
|
|
60
|
+
toolUse: true,
|
|
61
|
+
tier: 'small',
|
|
62
|
+
suitableFor: [
|
|
63
|
+
'routing',
|
|
64
|
+
'classification',
|
|
65
|
+
'extraction',
|
|
66
|
+
'summarization',
|
|
67
|
+
'local',
|
|
68
|
+
'lower-cost',
|
|
69
|
+
],
|
|
70
|
+
dtype: 'q4f16',
|
|
71
|
+
},
|
|
72
|
+
'qwen3-0.6b': {
|
|
73
|
+
hfName: 'onnx-community/Qwen3-0.6B-ONNX',
|
|
74
|
+
approxDownloadMb: 550,
|
|
75
|
+
contextWindow: 32_768,
|
|
76
|
+
toolUse: true,
|
|
77
|
+
tier: 'medium',
|
|
78
|
+
suitableFor: ['reasoning', 'routing', 'classification', 'extraction', 'long-context', 'local'],
|
|
79
|
+
dtype: 'q4f16',
|
|
80
|
+
},
|
|
81
|
+
};
|
|
82
|
+
|
|
83
|
+
/** Default model when none supplied — best quality-per-MB in the tier. */
|
|
84
|
+
export const DEFAULT_LOCAL_MODEL: LocalModel = 'smollm2-360m';
|
package/src/prepare.ts
ADDED
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
// Copyright (C) 2026 Kindgi Inc.
|
|
3
|
+
|
|
4
|
+
import type { PrepareEvent } from '@kindgi/capabilities';
|
|
5
|
+
|
|
6
|
+
import { DEFAULT_LOCAL_MODEL, type LocalModel, MODEL_SPECS } from './models.js';
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* `prepareInProcessModel` — the in-process adapter's download / warmup
|
|
10
|
+
* step, in the shape of `AdapterFactoryEntry.prepare()` from
|
|
11
|
+
* `@kindgi/capabilities`. Streams download / warmup progress as
|
|
12
|
+
* framework `PrepareEvent`s so an HTTP SSE endpoint (such as
|
|
13
|
+
* `POST /v1/adapters/:adapterId/prepare` in `@kindgi/api`) can stream
|
|
14
|
+
* them to a client.
|
|
15
|
+
*
|
|
16
|
+
* What happens under the hood:
|
|
17
|
+
* 0. Yield a first `progress` event naming the model and its
|
|
18
|
+
* approximate download size, before any callback fires.
|
|
19
|
+
* 1. Dynamically import `@huggingface/transformers` (kept
|
|
20
|
+
* dynamic so shape tests don't touch the ONNX runtime).
|
|
21
|
+
* 2. Call `pipeline('text-generation', hfName, {dtype,
|
|
22
|
+
* progress_callback})`. First run: transformers.js downloads
|
|
23
|
+
* the model into its cache (`cacheDir` when set, passed as
|
|
24
|
+
* `cache_dir`; otherwise transformers.js's default, see
|
|
25
|
+
* `models.ts`). Subsequent runs hit the cache — the callback
|
|
26
|
+
* still fires but reports zero-byte "downloads."
|
|
27
|
+
* 3. The pipeline's progress_callback fires with events like
|
|
28
|
+
* `{status: 'initiate' | 'download' | 'progress' | 'done',
|
|
29
|
+
* name, file, progress?, loaded?, total?}`. We translate each
|
|
30
|
+
* into a `PrepareEvent` and enqueue it for the async iterator
|
|
31
|
+
* to yield.
|
|
32
|
+
* 4. When the pipeline promise resolves → emit `{kind: 'ready'}`
|
|
33
|
+
* and end the stream. On rejection → emit
|
|
34
|
+
* `{kind: 'error', message}` and end.
|
|
35
|
+
*
|
|
36
|
+
* Consumer pattern (e.g. inside an SSE route):
|
|
37
|
+
* ```ts
|
|
38
|
+
* for await (const event of prepareInProcessModel({model: 'smollm2-360m'})) {
|
|
39
|
+
* sendSse(event);
|
|
40
|
+
* }
|
|
41
|
+
* ```
|
|
42
|
+
*/
|
|
43
|
+
export interface PrepareInProcessParams {
|
|
44
|
+
readonly model?: LocalModel;
|
|
45
|
+
/** Directory for downloaded model files, passed to transformers.js as `cache_dir`. */
|
|
46
|
+
readonly cacheDir?: string;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
interface RawProgressEvent {
|
|
50
|
+
readonly status?: string;
|
|
51
|
+
readonly name?: string;
|
|
52
|
+
readonly file?: string;
|
|
53
|
+
readonly progress?: number;
|
|
54
|
+
readonly loaded?: number;
|
|
55
|
+
readonly total?: number;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
export async function* prepareInProcessModel(
|
|
59
|
+
params: PrepareInProcessParams = {},
|
|
60
|
+
): AsyncIterable<PrepareEvent> {
|
|
61
|
+
const modelKey = params.model ?? DEFAULT_LOCAL_MODEL;
|
|
62
|
+
const spec = MODEL_SPECS[modelKey];
|
|
63
|
+
|
|
64
|
+
const queue: PrepareEvent[] = [];
|
|
65
|
+
let resolveWait: (() => void) | undefined;
|
|
66
|
+
let ended = false;
|
|
67
|
+
|
|
68
|
+
const enqueue = (event: PrepareEvent): void => {
|
|
69
|
+
queue.push(event);
|
|
70
|
+
const w = resolveWait;
|
|
71
|
+
resolveWait = undefined;
|
|
72
|
+
w?.();
|
|
73
|
+
};
|
|
74
|
+
const finish = (event: PrepareEvent): void => {
|
|
75
|
+
queue.push(event);
|
|
76
|
+
ended = true;
|
|
77
|
+
const w = resolveWait;
|
|
78
|
+
resolveWait = undefined;
|
|
79
|
+
w?.();
|
|
80
|
+
};
|
|
81
|
+
|
|
82
|
+
const progressCallback = (raw: unknown): void => {
|
|
83
|
+
if (raw === null || typeof raw !== 'object') return;
|
|
84
|
+
const r = raw as RawProgressEvent;
|
|
85
|
+
switch (r.status) {
|
|
86
|
+
case 'initiate':
|
|
87
|
+
enqueue({
|
|
88
|
+
kind: 'progress',
|
|
89
|
+
message: `Fetching ${r.file ?? r.name ?? spec.hfName}`,
|
|
90
|
+
...(r.file !== undefined && { file: r.file }),
|
|
91
|
+
});
|
|
92
|
+
return;
|
|
93
|
+
case 'download':
|
|
94
|
+
enqueue({
|
|
95
|
+
kind: 'progress',
|
|
96
|
+
message: `Downloading ${r.file ?? r.name ?? spec.hfName}`,
|
|
97
|
+
...(r.file !== undefined && { file: r.file }),
|
|
98
|
+
});
|
|
99
|
+
return;
|
|
100
|
+
case 'progress':
|
|
101
|
+
if (typeof r.progress === 'number') {
|
|
102
|
+
enqueue({
|
|
103
|
+
kind: 'progress',
|
|
104
|
+
message: `Downloading ${r.file ?? r.name ?? spec.hfName}`,
|
|
105
|
+
ratio: Math.max(0, Math.min(1, r.progress / 100)),
|
|
106
|
+
...(typeof r.loaded === 'number' && { loadedBytes: r.loaded }),
|
|
107
|
+
...(typeof r.total === 'number' && { totalBytes: r.total }),
|
|
108
|
+
...(r.file !== undefined && { file: r.file }),
|
|
109
|
+
});
|
|
110
|
+
}
|
|
111
|
+
return;
|
|
112
|
+
case 'done':
|
|
113
|
+
enqueue({
|
|
114
|
+
kind: 'progress',
|
|
115
|
+
message: `Completed ${r.file ?? r.name ?? spec.hfName}`,
|
|
116
|
+
ratio: 1,
|
|
117
|
+
...(r.file !== undefined && { file: r.file }),
|
|
118
|
+
});
|
|
119
|
+
return;
|
|
120
|
+
default:
|
|
121
|
+
// 'ready' + any unknown status — ignore; the load-promise
|
|
122
|
+
// resolution is the authoritative "done" signal.
|
|
123
|
+
return;
|
|
124
|
+
}
|
|
125
|
+
};
|
|
126
|
+
|
|
127
|
+
// Kick off the load in parallel with the yield loop.
|
|
128
|
+
const loadPromise = (async () => {
|
|
129
|
+
const mod = (await import('@huggingface/transformers')) as unknown as {
|
|
130
|
+
pipeline: (task: string, model: string, opts: Record<string, unknown>) => Promise<unknown>;
|
|
131
|
+
};
|
|
132
|
+
return mod.pipeline('text-generation', spec.hfName, {
|
|
133
|
+
dtype: spec.dtype,
|
|
134
|
+
progress_callback: progressCallback,
|
|
135
|
+
...(params.cacheDir !== undefined && { cache_dir: params.cacheDir }),
|
|
136
|
+
});
|
|
137
|
+
})();
|
|
138
|
+
|
|
139
|
+
loadPromise.then(
|
|
140
|
+
() => finish({ kind: 'ready', message: `${spec.hfName} ready` }),
|
|
141
|
+
(err) =>
|
|
142
|
+
finish({
|
|
143
|
+
kind: 'error',
|
|
144
|
+
message: err instanceof Error ? err.message : String(err),
|
|
145
|
+
}),
|
|
146
|
+
);
|
|
147
|
+
|
|
148
|
+
// Prologue — tell the consumer what's about to happen even before
|
|
149
|
+
// the callback fires (transformers.js can silently sit on a cache
|
|
150
|
+
// check for a second or two before emitting anything).
|
|
151
|
+
yield {
|
|
152
|
+
kind: 'progress',
|
|
153
|
+
message: `Preparing ${spec.hfName} (~${spec.approxDownloadMb} MB, ${spec.tier} tier)`,
|
|
154
|
+
};
|
|
155
|
+
|
|
156
|
+
while (!ended || queue.length > 0) {
|
|
157
|
+
const next = queue.shift();
|
|
158
|
+
if (next !== undefined) {
|
|
159
|
+
yield next;
|
|
160
|
+
continue;
|
|
161
|
+
}
|
|
162
|
+
await new Promise<void>((resolve) => {
|
|
163
|
+
resolveWait = resolve;
|
|
164
|
+
});
|
|
165
|
+
}
|
|
166
|
+
}
|