@kindgi/adapter-model-in-process 0.0.0-bootstrap.0 → 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,165 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ // Copyright (C) 2026 Kindgi Inc.
3
+ import { DEFAULT_LOCAL_MODEL, MODEL_SPECS } from './models.js';
4
+ /**
5
+ * Create an in-process `ModelProvider` backed by `@huggingface/transformers`.
6
+ *
7
+ * Pipelines are loaded lazily per model on the first `invoke()` call
8
+ * that names them — construction of the provider is cheap. Model
9
+ * weights download into the transformers.js cache (or `cacheDir`) on
10
+ * first use; later loads read from the cache.
11
+ *
12
+ * Cost is always `0` USD (no external service). Resource-usage recording
13
+ * still tracks token counts + duration so operators can see the local
14
+ * model's real load in aggregate reports.
15
+ *
16
+ * The returned provider is safe to share process-wide; each underlying
17
+ * pipeline serialises its own requests inside the ONNX runtime.
18
+ */
19
+ export function createInProcessModelProvider(options = {}) {
20
+ const modelKeys = options.models ?? [DEFAULT_LOCAL_MODEL];
21
+ if (modelKeys.length === 0) {
22
+ throw new Error('createInProcessModelProvider: `models` must not be empty.');
23
+ }
24
+ const providerId = options.providerId ?? `in-process/${modelKeys.join('+')}`;
25
+ const metadata = buildProviderMetadata(providerId, modelKeys);
26
+ // Cache loaded pipelines per model key. Uses a Promise to dedupe
27
+ // concurrent first-invocations on the same model.
28
+ const pipelines = new Map();
29
+ function loadPipeline(modelKey) {
30
+ const cached = pipelines.get(modelKey);
31
+ if (cached !== undefined)
32
+ return cached;
33
+ const spec = MODEL_SPECS[modelKey];
34
+ const loading = (async () => {
35
+ const mod = (await import('@huggingface/transformers'));
36
+ return (await mod.pipeline('text-generation', spec.hfName, {
37
+ dtype: spec.dtype,
38
+ ...(options.cacheDir !== undefined && { cache_dir: options.cacheDir }),
39
+ }));
40
+ })();
41
+ pipelines.set(modelKey, loading);
42
+ return loading;
43
+ }
44
+ return {
45
+ metadata,
46
+ async invoke(input) {
47
+ const modelKey = input.model;
48
+ if (!modelKeys.includes(modelKey)) {
49
+ throw new Error(`@kindgi/adapter-model-in-process: provider "${providerId}" does not expose model "${input.model}". ` +
50
+ `Available: ${modelKeys.join(', ') || '<none>'}.`);
51
+ }
52
+ const spec = MODEL_SPECS[modelKey];
53
+ const p = await loadPipeline(modelKey);
54
+ const messages = input.messages.map(toChatTemplateMessage);
55
+ const startedAt = Date.now();
56
+ const promptTokens = countPromptTokens(p, messages);
57
+ const output = await p(messages, {
58
+ max_new_tokens: input.maxOutputTokens ?? 512,
59
+ do_sample: input.temperature !== undefined && input.temperature > 0,
60
+ ...(input.temperature !== undefined &&
61
+ input.temperature > 0 && {
62
+ temperature: input.temperature,
63
+ }),
64
+ // No TextStreamer: `invoke` waits for the complete generation and
65
+ // returns it in one result.
66
+ });
67
+ const durationMs = Date.now() - startedAt;
68
+ const firstResult = output[0];
69
+ const chat = firstResult?.generated_text ?? [];
70
+ const assistantTurn = chat.at(-1);
71
+ const responseText = assistantTurn?.content ?? '';
72
+ const completionTokens = countTextTokens(p, responseText);
73
+ void spec;
74
+ return {
75
+ message: { role: 'assistant', content: responseText },
76
+ finishReason: 'stop',
77
+ usage: { promptTokens, completionTokens },
78
+ // In-process = zero direct USD cost. Ledger still records tokens.
79
+ costUsd: 0,
80
+ durationMs,
81
+ provider: { id: providerId, model: modelKey },
82
+ };
83
+ },
84
+ };
85
+ }
86
+ /**
87
+ * Build `ProviderMetadata` from the set of local models this provider
88
+ * exposes. Each `LocalModel` key maps to a `ModelInfo` entry —
89
+ * `ModelInfo.name` is the key itself (short label), not the fully-
90
+ * qualified Hugging Face model id, so provider listings (such as
91
+ * `GET /v1/providers`) show `smollm2-360m` alongside vendor model ids.
92
+ * Provider-level `attributes` are the union of every model's
93
+ * `suitableFor` + `tier`, deduped, plus a marker `in-process`.
94
+ */
95
+ function buildProviderMetadata(providerId, modelKeys) {
96
+ const attributes = new Set(['in-process']);
97
+ for (const key of modelKeys) {
98
+ const spec = MODEL_SPECS[key];
99
+ for (const attr of spec.suitableFor)
100
+ attributes.add(attr);
101
+ attributes.add(spec.tier);
102
+ }
103
+ return {
104
+ id: providerId,
105
+ region: 'in-process',
106
+ models: modelKeys.map((key) => buildModelInfo(key, MODEL_SPECS[key])),
107
+ attributes: [...attributes],
108
+ description: 'Local models via @huggingface/transformers — routing / classification / smoke-test tier.',
109
+ };
110
+ }
111
+ function buildModelInfo(key, spec) {
112
+ const features = ['streaming'];
113
+ if (spec.toolUse)
114
+ features.push('tool-use', 'structured-output');
115
+ if (spec.contextWindow >= 32_000)
116
+ features.push('long-context');
117
+ return {
118
+ name: key,
119
+ contextWindow: spec.contextWindow,
120
+ features,
121
+ cost: { promptUsdPer1kTokens: 0, completionUsdPer1kTokens: 0 },
122
+ description: `Local ${spec.tier}-tier model (${spec.approxDownloadMb} MB, ${spec.contextWindow}-token context, HF id: ${spec.hfName}).`,
123
+ };
124
+ }
125
+ /** Translate Kindgi ModelMessage → transformers.js Chat template message. */
126
+ function toChatTemplateMessage(m) {
127
+ return { role: m.role, content: m.content };
128
+ }
129
+ /**
130
+ * Token counting via the pipeline's tokenizer. Called before invocation
131
+ * for prompt tokens and after for completion tokens. The tokenizer is
132
+ * async-loaded with the pipeline; this must run after `loadPipeline()`.
133
+ */
134
+ function countPromptTokens(p, messages) {
135
+ const applyTemplate = p.tokenizer.apply_chat_template;
136
+ if (typeof applyTemplate !== 'function') {
137
+ return messages.reduce((sum, m) => sum + countTextTokens(p, m.content), 0);
138
+ }
139
+ try {
140
+ const ids = applyTemplate(messages, {
141
+ tokenize: true,
142
+ add_generation_prompt: true,
143
+ });
144
+ return Array.isArray(ids) ? ids.length : 0;
145
+ }
146
+ catch {
147
+ return messages.reduce((sum, m) => sum + countTextTokens(p, m.content), 0);
148
+ }
149
+ }
150
+ function countTextTokens(p, text) {
151
+ if (text.length === 0)
152
+ return 0;
153
+ const encode = p.tokenizer.encode;
154
+ if (typeof encode !== 'function') {
155
+ return Math.ceil(text.length / 4);
156
+ }
157
+ try {
158
+ return encode(text).length;
159
+ }
160
+ catch {
161
+ return Math.ceil(text.length / 4);
162
+ }
163
+ }
164
+ export { MODEL_SPECS, DEFAULT_LOCAL_MODEL } from './models.js';
165
+ //# sourceMappingURL=provider.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"provider.js","sourceRoot":"","sources":["../src/provider.ts"],"names":[],"mappings":"AAAA,sCAAsC;AACtC,iCAAiC;AAYjC,OAAO,EAAE,mBAAmB,EAAmB,WAAW,EAAkB,MAAM,aAAa,CAAC;AA2ChG;;;;;;;;;;;;;;GAcG;AACH,MAAM,UAAU,4BAA4B,CAC1C,UAAoC,EAAE;IAEtC,MAAM,SAAS,GAAG,OAAO,CAAC,MAAM,IAAI,CAAC,mBAAmB,CAAC,CAAC;IAC1D,IAAI,SAAS,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QAC3B,MAAM,IAAI,KAAK,CAAC,2DAA2D,CAAC,CAAC;IAC/E,CAAC;IACD,MAAM,UAAU,GAAG,OAAO,CAAC,UAAU,IAAI,cAAc,SAAS,CAAC,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC;IAC7E,MAAM,QAAQ,GAAqB,qBAAqB,CAAC,UAAU,EAAE,SAAS,CAAC,CAAC;IAChF,iEAAiE;IACjE,kDAAkD;IAClD,MAAM,SAAS,GAAG,IAAI,GAAG,EAA+C,CAAC;IAEzE,SAAS,YAAY,CAAC,QAAoB;QACxC,MAAM,MAAM,GAAG,SAAS,CAAC,GAAG,CAAC,QAAQ,CAAC,CAAC;QACvC,IAAI,MAAM,KAAK,SAAS;YAAE,OAAO,MAAM,CAAC;QACxC,MAAM,IAAI,GAAG,WAAW,CAAC,QAAQ,CAAC,CAAC;QACnC,MAAM,OAAO,GAAG,CAAC,KAAK,IAAqC,EAAE;YAC3D,MAAM,GAAG,GAAG,CAAC,MAAM,MAAM,CAAC,2BAA2B,CAAC,CAErD,CAAC;YACF,OAAO,CAAC,MAAM,GAAG,CAAC,QAAQ,CAAC,iBAAiB,EAAE,IAAI,CAAC,MAAM,EAAE;gBACzD,KAAK,EAAE,IAAI,CAAC,KAAK;gBACjB,GAAG,CAAC,OAAO,CAAC,QAAQ,KAAK,SAAS,IAAI,EAAE,SAAS,EAAE,OAAO,CAAC,QAAQ,EAAE,CAAC;aACvE,CAAC,CAA2B,CAAC;QAChC,CAAC,CAAC,EAAE,CAAC;QACL,SAAS,CAAC,GAAG,CAAC,QAAQ,EAAE,OAAO,CAAC,CAAC;QACjC,OAAO,OAAO,CAAC;IACjB,CAAC;IAED,OAAO;QACL,QAAQ;QACR,KAAK,CAAC,MAAM,CAAC,KAAqB;YAChC,MAAM,QAAQ,GAAG,KAAK,CAAC,KAAmB,CAAC;YAC3C,IAAI,CAAC,SAAS,CAAC,QAAQ,CAAC,QAAQ,CAAC,EAAE,CAAC;gBAClC,MAAM,IAAI,KAAK,CACb,+CAA+C,UAAU,4BAA4B,KAAK,CAAC,KAAK,KAAK;oBACnG,cAAc,SAAS,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,QAAQ,GAAG,CACpD,CAAC;YACJ,CAAC;YACD,MAAM,IAAI,GAAG,WAAW,CAAC,QAAQ,CAAC,CAAC;YACnC,MAAM,CAAC,GAAG,MAAM,YAAY,CAAC,QAAQ,CAAC,CAAC;YACvC,MAAM,QAAQ,GAAG,KAAK,CAAC,QAAQ,CAAC,GAAG,CAAC,qBAAqB,CAAC,CAAC;YAC3D,MAAM,SAAS,GAAG,IAAI,CAAC,GAAG,EAAE,CAAC;YAE7B,MAAM,YAAY,GAAG,iBAAiB,CAAC,CAAC,EAAE,QAAQ,CAAC,CAAC;YAEpD,MAAM,MAAM,GAAG,MAAM,CAAC,CAAC,QAAQ,EAAE;gBAC/B,cAAc,EAAE,KAAK,CAAC,eAAe,IAAI,GAAG;gBAC5C,SAAS,EAAE,KAAK,CAAC,WAAW,KAAK,SAAS,IAAI,KAAK,CAAC,WAAW,GAAG,CAAC;gBACnE,GAAG,CAAC,KAAK,CAAC,WAAW,KAAK,SAAS;oBACjC,KAAK,CAAC,WAAW,GAAG,CAAC,IAAI;oBACvB,WAAW,EAAE,KAAK,CAAC,WAAW;iBAC/B,CAAC;gBACJ,kEAAkE;gBAClE,4BAA4B;aAC7B,CAAC,CAAC;YAEH,MAAM,UAAU,GAAG,IAAI,CAAC,GAAG,EAAE,GAAG,SAAS,CAAC;YAC1C,MAAM,WAAW,GAAG,MAAM,CAAC,CAAC,CAAC,CAAC;YAC9B,MAAM,IAAI,GAAG,WAAW,EAAE,cAAc,IAAI,EAAE,CAAC;YAC/C,MAAM,aAAa,GAAG,IAAI,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC;YAClC,MAAM,YAAY,GAAG,aAAa,EAAE,OAAO,IAAI,EAAE,CAAC;YAClD,MAAM,gBAAgB,GAAG,eAAe,CAAC,CAAC,EAAE,YAAY,CAAC,CAAC;YAC1D,KAAK,IAAI,CAAC;YAEV,OAAO;gBACL,OAAO,EAAE,EAAE,IAAI,EAAE,WAAW,EAAE,OAAO,EAAE,YAAY,EAAE;gBACrD,YAAY,EAAE,MAAM;gBACpB,KAAK,EAAE,EAAE,YAAY,EAAE,gBAAgB,EAAE;gBACzC,kEAAkE;gBAClE,OAAO,EAAE,CAAC;gBACV,UAAU;gBACV,QAAQ,EAAE,EAAE,EAAE,EAAE,UAAU,EAAE,KAAK,EAAE,QAAQ,EAAE;aAC9C,CAAC;QACJ,CAAC;KACF,CAAC;AACJ,CAAC;AAED;;;;;;;;GAQG;AACH,SAAS,qBAAqB,CAC5B,UAAkB,EAClB,SAAgC;IAEhC,MAAM,UAAU,GAAG,IAAI,GAAG,CAAS,CAAC,YAAY,CAAC,CAAC,CAAC;IACnD,KAAK,MAAM,GAAG,IAAI,SAAS,EAAE,CAAC;QAC5B,MAAM,IAAI,GAAG,WAAW,CAAC,GAAG,CAAC,CAAC;QAC9B,KAAK,MAAM,IAAI,IAAI,IAAI,CAAC,WAAW;YAAE,UAAU,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC;QAC1D,UAAU,CAAC,GAAG,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;IAC5B,CAAC;IACD,OAAO;QACL,EAAE,EAAE,UAAU;QACd,MAAM,EAAE,YAAY;QACpB,MAAM,EAAE,SAAS,CAAC,GAAG,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,cAAc,CAAC,GAAG,EAAE,WAAW,CAAC,GAAG,CAAC,CAAC,CAAC;QACrE,UAAU,EAAE,CAAC,GAAG,UAAU,CAAC;QAC3B,WAAW,EACT,0FAA0F;KAC7F,CAAC;AACJ,CAAC;AAED,SAAS,cAAc,CAAC,GAAe,EAAE,IAAe;IACtD,MAAM,QAAQ,GAAc,CAAC,WAAW,CAAC,CAAC;IAC1C,IAAI,IAAI,CAAC,OAAO;QAAE,QAAQ,CAAC,IAAI,CAAC,UAAU,EAAE,mBAAmB,CAAC,CAAC;IACjE,IAAI,IAAI,CAAC,aAAa,IAAI,MAAM;QAAE,QAAQ,CAAC,IAAI,CAAC,cAAc,CAAC,CAAC;IAChE,OAAO;QACL,IAAI,EAAE,GAAG;QACT,aAAa,EAAE,IAAI,CAAC,aAAa;QACjC,QAAQ;QACR,IAAI,EAAE,EAAE,oBAAoB,EAAE,CAAC,EAAE,wBAAwB,EAAE,CAAC,EAAE;QAC9D,WAAW,EAAE,SAAS,IAAI,CAAC,IAAI,gBAAgB,IAAI,CAAC,gBAAgB,QAAQ,IAAI,CAAC,aAAa,0BAA0B,IAAI,CAAC,MAAM,IAAI;KACxI,CAAC;AACJ,CAAC;AAED,6EAA6E;AAC7E,SAAS,qBAAqB,CAAC,CAAe;IAC5C,OAAO,EAAE,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,OAAO,EAAE,CAAC,CAAC,OAAO,EAAE,CAAC;AAC9C,CAAC;AAED;;;;GAIG;AACH,SAAS,iBAAiB,CACxB,CAAyB,EACzB,QAAsD;IAEtD,MAAM,aAAa,GAAG,CAAC,CAAC,SAAS,CAAC,mBAAmB,CAAC;IACtD,IAAI,OAAO,aAAa,KAAK,UAAU,EAAE,CAAC;QACxC,OAAO,QAAQ,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,eAAe,CAAC,CAAC,EAAE,CAAC,CAAC,OAAO,CAAC,EAAE,CAAC,CAAC,CAAC;IAC7E,CAAC;IACD,IAAI,CAAC;QACH,MAAM,GAAG,GAAG,aAAa,CAAC,QAAQ,EAAE;YAClC,QAAQ,EAAE,IAAI;YACd,qBAAqB,EAAE,IAAI;SAC5B,CAAC,CAAC;QACH,OAAO,KAAK,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC;IAC7C,CAAC;IAAC,MAAM,CAAC;QACP,OAAO,QAAQ,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,eAAe,CAAC,CAAC,EAAE,CAAC,CAAC,OAAO,CAAC,EAAE,CAAC,CAAC,CAAC;IAC7E,CAAC;AACH,CAAC;AAED,SAAS,eAAe,CAAC,CAAyB,EAAE,IAAY;IAC9D,IAAI,IAAI,CAAC,MAAM,KAAK,CAAC;QAAE,OAAO,CAAC,CAAC;IAChC,MAAM,MAAM,GAAG,CAAC,CAAC,SAAS,CAAC,MAAM,CAAC;IAClC,IAAI,OAAO,MAAM,KAAK,UAAU,EAAE,CAAC;QACjC,OAAO,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;IACpC,CAAC;IACD,IAAI,CAAC;QACH,OAAO,MAAM,CAAC,IAAI,CAAC,CAAC,MAAM,CAAC;IAC7B,CAAC;IAAC,MAAM,CAAC;QACP,OAAO,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;IACpC,CAAC;AACH,CAAC;AAGD,OAAO,EAAE,WAAW,EAAE,mBAAmB,EAAE,MAAM,aAAa,CAAC"}
package/package.json CHANGED
@@ -1,7 +1,51 @@
1
1
  {
2
2
  "name": "@kindgi/adapter-model-in-process",
3
- "version": "0.0.0-bootstrap.0",
4
- "description": "Placeholder so a trusted publisher can be attached. Releases are published from https://github.com/kindgi/kindgi-sdk with provenance; use 0.1.0 or later.",
3
+ "version": "0.1.1",
4
+ "description": "In-process ModelProvider for @kindgi/capabilities. Runs small instruction-tuned LLMs inside the Node.js process via @huggingface/transformers (ONNX runtime; no external processes, no API keys). SmolLM2-360M by default (~273MB q4f16, Apache 2.0, tuned for function calling); SmolLM2-135M and Qwen3-0.6B selectable via the `models` option. prepareInProcessModel downloads a model ahead of first use and streams progress. Positioned as the cheapest tier for routing/classification/smoke-test workloads — not a replacement for hosted providers on real reasoning tasks.",
5
5
  "license": "Apache-2.0",
6
- "repository": { "type": "git", "url": "git+https://github.com/kindgi/kindgi-sdk.git" }
7
- }
6
+ "repository": {
7
+ "type": "git",
8
+ "url": "git+https://github.com/kindgi/kindgi-sdk.git",
9
+ "directory": "packages/adapters/model-in-process"
10
+ },
11
+ "homepage": "https://github.com/kindgi/kindgi-sdk/tree/main/packages/adapters/model-in-process#readme",
12
+ "bugs": {
13
+ "url": "https://github.com/kindgi/kindgi-sdk/issues"
14
+ },
15
+ "type": "module",
16
+ "main": "./dist/index.js",
17
+ "types": "./dist/index.d.ts",
18
+ "exports": {
19
+ ".": {
20
+ "types": "./dist/index.d.ts",
21
+ "import": "./dist/index.js"
22
+ }
23
+ },
24
+ "files": [
25
+ "dist",
26
+ "src",
27
+ "README.md"
28
+ ],
29
+ "dependencies": {
30
+ "@huggingface/transformers": "^4.3.0",
31
+ "@kindgi/capabilities": "0.1.1"
32
+ },
33
+ "engines": {
34
+ "node": ">=22.0.0"
35
+ },
36
+ "publishConfig": {
37
+ "access": "public",
38
+ "provenance": true
39
+ },
40
+ "devDependencies": {
41
+ "@types/node": "^22.10.5",
42
+ "typescript": "^5.7.3",
43
+ "vitest": "^2.1.8"
44
+ },
45
+ "scripts": {
46
+ "build": "tsc -p tsconfig.build.json",
47
+ "typecheck": "tsc --noEmit",
48
+ "test": "vitest run",
49
+ "clean": "rm -rf dist *.tsbuildinfo"
50
+ }
51
+ }
package/src/index.ts ADDED
@@ -0,0 +1,12 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ // Copyright (C) 2026 Kindgi Inc.
3
+
4
+ export {
5
+ DEFAULT_LOCAL_MODEL,
6
+ MODEL_SPECS,
7
+ createInProcessModelProvider,
8
+ } from './provider.js';
9
+ export type { InProcessProviderOptions, LocalModel, ModelSpec } from './provider.js';
10
+ export { prepareInProcessModel } from './prepare.js';
11
+ export type { PrepareInProcessParams } from './prepare.js';
12
+ export type { ModelProvider, PrepareEvent } from '@kindgi/capabilities';
package/src/models.ts ADDED
@@ -0,0 +1,84 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ // Copyright (C) 2026 Kindgi Inc.
3
+
4
+ /**
5
+ * Model registry — three selectable tiers, all Apache 2.0.
6
+ *
7
+ * Sizes are approximate q4f16 (4-bit quantization, float16 activations)
8
+ * download sizes. The first `invoke()` naming a model (or a
9
+ * `prepareInProcessModel` run for it) downloads the files into the
10
+ * transformers.js cache; later loads read from the cache. The default
11
+ * cache is transformers.js's own `env.cacheDir` — with
12
+ * `@huggingface/transformers` 4.x, a `.cache` directory inside that
13
+ * package's install location. The `cacheDir` option (provider and
14
+ * prepare) is passed to the pipeline as `cache_dir` and redirects the
15
+ * model files; transformers.js 4.3 can still write a model's
16
+ * `config.json` to its default cache.
17
+ *
18
+ * Positioning per tier:
19
+ * - `smollm2-135m` — ultra-light. Routing, tag extraction, short
20
+ * classification, simple JSON. Do not use for reasoning or math.
21
+ * - `smollm2-360m` — default. Instruction-tuned with function-calling,
22
+ * summarisation and rewriting data. Best quality-per-megabyte in the
23
+ * tier.
24
+ * - `qwen3-0.6b` — highest local quality. Larger download; better on
25
+ * reasoning / longer contexts.
26
+ */
27
+ export type LocalModel = 'smollm2-135m' | 'smollm2-360m' | 'qwen3-0.6b';
28
+
29
+ export interface ModelSpec {
30
+ /** Hugging Face hub repo id — must have transformers.js-compatible ONNX assets. */
31
+ readonly hfName: string;
32
+ /** Approximate q4f16 download size in MB. */
33
+ readonly approxDownloadMb: number;
34
+ /** Model context window in tokens. */
35
+ readonly contextWindow: number;
36
+ /** Whether the model was tuned for tool / function-calling use. */
37
+ readonly toolUse: boolean;
38
+ /** Rough size tier — added to the provider's `attributes` and each model's `description`. */
39
+ readonly tier: 'ultra-light' | 'small' | 'medium';
40
+ /** Suitable-for hints for capability router preference matching. */
41
+ readonly suitableFor: readonly string[];
42
+ /** dtype passed to transformers.js pipeline. Balances quality vs size. */
43
+ readonly dtype: 'q4f16' | 'q8' | 'fp16' | 'fp32';
44
+ }
45
+
46
+ export const MODEL_SPECS: Record<LocalModel, ModelSpec> = {
47
+ 'smollm2-135m': {
48
+ hfName: 'HuggingFaceTB/SmolLM2-135M-Instruct',
49
+ approxDownloadMb: 118,
50
+ contextWindow: 8192,
51
+ toolUse: false,
52
+ tier: 'ultra-light',
53
+ suitableFor: ['routing', 'classification', 'smoke-test', 'local', 'lower-cost'],
54
+ dtype: 'q4f16',
55
+ },
56
+ 'smollm2-360m': {
57
+ hfName: 'HuggingFaceTB/SmolLM2-360M-Instruct',
58
+ approxDownloadMb: 273,
59
+ contextWindow: 8192,
60
+ toolUse: true,
61
+ tier: 'small',
62
+ suitableFor: [
63
+ 'routing',
64
+ 'classification',
65
+ 'extraction',
66
+ 'summarization',
67
+ 'local',
68
+ 'lower-cost',
69
+ ],
70
+ dtype: 'q4f16',
71
+ },
72
+ 'qwen3-0.6b': {
73
+ hfName: 'onnx-community/Qwen3-0.6B-ONNX',
74
+ approxDownloadMb: 550,
75
+ contextWindow: 32_768,
76
+ toolUse: true,
77
+ tier: 'medium',
78
+ suitableFor: ['reasoning', 'routing', 'classification', 'extraction', 'long-context', 'local'],
79
+ dtype: 'q4f16',
80
+ },
81
+ };
82
+
83
+ /** Default model when none supplied — best quality-per-MB in the tier. */
84
+ export const DEFAULT_LOCAL_MODEL: LocalModel = 'smollm2-360m';
package/src/prepare.ts ADDED
@@ -0,0 +1,166 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ // Copyright (C) 2026 Kindgi Inc.
3
+
4
+ import type { PrepareEvent } from '@kindgi/capabilities';
5
+
6
+ import { DEFAULT_LOCAL_MODEL, type LocalModel, MODEL_SPECS } from './models.js';
7
+
8
+ /**
9
+ * `prepareInProcessModel` — the in-process adapter's download / warmup
10
+ * step, in the shape of `AdapterFactoryEntry.prepare()` from
11
+ * `@kindgi/capabilities`. Streams download / warmup progress as
12
+ * framework `PrepareEvent`s so an HTTP SSE endpoint (such as
13
+ * `POST /v1/adapters/:adapterId/prepare` in `@kindgi/api`) can stream
14
+ * them to a client.
15
+ *
16
+ * What happens under the hood:
17
+ * 0. Yield a first `progress` event naming the model and its
18
+ * approximate download size, before any callback fires.
19
+ * 1. Dynamically import `@huggingface/transformers` (kept
20
+ * dynamic so shape tests don't touch the ONNX runtime).
21
+ * 2. Call `pipeline('text-generation', hfName, {dtype,
22
+ * progress_callback})`. First run: transformers.js downloads
23
+ * the model into its cache (`cacheDir` when set, passed as
24
+ * `cache_dir`; otherwise transformers.js's default, see
25
+ * `models.ts`). Subsequent runs hit the cache — the callback
26
+ * still fires but reports zero-byte "downloads."
27
+ * 3. The pipeline's progress_callback fires with events like
28
+ * `{status: 'initiate' | 'download' | 'progress' | 'done',
29
+ * name, file, progress?, loaded?, total?}`. We translate each
30
+ * into a `PrepareEvent` and enqueue it for the async iterator
31
+ * to yield.
32
+ * 4. When the pipeline promise resolves → emit `{kind: 'ready'}`
33
+ * and end the stream. On rejection → emit
34
+ * `{kind: 'error', message}` and end.
35
+ *
36
+ * Consumer pattern (e.g. inside an SSE route):
37
+ * ```ts
38
+ * for await (const event of prepareInProcessModel({model: 'smollm2-360m'})) {
39
+ * sendSse(event);
40
+ * }
41
+ * ```
42
+ */
43
+ export interface PrepareInProcessParams {
44
+ readonly model?: LocalModel;
45
+ /** Directory for downloaded model files, passed to transformers.js as `cache_dir`. */
46
+ readonly cacheDir?: string;
47
+ }
48
+
49
+ interface RawProgressEvent {
50
+ readonly status?: string;
51
+ readonly name?: string;
52
+ readonly file?: string;
53
+ readonly progress?: number;
54
+ readonly loaded?: number;
55
+ readonly total?: number;
56
+ }
57
+
58
+ export async function* prepareInProcessModel(
59
+ params: PrepareInProcessParams = {},
60
+ ): AsyncIterable<PrepareEvent> {
61
+ const modelKey = params.model ?? DEFAULT_LOCAL_MODEL;
62
+ const spec = MODEL_SPECS[modelKey];
63
+
64
+ const queue: PrepareEvent[] = [];
65
+ let resolveWait: (() => void) | undefined;
66
+ let ended = false;
67
+
68
+ const enqueue = (event: PrepareEvent): void => {
69
+ queue.push(event);
70
+ const w = resolveWait;
71
+ resolveWait = undefined;
72
+ w?.();
73
+ };
74
+ const finish = (event: PrepareEvent): void => {
75
+ queue.push(event);
76
+ ended = true;
77
+ const w = resolveWait;
78
+ resolveWait = undefined;
79
+ w?.();
80
+ };
81
+
82
+ const progressCallback = (raw: unknown): void => {
83
+ if (raw === null || typeof raw !== 'object') return;
84
+ const r = raw as RawProgressEvent;
85
+ switch (r.status) {
86
+ case 'initiate':
87
+ enqueue({
88
+ kind: 'progress',
89
+ message: `Fetching ${r.file ?? r.name ?? spec.hfName}`,
90
+ ...(r.file !== undefined && { file: r.file }),
91
+ });
92
+ return;
93
+ case 'download':
94
+ enqueue({
95
+ kind: 'progress',
96
+ message: `Downloading ${r.file ?? r.name ?? spec.hfName}`,
97
+ ...(r.file !== undefined && { file: r.file }),
98
+ });
99
+ return;
100
+ case 'progress':
101
+ if (typeof r.progress === 'number') {
102
+ enqueue({
103
+ kind: 'progress',
104
+ message: `Downloading ${r.file ?? r.name ?? spec.hfName}`,
105
+ ratio: Math.max(0, Math.min(1, r.progress / 100)),
106
+ ...(typeof r.loaded === 'number' && { loadedBytes: r.loaded }),
107
+ ...(typeof r.total === 'number' && { totalBytes: r.total }),
108
+ ...(r.file !== undefined && { file: r.file }),
109
+ });
110
+ }
111
+ return;
112
+ case 'done':
113
+ enqueue({
114
+ kind: 'progress',
115
+ message: `Completed ${r.file ?? r.name ?? spec.hfName}`,
116
+ ratio: 1,
117
+ ...(r.file !== undefined && { file: r.file }),
118
+ });
119
+ return;
120
+ default:
121
+ // 'ready' + any unknown status — ignore; the load-promise
122
+ // resolution is the authoritative "done" signal.
123
+ return;
124
+ }
125
+ };
126
+
127
+ // Kick off the load in parallel with the yield loop.
128
+ const loadPromise = (async () => {
129
+ const mod = (await import('@huggingface/transformers')) as unknown as {
130
+ pipeline: (task: string, model: string, opts: Record<string, unknown>) => Promise<unknown>;
131
+ };
132
+ return mod.pipeline('text-generation', spec.hfName, {
133
+ dtype: spec.dtype,
134
+ progress_callback: progressCallback,
135
+ ...(params.cacheDir !== undefined && { cache_dir: params.cacheDir }),
136
+ });
137
+ })();
138
+
139
+ loadPromise.then(
140
+ () => finish({ kind: 'ready', message: `${spec.hfName} ready` }),
141
+ (err) =>
142
+ finish({
143
+ kind: 'error',
144
+ message: err instanceof Error ? err.message : String(err),
145
+ }),
146
+ );
147
+
148
+ // Prologue — tell the consumer what's about to happen even before
149
+ // the callback fires (transformers.js can silently sit on a cache
150
+ // check for a second or two before emitting anything).
151
+ yield {
152
+ kind: 'progress',
153
+ message: `Preparing ${spec.hfName} (~${spec.approxDownloadMb} MB, ${spec.tier} tier)`,
154
+ };
155
+
156
+ while (!ended || queue.length > 0) {
157
+ const next = queue.shift();
158
+ if (next !== undefined) {
159
+ yield next;
160
+ continue;
161
+ }
162
+ await new Promise<void>((resolve) => {
163
+ resolveWait = resolve;
164
+ });
165
+ }
166
+ }