@camstack/addon-post-analysis 1.2.285 → 1.2.286
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD_PARTY_MODELS.md +8 -0
- package/dist/{dist-ITESpHou.js → clip-model-registry-CY0FGcpQ.js} +1483 -403
- package/dist/{dist-D2tXUMfE.mjs → clip-model-registry-D4mC2M7F.mjs} +1374 -390
- package/dist/embedding-encoder/index.js +962 -428
- package/dist/embedding-encoder/index.mjs +952 -418
- package/dist/pipeline-analytics/_stub.js +2 -2
- package/dist/pipeline-analytics/{_virtual_mf-localSharedImportMap___mfe_internal__addon_pipeline_analytics_widgets-7oasHfhD.mjs → _virtual_mf-localSharedImportMap___mfe_internal__addon_pipeline_analytics_widgets-CkfvSztj.mjs} +2 -2
- package/dist/pipeline-analytics/_virtual_mf___mfe_internal__addon_pipeline_analytics_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-NKbCrEkH.mjs +26 -0
- package/dist/pipeline-analytics/_virtual_mf___mfe_internal__addon_pipeline_analytics_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-BAYoLHsN.mjs +26 -0
- package/dist/pipeline-analytics/{hostInit-B6m75Rnz.mjs → hostInit-CelYj4QN.mjs} +2 -2
- package/dist/pipeline-analytics/index.js +5863 -5053
- package/dist/pipeline-analytics/index.mjs +4510 -3700
- package/dist/pipeline-analytics/remoteEntry.js +1 -1
- package/package.json +1 -1
- package/python/test_text_encoder.py +49 -2
- package/python/text_encoder_inference.py +31 -14
- package/dist/pipeline-analytics/_virtual_mf___mfe_internal__addon_pipeline_analytics_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-CjeM5Bph.mjs +0 -26
- package/dist/pipeline-analytics/_virtual_mf___mfe_internal__addon_pipeline_analytics_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-BidXbeas.mjs +0 -26
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { Bn as normalizeNodeId, Ht as embeddingRequirementsUnavailableError, Rt as embeddingEncoderCapability, Vt as embeddingRequirementsFailedError, a as CLIP_EMBEDDING_STEP_ID, c as resolveClusterModelPin, ft as PoolMemoryWatchdog, i as CLIP_TEXT_MODELS, jn as BaseAddon, mn as resolvePoolMemoryPolicy, n as DEFAULT_CLIP_MODEL, r as CLIP_IMAGE_MODELS, sn as parseProcStatus, t as BUILTIN_CLIP_MODELS, zt as embeddingModelFailedError } from "../clip-model-registry-D4mC2M7F.mjs";
|
|
2
2
|
import { createRequire } from "node:module";
|
|
3
3
|
import * as fs from "node:fs";
|
|
4
4
|
import * as path$1 from "node:path";
|
|
@@ -64,232 +64,101 @@ function readViaPs(pid) {
|
|
|
64
64
|
});
|
|
65
65
|
}
|
|
66
66
|
//#endregion
|
|
67
|
-
//#region src/embedding-encoder/
|
|
68
|
-
var HF_REPO = "camstack/camstack-models";
|
|
69
|
-
var hf = (path) => hfModelUrl(HF_REPO, path);
|
|
67
|
+
//#region src/embedding-encoder/shared/framed-python-process.ts
|
|
70
68
|
/**
|
|
71
|
-
*
|
|
72
|
-
*
|
|
73
|
-
*
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
*
|
|
88
|
-
*
|
|
89
|
-
*/
|
|
90
|
-
var MLPACKAGE_FILES = [
|
|
91
|
-
"Manifest.json",
|
|
92
|
-
"Data/com.apple.CoreML/model.mlmodel",
|
|
93
|
-
"Data/com.apple.CoreML/weights/weight.bin"
|
|
94
|
-
];
|
|
95
|
-
/**
|
|
96
|
-
* NO onnx vision builds, deliberately (2026-08-21). The int8 ONNX vision
|
|
97
|
-
* exports were measured misaligned with the text encoders (matched image↔text
|
|
98
|
-
* cosine ≈ 0.01 vs ≈ 0.22 for the fp16 openvino/coreml builds) — see the
|
|
99
|
-
* catalog note in `addon-pipeline/.../model-catalogs.ts`. The image-encode leg
|
|
100
|
-
* that consumed them here (`embeddingEncoder.encode` → PythonRawTensorEngine)
|
|
101
|
-
* is retired with them: its only caller discarded the vector, and its
|
|
102
|
-
* `preprocessForClip` applied OpenAI mean/std that MobileCLIP never used.
|
|
103
|
-
* S0 was retired the same day (measured worst of the family on fleet crops).
|
|
104
|
-
*/
|
|
105
|
-
var CLIP_IMAGE_MODELS = [{
|
|
106
|
-
id: "mobileclip-s1",
|
|
107
|
-
name: "MobileCLIP S1",
|
|
108
|
-
description: "Apple MobileCLIP S1 — balanced vision encoder, 512-dim, 256×256 (fp16 OpenVINO/CoreML)",
|
|
109
|
-
inputSize: {
|
|
110
|
-
width: 256,
|
|
111
|
-
height: 256
|
|
112
|
-
},
|
|
113
|
-
labels: [],
|
|
114
|
-
inputNormalization: "none",
|
|
115
|
-
formats: {
|
|
116
|
-
openvino: ovFormat(hf("clip/mobileclip-s1/openvino/camstack-mobileclip-s1-vision.xml"), 55),
|
|
117
|
-
coreml: {
|
|
118
|
-
url: hf("clip/mobileclip-s1/coreml/camstack-mobileclip-s1-vision.mlpackage"),
|
|
119
|
-
sizeMB: 65,
|
|
120
|
-
isDirectory: true,
|
|
121
|
-
files: [...MLPACKAGE_FILES],
|
|
122
|
-
runtimes: ["python"]
|
|
123
|
-
}
|
|
124
|
-
}
|
|
125
|
-
}, {
|
|
126
|
-
id: "mobileclip-s2",
|
|
127
|
-
name: "MobileCLIP S2",
|
|
128
|
-
description: "Apple MobileCLIP S2 — high-accuracy vision encoder, 512-dim, 256×256 (fp16 OpenVINO/CoreML)",
|
|
129
|
-
inputSize: {
|
|
130
|
-
width: 256,
|
|
131
|
-
height: 256
|
|
132
|
-
},
|
|
133
|
-
labels: [],
|
|
134
|
-
inputNormalization: "none",
|
|
135
|
-
formats: {
|
|
136
|
-
openvino: ovFormat(hf("clip/mobileclip-s2/openvino/camstack-mobileclip-s2-vision.xml"), 90),
|
|
137
|
-
coreml: {
|
|
138
|
-
url: hf("clip/mobileclip-s2/coreml/camstack-mobileclip-s2-vision.mlpackage"),
|
|
139
|
-
sizeMB: 110,
|
|
140
|
-
isDirectory: true,
|
|
141
|
-
files: [...MLPACKAGE_FILES],
|
|
142
|
-
runtimes: ["python"]
|
|
143
|
-
}
|
|
144
|
-
}
|
|
145
|
-
}];
|
|
146
|
-
/**
|
|
147
|
-
* The int8 TEXT onnx encoders are healthy — unlike the retired int8 vision
|
|
148
|
-
* exports. Verified 2026-08-21: the live search path (fp16 openvino vision
|
|
149
|
-
* vectors ⋅ int8 onnx text queries) ranks correctly on the real index, and the
|
|
150
|
-
* local cross-check aligns them with the fp16 vision space (cos ≈ 0.22 on
|
|
151
|
-
* matched pairs).
|
|
152
|
-
*/
|
|
153
|
-
var CLIP_TEXT_MODELS = [{
|
|
154
|
-
id: "mobileclip-s1-text",
|
|
155
|
-
name: "MobileCLIP S1 Text Encoder",
|
|
156
|
-
description: "Text encoder for MobileCLIP S1, 512-dim, int8 quantized (61 MB)",
|
|
157
|
-
inputSize: {
|
|
158
|
-
width: 0,
|
|
159
|
-
height: 0
|
|
160
|
-
},
|
|
161
|
-
labels: [],
|
|
162
|
-
formats: {
|
|
163
|
-
onnx: {
|
|
164
|
-
url: hf("clip/mobileclip-s1/onnx/camstack-mobileclip-s1-text.onnx"),
|
|
165
|
-
sizeMB: 61,
|
|
166
|
-
files: [TOKENIZER_FILE]
|
|
167
|
-
},
|
|
168
|
-
openvino: ovFormat(hf("clip/mobileclip-s1/openvino/camstack-mobileclip-s1-text.xml"), 121)
|
|
169
|
-
}
|
|
170
|
-
}, {
|
|
171
|
-
id: "mobileclip-s2-text",
|
|
172
|
-
name: "MobileCLIP S2 Text Encoder",
|
|
173
|
-
description: "Text encoder for MobileCLIP S2, 512-dim, int8 quantized (61 MB)",
|
|
174
|
-
inputSize: {
|
|
175
|
-
width: 0,
|
|
176
|
-
height: 0
|
|
177
|
-
},
|
|
178
|
-
labels: [],
|
|
179
|
-
formats: {
|
|
180
|
-
onnx: {
|
|
181
|
-
url: hf("clip/mobileclip-s2/onnx/camstack-mobileclip-s2-text.onnx"),
|
|
182
|
-
sizeMB: 61,
|
|
183
|
-
files: [TOKENIZER_FILE]
|
|
184
|
-
},
|
|
185
|
-
openvino: ovFormat(hf("clip/mobileclip-s2/openvino/camstack-mobileclip-s2-text.xml"), 121)
|
|
186
|
-
}
|
|
187
|
-
}];
|
|
188
|
-
//#endregion
|
|
189
|
-
//#region src/embedding-encoder/shared/noop-logger.ts
|
|
190
|
-
var noop = () => {};
|
|
191
|
-
function createNoopLogger() {
|
|
192
|
-
const logger = {
|
|
193
|
-
debug: noop,
|
|
194
|
-
info: noop,
|
|
195
|
-
warn: noop,
|
|
196
|
-
error: noop,
|
|
197
|
-
child: () => logger,
|
|
198
|
-
withTags: (_tags) => logger
|
|
199
|
-
};
|
|
200
|
-
return logger;
|
|
201
|
-
}
|
|
202
|
-
//#endregion
|
|
203
|
-
//#region src/embedding-encoder/shared/python-raw-tensor-engine.ts
|
|
204
|
-
/**
|
|
205
|
-
* Raw-tensor ONNX engine backed by an embedded-Python subprocess
|
|
206
|
-
* (`raw_tensor_inference.py`). Replaces the Node `onnxruntime-node` raw-tensor
|
|
207
|
-
* engine so the platform ships no Node ONNX runtime. The caller preprocesses to
|
|
208
|
-
* a Float32Array; this engine ships it to Python, which runs onnxruntime and
|
|
209
|
-
* returns the output tensor. Wire protocol = length-prefixed binary frames
|
|
210
|
-
* ([4B LE length][payload]).
|
|
69
|
+
* One embedded-Python subprocess speaking the length-prefixed frame protocol
|
|
70
|
+
* (`[4B LE len][payload]`, ready = `[0x01]`) — the lifecycle both embedding
|
|
71
|
+
* engines share, written once.
|
|
72
|
+
*
|
|
73
|
+
* ## What it guarantees (D649 review)
|
|
74
|
+
*
|
|
75
|
+
* - **Every awaited frame settles.** Waiters are a FIFO queue (the process
|
|
76
|
+
* answers in order). On ANY exit — a crash, an OOM kill, a clean 0, the
|
|
77
|
+
* SIGTERM of `dispose` (code null) — and on a stdin error, every waiter is
|
|
78
|
+
* rejected. The old guard (`code !== 0 && code !== null`) hung the encode.
|
|
79
|
+
* - **A dead process is dead, by name.** After an unexpected exit the engine
|
|
80
|
+
* is marked dead with the exit code/signal, the exit is logged, and every
|
|
81
|
+
* later request rejects immediately with that reason instead of writing to
|
|
82
|
+
* a closed pipe and waiting forever. Holders check {@link isAlive} and
|
|
83
|
+
* rebuild — so a crash costs one failed request, not a silent stop.
|
|
84
|
+
* - **EPIPE never escapes.** stdin carries an `error` listener: a write into a
|
|
85
|
+
* process that just died becomes a rejected request, never an uncaught
|
|
86
|
+
* exception in the addon.
|
|
211
87
|
*/
|
|
212
|
-
var
|
|
88
|
+
var FramedPythonProcess = class {
|
|
89
|
+
name;
|
|
213
90
|
pythonPath;
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
runtime = "onnx";
|
|
217
|
-
device = "cpu";
|
|
91
|
+
args;
|
|
92
|
+
log;
|
|
218
93
|
process = null;
|
|
219
94
|
receiveBuffer = Buffer.alloc(0);
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
constructor(
|
|
95
|
+
pending = [];
|
|
96
|
+
dead = null;
|
|
97
|
+
disposing = false;
|
|
98
|
+
constructor(name, pythonPath, args, log) {
|
|
99
|
+
this.name = name;
|
|
224
100
|
this.pythonPath = pythonPath;
|
|
225
|
-
this.
|
|
226
|
-
this.
|
|
227
|
-
this.log = logger ?? createNoopLogger();
|
|
101
|
+
this.args = args;
|
|
102
|
+
this.log = log;
|
|
228
103
|
}
|
|
229
|
-
|
|
230
|
-
|
|
104
|
+
/** Spawn and wait for the ready frame. */
|
|
105
|
+
async start() {
|
|
106
|
+
const proc = spawn(this.pythonPath, [...this.args], { stdio: [
|
|
231
107
|
"pipe",
|
|
232
108
|
"pipe",
|
|
233
109
|
"pipe"
|
|
234
110
|
] });
|
|
235
|
-
this.process
|
|
111
|
+
this.process = proc;
|
|
112
|
+
proc.stderr?.on("data", (chunk) => {
|
|
236
113
|
const text = chunk.toString().trim();
|
|
237
114
|
if (text) this.log.warn(text);
|
|
238
115
|
});
|
|
239
|
-
|
|
240
|
-
this.log.error(
|
|
241
|
-
this.
|
|
242
|
-
this.pendingReject = null;
|
|
243
|
-
this.pendingResolve = null;
|
|
116
|
+
proc.on("error", (err) => {
|
|
117
|
+
this.log.error(`${this.name}: process error`, { meta: { error: err.message } });
|
|
118
|
+
this.markDead(`process error: ${err.message}`);
|
|
244
119
|
});
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
}
|
|
120
|
+
proc.stdin?.on("error", (err) => {
|
|
121
|
+
this.markDead(`stdin error: ${err.message}`);
|
|
122
|
+
});
|
|
123
|
+
proc.on("exit", (code, signal) => {
|
|
124
|
+
if (this.process === proc) this.process = null;
|
|
125
|
+
const reason = `process exited (code ${String(code)}, signal ${String(signal)})`;
|
|
126
|
+
if (!this.disposing) this.log.warn(`${this.name}: process exited unexpectedly — the engine is dead`, { meta: {
|
|
127
|
+
code,
|
|
128
|
+
signal,
|
|
129
|
+
pid: proc.pid ?? null
|
|
130
|
+
} });
|
|
131
|
+
this.markDead(reason);
|
|
252
132
|
});
|
|
253
|
-
|
|
133
|
+
proc.stdout?.on("data", (chunk) => {
|
|
254
134
|
this.receiveBuffer = Buffer.concat([this.receiveBuffer, chunk]);
|
|
255
135
|
this.tryReceive();
|
|
256
136
|
});
|
|
257
137
|
const ready = await this.receiveFrame();
|
|
258
|
-
if (ready.length !== 1 || ready[0] !== 1) throw new Error(
|
|
259
|
-
|
|
138
|
+
if (ready.length !== 1 || ready[0] !== 1) throw new Error(`${this.name}: unexpected ready frame`);
|
|
139
|
+
}
|
|
140
|
+
/** False once the process died or was disposed — a holder must rebuild. */
|
|
141
|
+
isAlive() {
|
|
142
|
+
return this.process !== null && this.dead === null;
|
|
260
143
|
}
|
|
261
|
-
/** Native pid of the Python subprocess — sampled by the pool memory
|
|
262
|
-
* watchdog (`/proc/<pid>/status`). */
|
|
263
144
|
getPid() {
|
|
264
145
|
return this.process?.pid ?? null;
|
|
265
146
|
}
|
|
266
|
-
/**
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
const meta = Buffer.allocUnsafe(1 + ndims * 4);
|
|
277
|
-
meta.writeUInt8(ndims, 0);
|
|
278
|
-
for (let i = 0; i < ndims; i++) meta.writeUInt32LE(inputShape[i], 1 + i * 4);
|
|
279
|
-
const dataBuf = Buffer.from(input.buffer, input.byteOffset, input.byteLength);
|
|
280
|
-
const payload = Buffer.concat([meta, dataBuf]);
|
|
281
|
-
const lenBuf = Buffer.allocUnsafe(4);
|
|
282
|
-
lenBuf.writeUInt32LE(payload.length, 0);
|
|
283
|
-
this.process.stdin.write(Buffer.concat([lenBuf, payload]));
|
|
284
|
-
const resp = await this.receiveFrame();
|
|
285
|
-
const floatStart = 1 + resp.readUInt8(0) * 4;
|
|
286
|
-
const count = (resp.length - floatStart) / 4;
|
|
287
|
-
const out = new Float32Array(count);
|
|
288
|
-
for (let i = 0; i < count; i++) out[i] = resp.readFloatLE(floatStart + i * 4);
|
|
289
|
-
return out;
|
|
147
|
+
/** Send one request frame and await its answer. Rejects AT ONCE when dead. */
|
|
148
|
+
async request(payload) {
|
|
149
|
+
if (this.dead !== null) throw new Error(`${this.name}: engine is dead — ${this.dead}`);
|
|
150
|
+
const stdin = this.process?.stdin;
|
|
151
|
+
if (!stdin) throw new Error(`${this.name}: not initialized — call initialize() first`);
|
|
152
|
+
const answer = this.receiveFrame();
|
|
153
|
+
const len = Buffer.allocUnsafe(4);
|
|
154
|
+
len.writeUInt32LE(payload.length, 0);
|
|
155
|
+
stdin.write(Buffer.concat([len, payload]));
|
|
156
|
+
return answer;
|
|
290
157
|
}
|
|
291
158
|
async dispose() {
|
|
292
159
|
const proc = this.process;
|
|
160
|
+
this.disposing = true;
|
|
161
|
+
this.markDead("disposed");
|
|
293
162
|
if (!proc) return;
|
|
294
163
|
this.process = null;
|
|
295
164
|
proc.stdin?.end();
|
|
@@ -307,185 +176,546 @@ var PythonRawTensorEngine = class {
|
|
|
307
176
|
});
|
|
308
177
|
});
|
|
309
178
|
}
|
|
179
|
+
markDead(reason) {
|
|
180
|
+
if (this.dead === null) this.dead = reason;
|
|
181
|
+
const err = /* @__PURE__ */ new Error(`${this.name}: ${reason}`);
|
|
182
|
+
for (const waiter of this.pending.splice(0)) waiter.reject(err);
|
|
183
|
+
}
|
|
310
184
|
receiveFrame() {
|
|
311
185
|
return new Promise((resolve, reject) => {
|
|
312
|
-
this.
|
|
313
|
-
|
|
186
|
+
this.pending.push({
|
|
187
|
+
resolve,
|
|
188
|
+
reject
|
|
189
|
+
});
|
|
314
190
|
});
|
|
315
191
|
}
|
|
316
192
|
tryReceive() {
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
this.pendingReject = null;
|
|
325
|
-
resolve?.(payload);
|
|
193
|
+
while (this.receiveBuffer.length >= 4) {
|
|
194
|
+
const length = this.receiveBuffer.readUInt32LE(0);
|
|
195
|
+
if (this.receiveBuffer.length < 4 + length) return;
|
|
196
|
+
const payload = Buffer.from(this.receiveBuffer.subarray(4, 4 + length));
|
|
197
|
+
this.receiveBuffer = this.receiveBuffer.subarray(4 + length);
|
|
198
|
+
this.pending.shift()?.resolve(payload);
|
|
199
|
+
}
|
|
326
200
|
}
|
|
327
201
|
};
|
|
328
202
|
//#endregion
|
|
203
|
+
//#region src/embedding-encoder/shared/noop-logger.ts
|
|
204
|
+
var noop = () => {};
|
|
205
|
+
function createNoopLogger() {
|
|
206
|
+
const logger = {
|
|
207
|
+
debug: noop,
|
|
208
|
+
info: noop,
|
|
209
|
+
warn: noop,
|
|
210
|
+
error: noop,
|
|
211
|
+
child: () => logger,
|
|
212
|
+
withTags: (_tags) => logger
|
|
213
|
+
};
|
|
214
|
+
return logger;
|
|
215
|
+
}
|
|
216
|
+
//#endregion
|
|
329
217
|
//#region src/embedding-encoder/shared/python-text-encoder-engine.ts
|
|
330
|
-
/**
|
|
331
|
-
* CLIP text-encoder engine backed by an embedded-Python subprocess
|
|
332
|
-
* (`text_encoder_inference.py`). Tokenization happens IN Python via the HF
|
|
333
|
-
* `tokenizers` Rust BPE (exact by construction) — this replaces the former
|
|
334
|
-
* hand-rolled TypeScript CLIP BPE.
|
|
335
|
-
*
|
|
336
|
-
* The caller sends raw UTF-8 text; Python tokenizes (truncate/pad to 77), runs
|
|
337
|
-
* onnxruntime, and returns the embedding tensor. Wire protocol = length-prefixed
|
|
338
|
-
* binary frames ([4B LE length][payload]):
|
|
339
|
-
* ready (in): [0x01]
|
|
340
|
-
* request (out): UTF-8 text bytes
|
|
341
|
-
* response (in): [1B ndims][dims × 4B LE uint32][float32 LE data]
|
|
342
|
-
*/
|
|
343
218
|
var PythonTextEncoderEngine = class {
|
|
344
|
-
pythonPath;
|
|
345
219
|
scriptPath;
|
|
346
220
|
modelPath;
|
|
347
221
|
tokenizerPath;
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
pendingResolve = null;
|
|
351
|
-
pendingReject = null;
|
|
222
|
+
window;
|
|
223
|
+
proc;
|
|
352
224
|
log;
|
|
353
|
-
|
|
354
|
-
|
|
225
|
+
requestCount = 0;
|
|
226
|
+
constructor(pythonPath, scriptPath, modelPath, tokenizerPath, window, logger) {
|
|
355
227
|
this.scriptPath = scriptPath;
|
|
356
228
|
this.modelPath = modelPath;
|
|
357
229
|
this.tokenizerPath = tokenizerPath;
|
|
230
|
+
this.window = window;
|
|
358
231
|
this.log = logger ?? createNoopLogger();
|
|
232
|
+
this.proc = new FramedPythonProcess("PythonTextEncoderEngine", pythonPath, this.spawnArgs(), this.log);
|
|
359
233
|
}
|
|
360
234
|
async initialize() {
|
|
361
|
-
this.
|
|
362
|
-
this.scriptPath,
|
|
363
|
-
this.modelPath,
|
|
364
|
-
this.tokenizerPath
|
|
365
|
-
], { stdio: [
|
|
366
|
-
"pipe",
|
|
367
|
-
"pipe",
|
|
368
|
-
"pipe"
|
|
369
|
-
] });
|
|
370
|
-
this.process.stderr?.on("data", (chunk) => {
|
|
371
|
-
const text = chunk.toString().trim();
|
|
372
|
-
if (text) this.log.warn(text);
|
|
373
|
-
});
|
|
374
|
-
this.process.on("error", (err) => {
|
|
375
|
-
this.log.error("Python text-encoder process error", { meta: { error: err.message } });
|
|
376
|
-
this.pendingReject?.(err);
|
|
377
|
-
this.pendingReject = null;
|
|
378
|
-
this.pendingResolve = null;
|
|
379
|
-
});
|
|
380
|
-
this.process.on("exit", (code) => {
|
|
381
|
-
if (code !== 0 && code !== null) {
|
|
382
|
-
const err = /* @__PURE__ */ new Error(`PythonTextEncoderEngine: process exited with code ${code}`);
|
|
383
|
-
this.pendingReject?.(err);
|
|
384
|
-
this.pendingReject = null;
|
|
385
|
-
this.pendingResolve = null;
|
|
386
|
-
}
|
|
387
|
-
});
|
|
388
|
-
this.process.stdout.on("data", (chunk) => {
|
|
389
|
-
this.receiveBuffer = Buffer.concat([this.receiveBuffer, chunk]);
|
|
390
|
-
this.tryReceive();
|
|
391
|
-
});
|
|
392
|
-
const ready = await this.receiveFrame();
|
|
393
|
-
if (ready.length !== 1 || ready[0] !== 1) throw new Error("PythonTextEncoderEngine: unexpected ready frame");
|
|
235
|
+
await this.proc.start();
|
|
394
236
|
this.log.info("CLIP text-encoder engine ready (embedded Python)", { meta: {
|
|
395
237
|
modelPath: this.modelPath,
|
|
396
|
-
tokenizerPath: this.tokenizerPath
|
|
238
|
+
tokenizerPath: this.tokenizerPath,
|
|
239
|
+
contextLength: this.window.contextLength,
|
|
240
|
+
padId: this.window.padId
|
|
397
241
|
} });
|
|
398
242
|
}
|
|
243
|
+
/** The subprocess argv after the interpreter — the window rides on it. */
|
|
244
|
+
spawnArgs() {
|
|
245
|
+
return [
|
|
246
|
+
this.scriptPath,
|
|
247
|
+
this.modelPath,
|
|
248
|
+
this.tokenizerPath,
|
|
249
|
+
"--context-length",
|
|
250
|
+
String(this.window.contextLength),
|
|
251
|
+
"--pad-id",
|
|
252
|
+
String(this.window.padId)
|
|
253
|
+
];
|
|
254
|
+
}
|
|
255
|
+
/** False once the process died (crash, OOM kill) or was disposed. */
|
|
256
|
+
isAlive() {
|
|
257
|
+
return this.proc.isAlive();
|
|
258
|
+
}
|
|
399
259
|
/** Native pid of the Python subprocess — sampled by the pool memory
|
|
400
260
|
* watchdog (`/proc/<pid>/status`). */
|
|
401
261
|
getPid() {
|
|
402
|
-
return this.
|
|
262
|
+
return this.proc.getPid();
|
|
403
263
|
}
|
|
404
264
|
/** Encode requests served since spawn — the denominator that separates
|
|
405
265
|
* "RSS grew per unit of work" (leak) from "RSS grew because it worked". */
|
|
406
266
|
getRequestCount() {
|
|
407
267
|
return this.requestCount;
|
|
408
268
|
}
|
|
409
|
-
requestCount = 0;
|
|
410
269
|
/** Tokenize + encode `text` into the model embedding (float32). */
|
|
411
270
|
async encode(text) {
|
|
412
|
-
if (!this.process?.stdin) throw new Error("PythonTextEncoderEngine: not initialized — call initialize() first");
|
|
413
271
|
this.requestCount++;
|
|
414
|
-
|
|
415
|
-
const lenBuf = Buffer.allocUnsafe(4);
|
|
416
|
-
lenBuf.writeUInt32LE(payload.length, 0);
|
|
417
|
-
this.process.stdin.write(Buffer.concat([lenBuf, payload]));
|
|
418
|
-
const resp = await this.receiveFrame();
|
|
419
|
-
const floatStart = 1 + resp.readUInt8(0) * 4;
|
|
420
|
-
const count = (resp.length - floatStart) / 4;
|
|
421
|
-
const out = new Float32Array(count);
|
|
422
|
-
for (let i = 0; i < count; i++) out[i] = resp.readFloatLE(floatStart + i * 4);
|
|
423
|
-
return out;
|
|
272
|
+
return decodeTensorFrame(await this.proc.request(Buffer.from(text, "utf-8")));
|
|
424
273
|
}
|
|
425
274
|
async dispose() {
|
|
426
|
-
|
|
427
|
-
if (!proc) return;
|
|
428
|
-
this.process = null;
|
|
429
|
-
proc.stdin?.end();
|
|
430
|
-
proc.kill("SIGTERM");
|
|
431
|
-
await new Promise((resolve) => {
|
|
432
|
-
const timer = setTimeout(() => {
|
|
433
|
-
try {
|
|
434
|
-
proc.kill("SIGKILL");
|
|
435
|
-
} catch {}
|
|
436
|
-
resolve();
|
|
437
|
-
}, 5e3);
|
|
438
|
-
proc.once("exit", () => {
|
|
439
|
-
clearTimeout(timer);
|
|
440
|
-
resolve();
|
|
441
|
-
});
|
|
442
|
-
});
|
|
275
|
+
await this.proc.dispose();
|
|
443
276
|
}
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
277
|
+
};
|
|
278
|
+
/** Decode `[1B ndims][dims × 4B][float32 data]`. */
|
|
279
|
+
function decodeTensorFrame(resp) {
|
|
280
|
+
const floatStart = 1 + resp.readUInt8(0) * 4;
|
|
281
|
+
const count = (resp.length - floatStart) / 4;
|
|
282
|
+
const out = new Float32Array(count);
|
|
283
|
+
for (let i = 0; i < count; i++) out[i] = resp.readFloatLE(floatStart + i * 4);
|
|
284
|
+
return out;
|
|
285
|
+
}
|
|
286
|
+
//#endregion
|
|
287
|
+
//#region src/embedding-encoder/shared/python-raw-tensor-engine.ts
|
|
288
|
+
/**
|
|
289
|
+
* Raw-tensor ONNX engine backed by an embedded-Python subprocess
|
|
290
|
+
* (`raw_tensor_inference.py`). Replaces the Node `onnxruntime-node` raw-tensor
|
|
291
|
+
* engine so the platform ships no Node ONNX runtime. The caller preprocesses to
|
|
292
|
+
* a Float32Array; this engine ships it to Python, which runs onnxruntime and
|
|
293
|
+
* returns the output tensor. Wire protocol = length-prefixed binary frames
|
|
294
|
+
* ([4B LE length][payload]); the process lifecycle is `FramedPythonProcess`.
|
|
295
|
+
*/
|
|
296
|
+
var PythonRawTensorEngine = class {
|
|
297
|
+
modelPath;
|
|
298
|
+
runtime = "onnx";
|
|
299
|
+
device = "cpu";
|
|
300
|
+
proc;
|
|
301
|
+
log;
|
|
302
|
+
requestCount = 0;
|
|
303
|
+
constructor(pythonPath, scriptPath, modelPath, logger) {
|
|
304
|
+
this.modelPath = modelPath;
|
|
305
|
+
this.log = logger ?? createNoopLogger();
|
|
306
|
+
this.proc = new FramedPythonProcess("PythonRawTensorEngine", pythonPath, [scriptPath, modelPath], this.log);
|
|
449
307
|
}
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
308
|
+
async initialize() {
|
|
309
|
+
await this.proc.start();
|
|
310
|
+
this.log.info("ONNX raw-tensor engine ready (embedded Python)", { meta: { modelPath: this.modelPath } });
|
|
311
|
+
}
|
|
312
|
+
/** False once the process died (crash, OOM kill) or was disposed. */
|
|
313
|
+
isAlive() {
|
|
314
|
+
return this.proc.isAlive();
|
|
315
|
+
}
|
|
316
|
+
/** Native pid of the Python subprocess — sampled by the pool memory
|
|
317
|
+
* watchdog (`/proc/<pid>/status`). */
|
|
318
|
+
getPid() {
|
|
319
|
+
return this.proc.getPid();
|
|
320
|
+
}
|
|
321
|
+
/** Inference requests served since spawn — the denominator that separates
|
|
322
|
+
* "RSS grew per unit of work" (leak) from "RSS grew because it worked". */
|
|
323
|
+
getRequestCount() {
|
|
324
|
+
return this.requestCount;
|
|
325
|
+
}
|
|
326
|
+
async run(input, inputShape) {
|
|
327
|
+
this.requestCount++;
|
|
328
|
+
const ndims = inputShape.length;
|
|
329
|
+
const meta = Buffer.allocUnsafe(1 + ndims * 4);
|
|
330
|
+
meta.writeUInt8(ndims, 0);
|
|
331
|
+
for (let i = 0; i < ndims; i++) meta.writeUInt32LE(inputShape[i], 1 + i * 4);
|
|
332
|
+
const dataBuf = Buffer.from(input.buffer, input.byteOffset, input.byteLength);
|
|
333
|
+
return decodeTensorFrame(await this.proc.request(Buffer.concat([meta, dataBuf])));
|
|
334
|
+
}
|
|
335
|
+
async dispose() {
|
|
336
|
+
await this.proc.dispose();
|
|
337
|
+
}
|
|
338
|
+
};
|
|
339
|
+
var ActiveClipModel = class {
|
|
340
|
+
deps;
|
|
341
|
+
cached = null;
|
|
342
|
+
lastRead = null;
|
|
343
|
+
constructor(deps) {
|
|
344
|
+
this.deps = deps;
|
|
345
|
+
}
|
|
346
|
+
/**
|
|
347
|
+
* The last row a read RETURNED, without reading — `null` while no read has
|
|
348
|
+
* ever succeeded (the config fallback is a guess, not a known model). The
|
|
349
|
+
* idle sweep exempts this model's text tower (`clip-text-engines.ts`).
|
|
350
|
+
*/
|
|
351
|
+
known() {
|
|
352
|
+
return this.lastRead;
|
|
353
|
+
}
|
|
354
|
+
async get() {
|
|
355
|
+
const now = this.deps.now();
|
|
356
|
+
if (this.cached !== null && now - this.cached.at < 3e4) return this.cached.modelId;
|
|
357
|
+
const row = await this.deps.readRow();
|
|
358
|
+
if (row === null) {
|
|
359
|
+
const kept = this.lastRead ?? this.deps.fallbackModelId();
|
|
360
|
+
this.deps.logger.warn("CLIP encoder kept its model — the cluster row could not be read", { meta: {
|
|
361
|
+
modelId: kept,
|
|
362
|
+
source: this.lastRead !== null ? "last-read" : "addon-config"
|
|
363
|
+
} });
|
|
364
|
+
return kept;
|
|
365
|
+
}
|
|
366
|
+
const previous = this.lastRead;
|
|
367
|
+
if (row !== previous) this.deps.logger.info("CLIP encoder follows the cluster model", { meta: {
|
|
368
|
+
modelId: row,
|
|
369
|
+
previous
|
|
370
|
+
} });
|
|
371
|
+
this.lastRead = row;
|
|
372
|
+
this.cached = {
|
|
373
|
+
modelId: row,
|
|
374
|
+
at: now
|
|
375
|
+
};
|
|
376
|
+
if (previous !== null && row !== previous) this.deps.onChange?.(row, previous);
|
|
377
|
+
return row;
|
|
378
|
+
}
|
|
379
|
+
};
|
|
380
|
+
var CrashBudget = class {
|
|
381
|
+
deps;
|
|
382
|
+
recent = /* @__PURE__ */ new Map();
|
|
383
|
+
failedModels = /* @__PURE__ */ new Map();
|
|
384
|
+
constructor(deps) {
|
|
385
|
+
this.deps = deps;
|
|
386
|
+
}
|
|
387
|
+
/** Throw the named refusal when the model is in `failed`. Spawns nothing. */
|
|
388
|
+
check(modelId) {
|
|
389
|
+
const failed = this.failedModels.get(modelId);
|
|
390
|
+
if (failed === void 0) return;
|
|
391
|
+
throw embeddingModelFailedError(failed);
|
|
392
|
+
}
|
|
393
|
+
/** Record one crash or load failure. Returns true when it made the model `failed`. */
|
|
394
|
+
record(modelId, reason) {
|
|
395
|
+
if (this.failedModels.has(modelId)) return false;
|
|
396
|
+
const now = this.now();
|
|
397
|
+
const windowMs = this.deps.windowMs ?? 6e5;
|
|
398
|
+
const times = (this.recent.get(modelId) ?? []).filter((t) => now - t < windowMs);
|
|
399
|
+
times.push(now);
|
|
400
|
+
this.recent.set(modelId, times);
|
|
401
|
+
if (times.length < (this.deps.maxCrashes ?? 3)) return false;
|
|
402
|
+
const failed = {
|
|
403
|
+
modelId,
|
|
404
|
+
tower: this.deps.tower,
|
|
405
|
+
nodeId: this.deps.nodeId(),
|
|
406
|
+
crashes: times.length,
|
|
407
|
+
windowMs,
|
|
408
|
+
sinceMs: now,
|
|
409
|
+
lastReason: reason
|
|
410
|
+
};
|
|
411
|
+
this.failedModels.set(modelId, failed);
|
|
412
|
+
this.recent.delete(modelId);
|
|
413
|
+
this.deps.logger.error("CLIP tower FAILED — crash budget exhausted, requests refused until reset", { meta: { ...failed } });
|
|
414
|
+
return true;
|
|
415
|
+
}
|
|
416
|
+
failed() {
|
|
417
|
+
return [...this.failedModels.values()];
|
|
418
|
+
}
|
|
419
|
+
/** Clear one model (or all). Returns the ids that were failed. */
|
|
420
|
+
reset(modelId, why = "operator") {
|
|
421
|
+
const ids = modelId !== void 0 ? [modelId] : [...this.failedModels.keys()];
|
|
422
|
+
const cleared = ids.filter((id) => this.failedModels.delete(id));
|
|
423
|
+
for (const id of ids) this.recent.delete(id);
|
|
424
|
+
if (cleared.length > 0) this.deps.logger.info("CLIP tower failed state cleared", { meta: {
|
|
425
|
+
tower: this.deps.tower,
|
|
426
|
+
nodeId: this.deps.nodeId(),
|
|
427
|
+
cleared,
|
|
428
|
+
why
|
|
429
|
+
} });
|
|
430
|
+
return cleared;
|
|
431
|
+
}
|
|
432
|
+
now() {
|
|
433
|
+
return (this.deps.now ?? Date.now)();
|
|
460
434
|
}
|
|
461
435
|
};
|
|
462
436
|
//#endregion
|
|
463
|
-
//#region src/embedding-encoder/addon/
|
|
437
|
+
//#region src/embedding-encoder/addon/failed-models-report.ts
|
|
464
438
|
/**
|
|
465
|
-
*
|
|
466
|
-
*
|
|
467
|
-
*
|
|
439
|
+
* Every failed tower on this node. Each row already names the node: the budget
|
|
440
|
+
* stamps it when the model goes `failed` (it is in the refusal too), and this
|
|
441
|
+
* report does not re-derive it — one authority for the fact.
|
|
468
442
|
*/
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
443
|
+
function reportFailedModels(budgets) {
|
|
444
|
+
return [...budgets.image.failed(), ...budgets.text.failed()];
|
|
445
|
+
}
|
|
446
|
+
/** Every model whose requirements are terminally unmeetable on this node; rows stamped by the gate. */
|
|
447
|
+
function reportFailedRequirements(requirements) {
|
|
448
|
+
return [...requirements.image.failed(), ...requirements.text.failed()];
|
|
449
|
+
}
|
|
450
|
+
/** The operator's clear (one model, or all) of BOTH terminal states on this node — the result names it. */
|
|
451
|
+
function resetFailedModelsOn(nodeId, towers, modelId) {
|
|
452
|
+
const cleared = [
|
|
453
|
+
...towers.budgets.image.reset(modelId),
|
|
454
|
+
...towers.budgets.text.reset(modelId),
|
|
455
|
+
...towers.requirements.image.reset(modelId),
|
|
456
|
+
...towers.requirements.text.reset(modelId)
|
|
457
|
+
];
|
|
458
|
+
return {
|
|
459
|
+
cleared: [...new Set(cleared)],
|
|
460
|
+
nodeId
|
|
461
|
+
};
|
|
462
|
+
}
|
|
463
|
+
/**
|
|
464
|
+
* The cluster row now names `modelId`: a fresh start for THAT model's towers
|
|
465
|
+
* only (D6's second exit from `failed`). The model the row left keeps its
|
|
466
|
+
* state — a rollback to a model that crashed three times must not buy three
|
|
467
|
+
* more spawns nobody asked for, and a search that still names it
|
|
468
|
+
* (`searchObjectEvents({ modelId })`) is refused with the reset hint, as
|
|
469
|
+
* before the flip.
|
|
470
|
+
*/
|
|
471
|
+
function clearFailedOnModelChange(towers, modelId) {
|
|
472
|
+
return [...new Set([
|
|
473
|
+
...towers.budgets.image.reset(modelId, "model-change"),
|
|
474
|
+
...towers.budgets.text.reset(modelId, "model-change"),
|
|
475
|
+
...towers.requirements.image.reset(modelId, "model-change"),
|
|
476
|
+
...towers.requirements.text.reset(modelId, "model-change")
|
|
477
|
+
])];
|
|
478
|
+
}
|
|
479
|
+
//#endregion
|
|
480
|
+
//#region src/embedding-encoder/addon/model-engine-slot.ts
|
|
481
|
+
var ModelEngineSlot = class {
|
|
482
|
+
build;
|
|
483
|
+
budget;
|
|
484
|
+
requirements;
|
|
485
|
+
onDead;
|
|
486
|
+
current = null;
|
|
487
|
+
inflight = /* @__PURE__ */ new Map();
|
|
488
|
+
/** The last model asked for — a build for any other model is stale on arrival. */
|
|
489
|
+
wanted = null;
|
|
490
|
+
/** Serialises switches so two models never tear the slot down concurrently. */
|
|
491
|
+
switching = Promise.resolve();
|
|
492
|
+
/**
|
|
493
|
+
* Bumped by {@link dispose}: a build that started before it is stale when it
|
|
494
|
+
* lands and is disposed on arrival — shutdown leaves no process behind.
|
|
495
|
+
*/
|
|
496
|
+
generation = 0;
|
|
497
|
+
constructor(build, budget, requirements, onDead = () => {}) {
|
|
498
|
+
this.build = build;
|
|
499
|
+
this.budget = budget;
|
|
500
|
+
this.requirements = requirements;
|
|
501
|
+
this.onDead = onDead;
|
|
502
|
+
}
|
|
503
|
+
held() {
|
|
504
|
+
return this.current;
|
|
505
|
+
}
|
|
506
|
+
async ensure(modelId) {
|
|
507
|
+
this.wanted = modelId;
|
|
508
|
+
this.dropIfDead();
|
|
509
|
+
if (this.current?.modelId === modelId) return this.current.engine;
|
|
510
|
+
this.budget.check(modelId);
|
|
511
|
+
this.requirements.check(modelId);
|
|
512
|
+
const pending = this.inflight.get(modelId);
|
|
513
|
+
if (pending !== void 0) return pending;
|
|
514
|
+
const generation = this.generation;
|
|
515
|
+
const guarded = this.switching.then(async () => {
|
|
516
|
+
this.dropIfDead();
|
|
517
|
+
if (this.current?.modelId === modelId) return this.current.engine;
|
|
518
|
+
this.budget.check(modelId);
|
|
519
|
+
this.requirements.check(modelId);
|
|
520
|
+
await this.disposeCurrent();
|
|
521
|
+
let engine;
|
|
522
|
+
try {
|
|
523
|
+
engine = await this.build(modelId);
|
|
524
|
+
} catch (err) {
|
|
525
|
+
throw this.requirements.refuse(modelId, err);
|
|
526
|
+
}
|
|
527
|
+
this.requirements.satisfied(modelId);
|
|
528
|
+
try {
|
|
529
|
+
await engine.initialize();
|
|
530
|
+
} catch (err) {
|
|
531
|
+
this.budget.record(modelId, `load failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
532
|
+
throw err;
|
|
533
|
+
}
|
|
534
|
+
if (this.generation !== generation) {
|
|
535
|
+
await engine.dispose();
|
|
536
|
+
throw new Error(`engine for "${modelId}" cancelled: the slot was disposed while building`);
|
|
537
|
+
}
|
|
538
|
+
if (this.wanted !== modelId) {
|
|
539
|
+
await engine.dispose();
|
|
540
|
+
throw new Error(`engine for "${modelId}" superseded while building`);
|
|
541
|
+
}
|
|
542
|
+
this.current = {
|
|
543
|
+
modelId,
|
|
544
|
+
engine
|
|
545
|
+
};
|
|
546
|
+
return engine;
|
|
547
|
+
}).finally(() => {
|
|
548
|
+
this.inflight.delete(modelId);
|
|
549
|
+
});
|
|
550
|
+
this.inflight.set(modelId, guarded);
|
|
551
|
+
this.switching = guarded.then(() => void 0, () => void 0);
|
|
552
|
+
return guarded;
|
|
553
|
+
}
|
|
554
|
+
/**
|
|
555
|
+
* Dispose the held engine AND cancel every build in flight: each is disposed
|
|
556
|
+
* when it lands (see `generation`). Resolves once in-flight builds settled.
|
|
557
|
+
*/
|
|
558
|
+
async dispose() {
|
|
559
|
+
this.generation += 1;
|
|
560
|
+
await this.disposeCurrent();
|
|
561
|
+
await this.switching;
|
|
562
|
+
}
|
|
563
|
+
async disposeCurrent() {
|
|
564
|
+
const held = this.current;
|
|
565
|
+
this.current = null;
|
|
566
|
+
await held?.engine.dispose();
|
|
567
|
+
}
|
|
568
|
+
dropIfDead() {
|
|
569
|
+
const held = this.current;
|
|
570
|
+
if (held === null || held.engine.isAlive()) return;
|
|
571
|
+
this.current = null;
|
|
572
|
+
this.onDead(held.modelId);
|
|
573
|
+
this.budget.record(held.modelId, "process died after loading");
|
|
574
|
+
held.engine.dispose().catch(() => {});
|
|
575
|
+
}
|
|
576
|
+
};
|
|
577
|
+
/**
|
|
578
|
+
* Thrown by the requirements stage when a retry can never help: the catalog
|
|
579
|
+
* does not know the id, the server said the file does not exist (404/410), a
|
|
580
|
+
* declared sibling is missing. Same process as the gate, so the class is the
|
|
581
|
+
* marker; the gate never matches wording.
|
|
582
|
+
*/
|
|
583
|
+
var PermanentRequirementsFailure = class extends Error {
|
|
584
|
+
constructor(message) {
|
|
585
|
+
super(message);
|
|
586
|
+
this.name = "PermanentRequirementsFailure";
|
|
483
587
|
}
|
|
484
588
|
};
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
return CLIP_MODEL_META[modelId] ?? CLIP_MODEL_META["mobileclip-s1"];
|
|
589
|
+
function isPermanentRequirementsFailure(err) {
|
|
590
|
+
return err instanceof PermanentRequirementsFailure;
|
|
488
591
|
}
|
|
592
|
+
var RequirementsGate = class {
|
|
593
|
+
deps;
|
|
594
|
+
unavailable = /* @__PURE__ */ new Map();
|
|
595
|
+
constructor(deps) {
|
|
596
|
+
this.deps = deps;
|
|
597
|
+
}
|
|
598
|
+
/**
|
|
599
|
+
* Consulted BEFORE a build. Throws the named refusal — and downloads nothing
|
|
600
|
+
* — while the model is terminal or inside its cooldown. Returns otherwise.
|
|
601
|
+
*/
|
|
602
|
+
check(modelId) {
|
|
603
|
+
const state = this.unavailable.get(modelId);
|
|
604
|
+
if (state === void 0) return;
|
|
605
|
+
if (state.terminal !== null) {
|
|
606
|
+
this.unavailable.set(modelId, {
|
|
607
|
+
...state,
|
|
608
|
+
refused: state.refused + 1
|
|
609
|
+
});
|
|
610
|
+
throw embeddingRequirementsFailedError(state.terminal);
|
|
611
|
+
}
|
|
612
|
+
const now = this.now();
|
|
613
|
+
if (now >= state.cooldownUntilMs) return;
|
|
614
|
+
this.unavailable.set(modelId, {
|
|
615
|
+
...state,
|
|
616
|
+
refused: state.refused + 1
|
|
617
|
+
});
|
|
618
|
+
throw embeddingRequirementsUnavailableError(this.on(modelId), `retry in ${String(Math.ceil((state.cooldownUntilMs - now) / 1e3))}s — ${state.reason}`);
|
|
619
|
+
}
|
|
620
|
+
/**
|
|
621
|
+
* A build failed before any process existed. Starts (or doubles) the
|
|
622
|
+
* cooldown; a PERMANENT failure past the bound makes the model terminal.
|
|
623
|
+
* Logs at the transitions only and returns the named refusal to throw.
|
|
624
|
+
*/
|
|
625
|
+
refuse(modelId, err) {
|
|
626
|
+
const reason = err instanceof Error ? err.message : String(err);
|
|
627
|
+
const now = this.now();
|
|
628
|
+
const prior = this.unavailable.get(modelId);
|
|
629
|
+
const attempts = (prior?.attempts ?? 0) + 1;
|
|
630
|
+
const permanent = isPermanentRequirementsFailure(err);
|
|
631
|
+
const permanentAttempts = (prior?.permanentAttempts ?? 0) + (permanent ? 1 : 0);
|
|
632
|
+
const cooldownMs = Math.min(this.deps.cooldownMaxMs ?? 3e5, (this.deps.cooldownBaseMs ?? 5e3) * 2 ** (attempts - 1));
|
|
633
|
+
const base = {
|
|
634
|
+
sinceMs: prior?.sinceMs ?? now,
|
|
635
|
+
reason,
|
|
636
|
+
refused: (prior?.refused ?? 0) + 1,
|
|
637
|
+
attempts,
|
|
638
|
+
permanentAttempts,
|
|
639
|
+
cooldownUntilMs: now + cooldownMs,
|
|
640
|
+
terminal: null
|
|
641
|
+
};
|
|
642
|
+
if (prior === void 0) this.deps.logger.warn("CLIP model requirements unavailable — requests refused until they are; the crash budget is untouched", { meta: {
|
|
643
|
+
modelId,
|
|
644
|
+
tower: this.deps.tower,
|
|
645
|
+
nodeId: this.deps.nodeId(),
|
|
646
|
+
reason,
|
|
647
|
+
cooldownMs
|
|
648
|
+
} });
|
|
649
|
+
if (permanent && permanentAttempts >= (this.deps.maxAttempts ?? 3)) {
|
|
650
|
+
const terminal = {
|
|
651
|
+
...this.on(modelId),
|
|
652
|
+
attempts: permanentAttempts,
|
|
653
|
+
sinceMs: now,
|
|
654
|
+
lastReason: reason
|
|
655
|
+
};
|
|
656
|
+
this.unavailable.set(modelId, {
|
|
657
|
+
...base,
|
|
658
|
+
terminal
|
|
659
|
+
});
|
|
660
|
+
this.deps.logger.error("CLIP model requirements can never be met — terminally failed, requests refused until reset", { meta: {
|
|
661
|
+
...terminal,
|
|
662
|
+
refusedSoFar: base.refused
|
|
663
|
+
} });
|
|
664
|
+
return embeddingRequirementsFailedError(terminal);
|
|
665
|
+
}
|
|
666
|
+
this.unavailable.set(modelId, base);
|
|
667
|
+
return embeddingRequirementsUnavailableError(this.on(modelId), reason);
|
|
668
|
+
}
|
|
669
|
+
/** A build got its engine: the requirements are met again. Logs the recovery once. */
|
|
670
|
+
satisfied(modelId) {
|
|
671
|
+
const prior = this.unavailable.get(modelId);
|
|
672
|
+
if (prior === void 0) return;
|
|
673
|
+
this.unavailable.delete(modelId);
|
|
674
|
+
this.deps.logger.info("CLIP model requirements available again", { meta: {
|
|
675
|
+
modelId,
|
|
676
|
+
tower: this.deps.tower,
|
|
677
|
+
nodeId: this.deps.nodeId(),
|
|
678
|
+
unavailableMs: this.now() - prior.sinceMs,
|
|
679
|
+
refused: prior.refused,
|
|
680
|
+
attempts: prior.attempts,
|
|
681
|
+
lastReason: prior.reason
|
|
682
|
+
} });
|
|
683
|
+
}
|
|
684
|
+
/** Models terminally `requirements-failed` on this node, as `getInfo` reports them. */
|
|
685
|
+
failed() {
|
|
686
|
+
return [...this.unavailable.values()].map((s) => s.terminal).filter((t) => t !== null);
|
|
687
|
+
}
|
|
688
|
+
/**
|
|
689
|
+
* Clear one model (or all): terminal state AND cooldown. Returns the ids
|
|
690
|
+
* that were TERMINAL — a cooldown is not a state an operator is told about.
|
|
691
|
+
*/
|
|
692
|
+
reset(modelId, why = "operator") {
|
|
693
|
+
const ids = modelId !== void 0 ? [modelId] : [...this.unavailable.keys()];
|
|
694
|
+
const cleared = ids.filter((id) => this.unavailable.get(id)?.terminal !== null && this.unavailable.has(id));
|
|
695
|
+
for (const id of ids) this.unavailable.delete(id);
|
|
696
|
+
if (cleared.length > 0) this.deps.logger.info("CLIP model requirements-failed state cleared", { meta: {
|
|
697
|
+
tower: this.deps.tower,
|
|
698
|
+
nodeId: this.deps.nodeId(),
|
|
699
|
+
cleared,
|
|
700
|
+
why
|
|
701
|
+
} });
|
|
702
|
+
return cleared;
|
|
703
|
+
}
|
|
704
|
+
/** Models currently refused (cooldown or terminal) — exposed for tests. */
|
|
705
|
+
pending() {
|
|
706
|
+
return [...this.unavailable.keys()];
|
|
707
|
+
}
|
|
708
|
+
on(modelId) {
|
|
709
|
+
return {
|
|
710
|
+
modelId,
|
|
711
|
+
tower: this.deps.tower,
|
|
712
|
+
nodeId: this.deps.nodeId()
|
|
713
|
+
};
|
|
714
|
+
}
|
|
715
|
+
now() {
|
|
716
|
+
return (this.deps.now ?? Date.now)();
|
|
717
|
+
}
|
|
718
|
+
};
|
|
489
719
|
//#endregion
|
|
490
720
|
//#region src/embedding-encoder/addon/clip-preprocessing.ts
|
|
491
721
|
var CLIP_MEAN = [
|
|
@@ -528,11 +758,286 @@ function l2Normalize(vec) {
|
|
|
528
758
|
if (norm > 0) for (let i = 0; i < vec.length; i++) vec[i] /= norm;
|
|
529
759
|
return vec;
|
|
530
760
|
}
|
|
761
|
+
/**
|
|
762
|
+
* A tower unused this long is disposed ({@link ClipTextEngines.evictIdle}) —
|
|
763
|
+
* EXCEPT the tower of the model the cluster row names, which stays loaded for
|
|
764
|
+
* the life of the process exactly as it did before D649: the steady-state
|
|
765
|
+
* search must not pay a Python spawn plus a model load after a quiet quarter
|
|
766
|
+
* hour. Only the OTHER towers (SigLIP2 after a comparison, the old model after
|
|
767
|
+
* a flip) are let go when idle. When the active model is unknown (the row was
|
|
768
|
+
* never read), nothing is evicted — the side that keeps memory as before.
|
|
769
|
+
* The encoder is `placement: any-node` and cannot tell locally whether it is
|
|
770
|
+
* the node that answers text queries, so towers are loaded ON DEMAND — never
|
|
771
|
+
* prefetched on a model flip (565 MB on every node). The first query after a
|
|
772
|
+
* flip pays the load; if it fails, the search fails loudly rather than reading
|
|
773
|
+
* as "no results".
|
|
774
|
+
*/
|
|
775
|
+
var TEXT_ENGINE_IDLE_MS = 15 * 6e4;
|
|
776
|
+
var ClipTextEngines = class {
|
|
777
|
+
deps;
|
|
778
|
+
/** Insertion order is recency: a hit is re-inserted at the end. */
|
|
779
|
+
engines = /* @__PURE__ */ new Map();
|
|
780
|
+
/** Single-flight per model: two concurrent first queries spawn one engine. */
|
|
781
|
+
inflight = /* @__PURE__ */ new Map();
|
|
782
|
+
/**
|
|
783
|
+
* Encodes in flight per model. An engine with any is NEVER evicted: disposing
|
|
784
|
+
* it would fail a search that is already running. When every held engine is
|
|
785
|
+
* busy the set goes over `maxEngines` temporarily and shrinks when one drains.
|
|
786
|
+
*/
|
|
787
|
+
busy = /* @__PURE__ */ new Map();
|
|
788
|
+
/** When each held tower last finished an encode (or was built). */
|
|
789
|
+
lastUsed = /* @__PURE__ */ new Map();
|
|
790
|
+
/** Bumped by {@link disposeAll}: a build landing after it is disposed on arrival. */
|
|
791
|
+
generation = 0;
|
|
792
|
+
constructor(deps) {
|
|
793
|
+
this.deps = deps;
|
|
794
|
+
}
|
|
795
|
+
/** Resolve a model id to its CLIP model, or throw naming the id. */
|
|
796
|
+
requireModel(modelId) {
|
|
797
|
+
const model = this.deps.registry.resolve(modelId);
|
|
798
|
+
if (model === null) throw new Error(`EmbeddingEncoder: "${modelId}" has no CLIP metadata — not a CLIP model`);
|
|
799
|
+
return model;
|
|
800
|
+
}
|
|
801
|
+
async encode(modelId, text) {
|
|
802
|
+
const model = this.requireModel(modelId);
|
|
803
|
+
this.busy.set(modelId, (this.busy.get(modelId) ?? 0) + 1);
|
|
804
|
+
let vector;
|
|
805
|
+
try {
|
|
806
|
+
vector = await (await this.engineFor(model)).encode(text);
|
|
807
|
+
} finally {
|
|
808
|
+
const left = (this.busy.get(modelId) ?? 1) - 1;
|
|
809
|
+
if (left > 0) this.busy.set(modelId, left);
|
|
810
|
+
else this.busy.delete(modelId);
|
|
811
|
+
if (this.engines.has(modelId)) this.lastUsed.set(modelId, this.now());
|
|
812
|
+
this.scheduleEviction();
|
|
813
|
+
}
|
|
814
|
+
if (vector.length !== model.meta.embeddingDim) throw new Error(`EmbeddingEncoder: ${model.meta.textModelId} produced ${String(vector.length)} dims, its metadata declares ${String(model.meta.embeddingDim)} — refusing the query rather than truncating it`);
|
|
815
|
+
return {
|
|
816
|
+
vector,
|
|
817
|
+
model
|
|
818
|
+
};
|
|
819
|
+
}
|
|
820
|
+
/**
|
|
821
|
+
* Dispose every IDLE tower unused for {@link TEXT_ENGINE_IDLE_MS} (a timer in
|
|
822
|
+
* the addon calls this), except the active model's. Returns the evicted
|
|
823
|
+
* model ids; each is logged. An unknown active model evicts nothing.
|
|
824
|
+
*/
|
|
825
|
+
async evictIdle(idleMs = TEXT_ENGINE_IDLE_MS) {
|
|
826
|
+
const active = this.deps.activeModelId();
|
|
827
|
+
if (active === null) return [];
|
|
828
|
+
const now = this.now();
|
|
829
|
+
const idle = [...this.engines.keys()].filter((id) => id !== active && (this.busy.get(id) ?? 0) === 0 && now - (this.lastUsed.get(id) ?? now) >= idleMs);
|
|
830
|
+
const evicted = [];
|
|
831
|
+
for (const modelId of idle) {
|
|
832
|
+
if ((this.busy.get(modelId) ?? 0) > 0 || !this.engines.has(modelId)) continue;
|
|
833
|
+
evicted.push(modelId);
|
|
834
|
+
this.deps.logger.info("CLIP text tower evicted — idle", { meta: {
|
|
835
|
+
modelId,
|
|
836
|
+
idleMs: now - (this.lastUsed.get(modelId) ?? now)
|
|
837
|
+
} });
|
|
838
|
+
await this.restart(modelId);
|
|
839
|
+
}
|
|
840
|
+
return evicted;
|
|
841
|
+
}
|
|
842
|
+
/** Live engines, for the memory watchdog. */
|
|
843
|
+
live() {
|
|
844
|
+
return this.engines;
|
|
845
|
+
}
|
|
846
|
+
/** Dispose one engine; the next query rebuilds it. */
|
|
847
|
+
async restart(modelId) {
|
|
848
|
+
const engine = this.engines.get(modelId);
|
|
849
|
+
if (engine === void 0) return;
|
|
850
|
+
this.engines.delete(modelId);
|
|
851
|
+
this.lastUsed.delete(modelId);
|
|
852
|
+
await engine.dispose();
|
|
853
|
+
}
|
|
854
|
+
/** Dispose every tower AND cancel the builds in flight (disposed as they land). */
|
|
855
|
+
async disposeAll() {
|
|
856
|
+
this.generation += 1;
|
|
857
|
+
const all = [...this.engines.values()];
|
|
858
|
+
this.engines.clear();
|
|
859
|
+
this.lastUsed.clear();
|
|
860
|
+
await Promise.all(all.map((e) => e.dispose()));
|
|
861
|
+
await Promise.allSettled(this.inflight.values());
|
|
862
|
+
}
|
|
863
|
+
async engineFor(model) {
|
|
864
|
+
const held = this.engines.get(model.modelId);
|
|
865
|
+
if (held !== void 0 && held.isAlive()) {
|
|
866
|
+
this.engines.delete(model.modelId);
|
|
867
|
+
this.engines.set(model.modelId, held);
|
|
868
|
+
return held;
|
|
869
|
+
}
|
|
870
|
+
if (held !== void 0) {
|
|
871
|
+
this.deps.logger.warn("CLIP text tower died — dropping it and rebuilding", { meta: {
|
|
872
|
+
modelId: model.modelId,
|
|
873
|
+
textModelId: model.meta.textModelId
|
|
874
|
+
} });
|
|
875
|
+
this.engines.delete(model.modelId);
|
|
876
|
+
this.lastUsed.delete(model.modelId);
|
|
877
|
+
this.deps.budget.record(model.modelId, "process died after loading");
|
|
878
|
+
held.dispose().catch(() => {});
|
|
879
|
+
}
|
|
880
|
+
const pending = this.inflight.get(model.modelId);
|
|
881
|
+
if (pending !== void 0) return pending;
|
|
882
|
+
this.deps.budget.check(model.modelId);
|
|
883
|
+
this.deps.requirements.check(model.modelId);
|
|
884
|
+
const textEntry = this.deps.textCatalog.find((e) => e.id === model.meta.textModelId);
|
|
885
|
+
if (textEntry === void 0) throw this.deps.requirements.refuse(model.modelId, new PermanentRequirementsFailure(`EmbeddingEncoder: text model "${model.meta.textModelId}" (for ${model.modelId}) is not in the text catalog`));
|
|
886
|
+
const generation = this.generation;
|
|
887
|
+
const build = this.buildAndLoad(model, textEntry);
|
|
888
|
+
this.inflight.set(model.modelId, build);
|
|
889
|
+
try {
|
|
890
|
+
const engine = await build;
|
|
891
|
+
if (this.generation !== generation) {
|
|
892
|
+
await engine.dispose();
|
|
893
|
+
throw new Error(`text tower for "${model.modelId}" cancelled: disposed while building`);
|
|
894
|
+
}
|
|
895
|
+
this.engines.set(model.modelId, engine);
|
|
896
|
+
this.lastUsed.set(model.modelId, this.now());
|
|
897
|
+
this.deps.logger.info("CLIP text tower loaded", { meta: {
|
|
898
|
+
modelId: model.modelId,
|
|
899
|
+
textModelId: model.meta.textModelId
|
|
900
|
+
} });
|
|
901
|
+
this.scheduleEviction();
|
|
902
|
+
return engine;
|
|
903
|
+
} finally {
|
|
904
|
+
this.inflight.delete(model.modelId);
|
|
905
|
+
}
|
|
906
|
+
}
|
|
907
|
+
/**
|
|
908
|
+
* Two stages, two owners (`model-requirements.ts`): `build` resolves files
|
|
909
|
+
* and runtime and constructs the tower — a throw there is refused by name
|
|
910
|
+
* and spends nothing; `initialize()` spawns and loads — a throw there is a
|
|
911
|
+
* load failure and spends the crash budget.
|
|
912
|
+
*/
|
|
913
|
+
async buildAndLoad(model, textEntry) {
|
|
914
|
+
let engine;
|
|
915
|
+
try {
|
|
916
|
+
engine = await this.deps.build({
|
|
917
|
+
model,
|
|
918
|
+
textEntry,
|
|
919
|
+
meta: model.meta
|
|
920
|
+
});
|
|
921
|
+
} catch (err) {
|
|
922
|
+
throw this.deps.requirements.refuse(model.modelId, err);
|
|
923
|
+
}
|
|
924
|
+
this.deps.requirements.satisfied(model.modelId);
|
|
925
|
+
try {
|
|
926
|
+
await engine.initialize();
|
|
927
|
+
} catch (err) {
|
|
928
|
+
this.deps.budget.record(model.modelId, `load failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
929
|
+
throw err;
|
|
930
|
+
}
|
|
931
|
+
return engine;
|
|
932
|
+
}
|
|
933
|
+
now() {
|
|
934
|
+
return (this.deps.now ?? Date.now)();
|
|
935
|
+
}
|
|
936
|
+
/** Fire-and-log: eviction errors never reach an encode. */
|
|
937
|
+
scheduleEviction() {
|
|
938
|
+
this.evictBeyond(this.deps.maxEngines ?? 2).catch((err) => {
|
|
939
|
+
this.deps.logger.warn("CLIP text tower eviction failed", { meta: { error: err instanceof Error ? err.message : String(err) } });
|
|
940
|
+
});
|
|
941
|
+
}
|
|
942
|
+
/** Evict the coldest IDLE engines until at most `max` remain, or none is idle. */
|
|
943
|
+
async evictBeyond(max) {
|
|
944
|
+
while (this.engines.size > max) {
|
|
945
|
+
const coldestIdle = [...this.engines.keys()].find((id) => (this.busy.get(id) ?? 0) === 0);
|
|
946
|
+
if (coldestIdle === void 0) return;
|
|
947
|
+
await this.restart(coldestIdle);
|
|
948
|
+
}
|
|
949
|
+
}
|
|
950
|
+
/** Encodes in flight on a model — exposed for the eviction tests. */
|
|
951
|
+
inFlight(modelId) {
|
|
952
|
+
return this.busy.get(modelId) ?? 0;
|
|
953
|
+
}
|
|
954
|
+
};
|
|
531
955
|
//#endregion
|
|
532
956
|
//#region src/embedding-encoder/addon/index.ts
|
|
957
|
+
/** Memory-watchdog key prefix of a text tower; the suffix is its CLIP model id. */
|
|
958
|
+
var TEXT_ENGINE_KEY_PREFIX = "clip-text:";
|
|
959
|
+
/** How often idle text towers are looked for (see `TEXT_ENGINE_IDLE_MS`). */
|
|
960
|
+
var IDLE_SWEEP_MS = 6e4;
|
|
961
|
+
/** A model URL answering one of these does not exist; retrying changes nothing. */
|
|
962
|
+
var PERMANENT_HTTP_STATUS = new Set([404, 410]);
|
|
533
963
|
var EmbeddingEncoderAddon = class extends BaseAddon {
|
|
534
|
-
|
|
535
|
-
|
|
964
|
+
/**
|
|
965
|
+
* Everything on this node that can be terminally refused (`failed-models-report.ts`):
|
|
966
|
+
* the respawn budgets per tower (D6, `crash-budget.ts`) and the requirements
|
|
967
|
+
* gates (`model-requirements.ts` — cooldown, and terminal for what can never
|
|
968
|
+
* be met). All in memory: a runner respawn starts them from zero. Every row
|
|
969
|
+
* and refusal names this node.
|
|
970
|
+
*/
|
|
971
|
+
towers = {
|
|
972
|
+
budgets: {
|
|
973
|
+
image: new CrashBudget({
|
|
974
|
+
tower: "image",
|
|
975
|
+
nodeId: () => this.localNodeId(),
|
|
976
|
+
logger: {
|
|
977
|
+
info: (message, extras) => this.ctx.logger.info(message, extras),
|
|
978
|
+
error: (message, extras) => this.ctx.logger.error(message, extras)
|
|
979
|
+
}
|
|
980
|
+
}),
|
|
981
|
+
text: new CrashBudget({
|
|
982
|
+
tower: "text",
|
|
983
|
+
nodeId: () => this.localNodeId(),
|
|
984
|
+
logger: {
|
|
985
|
+
info: (message, extras) => this.ctx.logger.info(message, extras),
|
|
986
|
+
error: (message, extras) => this.ctx.logger.error(message, extras)
|
|
987
|
+
}
|
|
988
|
+
})
|
|
989
|
+
},
|
|
990
|
+
requirements: {
|
|
991
|
+
image: new RequirementsGate({
|
|
992
|
+
tower: "image",
|
|
993
|
+
nodeId: () => this.localNodeId(),
|
|
994
|
+
logger: {
|
|
995
|
+
info: (message, extras) => this.ctx.logger.info(message, extras),
|
|
996
|
+
warn: (message, extras) => this.ctx.logger.warn(message, extras),
|
|
997
|
+
error: (message, extras) => this.ctx.logger.error(message, extras)
|
|
998
|
+
}
|
|
999
|
+
}),
|
|
1000
|
+
text: new RequirementsGate({
|
|
1001
|
+
tower: "text",
|
|
1002
|
+
nodeId: () => this.localNodeId(),
|
|
1003
|
+
logger: {
|
|
1004
|
+
info: (message, extras) => this.ctx.logger.info(message, extras),
|
|
1005
|
+
warn: (message, extras) => this.ctx.logger.warn(message, extras),
|
|
1006
|
+
error: (message, extras) => this.ctx.logger.error(message, extras)
|
|
1007
|
+
}
|
|
1008
|
+
})
|
|
1009
|
+
}
|
|
1010
|
+
};
|
|
1011
|
+
/** The image tower, one model at a time, single-flight per model (`model-engine-slot.ts`). */
|
|
1012
|
+
imageEngine = new ModelEngineSlot((modelId) => this.buildImageEngine(modelId), this.towers.budgets.image, this.towers.requirements.image, (modelId) => this.ctx.logger.warn("CLIP image tower died — dropping it and rebuilding", { meta: { modelId } }));
|
|
1013
|
+
/** One text tower per CLIP model, bounded (see `clip-text-engines.ts`). */
|
|
1014
|
+
textEngines = new ClipTextEngines({
|
|
1015
|
+
registry: BUILTIN_CLIP_MODELS,
|
|
1016
|
+
textCatalog: CLIP_TEXT_MODELS,
|
|
1017
|
+
build: (spec) => this.buildTextEngine(spec),
|
|
1018
|
+
budget: this.towers.budgets.text,
|
|
1019
|
+
requirements: this.towers.requirements.text,
|
|
1020
|
+
activeModelId: () => this.activeModel.known(),
|
|
1021
|
+
logger: {
|
|
1022
|
+
info: (message, extras) => this.ctx.logger.info(message, extras),
|
|
1023
|
+
warn: (message, extras) => this.ctx.logger.warn(message, extras)
|
|
1024
|
+
}
|
|
1025
|
+
});
|
|
1026
|
+
/** Disposes NON-active text towers idle for `TEXT_ENGINE_IDLE_MS` (never prefetched). */
|
|
1027
|
+
idleSweep = null;
|
|
1028
|
+
/** The cluster row's CLIP model — what an unnamed request encodes in (D649). */
|
|
1029
|
+
activeModel = new ActiveClipModel({
|
|
1030
|
+
readRow: () => this.readClusterRow(),
|
|
1031
|
+
fallbackModelId: () => BUILTIN_CLIP_MODELS.resolve(this.config.modelId) !== null ? this.config.modelId : DEFAULT_CLIP_MODEL,
|
|
1032
|
+
now: () => Date.now(),
|
|
1033
|
+
logger: {
|
|
1034
|
+
info: (message, extras) => this.ctx.logger.info(message, extras),
|
|
1035
|
+
warn: (message, extras) => this.ctx.logger.warn(message, extras)
|
|
1036
|
+
},
|
|
1037
|
+
onChange: (modelId) => {
|
|
1038
|
+
clearFailedOnModelChange(this.towers, modelId);
|
|
1039
|
+
}
|
|
1040
|
+
});
|
|
536
1041
|
models = null;
|
|
537
1042
|
/**
|
|
538
1043
|
* RSS bound + periodic memory telemetry for the two Python engines
|
|
@@ -544,12 +1049,6 @@ var EmbeddingEncoderAddon = class extends BaseAddon {
|
|
|
544
1049
|
* stopped in `onShutdown`.
|
|
545
1050
|
*/
|
|
546
1051
|
memoryGuard = null;
|
|
547
|
-
/** Single-flight guards for the lazy engine builds — a watchdog restart
|
|
548
|
-
* followed by two concurrent encodes must not spawn the engine twice
|
|
549
|
-
* (the same bug detection-pipeline fixed three times; see its
|
|
550
|
-
* `engineFactoryInflight`). */
|
|
551
|
-
imageEngineInflight = null;
|
|
552
|
-
textEngineInflight = null;
|
|
553
1052
|
constructor() {
|
|
554
1053
|
super({ modelId: DEFAULT_CLIP_MODEL });
|
|
555
1054
|
}
|
|
@@ -563,6 +1062,12 @@ var EmbeddingEncoderAddon = class extends BaseAddon {
|
|
|
563
1062
|
restart: (key) => this.restartEngineForMemory(key)
|
|
564
1063
|
});
|
|
565
1064
|
this.memoryGuard.start();
|
|
1065
|
+
this.idleSweep = setInterval(() => {
|
|
1066
|
+
this.textEngines.evictIdle().catch((err) => {
|
|
1067
|
+
this.ctx.logger.warn("CLIP text tower idle sweep failed", { meta: { error: err instanceof Error ? err.message : String(err) } });
|
|
1068
|
+
});
|
|
1069
|
+
}, IDLE_SWEEP_MS);
|
|
1070
|
+
this.idleSweep.unref?.();
|
|
566
1071
|
return [{
|
|
567
1072
|
capability: embeddingEncoderCapability,
|
|
568
1073
|
provider: this
|
|
@@ -573,13 +1078,15 @@ var EmbeddingEncoderAddon = class extends BaseAddon {
|
|
|
573
1078
|
async sampleEngineMemory() {
|
|
574
1079
|
const engines = [{
|
|
575
1080
|
key: "clip-image",
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
1081
|
+
modelId: this.imageEngine.held()?.modelId ?? null,
|
|
1082
|
+
pid: this.imageEngine.held()?.engine.getPid() ?? null,
|
|
1083
|
+
requests: this.imageEngine.held()?.engine.getRequestCount() ?? 0
|
|
1084
|
+
}, ...[...this.textEngines.live()].map(([modelId, engine]) => ({
|
|
1085
|
+
key: `${TEXT_ENGINE_KEY_PREFIX}${modelId}`,
|
|
1086
|
+
modelId,
|
|
1087
|
+
pid: engine.getPid(),
|
|
1088
|
+
requests: engine.getRequestCount()
|
|
1089
|
+
}))];
|
|
583
1090
|
const out = [];
|
|
584
1091
|
for (const engine of engines) {
|
|
585
1092
|
if (engine.pid === null) continue;
|
|
@@ -597,7 +1104,7 @@ var EmbeddingEncoderAddon = class extends BaseAddon {
|
|
|
597
1104
|
pids: [engine.pid],
|
|
598
1105
|
rssBytes: mem.rssBytes,
|
|
599
1106
|
meta: {
|
|
600
|
-
modelId:
|
|
1107
|
+
modelId: engine.modelId,
|
|
601
1108
|
vmMb: mb(mem.vmBytes),
|
|
602
1109
|
peakRssMb: mb(mem.hwmBytes),
|
|
603
1110
|
swapMb: mb(mem.swapBytes),
|
|
@@ -614,110 +1121,125 @@ var EmbeddingEncoderAddon = class extends BaseAddon {
|
|
|
614
1121
|
* surface errors, never silent loss. */
|
|
615
1122
|
async restartEngineForMemory(key) {
|
|
616
1123
|
if (key === "clip-image") {
|
|
617
|
-
|
|
618
|
-
if (!engine) return;
|
|
619
|
-
this.imageRawEngine = null;
|
|
620
|
-
await engine.dispose();
|
|
1124
|
+
await this.imageEngine.dispose();
|
|
621
1125
|
return;
|
|
622
1126
|
}
|
|
623
|
-
|
|
624
|
-
|
|
625
|
-
|
|
626
|
-
|
|
1127
|
+
if (key.startsWith(TEXT_ENGINE_KEY_PREFIX)) await this.textEngines.restart(key.slice(10));
|
|
1128
|
+
}
|
|
1129
|
+
/** The cluster row's `clip-embedding` model, or `null` when unreadable. */
|
|
1130
|
+
async readClusterRow() {
|
|
1131
|
+
return resolveClusterModelPin(this.ctx.api, CLIP_EMBEDDING_STEP_ID, { warn: (message, extras) => this.ctx.logger.warn(message, extras) });
|
|
627
1132
|
}
|
|
628
1133
|
async encode(input) {
|
|
629
1134
|
const { crop, width, height } = input;
|
|
630
|
-
await this.
|
|
631
|
-
const
|
|
1135
|
+
const model = this.textEngines.requireModel(await this.activeModel.get());
|
|
1136
|
+
const imageEngine = await this.imageEngine.ensure(model.modelId);
|
|
1137
|
+
const meta = model.meta;
|
|
632
1138
|
const start = Date.now();
|
|
633
|
-
const preprocessed = preprocessForClip(Buffer.isBuffer(crop) ? crop : Buffer.from(crop), width, height,
|
|
634
|
-
const output = await
|
|
1139
|
+
const preprocessed = preprocessForClip(Buffer.isBuffer(crop) ? crop : Buffer.from(crop), width, height, model.inputSize, model.inputSize);
|
|
1140
|
+
const output = await imageEngine.run(preprocessed, [
|
|
635
1141
|
1,
|
|
636
1142
|
3,
|
|
637
|
-
|
|
638
|
-
|
|
1143
|
+
model.inputSize,
|
|
1144
|
+
model.inputSize
|
|
639
1145
|
]);
|
|
640
|
-
|
|
641
|
-
const normalized = l2Normalize(new Float32Array(
|
|
1146
|
+
if (output.length !== meta.embeddingDim) throw new Error(`EmbeddingEncoder: ${model.modelId} image tower produced ${String(output.length)} dims, its metadata declares ${String(meta.embeddingDim)}`);
|
|
1147
|
+
const normalized = l2Normalize(new Float32Array(output));
|
|
642
1148
|
return {
|
|
643
1149
|
embedding: Array.from(normalized),
|
|
644
1150
|
inferenceMs: Date.now() - start
|
|
645
1151
|
};
|
|
646
1152
|
}
|
|
1153
|
+
/**
|
|
1154
|
+
* Encode a query in ONE model's space: the named `modelId`, else the cluster
|
|
1155
|
+
* row's (D649). The token window, pad id, tokenizer and dimension all come
|
|
1156
|
+
* from that model's catalog metadata.
|
|
1157
|
+
*/
|
|
647
1158
|
async encodeText(input) {
|
|
648
|
-
const
|
|
649
|
-
await this.ensureTextEngine();
|
|
650
|
-
const meta = getModelMeta(this.config.modelId);
|
|
1159
|
+
const modelId = input.modelId ?? await this.activeModel.get();
|
|
651
1160
|
const start = Date.now();
|
|
652
|
-
|
|
653
|
-
const output = await this.textEngine.encode(text);
|
|
654
|
-
const sliced = output.length > meta.embeddingDim ? output.slice(0, meta.embeddingDim) : output;
|
|
655
|
-
const normalized = l2Normalize(new Float32Array(sliced));
|
|
1161
|
+
const { vector } = await this.textEngines.encode(modelId, input.text);
|
|
656
1162
|
return {
|
|
657
|
-
embedding: Array.from(
|
|
1163
|
+
embedding: Array.from(l2Normalize(new Float32Array(vector))),
|
|
658
1164
|
inferenceMs: Date.now() - start
|
|
659
1165
|
};
|
|
660
1166
|
}
|
|
1167
|
+
/** THIS node's answer — the cap is one provider per node (see the cap's docblock). */
|
|
661
1168
|
async getInfo() {
|
|
662
|
-
const
|
|
1169
|
+
const model = this.textEngines.requireModel(await this.activeModel.get());
|
|
1170
|
+
const nodeId = this.localNodeId();
|
|
663
1171
|
return {
|
|
664
|
-
modelId:
|
|
665
|
-
embeddingDim: meta.embeddingDim,
|
|
666
|
-
ready: this.
|
|
1172
|
+
modelId: model.modelId,
|
|
1173
|
+
embeddingDim: model.meta.embeddingDim,
|
|
1174
|
+
ready: this.imageEngine.held()?.modelId === model.modelId,
|
|
1175
|
+
nodeId,
|
|
1176
|
+
failedModels: [...reportFailedModels(this.towers.budgets)],
|
|
1177
|
+
failedRequirements: [...reportFailedRequirements(this.towers.requirements)]
|
|
667
1178
|
};
|
|
668
1179
|
}
|
|
669
|
-
|
|
670
|
-
|
|
671
|
-
|
|
672
|
-
const meta = getModelMeta(this.config.modelId);
|
|
673
|
-
const imageEntry = CLIP_IMAGE_MODELS.find((m) => m.id === meta.imageModelId);
|
|
674
|
-
if (!imageEntry) throw new Error(`EmbeddingEncoderAddon: unknown image model "${meta.imageModelId}"`);
|
|
675
|
-
const inflight = this.resolveForEntry(imageEntry, "image");
|
|
676
|
-
this.imageEngineInflight = inflight;
|
|
677
|
-
try {
|
|
678
|
-
await inflight;
|
|
679
|
-
} finally {
|
|
680
|
-
this.imageEngineInflight = null;
|
|
681
|
-
}
|
|
1180
|
+
/** The operator's explicit clear of THIS node's `failed` state (D6); names the node. */
|
|
1181
|
+
async resetFailedModels(input) {
|
|
1182
|
+
return resetFailedModelsOn(this.localNodeId(), this.towers, input.modelId);
|
|
682
1183
|
}
|
|
683
|
-
|
|
684
|
-
|
|
685
|
-
|
|
686
|
-
const meta = getModelMeta(this.config.modelId);
|
|
687
|
-
const textEntry = CLIP_TEXT_MODELS.find((m) => m.id === meta.textModelId);
|
|
688
|
-
if (!textEntry) throw new Error(`EmbeddingEncoderAddon: unknown text model "${meta.textModelId}"`);
|
|
689
|
-
const inflight = this.resolveForEntry(textEntry, "text");
|
|
690
|
-
this.textEngineInflight = inflight;
|
|
691
|
-
try {
|
|
692
|
-
await inflight;
|
|
693
|
-
} finally {
|
|
694
|
-
this.textEngineInflight = null;
|
|
695
|
-
}
|
|
1184
|
+
/** The logical node (a forked child's `<node>/<addon>` id stripped to the node). */
|
|
1185
|
+
localNodeId() {
|
|
1186
|
+
return normalizeNodeId(this.ctx.kernel?.localNodeId);
|
|
696
1187
|
}
|
|
697
|
-
|
|
698
|
-
|
|
699
|
-
|
|
1188
|
+
/**
|
|
1189
|
+
* The REQUIREMENTS stage, shared by both towers: the embedded Python, its
|
|
1190
|
+
* requirements, then the model file (downloaded on first use). Anything that
|
|
1191
|
+
* throws here is refused as `model-requirements-unavailable` and spends no
|
|
1192
|
+
* crash budget (`model-requirements.ts`). The runtime is asked for first —
|
|
1193
|
+
* the cheap question before the 185–565 MB download (D54's rule).
|
|
1194
|
+
*/
|
|
1195
|
+
async prepareOnnx(entry) {
|
|
700
1196
|
const pythonPath = await this.ctx.deps.ensurePython();
|
|
701
1197
|
if (!pythonPath) throw new Error("EmbeddingEncoder: embedded Python is unavailable — cannot run ONNX embeddings. ctx.deps.ensurePython() returned null (portable Python download likely failed).");
|
|
702
1198
|
const pythonDir = resolveEmbeddingPythonDir();
|
|
703
1199
|
await this.ctx.deps.installPythonRequirements(path$1.join(pythonDir, "requirements-embedding.txt"));
|
|
704
|
-
if (
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
this.
|
|
708
|
-
|
|
1200
|
+
if (entry.formats.onnx === void 0) throw new PermanentRequirementsFailure(`EmbeddingEncoder: ${entry.id} has no onnx build in its catalog entry (formats: ${Object.keys(entry.formats).join(", ") || "none"}) — no download can supply it`);
|
|
1201
|
+
let modelPath;
|
|
1202
|
+
try {
|
|
1203
|
+
modelPath = await this.models.ensure(entry.id, "onnx");
|
|
1204
|
+
} catch (err) {
|
|
1205
|
+
const status = downloadHttpStatusOf(err);
|
|
1206
|
+
if (status !== null && PERMANENT_HTTP_STATUS.has(status)) throw new PermanentRequirementsFailure(`EmbeddingEncoder: ${entry.id} is not at its catalog URL (HTTP ${String(status)}) — fix the catalog entry`);
|
|
1207
|
+
throw err;
|
|
709
1208
|
}
|
|
710
|
-
|
|
711
|
-
|
|
712
|
-
|
|
713
|
-
|
|
714
|
-
|
|
1209
|
+
return {
|
|
1210
|
+
modelPath,
|
|
1211
|
+
pythonPath,
|
|
1212
|
+
pythonDir
|
|
1213
|
+
};
|
|
1214
|
+
}
|
|
1215
|
+
async buildImageEngine(modelId) {
|
|
1216
|
+
const entry = CLIP_IMAGE_MODELS.find((m) => m.id === modelId);
|
|
1217
|
+
if (!entry) throw new PermanentRequirementsFailure(`EmbeddingEncoderAddon: unknown image model "${modelId}" — not in the CLIP image catalog`);
|
|
1218
|
+
const { modelPath, pythonPath, pythonDir } = await this.prepareOnnx(entry);
|
|
1219
|
+
return new PythonRawTensorEngine(pythonPath, path$1.join(pythonDir, "raw_tensor_inference.py"), modelPath, this.ctx.logger.withTags({ modelId: entry.id }));
|
|
1220
|
+
}
|
|
1221
|
+
/**
|
|
1222
|
+
* One text tower. Tokenization happens IN Python via the HF `tokenizers`
|
|
1223
|
+
* library, with the tokenizer the model's metadata names — a declared sibling
|
|
1224
|
+
* of the text onnx, so `ModelDownloadService.ensure()` fetched it next to the
|
|
1225
|
+
* model file — and the model's own token window.
|
|
1226
|
+
*/
|
|
1227
|
+
async buildTextEngine(spec) {
|
|
1228
|
+
const { modelPath, pythonPath, pythonDir } = await this.prepareOnnx(spec.textEntry);
|
|
1229
|
+
const tokenizerPath = path$1.join(path$1.dirname(modelPath), spec.meta.tokenizerFile);
|
|
1230
|
+
if (!fs.existsSync(tokenizerPath)) throw new PermanentRequirementsFailure(`EmbeddingEncoder: tokenizer not found at "${tokenizerPath}" — the ${spec.meta.tokenizerFile} sibling download of ${spec.textEntry.id} likely failed.`);
|
|
1231
|
+
return new PythonTextEncoderEngine(pythonPath, path$1.join(pythonDir, "text_encoder_inference.py"), modelPath, tokenizerPath, {
|
|
1232
|
+
contextLength: spec.meta.contextLength,
|
|
1233
|
+
padId: spec.meta.padId
|
|
1234
|
+
}, this.ctx.logger.withTags({ modelId: spec.textEntry.id }));
|
|
715
1235
|
}
|
|
716
1236
|
async onShutdown() {
|
|
717
1237
|
this.memoryGuard?.stop();
|
|
1238
|
+
if (this.idleSweep !== null) clearInterval(this.idleSweep);
|
|
1239
|
+
this.idleSweep = null;
|
|
718
1240
|
this.memoryGuard = null;
|
|
719
|
-
await this.
|
|
720
|
-
await this.
|
|
1241
|
+
await this.imageEngine.dispose();
|
|
1242
|
+
await this.textEngines.disposeAll();
|
|
721
1243
|
}
|
|
722
1244
|
globalSettingsSchema() {
|
|
723
1245
|
return this.schema({ sections: [{
|
|
@@ -727,8 +1249,8 @@ var EmbeddingEncoderAddon = class extends BaseAddon {
|
|
|
727
1249
|
fields: [{
|
|
728
1250
|
type: "text",
|
|
729
1251
|
key: "modelId",
|
|
730
|
-
label: "
|
|
731
|
-
description: "
|
|
1252
|
+
label: "Fallback model ID",
|
|
1253
|
+
description: "Used only until the cluster \"Semantic search model\" row has been read once. The encoder follows that row (D649).",
|
|
732
1254
|
default: DEFAULT_CLIP_MODEL
|
|
733
1255
|
}]
|
|
734
1256
|
}] });
|
|
@@ -750,5 +1272,17 @@ function resolveEmbeddingPythonDir() {
|
|
|
750
1272
|
for (const c of candidates) if (fs.existsSync(path$1.join(c, "raw_tensor_inference.py"))) return c;
|
|
751
1273
|
throw new Error(`EmbeddingEncoder: python/ dir (raw_tensor_inference.py) not found. Searched:\n${candidates.join("\n")}`);
|
|
752
1274
|
}
|
|
1275
|
+
/**
|
|
1276
|
+
* HTTP status of a model-download failure, recognised by SHAPE (`name` +
|
|
1277
|
+
* numeric `status`), never by `instanceof` or a framework import.
|
|
1278
|
+
* `@camstack/system` is host-resolved (not bundled), so importing a guard it
|
|
1279
|
+
* only exports from a newer server would stop this whole entry from linking
|
|
1280
|
+
* on a node still running the older server — a deploy-order hazard for a
|
|
1281
|
+
* one-line classification. `null` = not an HTTP answer (unreachable, other).
|
|
1282
|
+
*/
|
|
1283
|
+
function downloadHttpStatusOf(err) {
|
|
1284
|
+
if (err instanceof Error && err.name === "ModelDownloadHttpError" && "status" in err && typeof err.status === "number") return err.status;
|
|
1285
|
+
return null;
|
|
1286
|
+
}
|
|
753
1287
|
//#endregion
|
|
754
1288
|
export { EmbeddingEncoderAddon, EmbeddingEncoderAddon as default, __exportAll as t };
|