@camstack/addon-post-analysis 1.2.285 → 1.2.287

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (18) hide show
  1. package/THIRD_PARTY_MODELS.md +8 -0
  2. package/dist/{dist-D2tXUMfE.mjs → clip-model-registry-BRHeTbFV.mjs} +1405 -427
  3. package/dist/{dist-ITESpHou.js → clip-model-registry-Cq1X8kTo.js} +1514 -440
  4. package/dist/embedding-encoder/index.js +962 -428
  5. package/dist/embedding-encoder/index.mjs +952 -418
  6. package/dist/pipeline-analytics/_stub.js +2 -2
  7. package/dist/pipeline-analytics/{_virtual_mf-localSharedImportMap___mfe_internal__addon_pipeline_analytics_widgets-7oasHfhD.mjs → _virtual_mf-localSharedImportMap___mfe_internal__addon_pipeline_analytics_widgets-BpJC8P0K.mjs} +2 -2
  8. package/dist/pipeline-analytics/_virtual_mf___mfe_internal__addon_pipeline_analytics_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-3dZf67Z7.mjs +26 -0
  9. package/dist/pipeline-analytics/_virtual_mf___mfe_internal__addon_pipeline_analytics_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-r4OU_ahc.mjs +26 -0
  10. package/dist/pipeline-analytics/{hostInit-B6m75Rnz.mjs → hostInit-BIUCfS5T.mjs} +2 -2
  11. package/dist/pipeline-analytics/index.js +6449 -5096
  12. package/dist/pipeline-analytics/index.mjs +5109 -3756
  13. package/dist/pipeline-analytics/remoteEntry.js +1 -1
  14. package/package.json +1 -1
  15. package/python/test_text_encoder.py +49 -2
  16. package/python/text_encoder_inference.py +31 -14
  17. package/dist/pipeline-analytics/_virtual_mf___mfe_internal__addon_pipeline_analytics_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-CjeM5Bph.mjs +0 -26
  18. package/dist/pipeline-analytics/_virtual_mf___mfe_internal__addon_pipeline_analytics_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-BidXbeas.mjs +0 -26
@@ -2,11 +2,11 @@ Object.defineProperties(exports, {
2
2
  __esModule: { value: true },
3
3
  [Symbol.toStringTag]: { value: "Module" }
4
4
  });
5
- const require_dist = require("../dist-ITESpHou.js");
5
+ const require_clip_model_registry = require("../clip-model-registry-Cq1X8kTo.js");
6
6
  let node_fs = require("node:fs");
7
- node_fs = require_dist.__toESM(node_fs);
7
+ node_fs = require_clip_model_registry.__toESM(node_fs);
8
8
  let node_path = require("node:path");
9
- node_path = require_dist.__toESM(node_path);
9
+ node_path = require_clip_model_registry.__toESM(node_path);
10
10
  let node_child_process = require("node:child_process");
11
11
  let _camstack_system_addon_utils = require("@camstack/system/addon-utils");
12
12
  //#region src/embedding-encoder/shared/process-memory.ts
@@ -28,7 +28,7 @@ let _camstack_system_addon_utils = require("@camstack/system/addon-utils");
28
28
  var IS_LINUX = process.platform === "linux";
29
29
  async function readProcessMemory(pid) {
30
30
  if (IS_LINUX) try {
31
- return require_dist.parseProcStatus(await node_fs.promises.readFile(`/proc/${pid}/status`, "utf8"));
31
+ return require_clip_model_registry.parseProcStatus(await node_fs.promises.readFile(`/proc/${pid}/status`, "utf8"));
32
32
  } catch {
33
33
  return null;
34
34
  }
@@ -56,232 +56,101 @@ function readViaPs(pid) {
56
56
  });
57
57
  }
58
58
  //#endregion
59
- //#region src/embedding-encoder/catalogs/embedding-models.ts
60
- var HF_REPO = "camstack/camstack-models";
61
- var hf = (path) => require_dist.hfModelUrl(HF_REPO, path);
59
+ //#region src/embedding-encoder/shared/framed-python-process.ts
62
60
  /**
63
- * The CLIP BPE tokenizer (HF `tokenizers` format: vocab 49408 + 48894 merges).
64
- * Hosted next to every text-encoder onnx (`.../onnx/tokenizer.json`) and fetched
65
- * as a sibling file so it lands beside the model in the shared models dir.
66
- */
67
- var TOKENIZER_FILE = "tokenizer.json";
68
- var ovFormat = (url, sizeMB) => {
69
- const base = url.split("/").pop() ?? "";
70
- const files = base.endsWith(".xml") ? [base.replace(/\.xml$/, ".bin")] : void 0;
71
- return {
72
- url,
73
- sizeMB,
74
- runtimes: ["python"],
75
- ...files ? { files } : {}
76
- };
77
- };
78
- /**
79
- * Files inside an .mlpackage directory bundle.
80
- * Must be fetched alongside the package root when isDirectory is true.
81
- */
82
- var MLPACKAGE_FILES = [
83
- "Manifest.json",
84
- "Data/com.apple.CoreML/model.mlmodel",
85
- "Data/com.apple.CoreML/weights/weight.bin"
86
- ];
87
- /**
88
- * NO onnx vision builds, deliberately (2026-08-21). The int8 ONNX vision
89
- * exports were measured misaligned with the text encoders (matched image↔text
90
- * cosine ≈ 0.01 vs ≈ 0.22 for the fp16 openvino/coreml builds) — see the
91
- * catalog note in `addon-pipeline/.../model-catalogs.ts`. The image-encode leg
92
- * that consumed them here (`embeddingEncoder.encode` → PythonRawTensorEngine)
93
- * is retired with them: its only caller discarded the vector, and its
94
- * `preprocessForClip` applied OpenAI mean/std that MobileCLIP never used.
95
- * S0 was retired the same day (measured worst of the family on fleet crops).
96
- */
97
- var CLIP_IMAGE_MODELS = [{
98
- id: "mobileclip-s1",
99
- name: "MobileCLIP S1",
100
- description: "Apple MobileCLIP S1 — balanced vision encoder, 512-dim, 256×256 (fp16 OpenVINO/CoreML)",
101
- inputSize: {
102
- width: 256,
103
- height: 256
104
- },
105
- labels: [],
106
- inputNormalization: "none",
107
- formats: {
108
- openvino: ovFormat(hf("clip/mobileclip-s1/openvino/camstack-mobileclip-s1-vision.xml"), 55),
109
- coreml: {
110
- url: hf("clip/mobileclip-s1/coreml/camstack-mobileclip-s1-vision.mlpackage"),
111
- sizeMB: 65,
112
- isDirectory: true,
113
- files: [...MLPACKAGE_FILES],
114
- runtimes: ["python"]
115
- }
116
- }
117
- }, {
118
- id: "mobileclip-s2",
119
- name: "MobileCLIP S2",
120
- description: "Apple MobileCLIP S2 — high-accuracy vision encoder, 512-dim, 256×256 (fp16 OpenVINO/CoreML)",
121
- inputSize: {
122
- width: 256,
123
- height: 256
124
- },
125
- labels: [],
126
- inputNormalization: "none",
127
- formats: {
128
- openvino: ovFormat(hf("clip/mobileclip-s2/openvino/camstack-mobileclip-s2-vision.xml"), 90),
129
- coreml: {
130
- url: hf("clip/mobileclip-s2/coreml/camstack-mobileclip-s2-vision.mlpackage"),
131
- sizeMB: 110,
132
- isDirectory: true,
133
- files: [...MLPACKAGE_FILES],
134
- runtimes: ["python"]
135
- }
136
- }
137
- }];
138
- /**
139
- * The int8 TEXT onnx encoders are healthy — unlike the retired int8 vision
140
- * exports. Verified 2026-08-21: the live search path (fp16 openvino vision
141
- * vectors ⋅ int8 onnx text queries) ranks correctly on the real index, and the
142
- * local cross-check aligns them with the fp16 vision space (cos ≈ 0.22 on
143
- * matched pairs).
144
- */
145
- var CLIP_TEXT_MODELS = [{
146
- id: "mobileclip-s1-text",
147
- name: "MobileCLIP S1 Text Encoder",
148
- description: "Text encoder for MobileCLIP S1, 512-dim, int8 quantized (61 MB)",
149
- inputSize: {
150
- width: 0,
151
- height: 0
152
- },
153
- labels: [],
154
- formats: {
155
- onnx: {
156
- url: hf("clip/mobileclip-s1/onnx/camstack-mobileclip-s1-text.onnx"),
157
- sizeMB: 61,
158
- files: [TOKENIZER_FILE]
159
- },
160
- openvino: ovFormat(hf("clip/mobileclip-s1/openvino/camstack-mobileclip-s1-text.xml"), 121)
161
- }
162
- }, {
163
- id: "mobileclip-s2-text",
164
- name: "MobileCLIP S2 Text Encoder",
165
- description: "Text encoder for MobileCLIP S2, 512-dim, int8 quantized (61 MB)",
166
- inputSize: {
167
- width: 0,
168
- height: 0
169
- },
170
- labels: [],
171
- formats: {
172
- onnx: {
173
- url: hf("clip/mobileclip-s2/onnx/camstack-mobileclip-s2-text.onnx"),
174
- sizeMB: 61,
175
- files: [TOKENIZER_FILE]
176
- },
177
- openvino: ovFormat(hf("clip/mobileclip-s2/openvino/camstack-mobileclip-s2-text.xml"), 121)
178
- }
179
- }];
180
- //#endregion
181
- //#region src/embedding-encoder/shared/noop-logger.ts
182
- var noop = () => {};
183
- function createNoopLogger() {
184
- const logger = {
185
- debug: noop,
186
- info: noop,
187
- warn: noop,
188
- error: noop,
189
- child: () => logger,
190
- withTags: (_tags) => logger
191
- };
192
- return logger;
193
- }
194
- //#endregion
195
- //#region src/embedding-encoder/shared/python-raw-tensor-engine.ts
196
- /**
197
- * Raw-tensor ONNX engine backed by an embedded-Python subprocess
198
- * (`raw_tensor_inference.py`). Replaces the Node `onnxruntime-node` raw-tensor
199
- * engine so the platform ships no Node ONNX runtime. The caller preprocesses to
200
- * a Float32Array; this engine ships it to Python, which runs onnxruntime and
201
- * returns the output tensor. Wire protocol = length-prefixed binary frames
202
- * ([4B LE length][payload]).
61
+ * One embedded-Python subprocess speaking the length-prefixed frame protocol
62
+ * (`[4B LE len][payload]`, ready = `[0x01]`) — the lifecycle both embedding
63
+ * engines share, written once.
64
+ *
65
+ * ## What it guarantees (D649 review)
66
+ *
67
+ * - **Every awaited frame settles.** Waiters are a FIFO queue (the process
68
+ * answers in order). On ANY exit — a crash, an OOM kill, a clean 0, the
69
+ * SIGTERM of `dispose` (code null) — and on a stdin error, every waiter is
70
+ * rejected. The old guard (`code !== 0 && code !== null`) hung the encode.
71
+ * - **A dead process is dead, by name.** After an unexpected exit the engine
72
+ * is marked dead with the exit code/signal, the exit is logged, and every
73
+ * later request rejects immediately with that reason instead of writing to
74
+ * a closed pipe and waiting forever. Holders check {@link isAlive} and
75
+ * rebuild — so a crash costs one failed request, not a silent stop.
76
+ * - **EPIPE never escapes.** stdin carries an `error` listener: a write into a
77
+ * process that just died becomes a rejected request, never an uncaught
78
+ * exception in the addon.
203
79
  */
204
- var PythonRawTensorEngine = class {
80
+ var FramedPythonProcess = class {
81
+ name;
205
82
  pythonPath;
206
- scriptPath;
207
- modelPath;
208
- runtime = "onnx";
209
- device = "cpu";
83
+ args;
84
+ log;
210
85
  process = null;
211
86
  receiveBuffer = Buffer.alloc(0);
212
- pendingResolve = null;
213
- pendingReject = null;
214
- log;
215
- constructor(pythonPath, scriptPath, modelPath, logger) {
87
+ pending = [];
88
+ dead = null;
89
+ disposing = false;
90
+ constructor(name, pythonPath, args, log) {
91
+ this.name = name;
216
92
  this.pythonPath = pythonPath;
217
- this.scriptPath = scriptPath;
218
- this.modelPath = modelPath;
219
- this.log = logger ?? createNoopLogger();
93
+ this.args = args;
94
+ this.log = log;
220
95
  }
221
- async initialize() {
222
- this.process = (0, node_child_process.spawn)(this.pythonPath, [this.scriptPath, this.modelPath], { stdio: [
96
+ /** Spawn and wait for the ready frame. */
97
+ async start() {
98
+ const proc = (0, node_child_process.spawn)(this.pythonPath, [...this.args], { stdio: [
223
99
  "pipe",
224
100
  "pipe",
225
101
  "pipe"
226
102
  ] });
227
- this.process.stderr?.on("data", (chunk) => {
103
+ this.process = proc;
104
+ proc.stderr?.on("data", (chunk) => {
228
105
  const text = chunk.toString().trim();
229
106
  if (text) this.log.warn(text);
230
107
  });
231
- this.process.on("error", (err) => {
232
- this.log.error("Python raw-tensor process error", { meta: { error: err.message } });
233
- this.pendingReject?.(err);
234
- this.pendingReject = null;
235
- this.pendingResolve = null;
108
+ proc.on("error", (err) => {
109
+ this.log.error(`${this.name}: process error`, { meta: { error: err.message } });
110
+ this.markDead(`process error: ${err.message}`);
236
111
  });
237
- this.process.on("exit", (code) => {
238
- if (code !== 0 && code !== null) {
239
- const err = /* @__PURE__ */ new Error(`PythonRawTensorEngine: process exited with code ${code}`);
240
- this.pendingReject?.(err);
241
- this.pendingReject = null;
242
- this.pendingResolve = null;
243
- }
112
+ proc.stdin?.on("error", (err) => {
113
+ this.markDead(`stdin error: ${err.message}`);
114
+ });
115
+ proc.on("exit", (code, signal) => {
116
+ if (this.process === proc) this.process = null;
117
+ const reason = `process exited (code ${String(code)}, signal ${String(signal)})`;
118
+ if (!this.disposing) this.log.warn(`${this.name}: process exited unexpectedly — the engine is dead`, { meta: {
119
+ code,
120
+ signal,
121
+ pid: proc.pid ?? null
122
+ } });
123
+ this.markDead(reason);
244
124
  });
245
- this.process.stdout.on("data", (chunk) => {
125
+ proc.stdout?.on("data", (chunk) => {
246
126
  this.receiveBuffer = Buffer.concat([this.receiveBuffer, chunk]);
247
127
  this.tryReceive();
248
128
  });
249
129
  const ready = await this.receiveFrame();
250
- if (ready.length !== 1 || ready[0] !== 1) throw new Error("PythonRawTensorEngine: unexpected ready frame");
251
- this.log.info("ONNX raw-tensor engine ready (embedded Python)", { meta: { modelPath: this.modelPath } });
130
+ if (ready.length !== 1 || ready[0] !== 1) throw new Error(`${this.name}: unexpected ready frame`);
131
+ }
132
+ /** False once the process died or was disposed — a holder must rebuild. */
133
+ isAlive() {
134
+ return this.process !== null && this.dead === null;
252
135
  }
253
- /** Native pid of the Python subprocess — sampled by the pool memory
254
- * watchdog (`/proc/<pid>/status`). */
255
136
  getPid() {
256
137
  return this.process?.pid ?? null;
257
138
  }
258
- /** Inference requests served since spawn — the denominator that separates
259
- * "RSS grew per unit of work" (leak) from "RSS grew because it worked". */
260
- getRequestCount() {
261
- return this.requestCount;
262
- }
263
- requestCount = 0;
264
- async run(input, inputShape) {
265
- if (!this.process?.stdin) throw new Error("PythonRawTensorEngine: not initialized — call initialize() first");
266
- this.requestCount++;
267
- const ndims = inputShape.length;
268
- const meta = Buffer.allocUnsafe(1 + ndims * 4);
269
- meta.writeUInt8(ndims, 0);
270
- for (let i = 0; i < ndims; i++) meta.writeUInt32LE(inputShape[i], 1 + i * 4);
271
- const dataBuf = Buffer.from(input.buffer, input.byteOffset, input.byteLength);
272
- const payload = Buffer.concat([meta, dataBuf]);
273
- const lenBuf = Buffer.allocUnsafe(4);
274
- lenBuf.writeUInt32LE(payload.length, 0);
275
- this.process.stdin.write(Buffer.concat([lenBuf, payload]));
276
- const resp = await this.receiveFrame();
277
- const floatStart = 1 + resp.readUInt8(0) * 4;
278
- const count = (resp.length - floatStart) / 4;
279
- const out = new Float32Array(count);
280
- for (let i = 0; i < count; i++) out[i] = resp.readFloatLE(floatStart + i * 4);
281
- return out;
139
+ /** Send one request frame and await its answer. Rejects AT ONCE when dead. */
140
+ async request(payload) {
141
+ if (this.dead !== null) throw new Error(`${this.name}: engine is dead — ${this.dead}`);
142
+ const stdin = this.process?.stdin;
143
+ if (!stdin) throw new Error(`${this.name}: not initialized — call initialize() first`);
144
+ const answer = this.receiveFrame();
145
+ const len = Buffer.allocUnsafe(4);
146
+ len.writeUInt32LE(payload.length, 0);
147
+ stdin.write(Buffer.concat([len, payload]));
148
+ return answer;
282
149
  }
283
150
  async dispose() {
284
151
  const proc = this.process;
152
+ this.disposing = true;
153
+ this.markDead("disposed");
285
154
  if (!proc) return;
286
155
  this.process = null;
287
156
  proc.stdin?.end();
@@ -299,185 +168,546 @@ var PythonRawTensorEngine = class {
299
168
  });
300
169
  });
301
170
  }
171
+ markDead(reason) {
172
+ if (this.dead === null) this.dead = reason;
173
+ const err = /* @__PURE__ */ new Error(`${this.name}: ${reason}`);
174
+ for (const waiter of this.pending.splice(0)) waiter.reject(err);
175
+ }
302
176
  receiveFrame() {
303
177
  return new Promise((resolve, reject) => {
304
- this.pendingResolve = resolve;
305
- this.pendingReject = reject;
178
+ this.pending.push({
179
+ resolve,
180
+ reject
181
+ });
306
182
  });
307
183
  }
308
184
  tryReceive() {
309
- if (this.receiveBuffer.length < 4) return;
310
- const length = this.receiveBuffer.readUInt32LE(0);
311
- if (this.receiveBuffer.length < 4 + length) return;
312
- const payload = Buffer.from(this.receiveBuffer.subarray(4, 4 + length));
313
- this.receiveBuffer = this.receiveBuffer.subarray(4 + length);
314
- const resolve = this.pendingResolve;
315
- this.pendingResolve = null;
316
- this.pendingReject = null;
317
- resolve?.(payload);
185
+ while (this.receiveBuffer.length >= 4) {
186
+ const length = this.receiveBuffer.readUInt32LE(0);
187
+ if (this.receiveBuffer.length < 4 + length) return;
188
+ const payload = Buffer.from(this.receiveBuffer.subarray(4, 4 + length));
189
+ this.receiveBuffer = this.receiveBuffer.subarray(4 + length);
190
+ this.pending.shift()?.resolve(payload);
191
+ }
318
192
  }
319
193
  };
320
194
  //#endregion
195
+ //#region src/embedding-encoder/shared/noop-logger.ts
196
+ var noop = () => {};
197
+ function createNoopLogger() {
198
+ const logger = {
199
+ debug: noop,
200
+ info: noop,
201
+ warn: noop,
202
+ error: noop,
203
+ child: () => logger,
204
+ withTags: (_tags) => logger
205
+ };
206
+ return logger;
207
+ }
208
+ //#endregion
321
209
  //#region src/embedding-encoder/shared/python-text-encoder-engine.ts
322
- /**
323
- * CLIP text-encoder engine backed by an embedded-Python subprocess
324
- * (`text_encoder_inference.py`). Tokenization happens IN Python via the HF
325
- * `tokenizers` Rust BPE (exact by construction) — this replaces the former
326
- * hand-rolled TypeScript CLIP BPE.
327
- *
328
- * The caller sends raw UTF-8 text; Python tokenizes (truncate/pad to 77), runs
329
- * onnxruntime, and returns the embedding tensor. Wire protocol = length-prefixed
330
- * binary frames ([4B LE length][payload]):
331
- * ready (in): [0x01]
332
- * request (out): UTF-8 text bytes
333
- * response (in): [1B ndims][dims × 4B LE uint32][float32 LE data]
334
- */
335
210
  var PythonTextEncoderEngine = class {
336
- pythonPath;
337
211
  scriptPath;
338
212
  modelPath;
339
213
  tokenizerPath;
340
- process = null;
341
- receiveBuffer = Buffer.alloc(0);
342
- pendingResolve = null;
343
- pendingReject = null;
214
+ window;
215
+ proc;
344
216
  log;
345
- constructor(pythonPath, scriptPath, modelPath, tokenizerPath, logger) {
346
- this.pythonPath = pythonPath;
217
+ requestCount = 0;
218
+ constructor(pythonPath, scriptPath, modelPath, tokenizerPath, window, logger) {
347
219
  this.scriptPath = scriptPath;
348
220
  this.modelPath = modelPath;
349
221
  this.tokenizerPath = tokenizerPath;
222
+ this.window = window;
350
223
  this.log = logger ?? createNoopLogger();
224
+ this.proc = new FramedPythonProcess("PythonTextEncoderEngine", pythonPath, this.spawnArgs(), this.log);
351
225
  }
352
226
  async initialize() {
353
- this.process = (0, node_child_process.spawn)(this.pythonPath, [
354
- this.scriptPath,
355
- this.modelPath,
356
- this.tokenizerPath
357
- ], { stdio: [
358
- "pipe",
359
- "pipe",
360
- "pipe"
361
- ] });
362
- this.process.stderr?.on("data", (chunk) => {
363
- const text = chunk.toString().trim();
364
- if (text) this.log.warn(text);
365
- });
366
- this.process.on("error", (err) => {
367
- this.log.error("Python text-encoder process error", { meta: { error: err.message } });
368
- this.pendingReject?.(err);
369
- this.pendingReject = null;
370
- this.pendingResolve = null;
371
- });
372
- this.process.on("exit", (code) => {
373
- if (code !== 0 && code !== null) {
374
- const err = /* @__PURE__ */ new Error(`PythonTextEncoderEngine: process exited with code ${code}`);
375
- this.pendingReject?.(err);
376
- this.pendingReject = null;
377
- this.pendingResolve = null;
378
- }
379
- });
380
- this.process.stdout.on("data", (chunk) => {
381
- this.receiveBuffer = Buffer.concat([this.receiveBuffer, chunk]);
382
- this.tryReceive();
383
- });
384
- const ready = await this.receiveFrame();
385
- if (ready.length !== 1 || ready[0] !== 1) throw new Error("PythonTextEncoderEngine: unexpected ready frame");
227
+ await this.proc.start();
386
228
  this.log.info("CLIP text-encoder engine ready (embedded Python)", { meta: {
387
229
  modelPath: this.modelPath,
388
- tokenizerPath: this.tokenizerPath
230
+ tokenizerPath: this.tokenizerPath,
231
+ contextLength: this.window.contextLength,
232
+ padId: this.window.padId
389
233
  } });
390
234
  }
235
+ /** The subprocess argv after the interpreter — the window rides on it. */
236
+ spawnArgs() {
237
+ return [
238
+ this.scriptPath,
239
+ this.modelPath,
240
+ this.tokenizerPath,
241
+ "--context-length",
242
+ String(this.window.contextLength),
243
+ "--pad-id",
244
+ String(this.window.padId)
245
+ ];
246
+ }
247
+ /** False once the process died (crash, OOM kill) or was disposed. */
248
+ isAlive() {
249
+ return this.proc.isAlive();
250
+ }
391
251
  /** Native pid of the Python subprocess — sampled by the pool memory
392
252
  * watchdog (`/proc/<pid>/status`). */
393
253
  getPid() {
394
- return this.process?.pid ?? null;
254
+ return this.proc.getPid();
395
255
  }
396
256
  /** Encode requests served since spawn — the denominator that separates
397
257
  * "RSS grew per unit of work" (leak) from "RSS grew because it worked". */
398
258
  getRequestCount() {
399
259
  return this.requestCount;
400
260
  }
401
- requestCount = 0;
402
261
  /** Tokenize + encode `text` into the model embedding (float32). */
403
262
  async encode(text) {
404
- if (!this.process?.stdin) throw new Error("PythonTextEncoderEngine: not initialized — call initialize() first");
405
263
  this.requestCount++;
406
- const payload = Buffer.from(text, "utf-8");
407
- const lenBuf = Buffer.allocUnsafe(4);
408
- lenBuf.writeUInt32LE(payload.length, 0);
409
- this.process.stdin.write(Buffer.concat([lenBuf, payload]));
410
- const resp = await this.receiveFrame();
411
- const floatStart = 1 + resp.readUInt8(0) * 4;
412
- const count = (resp.length - floatStart) / 4;
413
- const out = new Float32Array(count);
414
- for (let i = 0; i < count; i++) out[i] = resp.readFloatLE(floatStart + i * 4);
415
- return out;
264
+ return decodeTensorFrame(await this.proc.request(Buffer.from(text, "utf-8")));
416
265
  }
417
266
  async dispose() {
418
- const proc = this.process;
419
- if (!proc) return;
420
- this.process = null;
421
- proc.stdin?.end();
422
- proc.kill("SIGTERM");
423
- await new Promise((resolve) => {
424
- const timer = setTimeout(() => {
425
- try {
426
- proc.kill("SIGKILL");
427
- } catch {}
428
- resolve();
429
- }, 5e3);
430
- proc.once("exit", () => {
431
- clearTimeout(timer);
432
- resolve();
433
- });
434
- });
267
+ await this.proc.dispose();
435
268
  }
436
- receiveFrame() {
437
- return new Promise((resolve, reject) => {
438
- this.pendingResolve = resolve;
439
- this.pendingReject = reject;
440
- });
269
+ };
270
+ /** Decode `[1B ndims][dims × 4B][float32 data]`. */
271
+ function decodeTensorFrame(resp) {
272
+ const floatStart = 1 + resp.readUInt8(0) * 4;
273
+ const count = (resp.length - floatStart) / 4;
274
+ const out = new Float32Array(count);
275
+ for (let i = 0; i < count; i++) out[i] = resp.readFloatLE(floatStart + i * 4);
276
+ return out;
277
+ }
278
+ //#endregion
279
+ //#region src/embedding-encoder/shared/python-raw-tensor-engine.ts
280
+ /**
281
+ * Raw-tensor ONNX engine backed by an embedded-Python subprocess
282
+ * (`raw_tensor_inference.py`). Replaces the Node `onnxruntime-node` raw-tensor
283
+ * engine so the platform ships no Node ONNX runtime. The caller preprocesses to
284
+ * a Float32Array; this engine ships it to Python, which runs onnxruntime and
285
+ * returns the output tensor. Wire protocol = length-prefixed binary frames
286
+ * ([4B LE length][payload]); the process lifecycle is `FramedPythonProcess`.
287
+ */
288
+ var PythonRawTensorEngine = class {
289
+ modelPath;
290
+ runtime = "onnx";
291
+ device = "cpu";
292
+ proc;
293
+ log;
294
+ requestCount = 0;
295
+ constructor(pythonPath, scriptPath, modelPath, logger) {
296
+ this.modelPath = modelPath;
297
+ this.log = logger ?? createNoopLogger();
298
+ this.proc = new FramedPythonProcess("PythonRawTensorEngine", pythonPath, [scriptPath, modelPath], this.log);
441
299
  }
442
- tryReceive() {
443
- if (this.receiveBuffer.length < 4) return;
444
- const length = this.receiveBuffer.readUInt32LE(0);
445
- if (this.receiveBuffer.length < 4 + length) return;
446
- const payload = Buffer.from(this.receiveBuffer.subarray(4, 4 + length));
447
- this.receiveBuffer = this.receiveBuffer.subarray(4 + length);
448
- const resolve = this.pendingResolve;
449
- this.pendingResolve = null;
450
- this.pendingReject = null;
451
- resolve?.(payload);
300
+ async initialize() {
301
+ await this.proc.start();
302
+ this.log.info("ONNX raw-tensor engine ready (embedded Python)", { meta: { modelPath: this.modelPath } });
303
+ }
304
+ /** False once the process died (crash, OOM kill) or was disposed. */
305
+ isAlive() {
306
+ return this.proc.isAlive();
307
+ }
308
+ /** Native pid of the Python subprocess — sampled by the pool memory
309
+ * watchdog (`/proc/<pid>/status`). */
310
+ getPid() {
311
+ return this.proc.getPid();
312
+ }
313
+ /** Inference requests served since spawn — the denominator that separates
314
+ * "RSS grew per unit of work" (leak) from "RSS grew because it worked". */
315
+ getRequestCount() {
316
+ return this.requestCount;
317
+ }
318
+ async run(input, inputShape) {
319
+ this.requestCount++;
320
+ const ndims = inputShape.length;
321
+ const meta = Buffer.allocUnsafe(1 + ndims * 4);
322
+ meta.writeUInt8(ndims, 0);
323
+ for (let i = 0; i < ndims; i++) meta.writeUInt32LE(inputShape[i], 1 + i * 4);
324
+ const dataBuf = Buffer.from(input.buffer, input.byteOffset, input.byteLength);
325
+ return decodeTensorFrame(await this.proc.request(Buffer.concat([meta, dataBuf])));
326
+ }
327
+ async dispose() {
328
+ await this.proc.dispose();
329
+ }
330
+ };
331
+ var ActiveClipModel = class {
332
+ deps;
333
+ cached = null;
334
+ lastRead = null;
335
+ constructor(deps) {
336
+ this.deps = deps;
337
+ }
338
+ /**
339
+ * The last row a read RETURNED, without reading — `null` while no read has
340
+ * ever succeeded (the config fallback is a guess, not a known model). The
341
+ * idle sweep exempts this model's text tower (`clip-text-engines.ts`).
342
+ */
343
+ known() {
344
+ return this.lastRead;
345
+ }
346
+ async get() {
347
+ const now = this.deps.now();
348
+ if (this.cached !== null && now - this.cached.at < 3e4) return this.cached.modelId;
349
+ const row = await this.deps.readRow();
350
+ if (row === null) {
351
+ const kept = this.lastRead ?? this.deps.fallbackModelId();
352
+ this.deps.logger.warn("CLIP encoder kept its model — the cluster row could not be read", { meta: {
353
+ modelId: kept,
354
+ source: this.lastRead !== null ? "last-read" : "addon-config"
355
+ } });
356
+ return kept;
357
+ }
358
+ const previous = this.lastRead;
359
+ if (row !== previous) this.deps.logger.info("CLIP encoder follows the cluster model", { meta: {
360
+ modelId: row,
361
+ previous
362
+ } });
363
+ this.lastRead = row;
364
+ this.cached = {
365
+ modelId: row,
366
+ at: now
367
+ };
368
+ if (previous !== null && row !== previous) this.deps.onChange?.(row, previous);
369
+ return row;
370
+ }
371
+ };
372
+ var CrashBudget = class {
373
+ deps;
374
+ recent = /* @__PURE__ */ new Map();
375
+ failedModels = /* @__PURE__ */ new Map();
376
+ constructor(deps) {
377
+ this.deps = deps;
378
+ }
379
+ /** Throw the named refusal when the model is in `failed`. Spawns nothing. */
380
+ check(modelId) {
381
+ const failed = this.failedModels.get(modelId);
382
+ if (failed === void 0) return;
383
+ throw require_clip_model_registry.embeddingModelFailedError(failed);
384
+ }
385
+ /** Record one crash or load failure. Returns true when it made the model `failed`. */
386
+ record(modelId, reason) {
387
+ if (this.failedModels.has(modelId)) return false;
388
+ const now = this.now();
389
+ const windowMs = this.deps.windowMs ?? 6e5;
390
+ const times = (this.recent.get(modelId) ?? []).filter((t) => now - t < windowMs);
391
+ times.push(now);
392
+ this.recent.set(modelId, times);
393
+ if (times.length < (this.deps.maxCrashes ?? 3)) return false;
394
+ const failed = {
395
+ modelId,
396
+ tower: this.deps.tower,
397
+ nodeId: this.deps.nodeId(),
398
+ crashes: times.length,
399
+ windowMs,
400
+ sinceMs: now,
401
+ lastReason: reason
402
+ };
403
+ this.failedModels.set(modelId, failed);
404
+ this.recent.delete(modelId);
405
+ this.deps.logger.error("CLIP tower FAILED — crash budget exhausted, requests refused until reset", { meta: { ...failed } });
406
+ return true;
407
+ }
408
+ failed() {
409
+ return [...this.failedModels.values()];
410
+ }
411
+ /** Clear one model (or all). Returns the ids that were failed. */
412
+ reset(modelId, why = "operator") {
413
+ const ids = modelId !== void 0 ? [modelId] : [...this.failedModels.keys()];
414
+ const cleared = ids.filter((id) => this.failedModels.delete(id));
415
+ for (const id of ids) this.recent.delete(id);
416
+ if (cleared.length > 0) this.deps.logger.info("CLIP tower failed state cleared", { meta: {
417
+ tower: this.deps.tower,
418
+ nodeId: this.deps.nodeId(),
419
+ cleared,
420
+ why
421
+ } });
422
+ return cleared;
423
+ }
424
+ now() {
425
+ return (this.deps.now ?? Date.now)();
452
426
  }
453
427
  };
454
428
  //#endregion
455
- //#region src/embedding-encoder/addon/clip-models.ts
429
+ //#region src/embedding-encoder/addon/failed-models-report.ts
456
430
  /**
457
- * `mobileclip-s0` was retired on 2026-08-21 (measured worst of the family on
458
- * real fleet crops). A stored config still naming it resolves to the default
459
- * via {@link getModelMeta}'s fallback — never to a dangling id.
431
+ * Every failed tower on this node. Each row already names the node: the budget
432
+ * stamps it when the model goes `failed` (it is in the refusal too), and this
433
+ * report does not re-derive it — one authority for the fact.
460
434
  */
461
- var CLIP_MODEL_META = {
462
- "mobileclip-s1": {
463
- imageModelId: "mobileclip-s1",
464
- textModelId: "mobileclip-s1-text",
465
- embeddingDim: 512,
466
- inputSize: 256,
467
- tokenizerType: "clip"
468
- },
469
- "mobileclip-s2": {
470
- imageModelId: "mobileclip-s2",
471
- textModelId: "mobileclip-s2-text",
472
- embeddingDim: 512,
473
- inputSize: 256,
474
- tokenizerType: "clip"
435
+ function reportFailedModels(budgets) {
436
+ return [...budgets.image.failed(), ...budgets.text.failed()];
437
+ }
438
+ /** Every model whose requirements are terminally unmeetable on this node; rows stamped by the gate. */
439
+ function reportFailedRequirements(requirements) {
440
+ return [...requirements.image.failed(), ...requirements.text.failed()];
441
+ }
442
+ /** The operator's clear (one model, or all) of BOTH terminal states on this node — the result names it. */
443
+ function resetFailedModelsOn(nodeId, towers, modelId) {
444
+ const cleared = [
445
+ ...towers.budgets.image.reset(modelId),
446
+ ...towers.budgets.text.reset(modelId),
447
+ ...towers.requirements.image.reset(modelId),
448
+ ...towers.requirements.text.reset(modelId)
449
+ ];
450
+ return {
451
+ cleared: [...new Set(cleared)],
452
+ nodeId
453
+ };
454
+ }
455
+ /**
456
+ * The cluster row now names `modelId`: a fresh start for THAT model's towers
457
+ * only (D6's second exit from `failed`). The model the row left keeps its
458
+ * state — a rollback to a model that crashed three times must not buy three
459
+ * more spawns nobody asked for, and a search that still names it
460
+ * (`searchObjectEvents({ modelId })`) is refused with the reset hint, as
461
+ * before the flip.
462
+ */
463
+ function clearFailedOnModelChange(towers, modelId) {
464
+ return [...new Set([
465
+ ...towers.budgets.image.reset(modelId, "model-change"),
466
+ ...towers.budgets.text.reset(modelId, "model-change"),
467
+ ...towers.requirements.image.reset(modelId, "model-change"),
468
+ ...towers.requirements.text.reset(modelId, "model-change")
469
+ ])];
470
+ }
471
+ //#endregion
472
+ //#region src/embedding-encoder/addon/model-engine-slot.ts
473
+ var ModelEngineSlot = class {
474
+ build;
475
+ budget;
476
+ requirements;
477
+ onDead;
478
+ current = null;
479
+ inflight = /* @__PURE__ */ new Map();
480
+ /** The last model asked for — a build for any other model is stale on arrival. */
481
+ wanted = null;
482
+ /** Serialises switches so two models never tear the slot down concurrently. */
483
+ switching = Promise.resolve();
484
+ /**
485
+ * Bumped by {@link dispose}: a build that started before it is stale when it
486
+ * lands and is disposed on arrival — shutdown leaves no process behind.
487
+ */
488
+ generation = 0;
489
+ constructor(build, budget, requirements, onDead = () => {}) {
490
+ this.build = build;
491
+ this.budget = budget;
492
+ this.requirements = requirements;
493
+ this.onDead = onDead;
494
+ }
495
+ held() {
496
+ return this.current;
497
+ }
498
+ async ensure(modelId) {
499
+ this.wanted = modelId;
500
+ this.dropIfDead();
501
+ if (this.current?.modelId === modelId) return this.current.engine;
502
+ this.budget.check(modelId);
503
+ this.requirements.check(modelId);
504
+ const pending = this.inflight.get(modelId);
505
+ if (pending !== void 0) return pending;
506
+ const generation = this.generation;
507
+ const guarded = this.switching.then(async () => {
508
+ this.dropIfDead();
509
+ if (this.current?.modelId === modelId) return this.current.engine;
510
+ this.budget.check(modelId);
511
+ this.requirements.check(modelId);
512
+ await this.disposeCurrent();
513
+ let engine;
514
+ try {
515
+ engine = await this.build(modelId);
516
+ } catch (err) {
517
+ throw this.requirements.refuse(modelId, err);
518
+ }
519
+ this.requirements.satisfied(modelId);
520
+ try {
521
+ await engine.initialize();
522
+ } catch (err) {
523
+ this.budget.record(modelId, `load failed: ${err instanceof Error ? err.message : String(err)}`);
524
+ throw err;
525
+ }
526
+ if (this.generation !== generation) {
527
+ await engine.dispose();
528
+ throw new Error(`engine for "${modelId}" cancelled: the slot was disposed while building`);
529
+ }
530
+ if (this.wanted !== modelId) {
531
+ await engine.dispose();
532
+ throw new Error(`engine for "${modelId}" superseded while building`);
533
+ }
534
+ this.current = {
535
+ modelId,
536
+ engine
537
+ };
538
+ return engine;
539
+ }).finally(() => {
540
+ this.inflight.delete(modelId);
541
+ });
542
+ this.inflight.set(modelId, guarded);
543
+ this.switching = guarded.then(() => void 0, () => void 0);
544
+ return guarded;
545
+ }
546
+ /**
547
+ * Dispose the held engine AND cancel every build in flight: each is disposed
548
+ * when it lands (see `generation`). Resolves once in-flight builds settled.
549
+ */
550
+ async dispose() {
551
+ this.generation += 1;
552
+ await this.disposeCurrent();
553
+ await this.switching;
554
+ }
555
+ async disposeCurrent() {
556
+ const held = this.current;
557
+ this.current = null;
558
+ await held?.engine.dispose();
559
+ }
560
+ dropIfDead() {
561
+ const held = this.current;
562
+ if (held === null || held.engine.isAlive()) return;
563
+ this.current = null;
564
+ this.onDead(held.modelId);
565
+ this.budget.record(held.modelId, "process died after loading");
566
+ held.engine.dispose().catch(() => {});
567
+ }
568
+ };
569
+ /**
570
+ * Thrown by the requirements stage when a retry can never help: the catalog
571
+ * does not know the id, the server said the file does not exist (404/410), a
572
+ * declared sibling is missing. Same process as the gate, so the class is the
573
+ * marker; the gate never matches wording.
574
+ */
575
+ var PermanentRequirementsFailure = class extends Error {
576
+ constructor(message) {
577
+ super(message);
578
+ this.name = "PermanentRequirementsFailure";
475
579
  }
476
580
  };
477
- var DEFAULT_CLIP_MODEL = "mobileclip-s1";
478
- function getModelMeta(modelId) {
479
- return CLIP_MODEL_META[modelId] ?? CLIP_MODEL_META["mobileclip-s1"];
581
+ function isPermanentRequirementsFailure(err) {
582
+ return err instanceof PermanentRequirementsFailure;
480
583
  }
584
+ var RequirementsGate = class {
585
+ deps;
586
+ unavailable = /* @__PURE__ */ new Map();
587
+ constructor(deps) {
588
+ this.deps = deps;
589
+ }
590
+ /**
591
+ * Consulted BEFORE a build. Throws the named refusal — and downloads nothing
592
+ * — while the model is terminal or inside its cooldown. Returns otherwise.
593
+ */
594
+ check(modelId) {
595
+ const state = this.unavailable.get(modelId);
596
+ if (state === void 0) return;
597
+ if (state.terminal !== null) {
598
+ this.unavailable.set(modelId, {
599
+ ...state,
600
+ refused: state.refused + 1
601
+ });
602
+ throw require_clip_model_registry.embeddingRequirementsFailedError(state.terminal);
603
+ }
604
+ const now = this.now();
605
+ if (now >= state.cooldownUntilMs) return;
606
+ this.unavailable.set(modelId, {
607
+ ...state,
608
+ refused: state.refused + 1
609
+ });
610
+ throw require_clip_model_registry.embeddingRequirementsUnavailableError(this.on(modelId), `retry in ${String(Math.ceil((state.cooldownUntilMs - now) / 1e3))}s — ${state.reason}`);
611
+ }
612
+ /**
613
+ * A build failed before any process existed. Starts (or doubles) the
614
+ * cooldown; a PERMANENT failure past the bound makes the model terminal.
615
+ * Logs at the transitions only and returns the named refusal to throw.
616
+ */
617
+ refuse(modelId, err) {
618
+ const reason = err instanceof Error ? err.message : String(err);
619
+ const now = this.now();
620
+ const prior = this.unavailable.get(modelId);
621
+ const attempts = (prior?.attempts ?? 0) + 1;
622
+ const permanent = isPermanentRequirementsFailure(err);
623
+ const permanentAttempts = (prior?.permanentAttempts ?? 0) + (permanent ? 1 : 0);
624
+ const cooldownMs = Math.min(this.deps.cooldownMaxMs ?? 3e5, (this.deps.cooldownBaseMs ?? 5e3) * 2 ** (attempts - 1));
625
+ const base = {
626
+ sinceMs: prior?.sinceMs ?? now,
627
+ reason,
628
+ refused: (prior?.refused ?? 0) + 1,
629
+ attempts,
630
+ permanentAttempts,
631
+ cooldownUntilMs: now + cooldownMs,
632
+ terminal: null
633
+ };
634
+ if (prior === void 0) this.deps.logger.warn("CLIP model requirements unavailable — requests refused until they are; the crash budget is untouched", { meta: {
635
+ modelId,
636
+ tower: this.deps.tower,
637
+ nodeId: this.deps.nodeId(),
638
+ reason,
639
+ cooldownMs
640
+ } });
641
+ if (permanent && permanentAttempts >= (this.deps.maxAttempts ?? 3)) {
642
+ const terminal = {
643
+ ...this.on(modelId),
644
+ attempts: permanentAttempts,
645
+ sinceMs: now,
646
+ lastReason: reason
647
+ };
648
+ this.unavailable.set(modelId, {
649
+ ...base,
650
+ terminal
651
+ });
652
+ this.deps.logger.error("CLIP model requirements can never be met — terminally failed, requests refused until reset", { meta: {
653
+ ...terminal,
654
+ refusedSoFar: base.refused
655
+ } });
656
+ return require_clip_model_registry.embeddingRequirementsFailedError(terminal);
657
+ }
658
+ this.unavailable.set(modelId, base);
659
+ return require_clip_model_registry.embeddingRequirementsUnavailableError(this.on(modelId), reason);
660
+ }
661
+ /** A build got its engine: the requirements are met again. Logs the recovery once. */
662
+ satisfied(modelId) {
663
+ const prior = this.unavailable.get(modelId);
664
+ if (prior === void 0) return;
665
+ this.unavailable.delete(modelId);
666
+ this.deps.logger.info("CLIP model requirements available again", { meta: {
667
+ modelId,
668
+ tower: this.deps.tower,
669
+ nodeId: this.deps.nodeId(),
670
+ unavailableMs: this.now() - prior.sinceMs,
671
+ refused: prior.refused,
672
+ attempts: prior.attempts,
673
+ lastReason: prior.reason
674
+ } });
675
+ }
676
+ /** Models terminally `requirements-failed` on this node, as `getInfo` reports them. */
677
+ failed() {
678
+ return [...this.unavailable.values()].map((s) => s.terminal).filter((t) => t !== null);
679
+ }
680
+ /**
681
+ * Clear one model (or all): terminal state AND cooldown. Returns the ids
682
+ * that were TERMINAL — a cooldown is not a state an operator is told about.
683
+ */
684
+ reset(modelId, why = "operator") {
685
+ const ids = modelId !== void 0 ? [modelId] : [...this.unavailable.keys()];
686
+ const cleared = ids.filter((id) => this.unavailable.get(id)?.terminal !== null && this.unavailable.has(id));
687
+ for (const id of ids) this.unavailable.delete(id);
688
+ if (cleared.length > 0) this.deps.logger.info("CLIP model requirements-failed state cleared", { meta: {
689
+ tower: this.deps.tower,
690
+ nodeId: this.deps.nodeId(),
691
+ cleared,
692
+ why
693
+ } });
694
+ return cleared;
695
+ }
696
+ /** Models currently refused (cooldown or terminal) — exposed for tests. */
697
+ pending() {
698
+ return [...this.unavailable.keys()];
699
+ }
700
+ on(modelId) {
701
+ return {
702
+ modelId,
703
+ tower: this.deps.tower,
704
+ nodeId: this.deps.nodeId()
705
+ };
706
+ }
707
+ now() {
708
+ return (this.deps.now ?? Date.now)();
709
+ }
710
+ };
481
711
  //#endregion
482
712
  //#region src/embedding-encoder/addon/clip-preprocessing.ts
483
713
  var CLIP_MEAN = [
@@ -520,11 +750,286 @@ function l2Normalize(vec) {
520
750
  if (norm > 0) for (let i = 0; i < vec.length; i++) vec[i] /= norm;
521
751
  return vec;
522
752
  }
753
+ /**
754
+ * A tower unused this long is disposed ({@link ClipTextEngines.evictIdle}) —
755
+ * EXCEPT the tower of the model the cluster row names, which stays loaded for
756
+ * the life of the process exactly as it did before D649: the steady-state
757
+ * search must not pay a Python spawn plus a model load after a quiet quarter
758
+ * hour. Only the OTHER towers (SigLIP2 after a comparison, the old model after
759
+ * a flip) are let go when idle. When the active model is unknown (the row was
760
+ * never read), nothing is evicted — the side that keeps memory as before.
761
+ * The encoder is `placement: any-node` and cannot tell locally whether it is
762
+ * the node that answers text queries, so towers are loaded ON DEMAND — never
763
+ * prefetched on a model flip (565 MB on every node). The first query after a
764
+ * flip pays the load; if it fails, the search fails loudly rather than reading
765
+ * as "no results".
766
+ */
767
+ var TEXT_ENGINE_IDLE_MS = 15 * 6e4;
768
+ var ClipTextEngines = class {
769
+ deps;
770
+ /** Insertion order is recency: a hit is re-inserted at the end. */
771
+ engines = /* @__PURE__ */ new Map();
772
+ /** Single-flight per model: two concurrent first queries spawn one engine. */
773
+ inflight = /* @__PURE__ */ new Map();
774
+ /**
775
+ * Encodes in flight per model. An engine with any is NEVER evicted: disposing
776
+ * it would fail a search that is already running. When every held engine is
777
+ * busy the set goes over `maxEngines` temporarily and shrinks when one drains.
778
+ */
779
+ busy = /* @__PURE__ */ new Map();
780
+ /** When each held tower last finished an encode (or was built). */
781
+ lastUsed = /* @__PURE__ */ new Map();
782
+ /** Bumped by {@link disposeAll}: a build landing after it is disposed on arrival. */
783
+ generation = 0;
784
+ constructor(deps) {
785
+ this.deps = deps;
786
+ }
787
+ /** Resolve a model id to its CLIP model, or throw naming the id. */
788
+ requireModel(modelId) {
789
+ const model = this.deps.registry.resolve(modelId);
790
+ if (model === null) throw new Error(`EmbeddingEncoder: "${modelId}" has no CLIP metadata — not a CLIP model`);
791
+ return model;
792
+ }
793
+ async encode(modelId, text) {
794
+ const model = this.requireModel(modelId);
795
+ this.busy.set(modelId, (this.busy.get(modelId) ?? 0) + 1);
796
+ let vector;
797
+ try {
798
+ vector = await (await this.engineFor(model)).encode(text);
799
+ } finally {
800
+ const left = (this.busy.get(modelId) ?? 1) - 1;
801
+ if (left > 0) this.busy.set(modelId, left);
802
+ else this.busy.delete(modelId);
803
+ if (this.engines.has(modelId)) this.lastUsed.set(modelId, this.now());
804
+ this.scheduleEviction();
805
+ }
806
+ if (vector.length !== model.meta.embeddingDim) throw new Error(`EmbeddingEncoder: ${model.meta.textModelId} produced ${String(vector.length)} dims, its metadata declares ${String(model.meta.embeddingDim)} — refusing the query rather than truncating it`);
807
+ return {
808
+ vector,
809
+ model
810
+ };
811
+ }
812
+ /**
813
+ * Dispose every IDLE tower unused for {@link TEXT_ENGINE_IDLE_MS} (a timer in
814
+ * the addon calls this), except the active model's. Returns the evicted
815
+ * model ids; each is logged. An unknown active model evicts nothing.
816
+ */
817
+ async evictIdle(idleMs = TEXT_ENGINE_IDLE_MS) {
818
+ const active = this.deps.activeModelId();
819
+ if (active === null) return [];
820
+ const now = this.now();
821
+ const idle = [...this.engines.keys()].filter((id) => id !== active && (this.busy.get(id) ?? 0) === 0 && now - (this.lastUsed.get(id) ?? now) >= idleMs);
822
+ const evicted = [];
823
+ for (const modelId of idle) {
824
+ if ((this.busy.get(modelId) ?? 0) > 0 || !this.engines.has(modelId)) continue;
825
+ evicted.push(modelId);
826
+ this.deps.logger.info("CLIP text tower evicted — idle", { meta: {
827
+ modelId,
828
+ idleMs: now - (this.lastUsed.get(modelId) ?? now)
829
+ } });
830
+ await this.restart(modelId);
831
+ }
832
+ return evicted;
833
+ }
834
+ /** Live engines, for the memory watchdog. */
835
+ live() {
836
+ return this.engines;
837
+ }
838
+ /** Dispose one engine; the next query rebuilds it. */
839
+ async restart(modelId) {
840
+ const engine = this.engines.get(modelId);
841
+ if (engine === void 0) return;
842
+ this.engines.delete(modelId);
843
+ this.lastUsed.delete(modelId);
844
+ await engine.dispose();
845
+ }
846
+ /** Dispose every tower AND cancel the builds in flight (disposed as they land). */
847
+ async disposeAll() {
848
+ this.generation += 1;
849
+ const all = [...this.engines.values()];
850
+ this.engines.clear();
851
+ this.lastUsed.clear();
852
+ await Promise.all(all.map((e) => e.dispose()));
853
+ await Promise.allSettled(this.inflight.values());
854
+ }
855
+ async engineFor(model) {
856
+ const held = this.engines.get(model.modelId);
857
+ if (held !== void 0 && held.isAlive()) {
858
+ this.engines.delete(model.modelId);
859
+ this.engines.set(model.modelId, held);
860
+ return held;
861
+ }
862
+ if (held !== void 0) {
863
+ this.deps.logger.warn("CLIP text tower died — dropping it and rebuilding", { meta: {
864
+ modelId: model.modelId,
865
+ textModelId: model.meta.textModelId
866
+ } });
867
+ this.engines.delete(model.modelId);
868
+ this.lastUsed.delete(model.modelId);
869
+ this.deps.budget.record(model.modelId, "process died after loading");
870
+ held.dispose().catch(() => {});
871
+ }
872
+ const pending = this.inflight.get(model.modelId);
873
+ if (pending !== void 0) return pending;
874
+ this.deps.budget.check(model.modelId);
875
+ this.deps.requirements.check(model.modelId);
876
+ const textEntry = this.deps.textCatalog.find((e) => e.id === model.meta.textModelId);
877
+ if (textEntry === void 0) throw this.deps.requirements.refuse(model.modelId, new PermanentRequirementsFailure(`EmbeddingEncoder: text model "${model.meta.textModelId}" (for ${model.modelId}) is not in the text catalog`));
878
+ const generation = this.generation;
879
+ const build = this.buildAndLoad(model, textEntry);
880
+ this.inflight.set(model.modelId, build);
881
+ try {
882
+ const engine = await build;
883
+ if (this.generation !== generation) {
884
+ await engine.dispose();
885
+ throw new Error(`text tower for "${model.modelId}" cancelled: disposed while building`);
886
+ }
887
+ this.engines.set(model.modelId, engine);
888
+ this.lastUsed.set(model.modelId, this.now());
889
+ this.deps.logger.info("CLIP text tower loaded", { meta: {
890
+ modelId: model.modelId,
891
+ textModelId: model.meta.textModelId
892
+ } });
893
+ this.scheduleEviction();
894
+ return engine;
895
+ } finally {
896
+ this.inflight.delete(model.modelId);
897
+ }
898
+ }
899
+ /**
900
+ * Two stages, two owners (`model-requirements.ts`): `build` resolves files
901
+ * and runtime and constructs the tower — a throw there is refused by name
902
+ * and spends nothing; `initialize()` spawns and loads — a throw there is a
903
+ * load failure and spends the crash budget.
904
+ */
905
+ async buildAndLoad(model, textEntry) {
906
+ let engine;
907
+ try {
908
+ engine = await this.deps.build({
909
+ model,
910
+ textEntry,
911
+ meta: model.meta
912
+ });
913
+ } catch (err) {
914
+ throw this.deps.requirements.refuse(model.modelId, err);
915
+ }
916
+ this.deps.requirements.satisfied(model.modelId);
917
+ try {
918
+ await engine.initialize();
919
+ } catch (err) {
920
+ this.deps.budget.record(model.modelId, `load failed: ${err instanceof Error ? err.message : String(err)}`);
921
+ throw err;
922
+ }
923
+ return engine;
924
+ }
925
+ now() {
926
+ return (this.deps.now ?? Date.now)();
927
+ }
928
+ /** Fire-and-log: eviction errors never reach an encode. */
929
+ scheduleEviction() {
930
+ this.evictBeyond(this.deps.maxEngines ?? 2).catch((err) => {
931
+ this.deps.logger.warn("CLIP text tower eviction failed", { meta: { error: err instanceof Error ? err.message : String(err) } });
932
+ });
933
+ }
934
+ /** Evict the coldest IDLE engines until at most `max` remain, or none is idle. */
935
+ async evictBeyond(max) {
936
+ while (this.engines.size > max) {
937
+ const coldestIdle = [...this.engines.keys()].find((id) => (this.busy.get(id) ?? 0) === 0);
938
+ if (coldestIdle === void 0) return;
939
+ await this.restart(coldestIdle);
940
+ }
941
+ }
942
+ /** Encodes in flight on a model — exposed for the eviction tests. */
943
+ inFlight(modelId) {
944
+ return this.busy.get(modelId) ?? 0;
945
+ }
946
+ };
523
947
  //#endregion
524
948
  //#region src/embedding-encoder/addon/index.ts
525
- var EmbeddingEncoderAddon = class extends require_dist.BaseAddon {
526
- imageRawEngine = null;
527
- textEngine = null;
949
+ /** Memory-watchdog key prefix of a text tower; the suffix is its CLIP model id. */
950
+ var TEXT_ENGINE_KEY_PREFIX = "clip-text:";
951
+ /** How often idle text towers are looked for (see `TEXT_ENGINE_IDLE_MS`). */
952
+ var IDLE_SWEEP_MS = 6e4;
953
+ /** A model URL answering one of these does not exist; retrying changes nothing. */
954
+ var PERMANENT_HTTP_STATUS = new Set([404, 410]);
955
+ var EmbeddingEncoderAddon = class extends require_clip_model_registry.BaseAddon {
956
+ /**
957
+ * Everything on this node that can be terminally refused (`failed-models-report.ts`):
958
+ * the respawn budgets per tower (D6, `crash-budget.ts`) and the requirements
959
+ * gates (`model-requirements.ts` — cooldown, and terminal for what can never
960
+ * be met). All in memory: a runner respawn starts them from zero. Every row
961
+ * and refusal names this node.
962
+ */
963
+ towers = {
964
+ budgets: {
965
+ image: new CrashBudget({
966
+ tower: "image",
967
+ nodeId: () => this.localNodeId(),
968
+ logger: {
969
+ info: (message, extras) => this.ctx.logger.info(message, extras),
970
+ error: (message, extras) => this.ctx.logger.error(message, extras)
971
+ }
972
+ }),
973
+ text: new CrashBudget({
974
+ tower: "text",
975
+ nodeId: () => this.localNodeId(),
976
+ logger: {
977
+ info: (message, extras) => this.ctx.logger.info(message, extras),
978
+ error: (message, extras) => this.ctx.logger.error(message, extras)
979
+ }
980
+ })
981
+ },
982
+ requirements: {
983
+ image: new RequirementsGate({
984
+ tower: "image",
985
+ nodeId: () => this.localNodeId(),
986
+ logger: {
987
+ info: (message, extras) => this.ctx.logger.info(message, extras),
988
+ warn: (message, extras) => this.ctx.logger.warn(message, extras),
989
+ error: (message, extras) => this.ctx.logger.error(message, extras)
990
+ }
991
+ }),
992
+ text: new RequirementsGate({
993
+ tower: "text",
994
+ nodeId: () => this.localNodeId(),
995
+ logger: {
996
+ info: (message, extras) => this.ctx.logger.info(message, extras),
997
+ warn: (message, extras) => this.ctx.logger.warn(message, extras),
998
+ error: (message, extras) => this.ctx.logger.error(message, extras)
999
+ }
1000
+ })
1001
+ }
1002
+ };
1003
+ /** The image tower, one model at a time, single-flight per model (`model-engine-slot.ts`). */
1004
+ imageEngine = new ModelEngineSlot((modelId) => this.buildImageEngine(modelId), this.towers.budgets.image, this.towers.requirements.image, (modelId) => this.ctx.logger.warn("CLIP image tower died — dropping it and rebuilding", { meta: { modelId } }));
1005
+ /** One text tower per CLIP model, bounded (see `clip-text-engines.ts`). */
1006
+ textEngines = new ClipTextEngines({
1007
+ registry: require_clip_model_registry.BUILTIN_CLIP_MODELS,
1008
+ textCatalog: require_clip_model_registry.CLIP_TEXT_MODELS,
1009
+ build: (spec) => this.buildTextEngine(spec),
1010
+ budget: this.towers.budgets.text,
1011
+ requirements: this.towers.requirements.text,
1012
+ activeModelId: () => this.activeModel.known(),
1013
+ logger: {
1014
+ info: (message, extras) => this.ctx.logger.info(message, extras),
1015
+ warn: (message, extras) => this.ctx.logger.warn(message, extras)
1016
+ }
1017
+ });
1018
+ /** Disposes NON-active text towers idle for `TEXT_ENGINE_IDLE_MS` (never prefetched). */
1019
+ idleSweep = null;
1020
+ /** The cluster row's CLIP model — what an unnamed request encodes in (D649). */
1021
+ activeModel = new ActiveClipModel({
1022
+ readRow: () => this.readClusterRow(),
1023
+ fallbackModelId: () => require_clip_model_registry.BUILTIN_CLIP_MODELS.resolve(this.config.modelId) !== null ? this.config.modelId : require_clip_model_registry.DEFAULT_CLIP_MODEL,
1024
+ now: () => Date.now(),
1025
+ logger: {
1026
+ info: (message, extras) => this.ctx.logger.info(message, extras),
1027
+ warn: (message, extras) => this.ctx.logger.warn(message, extras)
1028
+ },
1029
+ onChange: (modelId) => {
1030
+ clearFailedOnModelChange(this.towers, modelId);
1031
+ }
1032
+ });
528
1033
  models = null;
529
1034
  /**
530
1035
  * RSS bound + periodic memory telemetry for the two Python engines
@@ -536,27 +1041,27 @@ var EmbeddingEncoderAddon = class extends require_dist.BaseAddon {
536
1041
  * stopped in `onShutdown`.
537
1042
  */
538
1043
  memoryGuard = null;
539
- /** Single-flight guards for the lazy engine builds — a watchdog restart
540
- * followed by two concurrent encodes must not spawn the engine twice
541
- * (the same bug detection-pipeline fixed three times; see its
542
- * `engineFactoryInflight`). */
543
- imageEngineInflight = null;
544
- textEngineInflight = null;
545
1044
  constructor() {
546
- super({ modelId: DEFAULT_CLIP_MODEL });
1045
+ super({ modelId: require_clip_model_registry.DEFAULT_CLIP_MODEL });
547
1046
  }
548
1047
  async onInitialize() {
549
1048
  const modelsDir = await this.resolveModelsDir();
550
- this.models = new _camstack_system_addon_utils.ModelDownloadService(modelsDir, [...CLIP_IMAGE_MODELS, ...CLIP_TEXT_MODELS]);
551
- this.memoryGuard = new require_dist.PoolMemoryWatchdog({
552
- policy: require_dist.resolvePoolMemoryPolicy(process.env),
1049
+ this.models = new _camstack_system_addon_utils.ModelDownloadService(modelsDir, [...require_clip_model_registry.CLIP_IMAGE_MODELS, ...require_clip_model_registry.CLIP_TEXT_MODELS]);
1050
+ this.memoryGuard = new require_clip_model_registry.PoolMemoryWatchdog({
1051
+ policy: require_clip_model_registry.resolvePoolMemoryPolicy(process.env),
553
1052
  log: this.ctx.logger.child("pool-memory"),
554
1053
  sample: () => this.sampleEngineMemory(),
555
1054
  restart: (key) => this.restartEngineForMemory(key)
556
1055
  });
557
1056
  this.memoryGuard.start();
1057
+ this.idleSweep = setInterval(() => {
1058
+ this.textEngines.evictIdle().catch((err) => {
1059
+ this.ctx.logger.warn("CLIP text tower idle sweep failed", { meta: { error: err instanceof Error ? err.message : String(err) } });
1060
+ });
1061
+ }, IDLE_SWEEP_MS);
1062
+ this.idleSweep.unref?.();
558
1063
  return [{
559
- capability: require_dist.embeddingEncoderCapability,
1064
+ capability: require_clip_model_registry.embeddingEncoderCapability,
560
1065
  provider: this
561
1066
  }];
562
1067
  }
@@ -565,13 +1070,15 @@ var EmbeddingEncoderAddon = class extends require_dist.BaseAddon {
565
1070
  async sampleEngineMemory() {
566
1071
  const engines = [{
567
1072
  key: "clip-image",
568
- pid: this.imageRawEngine?.getPid() ?? null,
569
- requests: this.imageRawEngine?.getRequestCount() ?? 0
570
- }, {
571
- key: "clip-text",
572
- pid: this.textEngine?.getPid() ?? null,
573
- requests: this.textEngine?.getRequestCount() ?? 0
574
- }];
1073
+ modelId: this.imageEngine.held()?.modelId ?? null,
1074
+ pid: this.imageEngine.held()?.engine.getPid() ?? null,
1075
+ requests: this.imageEngine.held()?.engine.getRequestCount() ?? 0
1076
+ }, ...[...this.textEngines.live()].map(([modelId, engine]) => ({
1077
+ key: `${TEXT_ENGINE_KEY_PREFIX}${modelId}`,
1078
+ modelId,
1079
+ pid: engine.getPid(),
1080
+ requests: engine.getRequestCount()
1081
+ }))];
575
1082
  const out = [];
576
1083
  for (const engine of engines) {
577
1084
  if (engine.pid === null) continue;
@@ -589,7 +1096,7 @@ var EmbeddingEncoderAddon = class extends require_dist.BaseAddon {
589
1096
  pids: [engine.pid],
590
1097
  rssBytes: mem.rssBytes,
591
1098
  meta: {
592
- modelId: this.config.modelId,
1099
+ modelId: engine.modelId,
593
1100
  vmMb: mb(mem.vmBytes),
594
1101
  peakRssMb: mb(mem.hwmBytes),
595
1102
  swapMb: mb(mem.swapBytes),
@@ -606,110 +1113,125 @@ var EmbeddingEncoderAddon = class extends require_dist.BaseAddon {
606
1113
  * surface errors, never silent loss. */
607
1114
  async restartEngineForMemory(key) {
608
1115
  if (key === "clip-image") {
609
- const engine = this.imageRawEngine;
610
- if (!engine) return;
611
- this.imageRawEngine = null;
612
- await engine.dispose();
1116
+ await this.imageEngine.dispose();
613
1117
  return;
614
1118
  }
615
- const engine = this.textEngine;
616
- if (!engine) return;
617
- this.textEngine = null;
618
- await engine.dispose();
1119
+ if (key.startsWith(TEXT_ENGINE_KEY_PREFIX)) await this.textEngines.restart(key.slice(10));
1120
+ }
1121
+ /** The cluster row's `clip-embedding` model, or `null` when unreadable. */
1122
+ async readClusterRow() {
1123
+ return require_clip_model_registry.resolveClusterModelPin(this.ctx.api, require_clip_model_registry.CLIP_EMBEDDING_STEP_ID, { warn: (message, extras) => this.ctx.logger.warn(message, extras) });
619
1124
  }
620
1125
  async encode(input) {
621
1126
  const { crop, width, height } = input;
622
- await this.ensureImageEngine();
623
- const meta = getModelMeta(this.config.modelId);
1127
+ const model = this.textEngines.requireModel(await this.activeModel.get());
1128
+ const imageEngine = await this.imageEngine.ensure(model.modelId);
1129
+ const meta = model.meta;
624
1130
  const start = Date.now();
625
- const preprocessed = preprocessForClip(Buffer.isBuffer(crop) ? crop : Buffer.from(crop), width, height, meta.inputSize, meta.inputSize);
626
- const output = await this.imageRawEngine.run(preprocessed, [
1131
+ const preprocessed = preprocessForClip(Buffer.isBuffer(crop) ? crop : Buffer.from(crop), width, height, model.inputSize, model.inputSize);
1132
+ const output = await imageEngine.run(preprocessed, [
627
1133
  1,
628
1134
  3,
629
- meta.inputSize,
630
- meta.inputSize
1135
+ model.inputSize,
1136
+ model.inputSize
631
1137
  ]);
632
- const sliced = output.length > meta.embeddingDim ? output.slice(0, meta.embeddingDim) : output;
633
- const normalized = l2Normalize(new Float32Array(sliced));
1138
+ if (output.length !== meta.embeddingDim) throw new Error(`EmbeddingEncoder: ${model.modelId} image tower produced ${String(output.length)} dims, its metadata declares ${String(meta.embeddingDim)}`);
1139
+ const normalized = l2Normalize(new Float32Array(output));
634
1140
  return {
635
1141
  embedding: Array.from(normalized),
636
1142
  inferenceMs: Date.now() - start
637
1143
  };
638
1144
  }
1145
+ /**
1146
+ * Encode a query in ONE model's space: the named `modelId`, else the cluster
1147
+ * row's (D649). The token window, pad id, tokenizer and dimension all come
1148
+ * from that model's catalog metadata.
1149
+ */
639
1150
  async encodeText(input) {
640
- const { text } = input;
641
- await this.ensureTextEngine();
642
- const meta = getModelMeta(this.config.modelId);
1151
+ const modelId = input.modelId ?? await this.activeModel.get();
643
1152
  const start = Date.now();
644
- if (!this.textEngine) throw new Error("EmbeddingEncoder: text engine not loaded — ensureTextEngine() must run first");
645
- const output = await this.textEngine.encode(text);
646
- const sliced = output.length > meta.embeddingDim ? output.slice(0, meta.embeddingDim) : output;
647
- const normalized = l2Normalize(new Float32Array(sliced));
1153
+ const { vector } = await this.textEngines.encode(modelId, input.text);
648
1154
  return {
649
- embedding: Array.from(normalized),
1155
+ embedding: Array.from(l2Normalize(new Float32Array(vector))),
650
1156
  inferenceMs: Date.now() - start
651
1157
  };
652
1158
  }
1159
+ /** THIS node's answer — the cap is one provider per node (see the cap's docblock). */
653
1160
  async getInfo() {
654
- const meta = getModelMeta(this.config.modelId);
1161
+ const model = this.textEngines.requireModel(await this.activeModel.get());
1162
+ const nodeId = this.localNodeId();
655
1163
  return {
656
- modelId: this.config.modelId,
657
- embeddingDim: meta.embeddingDim,
658
- ready: this.imageRawEngine !== null
1164
+ modelId: model.modelId,
1165
+ embeddingDim: model.meta.embeddingDim,
1166
+ ready: this.imageEngine.held()?.modelId === model.modelId,
1167
+ nodeId,
1168
+ failedModels: [...reportFailedModels(this.towers.budgets)],
1169
+ failedRequirements: [...reportFailedRequirements(this.towers.requirements)]
659
1170
  };
660
1171
  }
661
- async ensureImageEngine() {
662
- if (this.imageRawEngine) return;
663
- if (this.imageEngineInflight) return this.imageEngineInflight;
664
- const meta = getModelMeta(this.config.modelId);
665
- const imageEntry = CLIP_IMAGE_MODELS.find((m) => m.id === meta.imageModelId);
666
- if (!imageEntry) throw new Error(`EmbeddingEncoderAddon: unknown image model "${meta.imageModelId}"`);
667
- const inflight = this.resolveForEntry(imageEntry, "image");
668
- this.imageEngineInflight = inflight;
669
- try {
670
- await inflight;
671
- } finally {
672
- this.imageEngineInflight = null;
673
- }
1172
+ /** The operator's explicit clear of THIS node's `failed` state (D6); names the node. */
1173
+ async resetFailedModels(input) {
1174
+ return resetFailedModelsOn(this.localNodeId(), this.towers, input.modelId);
674
1175
  }
675
- async ensureTextEngine() {
676
- if (this.textEngine) return;
677
- if (this.textEngineInflight) return this.textEngineInflight;
678
- const meta = getModelMeta(this.config.modelId);
679
- const textEntry = CLIP_TEXT_MODELS.find((m) => m.id === meta.textModelId);
680
- if (!textEntry) throw new Error(`EmbeddingEncoderAddon: unknown text model "${meta.textModelId}"`);
681
- const inflight = this.resolveForEntry(textEntry, "text");
682
- this.textEngineInflight = inflight;
683
- try {
684
- await inflight;
685
- } finally {
686
- this.textEngineInflight = null;
687
- }
1176
+ /** The logical node (a forked child's `<node>/<addon>` id stripped to the node). */
1177
+ localNodeId() {
1178
+ return require_clip_model_registry.normalizeNodeId(this.ctx.kernel?.localNodeId);
688
1179
  }
689
- async resolveForEntry(entry, target) {
690
- const engineLogger = this.ctx.logger.withTags({ modelId: entry.id });
691
- const modelPath = await this.models.ensure(entry.id, "onnx");
1180
+ /**
1181
+ * The REQUIREMENTS stage, shared by both towers: the embedded Python, its
1182
+ * requirements, then the model file (downloaded on first use). Anything that
1183
+ * throws here is refused as `model-requirements-unavailable` and spends no
1184
+ * crash budget (`model-requirements.ts`). The runtime is asked for first —
1185
+ * the cheap question before the 185–565 MB download (D54's rule).
1186
+ */
1187
+ async prepareOnnx(entry) {
692
1188
  const pythonPath = await this.ctx.deps.ensurePython();
693
1189
  if (!pythonPath) throw new Error("EmbeddingEncoder: embedded Python is unavailable — cannot run ONNX embeddings. ctx.deps.ensurePython() returned null (portable Python download likely failed).");
694
1190
  const pythonDir = resolveEmbeddingPythonDir();
695
1191
  await this.ctx.deps.installPythonRequirements(node_path.join(pythonDir, "requirements-embedding.txt"));
696
- if (target === "image") {
697
- const rawEngine = new PythonRawTensorEngine(pythonPath, node_path.join(pythonDir, "raw_tensor_inference.py"), modelPath, engineLogger);
698
- await rawEngine.initialize();
699
- this.imageRawEngine = rawEngine;
700
- return;
1192
+ if (entry.formats.onnx === void 0) throw new PermanentRequirementsFailure(`EmbeddingEncoder: ${entry.id} has no onnx build in its catalog entry (formats: ${Object.keys(entry.formats).join(", ") || "none"}) — no download can supply it`);
1193
+ let modelPath;
1194
+ try {
1195
+ modelPath = await this.models.ensure(entry.id, "onnx");
1196
+ } catch (err) {
1197
+ const status = downloadHttpStatusOf(err);
1198
+ if (status !== null && PERMANENT_HTTP_STATUS.has(status)) throw new PermanentRequirementsFailure(`EmbeddingEncoder: ${entry.id} is not at its catalog URL (HTTP ${String(status)}) — fix the catalog entry`);
1199
+ throw err;
701
1200
  }
702
- const tokenizerPath = node_path.join(node_path.dirname(modelPath), TOKENIZER_FILE);
703
- if (!node_fs.existsSync(tokenizerPath)) throw new Error(`EmbeddingEncoder: CLIP tokenizer not found at "${tokenizerPath}" — the tokenizer.json sibling download likely failed.`);
704
- const textEngine = new PythonTextEncoderEngine(pythonPath, node_path.join(pythonDir, "text_encoder_inference.py"), modelPath, tokenizerPath, engineLogger);
705
- await textEngine.initialize();
706
- this.textEngine = textEngine;
1201
+ return {
1202
+ modelPath,
1203
+ pythonPath,
1204
+ pythonDir
1205
+ };
1206
+ }
1207
+ async buildImageEngine(modelId) {
1208
+ const entry = require_clip_model_registry.CLIP_IMAGE_MODELS.find((m) => m.id === modelId);
1209
+ if (!entry) throw new PermanentRequirementsFailure(`EmbeddingEncoderAddon: unknown image model "${modelId}" — not in the CLIP image catalog`);
1210
+ const { modelPath, pythonPath, pythonDir } = await this.prepareOnnx(entry);
1211
+ return new PythonRawTensorEngine(pythonPath, node_path.join(pythonDir, "raw_tensor_inference.py"), modelPath, this.ctx.logger.withTags({ modelId: entry.id }));
1212
+ }
1213
+ /**
1214
+ * One text tower. Tokenization happens IN Python via the HF `tokenizers`
1215
+ * library, with the tokenizer the model's metadata names — a declared sibling
1216
+ * of the text onnx, so `ModelDownloadService.ensure()` fetched it next to the
1217
+ * model file — and the model's own token window.
1218
+ */
1219
+ async buildTextEngine(spec) {
1220
+ const { modelPath, pythonPath, pythonDir } = await this.prepareOnnx(spec.textEntry);
1221
+ const tokenizerPath = node_path.join(node_path.dirname(modelPath), spec.meta.tokenizerFile);
1222
+ if (!node_fs.existsSync(tokenizerPath)) throw new PermanentRequirementsFailure(`EmbeddingEncoder: tokenizer not found at "${tokenizerPath}" — the ${spec.meta.tokenizerFile} sibling download of ${spec.textEntry.id} likely failed.`);
1223
+ return new PythonTextEncoderEngine(pythonPath, node_path.join(pythonDir, "text_encoder_inference.py"), modelPath, tokenizerPath, {
1224
+ contextLength: spec.meta.contextLength,
1225
+ padId: spec.meta.padId
1226
+ }, this.ctx.logger.withTags({ modelId: spec.textEntry.id }));
707
1227
  }
708
1228
  async onShutdown() {
709
1229
  this.memoryGuard?.stop();
1230
+ if (this.idleSweep !== null) clearInterval(this.idleSweep);
1231
+ this.idleSweep = null;
710
1232
  this.memoryGuard = null;
711
- await this.imageRawEngine?.dispose();
712
- await this.textEngine?.dispose();
1233
+ await this.imageEngine.dispose();
1234
+ await this.textEngines.disposeAll();
713
1235
  }
714
1236
  globalSettingsSchema() {
715
1237
  return this.schema({ sections: [{
@@ -719,9 +1241,9 @@ var EmbeddingEncoderAddon = class extends require_dist.BaseAddon {
719
1241
  fields: [{
720
1242
  type: "text",
721
1243
  key: "modelId",
722
- label: "Model ID",
723
- description: "CLIP model identifier to use for image/text embedding",
724
- default: DEFAULT_CLIP_MODEL
1244
+ label: "Fallback model ID",
1245
+ description: "Used only until the cluster \"Semantic search model\" row has been read once. The encoder follows that row (D649).",
1246
+ default: require_clip_model_registry.DEFAULT_CLIP_MODEL
725
1247
  }]
726
1248
  }] });
727
1249
  }
@@ -742,6 +1264,18 @@ function resolveEmbeddingPythonDir() {
742
1264
  for (const c of candidates) if (node_fs.existsSync(node_path.join(c, "raw_tensor_inference.py"))) return c;
743
1265
  throw new Error(`EmbeddingEncoder: python/ dir (raw_tensor_inference.py) not found. Searched:\n${candidates.join("\n")}`);
744
1266
  }
1267
+ /**
1268
+ * HTTP status of a model-download failure, recognised by SHAPE (`name` +
1269
+ * numeric `status`), never by `instanceof` or a framework import.
1270
+ * `@camstack/system` is host-resolved (not bundled), so importing a guard it
1271
+ * only exports from a newer server would stop this whole entry from linking
1272
+ * on a node still running the older server — a deploy-order hazard for a
1273
+ * one-line classification. `null` = not an HTTP answer (unreachable, other).
1274
+ */
1275
+ function downloadHttpStatusOf(err) {
1276
+ if (err instanceof Error && err.name === "ModelDownloadHttpError" && "status" in err && typeof err.status === "number") return err.status;
1277
+ return null;
1278
+ }
745
1279
  //#endregion
746
1280
  exports.EmbeddingEncoderAddon = EmbeddingEncoderAddon;
747
1281
  exports.default = EmbeddingEncoderAddon;