@camstack/addon-post-analysis 1.2.285 → 1.2.286

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (18) hide show
  1. package/THIRD_PARTY_MODELS.md +8 -0
  2. package/dist/{dist-ITESpHou.js → clip-model-registry-CY0FGcpQ.js} +1483 -403
  3. package/dist/{dist-D2tXUMfE.mjs → clip-model-registry-D4mC2M7F.mjs} +1374 -390
  4. package/dist/embedding-encoder/index.js +962 -428
  5. package/dist/embedding-encoder/index.mjs +952 -418
  6. package/dist/pipeline-analytics/_stub.js +2 -2
  7. package/dist/pipeline-analytics/{_virtual_mf-localSharedImportMap___mfe_internal__addon_pipeline_analytics_widgets-7oasHfhD.mjs → _virtual_mf-localSharedImportMap___mfe_internal__addon_pipeline_analytics_widgets-CkfvSztj.mjs} +2 -2
  8. package/dist/pipeline-analytics/_virtual_mf___mfe_internal__addon_pipeline_analytics_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-NKbCrEkH.mjs +26 -0
  9. package/dist/pipeline-analytics/_virtual_mf___mfe_internal__addon_pipeline_analytics_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-BAYoLHsN.mjs +26 -0
  10. package/dist/pipeline-analytics/{hostInit-B6m75Rnz.mjs → hostInit-CelYj4QN.mjs} +2 -2
  11. package/dist/pipeline-analytics/index.js +5863 -5053
  12. package/dist/pipeline-analytics/index.mjs +4510 -3700
  13. package/dist/pipeline-analytics/remoteEntry.js +1 -1
  14. package/package.json +1 -1
  15. package/python/test_text_encoder.py +49 -2
  16. package/python/text_encoder_inference.py +31 -14
  17. package/dist/pipeline-analytics/_virtual_mf___mfe_internal__addon_pipeline_analytics_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-CjeM5Bph.mjs +0 -26
  18. package/dist/pipeline-analytics/_virtual_mf___mfe_internal__addon_pipeline_analytics_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-BidXbeas.mjs +0 -26
@@ -1,4 +1,4 @@
1
- import { $t as resolvePoolMemoryPolicy, Dt as embeddingEncoderCapability, Kt as parseProcStatus, Nt as hfModelUrl, gn as BaseAddon, nt as PoolMemoryWatchdog } from "../dist-D2tXUMfE.mjs";
1
+ import { Bn as normalizeNodeId, Ht as embeddingRequirementsUnavailableError, Rt as embeddingEncoderCapability, Vt as embeddingRequirementsFailedError, a as CLIP_EMBEDDING_STEP_ID, c as resolveClusterModelPin, ft as PoolMemoryWatchdog, i as CLIP_TEXT_MODELS, jn as BaseAddon, mn as resolvePoolMemoryPolicy, n as DEFAULT_CLIP_MODEL, r as CLIP_IMAGE_MODELS, sn as parseProcStatus, t as BUILTIN_CLIP_MODELS, zt as embeddingModelFailedError } from "../clip-model-registry-D4mC2M7F.mjs";
2
2
  import { createRequire } from "node:module";
3
3
  import * as fs from "node:fs";
4
4
  import * as path$1 from "node:path";
@@ -64,232 +64,101 @@ function readViaPs(pid) {
64
64
  });
65
65
  }
66
66
  //#endregion
67
- //#region src/embedding-encoder/catalogs/embedding-models.ts
68
- var HF_REPO = "camstack/camstack-models";
69
- var hf = (path) => hfModelUrl(HF_REPO, path);
67
+ //#region src/embedding-encoder/shared/framed-python-process.ts
70
68
  /**
71
- * The CLIP BPE tokenizer (HF `tokenizers` format: vocab 49408 + 48894 merges).
72
- * Hosted next to every text-encoder onnx (`.../onnx/tokenizer.json`) and fetched
73
- * as a sibling file so it lands beside the model in the shared models dir.
74
- */
75
- var TOKENIZER_FILE = "tokenizer.json";
76
- var ovFormat = (url, sizeMB) => {
77
- const base = url.split("/").pop() ?? "";
78
- const files = base.endsWith(".xml") ? [base.replace(/\.xml$/, ".bin")] : void 0;
79
- return {
80
- url,
81
- sizeMB,
82
- runtimes: ["python"],
83
- ...files ? { files } : {}
84
- };
85
- };
86
- /**
87
- * Files inside an .mlpackage directory bundle.
88
- * Must be fetched alongside the package root when isDirectory is true.
89
- */
90
- var MLPACKAGE_FILES = [
91
- "Manifest.json",
92
- "Data/com.apple.CoreML/model.mlmodel",
93
- "Data/com.apple.CoreML/weights/weight.bin"
94
- ];
95
- /**
96
- * NO onnx vision builds, deliberately (2026-08-21). The int8 ONNX vision
97
- * exports were measured misaligned with the text encoders (matched image↔text
98
- * cosine ≈ 0.01 vs ≈ 0.22 for the fp16 openvino/coreml builds) — see the
99
- * catalog note in `addon-pipeline/.../model-catalogs.ts`. The image-encode leg
100
- * that consumed them here (`embeddingEncoder.encode` → PythonRawTensorEngine)
101
- * is retired with them: its only caller discarded the vector, and its
102
- * `preprocessForClip` applied OpenAI mean/std that MobileCLIP never used.
103
- * S0 was retired the same day (measured worst of the family on fleet crops).
104
- */
105
- var CLIP_IMAGE_MODELS = [{
106
- id: "mobileclip-s1",
107
- name: "MobileCLIP S1",
108
- description: "Apple MobileCLIP S1 — balanced vision encoder, 512-dim, 256×256 (fp16 OpenVINO/CoreML)",
109
- inputSize: {
110
- width: 256,
111
- height: 256
112
- },
113
- labels: [],
114
- inputNormalization: "none",
115
- formats: {
116
- openvino: ovFormat(hf("clip/mobileclip-s1/openvino/camstack-mobileclip-s1-vision.xml"), 55),
117
- coreml: {
118
- url: hf("clip/mobileclip-s1/coreml/camstack-mobileclip-s1-vision.mlpackage"),
119
- sizeMB: 65,
120
- isDirectory: true,
121
- files: [...MLPACKAGE_FILES],
122
- runtimes: ["python"]
123
- }
124
- }
125
- }, {
126
- id: "mobileclip-s2",
127
- name: "MobileCLIP S2",
128
- description: "Apple MobileCLIP S2 — high-accuracy vision encoder, 512-dim, 256×256 (fp16 OpenVINO/CoreML)",
129
- inputSize: {
130
- width: 256,
131
- height: 256
132
- },
133
- labels: [],
134
- inputNormalization: "none",
135
- formats: {
136
- openvino: ovFormat(hf("clip/mobileclip-s2/openvino/camstack-mobileclip-s2-vision.xml"), 90),
137
- coreml: {
138
- url: hf("clip/mobileclip-s2/coreml/camstack-mobileclip-s2-vision.mlpackage"),
139
- sizeMB: 110,
140
- isDirectory: true,
141
- files: [...MLPACKAGE_FILES],
142
- runtimes: ["python"]
143
- }
144
- }
145
- }];
146
- /**
147
- * The int8 TEXT onnx encoders are healthy — unlike the retired int8 vision
148
- * exports. Verified 2026-08-21: the live search path (fp16 openvino vision
149
- * vectors ⋅ int8 onnx text queries) ranks correctly on the real index, and the
150
- * local cross-check aligns them with the fp16 vision space (cos ≈ 0.22 on
151
- * matched pairs).
152
- */
153
- var CLIP_TEXT_MODELS = [{
154
- id: "mobileclip-s1-text",
155
- name: "MobileCLIP S1 Text Encoder",
156
- description: "Text encoder for MobileCLIP S1, 512-dim, int8 quantized (61 MB)",
157
- inputSize: {
158
- width: 0,
159
- height: 0
160
- },
161
- labels: [],
162
- formats: {
163
- onnx: {
164
- url: hf("clip/mobileclip-s1/onnx/camstack-mobileclip-s1-text.onnx"),
165
- sizeMB: 61,
166
- files: [TOKENIZER_FILE]
167
- },
168
- openvino: ovFormat(hf("clip/mobileclip-s1/openvino/camstack-mobileclip-s1-text.xml"), 121)
169
- }
170
- }, {
171
- id: "mobileclip-s2-text",
172
- name: "MobileCLIP S2 Text Encoder",
173
- description: "Text encoder for MobileCLIP S2, 512-dim, int8 quantized (61 MB)",
174
- inputSize: {
175
- width: 0,
176
- height: 0
177
- },
178
- labels: [],
179
- formats: {
180
- onnx: {
181
- url: hf("clip/mobileclip-s2/onnx/camstack-mobileclip-s2-text.onnx"),
182
- sizeMB: 61,
183
- files: [TOKENIZER_FILE]
184
- },
185
- openvino: ovFormat(hf("clip/mobileclip-s2/openvino/camstack-mobileclip-s2-text.xml"), 121)
186
- }
187
- }];
188
- //#endregion
189
- //#region src/embedding-encoder/shared/noop-logger.ts
190
- var noop = () => {};
191
- function createNoopLogger() {
192
- const logger = {
193
- debug: noop,
194
- info: noop,
195
- warn: noop,
196
- error: noop,
197
- child: () => logger,
198
- withTags: (_tags) => logger
199
- };
200
- return logger;
201
- }
202
- //#endregion
203
- //#region src/embedding-encoder/shared/python-raw-tensor-engine.ts
204
- /**
205
- * Raw-tensor ONNX engine backed by an embedded-Python subprocess
206
- * (`raw_tensor_inference.py`). Replaces the Node `onnxruntime-node` raw-tensor
207
- * engine so the platform ships no Node ONNX runtime. The caller preprocesses to
208
- * a Float32Array; this engine ships it to Python, which runs onnxruntime and
209
- * returns the output tensor. Wire protocol = length-prefixed binary frames
210
- * ([4B LE length][payload]).
69
+ * One embedded-Python subprocess speaking the length-prefixed frame protocol
70
+ * (`[4B LE len][payload]`, ready = `[0x01]`) — the lifecycle both embedding
71
+ * engines share, written once.
72
+ *
73
+ * ## What it guarantees (D649 review)
74
+ *
75
+ * - **Every awaited frame settles.** Waiters are a FIFO queue (the process
76
+ * answers in order). On ANY exit — a crash, an OOM kill, a clean 0, the
77
+ * SIGTERM of `dispose` (code null) — and on a stdin error, every waiter is
78
+ * rejected. The old guard (`code !== 0 && code !== null`) hung the encode.
79
+ * - **A dead process is dead, by name.** After an unexpected exit the engine
80
+ * is marked dead with the exit code/signal, the exit is logged, and every
81
+ * later request rejects immediately with that reason instead of writing to
82
+ * a closed pipe and waiting forever. Holders check {@link isAlive} and
83
+ * rebuild — so a crash costs one failed request, not a silent stop.
84
+ * - **EPIPE never escapes.** stdin carries an `error` listener: a write into a
85
+ * process that just died becomes a rejected request, never an uncaught
86
+ * exception in the addon.
211
87
  */
212
- var PythonRawTensorEngine = class {
88
+ var FramedPythonProcess = class {
89
+ name;
213
90
  pythonPath;
214
- scriptPath;
215
- modelPath;
216
- runtime = "onnx";
217
- device = "cpu";
91
+ args;
92
+ log;
218
93
  process = null;
219
94
  receiveBuffer = Buffer.alloc(0);
220
- pendingResolve = null;
221
- pendingReject = null;
222
- log;
223
- constructor(pythonPath, scriptPath, modelPath, logger) {
95
+ pending = [];
96
+ dead = null;
97
+ disposing = false;
98
+ constructor(name, pythonPath, args, log) {
99
+ this.name = name;
224
100
  this.pythonPath = pythonPath;
225
- this.scriptPath = scriptPath;
226
- this.modelPath = modelPath;
227
- this.log = logger ?? createNoopLogger();
101
+ this.args = args;
102
+ this.log = log;
228
103
  }
229
- async initialize() {
230
- this.process = spawn(this.pythonPath, [this.scriptPath, this.modelPath], { stdio: [
104
+ /** Spawn and wait for the ready frame. */
105
+ async start() {
106
+ const proc = spawn(this.pythonPath, [...this.args], { stdio: [
231
107
  "pipe",
232
108
  "pipe",
233
109
  "pipe"
234
110
  ] });
235
- this.process.stderr?.on("data", (chunk) => {
111
+ this.process = proc;
112
+ proc.stderr?.on("data", (chunk) => {
236
113
  const text = chunk.toString().trim();
237
114
  if (text) this.log.warn(text);
238
115
  });
239
- this.process.on("error", (err) => {
240
- this.log.error("Python raw-tensor process error", { meta: { error: err.message } });
241
- this.pendingReject?.(err);
242
- this.pendingReject = null;
243
- this.pendingResolve = null;
116
+ proc.on("error", (err) => {
117
+ this.log.error(`${this.name}: process error`, { meta: { error: err.message } });
118
+ this.markDead(`process error: ${err.message}`);
244
119
  });
245
- this.process.on("exit", (code) => {
246
- if (code !== 0 && code !== null) {
247
- const err = /* @__PURE__ */ new Error(`PythonRawTensorEngine: process exited with code ${code}`);
248
- this.pendingReject?.(err);
249
- this.pendingReject = null;
250
- this.pendingResolve = null;
251
- }
120
+ proc.stdin?.on("error", (err) => {
121
+ this.markDead(`stdin error: ${err.message}`);
122
+ });
123
+ proc.on("exit", (code, signal) => {
124
+ if (this.process === proc) this.process = null;
125
+ const reason = `process exited (code ${String(code)}, signal ${String(signal)})`;
126
+ if (!this.disposing) this.log.warn(`${this.name}: process exited unexpectedly — the engine is dead`, { meta: {
127
+ code,
128
+ signal,
129
+ pid: proc.pid ?? null
130
+ } });
131
+ this.markDead(reason);
252
132
  });
253
- this.process.stdout.on("data", (chunk) => {
133
+ proc.stdout?.on("data", (chunk) => {
254
134
  this.receiveBuffer = Buffer.concat([this.receiveBuffer, chunk]);
255
135
  this.tryReceive();
256
136
  });
257
137
  const ready = await this.receiveFrame();
258
- if (ready.length !== 1 || ready[0] !== 1) throw new Error("PythonRawTensorEngine: unexpected ready frame");
259
- this.log.info("ONNX raw-tensor engine ready (embedded Python)", { meta: { modelPath: this.modelPath } });
138
+ if (ready.length !== 1 || ready[0] !== 1) throw new Error(`${this.name}: unexpected ready frame`);
139
+ }
140
+ /** False once the process died or was disposed — a holder must rebuild. */
141
+ isAlive() {
142
+ return this.process !== null && this.dead === null;
260
143
  }
261
- /** Native pid of the Python subprocess — sampled by the pool memory
262
- * watchdog (`/proc/<pid>/status`). */
263
144
  getPid() {
264
145
  return this.process?.pid ?? null;
265
146
  }
266
- /** Inference requests served since spawn — the denominator that separates
267
- * "RSS grew per unit of work" (leak) from "RSS grew because it worked". */
268
- getRequestCount() {
269
- return this.requestCount;
270
- }
271
- requestCount = 0;
272
- async run(input, inputShape) {
273
- if (!this.process?.stdin) throw new Error("PythonRawTensorEngine: not initialized — call initialize() first");
274
- this.requestCount++;
275
- const ndims = inputShape.length;
276
- const meta = Buffer.allocUnsafe(1 + ndims * 4);
277
- meta.writeUInt8(ndims, 0);
278
- for (let i = 0; i < ndims; i++) meta.writeUInt32LE(inputShape[i], 1 + i * 4);
279
- const dataBuf = Buffer.from(input.buffer, input.byteOffset, input.byteLength);
280
- const payload = Buffer.concat([meta, dataBuf]);
281
- const lenBuf = Buffer.allocUnsafe(4);
282
- lenBuf.writeUInt32LE(payload.length, 0);
283
- this.process.stdin.write(Buffer.concat([lenBuf, payload]));
284
- const resp = await this.receiveFrame();
285
- const floatStart = 1 + resp.readUInt8(0) * 4;
286
- const count = (resp.length - floatStart) / 4;
287
- const out = new Float32Array(count);
288
- for (let i = 0; i < count; i++) out[i] = resp.readFloatLE(floatStart + i * 4);
289
- return out;
147
+ /** Send one request frame and await its answer. Rejects AT ONCE when dead. */
148
+ async request(payload) {
149
+ if (this.dead !== null) throw new Error(`${this.name}: engine is dead — ${this.dead}`);
150
+ const stdin = this.process?.stdin;
151
+ if (!stdin) throw new Error(`${this.name}: not initialized — call initialize() first`);
152
+ const answer = this.receiveFrame();
153
+ const len = Buffer.allocUnsafe(4);
154
+ len.writeUInt32LE(payload.length, 0);
155
+ stdin.write(Buffer.concat([len, payload]));
156
+ return answer;
290
157
  }
291
158
  async dispose() {
292
159
  const proc = this.process;
160
+ this.disposing = true;
161
+ this.markDead("disposed");
293
162
  if (!proc) return;
294
163
  this.process = null;
295
164
  proc.stdin?.end();
@@ -307,185 +176,546 @@ var PythonRawTensorEngine = class {
307
176
  });
308
177
  });
309
178
  }
179
+ markDead(reason) {
180
+ if (this.dead === null) this.dead = reason;
181
+ const err = /* @__PURE__ */ new Error(`${this.name}: ${reason}`);
182
+ for (const waiter of this.pending.splice(0)) waiter.reject(err);
183
+ }
310
184
  receiveFrame() {
311
185
  return new Promise((resolve, reject) => {
312
- this.pendingResolve = resolve;
313
- this.pendingReject = reject;
186
+ this.pending.push({
187
+ resolve,
188
+ reject
189
+ });
314
190
  });
315
191
  }
316
192
  tryReceive() {
317
- if (this.receiveBuffer.length < 4) return;
318
- const length = this.receiveBuffer.readUInt32LE(0);
319
- if (this.receiveBuffer.length < 4 + length) return;
320
- const payload = Buffer.from(this.receiveBuffer.subarray(4, 4 + length));
321
- this.receiveBuffer = this.receiveBuffer.subarray(4 + length);
322
- const resolve = this.pendingResolve;
323
- this.pendingResolve = null;
324
- this.pendingReject = null;
325
- resolve?.(payload);
193
+ while (this.receiveBuffer.length >= 4) {
194
+ const length = this.receiveBuffer.readUInt32LE(0);
195
+ if (this.receiveBuffer.length < 4 + length) return;
196
+ const payload = Buffer.from(this.receiveBuffer.subarray(4, 4 + length));
197
+ this.receiveBuffer = this.receiveBuffer.subarray(4 + length);
198
+ this.pending.shift()?.resolve(payload);
199
+ }
326
200
  }
327
201
  };
328
202
  //#endregion
203
+ //#region src/embedding-encoder/shared/noop-logger.ts
204
+ var noop = () => {};
205
+ function createNoopLogger() {
206
+ const logger = {
207
+ debug: noop,
208
+ info: noop,
209
+ warn: noop,
210
+ error: noop,
211
+ child: () => logger,
212
+ withTags: (_tags) => logger
213
+ };
214
+ return logger;
215
+ }
216
+ //#endregion
329
217
  //#region src/embedding-encoder/shared/python-text-encoder-engine.ts
330
- /**
331
- * CLIP text-encoder engine backed by an embedded-Python subprocess
332
- * (`text_encoder_inference.py`). Tokenization happens IN Python via the HF
333
- * `tokenizers` Rust BPE (exact by construction) — this replaces the former
334
- * hand-rolled TypeScript CLIP BPE.
335
- *
336
- * The caller sends raw UTF-8 text; Python tokenizes (truncate/pad to 77), runs
337
- * onnxruntime, and returns the embedding tensor. Wire protocol = length-prefixed
338
- * binary frames ([4B LE length][payload]):
339
- * ready (in): [0x01]
340
- * request (out): UTF-8 text bytes
341
- * response (in): [1B ndims][dims × 4B LE uint32][float32 LE data]
342
- */
343
218
  var PythonTextEncoderEngine = class {
344
- pythonPath;
345
219
  scriptPath;
346
220
  modelPath;
347
221
  tokenizerPath;
348
- process = null;
349
- receiveBuffer = Buffer.alloc(0);
350
- pendingResolve = null;
351
- pendingReject = null;
222
+ window;
223
+ proc;
352
224
  log;
353
- constructor(pythonPath, scriptPath, modelPath, tokenizerPath, logger) {
354
- this.pythonPath = pythonPath;
225
+ requestCount = 0;
226
+ constructor(pythonPath, scriptPath, modelPath, tokenizerPath, window, logger) {
355
227
  this.scriptPath = scriptPath;
356
228
  this.modelPath = modelPath;
357
229
  this.tokenizerPath = tokenizerPath;
230
+ this.window = window;
358
231
  this.log = logger ?? createNoopLogger();
232
+ this.proc = new FramedPythonProcess("PythonTextEncoderEngine", pythonPath, this.spawnArgs(), this.log);
359
233
  }
360
234
  async initialize() {
361
- this.process = spawn(this.pythonPath, [
362
- this.scriptPath,
363
- this.modelPath,
364
- this.tokenizerPath
365
- ], { stdio: [
366
- "pipe",
367
- "pipe",
368
- "pipe"
369
- ] });
370
- this.process.stderr?.on("data", (chunk) => {
371
- const text = chunk.toString().trim();
372
- if (text) this.log.warn(text);
373
- });
374
- this.process.on("error", (err) => {
375
- this.log.error("Python text-encoder process error", { meta: { error: err.message } });
376
- this.pendingReject?.(err);
377
- this.pendingReject = null;
378
- this.pendingResolve = null;
379
- });
380
- this.process.on("exit", (code) => {
381
- if (code !== 0 && code !== null) {
382
- const err = /* @__PURE__ */ new Error(`PythonTextEncoderEngine: process exited with code ${code}`);
383
- this.pendingReject?.(err);
384
- this.pendingReject = null;
385
- this.pendingResolve = null;
386
- }
387
- });
388
- this.process.stdout.on("data", (chunk) => {
389
- this.receiveBuffer = Buffer.concat([this.receiveBuffer, chunk]);
390
- this.tryReceive();
391
- });
392
- const ready = await this.receiveFrame();
393
- if (ready.length !== 1 || ready[0] !== 1) throw new Error("PythonTextEncoderEngine: unexpected ready frame");
235
+ await this.proc.start();
394
236
  this.log.info("CLIP text-encoder engine ready (embedded Python)", { meta: {
395
237
  modelPath: this.modelPath,
396
- tokenizerPath: this.tokenizerPath
238
+ tokenizerPath: this.tokenizerPath,
239
+ contextLength: this.window.contextLength,
240
+ padId: this.window.padId
397
241
  } });
398
242
  }
243
+ /** The subprocess argv after the interpreter — the window rides on it. */
244
+ spawnArgs() {
245
+ return [
246
+ this.scriptPath,
247
+ this.modelPath,
248
+ this.tokenizerPath,
249
+ "--context-length",
250
+ String(this.window.contextLength),
251
+ "--pad-id",
252
+ String(this.window.padId)
253
+ ];
254
+ }
255
+ /** False once the process died (crash, OOM kill) or was disposed. */
256
+ isAlive() {
257
+ return this.proc.isAlive();
258
+ }
399
259
  /** Native pid of the Python subprocess — sampled by the pool memory
400
260
  * watchdog (`/proc/<pid>/status`). */
401
261
  getPid() {
402
- return this.process?.pid ?? null;
262
+ return this.proc.getPid();
403
263
  }
404
264
  /** Encode requests served since spawn — the denominator that separates
405
265
  * "RSS grew per unit of work" (leak) from "RSS grew because it worked". */
406
266
  getRequestCount() {
407
267
  return this.requestCount;
408
268
  }
409
- requestCount = 0;
410
269
  /** Tokenize + encode `text` into the model embedding (float32). */
411
270
  async encode(text) {
412
- if (!this.process?.stdin) throw new Error("PythonTextEncoderEngine: not initialized — call initialize() first");
413
271
  this.requestCount++;
414
- const payload = Buffer.from(text, "utf-8");
415
- const lenBuf = Buffer.allocUnsafe(4);
416
- lenBuf.writeUInt32LE(payload.length, 0);
417
- this.process.stdin.write(Buffer.concat([lenBuf, payload]));
418
- const resp = await this.receiveFrame();
419
- const floatStart = 1 + resp.readUInt8(0) * 4;
420
- const count = (resp.length - floatStart) / 4;
421
- const out = new Float32Array(count);
422
- for (let i = 0; i < count; i++) out[i] = resp.readFloatLE(floatStart + i * 4);
423
- return out;
272
+ return decodeTensorFrame(await this.proc.request(Buffer.from(text, "utf-8")));
424
273
  }
425
274
  async dispose() {
426
- const proc = this.process;
427
- if (!proc) return;
428
- this.process = null;
429
- proc.stdin?.end();
430
- proc.kill("SIGTERM");
431
- await new Promise((resolve) => {
432
- const timer = setTimeout(() => {
433
- try {
434
- proc.kill("SIGKILL");
435
- } catch {}
436
- resolve();
437
- }, 5e3);
438
- proc.once("exit", () => {
439
- clearTimeout(timer);
440
- resolve();
441
- });
442
- });
275
+ await this.proc.dispose();
443
276
  }
444
- receiveFrame() {
445
- return new Promise((resolve, reject) => {
446
- this.pendingResolve = resolve;
447
- this.pendingReject = reject;
448
- });
277
+ };
278
+ /** Decode `[1B ndims][dims × 4B][float32 data]`. */
279
+ function decodeTensorFrame(resp) {
280
+ const floatStart = 1 + resp.readUInt8(0) * 4;
281
+ const count = (resp.length - floatStart) / 4;
282
+ const out = new Float32Array(count);
283
+ for (let i = 0; i < count; i++) out[i] = resp.readFloatLE(floatStart + i * 4);
284
+ return out;
285
+ }
286
+ //#endregion
287
+ //#region src/embedding-encoder/shared/python-raw-tensor-engine.ts
288
+ /**
289
+ * Raw-tensor ONNX engine backed by an embedded-Python subprocess
290
+ * (`raw_tensor_inference.py`). Replaces the Node `onnxruntime-node` raw-tensor
291
+ * engine so the platform ships no Node ONNX runtime. The caller preprocesses to
292
+ * a Float32Array; this engine ships it to Python, which runs onnxruntime and
293
+ * returns the output tensor. Wire protocol = length-prefixed binary frames
294
+ * ([4B LE length][payload]); the process lifecycle is `FramedPythonProcess`.
295
+ */
296
+ var PythonRawTensorEngine = class {
297
+ modelPath;
298
+ runtime = "onnx";
299
+ device = "cpu";
300
+ proc;
301
+ log;
302
+ requestCount = 0;
303
+ constructor(pythonPath, scriptPath, modelPath, logger) {
304
+ this.modelPath = modelPath;
305
+ this.log = logger ?? createNoopLogger();
306
+ this.proc = new FramedPythonProcess("PythonRawTensorEngine", pythonPath, [scriptPath, modelPath], this.log);
449
307
  }
450
- tryReceive() {
451
- if (this.receiveBuffer.length < 4) return;
452
- const length = this.receiveBuffer.readUInt32LE(0);
453
- if (this.receiveBuffer.length < 4 + length) return;
454
- const payload = Buffer.from(this.receiveBuffer.subarray(4, 4 + length));
455
- this.receiveBuffer = this.receiveBuffer.subarray(4 + length);
456
- const resolve = this.pendingResolve;
457
- this.pendingResolve = null;
458
- this.pendingReject = null;
459
- resolve?.(payload);
308
+ async initialize() {
309
+ await this.proc.start();
310
+ this.log.info("ONNX raw-tensor engine ready (embedded Python)", { meta: { modelPath: this.modelPath } });
311
+ }
312
+ /** False once the process died (crash, OOM kill) or was disposed. */
313
+ isAlive() {
314
+ return this.proc.isAlive();
315
+ }
316
+ /** Native pid of the Python subprocess — sampled by the pool memory
317
+ * watchdog (`/proc/<pid>/status`). */
318
+ getPid() {
319
+ return this.proc.getPid();
320
+ }
321
+ /** Inference requests served since spawn — the denominator that separates
322
+ * "RSS grew per unit of work" (leak) from "RSS grew because it worked". */
323
+ getRequestCount() {
324
+ return this.requestCount;
325
+ }
326
+ async run(input, inputShape) {
327
+ this.requestCount++;
328
+ const ndims = inputShape.length;
329
+ const meta = Buffer.allocUnsafe(1 + ndims * 4);
330
+ meta.writeUInt8(ndims, 0);
331
+ for (let i = 0; i < ndims; i++) meta.writeUInt32LE(inputShape[i], 1 + i * 4);
332
+ const dataBuf = Buffer.from(input.buffer, input.byteOffset, input.byteLength);
333
+ return decodeTensorFrame(await this.proc.request(Buffer.concat([meta, dataBuf])));
334
+ }
335
+ async dispose() {
336
+ await this.proc.dispose();
337
+ }
338
+ };
339
+ var ActiveClipModel = class {
340
+ deps;
341
+ cached = null;
342
+ lastRead = null;
343
+ constructor(deps) {
344
+ this.deps = deps;
345
+ }
346
+ /**
347
+ * The last row a read RETURNED, without reading — `null` while no read has
348
+ * ever succeeded (the config fallback is a guess, not a known model). The
349
+ * idle sweep exempts this model's text tower (`clip-text-engines.ts`).
350
+ */
351
+ known() {
352
+ return this.lastRead;
353
+ }
354
+ async get() {
355
+ const now = this.deps.now();
356
+ if (this.cached !== null && now - this.cached.at < 3e4) return this.cached.modelId;
357
+ const row = await this.deps.readRow();
358
+ if (row === null) {
359
+ const kept = this.lastRead ?? this.deps.fallbackModelId();
360
+ this.deps.logger.warn("CLIP encoder kept its model — the cluster row could not be read", { meta: {
361
+ modelId: kept,
362
+ source: this.lastRead !== null ? "last-read" : "addon-config"
363
+ } });
364
+ return kept;
365
+ }
366
+ const previous = this.lastRead;
367
+ if (row !== previous) this.deps.logger.info("CLIP encoder follows the cluster model", { meta: {
368
+ modelId: row,
369
+ previous
370
+ } });
371
+ this.lastRead = row;
372
+ this.cached = {
373
+ modelId: row,
374
+ at: now
375
+ };
376
+ if (previous !== null && row !== previous) this.deps.onChange?.(row, previous);
377
+ return row;
378
+ }
379
+ };
380
+ var CrashBudget = class {
381
+ deps;
382
+ recent = /* @__PURE__ */ new Map();
383
+ failedModels = /* @__PURE__ */ new Map();
384
+ constructor(deps) {
385
+ this.deps = deps;
386
+ }
387
+ /** Throw the named refusal when the model is in `failed`. Spawns nothing. */
388
+ check(modelId) {
389
+ const failed = this.failedModels.get(modelId);
390
+ if (failed === void 0) return;
391
+ throw embeddingModelFailedError(failed);
392
+ }
393
+ /** Record one crash or load failure. Returns true when it made the model `failed`. */
394
+ record(modelId, reason) {
395
+ if (this.failedModels.has(modelId)) return false;
396
+ const now = this.now();
397
+ const windowMs = this.deps.windowMs ?? 6e5;
398
+ const times = (this.recent.get(modelId) ?? []).filter((t) => now - t < windowMs);
399
+ times.push(now);
400
+ this.recent.set(modelId, times);
401
+ if (times.length < (this.deps.maxCrashes ?? 3)) return false;
402
+ const failed = {
403
+ modelId,
404
+ tower: this.deps.tower,
405
+ nodeId: this.deps.nodeId(),
406
+ crashes: times.length,
407
+ windowMs,
408
+ sinceMs: now,
409
+ lastReason: reason
410
+ };
411
+ this.failedModels.set(modelId, failed);
412
+ this.recent.delete(modelId);
413
+ this.deps.logger.error("CLIP tower FAILED — crash budget exhausted, requests refused until reset", { meta: { ...failed } });
414
+ return true;
415
+ }
416
+ failed() {
417
+ return [...this.failedModels.values()];
418
+ }
419
+ /** Clear one model (or all). Returns the ids that were failed. */
420
+ reset(modelId, why = "operator") {
421
+ const ids = modelId !== void 0 ? [modelId] : [...this.failedModels.keys()];
422
+ const cleared = ids.filter((id) => this.failedModels.delete(id));
423
+ for (const id of ids) this.recent.delete(id);
424
+ if (cleared.length > 0) this.deps.logger.info("CLIP tower failed state cleared", { meta: {
425
+ tower: this.deps.tower,
426
+ nodeId: this.deps.nodeId(),
427
+ cleared,
428
+ why
429
+ } });
430
+ return cleared;
431
+ }
432
+ now() {
433
+ return (this.deps.now ?? Date.now)();
460
434
  }
461
435
  };
462
436
  //#endregion
463
- //#region src/embedding-encoder/addon/clip-models.ts
437
+ //#region src/embedding-encoder/addon/failed-models-report.ts
464
438
  /**
465
- * `mobileclip-s0` was retired on 2026-08-21 (measured worst of the family on
466
- * real fleet crops). A stored config still naming it resolves to the default
467
- * via {@link getModelMeta}'s fallback — never to a dangling id.
439
+ * Every failed tower on this node. Each row already names the node: the budget
440
+ * stamps it when the model goes `failed` (it is in the refusal too), and this
441
+ * report does not re-derive it — one authority for the fact.
468
442
  */
469
- var CLIP_MODEL_META = {
470
- "mobileclip-s1": {
471
- imageModelId: "mobileclip-s1",
472
- textModelId: "mobileclip-s1-text",
473
- embeddingDim: 512,
474
- inputSize: 256,
475
- tokenizerType: "clip"
476
- },
477
- "mobileclip-s2": {
478
- imageModelId: "mobileclip-s2",
479
- textModelId: "mobileclip-s2-text",
480
- embeddingDim: 512,
481
- inputSize: 256,
482
- tokenizerType: "clip"
443
+ function reportFailedModels(budgets) {
444
+ return [...budgets.image.failed(), ...budgets.text.failed()];
445
+ }
446
+ /** Every model whose requirements are terminally unmeetable on this node; rows stamped by the gate. */
447
+ function reportFailedRequirements(requirements) {
448
+ return [...requirements.image.failed(), ...requirements.text.failed()];
449
+ }
450
+ /** The operator's clear (one model, or all) of BOTH terminal states on this node — the result names it. */
451
+ function resetFailedModelsOn(nodeId, towers, modelId) {
452
+ const cleared = [
453
+ ...towers.budgets.image.reset(modelId),
454
+ ...towers.budgets.text.reset(modelId),
455
+ ...towers.requirements.image.reset(modelId),
456
+ ...towers.requirements.text.reset(modelId)
457
+ ];
458
+ return {
459
+ cleared: [...new Set(cleared)],
460
+ nodeId
461
+ };
462
+ }
463
+ /**
464
+ * The cluster row now names `modelId`: a fresh start for THAT model's towers
465
+ * only (D6's second exit from `failed`). The model the row left keeps its
466
+ * state — a rollback to a model that crashed three times must not buy three
467
+ * more spawns nobody asked for, and a search that still names it
468
+ * (`searchObjectEvents({ modelId })`) is refused with the reset hint, as
469
+ * before the flip.
470
+ */
471
+ function clearFailedOnModelChange(towers, modelId) {
472
+ return [...new Set([
473
+ ...towers.budgets.image.reset(modelId, "model-change"),
474
+ ...towers.budgets.text.reset(modelId, "model-change"),
475
+ ...towers.requirements.image.reset(modelId, "model-change"),
476
+ ...towers.requirements.text.reset(modelId, "model-change")
477
+ ])];
478
+ }
479
+ //#endregion
480
+ //#region src/embedding-encoder/addon/model-engine-slot.ts
481
+ var ModelEngineSlot = class {
482
+ build;
483
+ budget;
484
+ requirements;
485
+ onDead;
486
+ current = null;
487
+ inflight = /* @__PURE__ */ new Map();
488
+ /** The last model asked for — a build for any other model is stale on arrival. */
489
+ wanted = null;
490
+ /** Serialises switches so two models never tear the slot down concurrently. */
491
+ switching = Promise.resolve();
492
+ /**
493
+ * Bumped by {@link dispose}: a build that started before it is stale when it
494
+ * lands and is disposed on arrival — shutdown leaves no process behind.
495
+ */
496
+ generation = 0;
497
+ constructor(build, budget, requirements, onDead = () => {}) {
498
+ this.build = build;
499
+ this.budget = budget;
500
+ this.requirements = requirements;
501
+ this.onDead = onDead;
502
+ }
503
+ held() {
504
+ return this.current;
505
+ }
506
+ async ensure(modelId) {
507
+ this.wanted = modelId;
508
+ this.dropIfDead();
509
+ if (this.current?.modelId === modelId) return this.current.engine;
510
+ this.budget.check(modelId);
511
+ this.requirements.check(modelId);
512
+ const pending = this.inflight.get(modelId);
513
+ if (pending !== void 0) return pending;
514
+ const generation = this.generation;
515
+ const guarded = this.switching.then(async () => {
516
+ this.dropIfDead();
517
+ if (this.current?.modelId === modelId) return this.current.engine;
518
+ this.budget.check(modelId);
519
+ this.requirements.check(modelId);
520
+ await this.disposeCurrent();
521
+ let engine;
522
+ try {
523
+ engine = await this.build(modelId);
524
+ } catch (err) {
525
+ throw this.requirements.refuse(modelId, err);
526
+ }
527
+ this.requirements.satisfied(modelId);
528
+ try {
529
+ await engine.initialize();
530
+ } catch (err) {
531
+ this.budget.record(modelId, `load failed: ${err instanceof Error ? err.message : String(err)}`);
532
+ throw err;
533
+ }
534
+ if (this.generation !== generation) {
535
+ await engine.dispose();
536
+ throw new Error(`engine for "${modelId}" cancelled: the slot was disposed while building`);
537
+ }
538
+ if (this.wanted !== modelId) {
539
+ await engine.dispose();
540
+ throw new Error(`engine for "${modelId}" superseded while building`);
541
+ }
542
+ this.current = {
543
+ modelId,
544
+ engine
545
+ };
546
+ return engine;
547
+ }).finally(() => {
548
+ this.inflight.delete(modelId);
549
+ });
550
+ this.inflight.set(modelId, guarded);
551
+ this.switching = guarded.then(() => void 0, () => void 0);
552
+ return guarded;
553
+ }
554
+ /**
555
+ * Dispose the held engine AND cancel every build in flight: each is disposed
556
+ * when it lands (see `generation`). Resolves once in-flight builds settled.
557
+ */
558
+ async dispose() {
559
+ this.generation += 1;
560
+ await this.disposeCurrent();
561
+ await this.switching;
562
+ }
563
+ async disposeCurrent() {
564
+ const held = this.current;
565
+ this.current = null;
566
+ await held?.engine.dispose();
567
+ }
568
+ dropIfDead() {
569
+ const held = this.current;
570
+ if (held === null || held.engine.isAlive()) return;
571
+ this.current = null;
572
+ this.onDead(held.modelId);
573
+ this.budget.record(held.modelId, "process died after loading");
574
+ held.engine.dispose().catch(() => {});
575
+ }
576
+ };
577
+ /**
578
+ * Thrown by the requirements stage when a retry can never help: the catalog
579
+ * does not know the id, the server said the file does not exist (404/410), a
580
+ * declared sibling is missing. Same process as the gate, so the class is the
581
+ * marker; the gate never matches wording.
582
+ */
583
+ var PermanentRequirementsFailure = class extends Error {
584
+ constructor(message) {
585
+ super(message);
586
+ this.name = "PermanentRequirementsFailure";
483
587
  }
484
588
  };
485
- var DEFAULT_CLIP_MODEL = "mobileclip-s1";
486
- function getModelMeta(modelId) {
487
- return CLIP_MODEL_META[modelId] ?? CLIP_MODEL_META["mobileclip-s1"];
589
+ function isPermanentRequirementsFailure(err) {
590
+ return err instanceof PermanentRequirementsFailure;
488
591
  }
592
+ var RequirementsGate = class {
593
+ deps;
594
+ unavailable = /* @__PURE__ */ new Map();
595
+ constructor(deps) {
596
+ this.deps = deps;
597
+ }
598
+ /**
599
+ * Consulted BEFORE a build. Throws the named refusal — and downloads nothing
600
+ * — while the model is terminal or inside its cooldown. Returns otherwise.
601
+ */
602
+ check(modelId) {
603
+ const state = this.unavailable.get(modelId);
604
+ if (state === void 0) return;
605
+ if (state.terminal !== null) {
606
+ this.unavailable.set(modelId, {
607
+ ...state,
608
+ refused: state.refused + 1
609
+ });
610
+ throw embeddingRequirementsFailedError(state.terminal);
611
+ }
612
+ const now = this.now();
613
+ if (now >= state.cooldownUntilMs) return;
614
+ this.unavailable.set(modelId, {
615
+ ...state,
616
+ refused: state.refused + 1
617
+ });
618
+ throw embeddingRequirementsUnavailableError(this.on(modelId), `retry in ${String(Math.ceil((state.cooldownUntilMs - now) / 1e3))}s — ${state.reason}`);
619
+ }
620
+ /**
621
+ * A build failed before any process existed. Starts (or doubles) the
622
+ * cooldown; a PERMANENT failure past the bound makes the model terminal.
623
+ * Logs at the transitions only and returns the named refusal to throw.
624
+ */
625
+ refuse(modelId, err) {
626
+ const reason = err instanceof Error ? err.message : String(err);
627
+ const now = this.now();
628
+ const prior = this.unavailable.get(modelId);
629
+ const attempts = (prior?.attempts ?? 0) + 1;
630
+ const permanent = isPermanentRequirementsFailure(err);
631
+ const permanentAttempts = (prior?.permanentAttempts ?? 0) + (permanent ? 1 : 0);
632
+ const cooldownMs = Math.min(this.deps.cooldownMaxMs ?? 3e5, (this.deps.cooldownBaseMs ?? 5e3) * 2 ** (attempts - 1));
633
+ const base = {
634
+ sinceMs: prior?.sinceMs ?? now,
635
+ reason,
636
+ refused: (prior?.refused ?? 0) + 1,
637
+ attempts,
638
+ permanentAttempts,
639
+ cooldownUntilMs: now + cooldownMs,
640
+ terminal: null
641
+ };
642
+ if (prior === void 0) this.deps.logger.warn("CLIP model requirements unavailable — requests refused until they are; the crash budget is untouched", { meta: {
643
+ modelId,
644
+ tower: this.deps.tower,
645
+ nodeId: this.deps.nodeId(),
646
+ reason,
647
+ cooldownMs
648
+ } });
649
+ if (permanent && permanentAttempts >= (this.deps.maxAttempts ?? 3)) {
650
+ const terminal = {
651
+ ...this.on(modelId),
652
+ attempts: permanentAttempts,
653
+ sinceMs: now,
654
+ lastReason: reason
655
+ };
656
+ this.unavailable.set(modelId, {
657
+ ...base,
658
+ terminal
659
+ });
660
+ this.deps.logger.error("CLIP model requirements can never be met — terminally failed, requests refused until reset", { meta: {
661
+ ...terminal,
662
+ refusedSoFar: base.refused
663
+ } });
664
+ return embeddingRequirementsFailedError(terminal);
665
+ }
666
+ this.unavailable.set(modelId, base);
667
+ return embeddingRequirementsUnavailableError(this.on(modelId), reason);
668
+ }
669
+ /** A build got its engine: the requirements are met again. Logs the recovery once. */
670
+ satisfied(modelId) {
671
+ const prior = this.unavailable.get(modelId);
672
+ if (prior === void 0) return;
673
+ this.unavailable.delete(modelId);
674
+ this.deps.logger.info("CLIP model requirements available again", { meta: {
675
+ modelId,
676
+ tower: this.deps.tower,
677
+ nodeId: this.deps.nodeId(),
678
+ unavailableMs: this.now() - prior.sinceMs,
679
+ refused: prior.refused,
680
+ attempts: prior.attempts,
681
+ lastReason: prior.reason
682
+ } });
683
+ }
684
+ /** Models terminally `requirements-failed` on this node, as `getInfo` reports them. */
685
+ failed() {
686
+ return [...this.unavailable.values()].map((s) => s.terminal).filter((t) => t !== null);
687
+ }
688
+ /**
689
+ * Clear one model (or all): terminal state AND cooldown. Returns the ids
690
+ * that were TERMINAL — a cooldown is not a state an operator is told about.
691
+ */
692
+ reset(modelId, why = "operator") {
693
+ const ids = modelId !== void 0 ? [modelId] : [...this.unavailable.keys()];
694
+ const cleared = ids.filter((id) => this.unavailable.get(id)?.terminal !== null && this.unavailable.has(id));
695
+ for (const id of ids) this.unavailable.delete(id);
696
+ if (cleared.length > 0) this.deps.logger.info("CLIP model requirements-failed state cleared", { meta: {
697
+ tower: this.deps.tower,
698
+ nodeId: this.deps.nodeId(),
699
+ cleared,
700
+ why
701
+ } });
702
+ return cleared;
703
+ }
704
+ /** Models currently refused (cooldown or terminal) — exposed for tests. */
705
+ pending() {
706
+ return [...this.unavailable.keys()];
707
+ }
708
+ on(modelId) {
709
+ return {
710
+ modelId,
711
+ tower: this.deps.tower,
712
+ nodeId: this.deps.nodeId()
713
+ };
714
+ }
715
+ now() {
716
+ return (this.deps.now ?? Date.now)();
717
+ }
718
+ };
489
719
  //#endregion
490
720
  //#region src/embedding-encoder/addon/clip-preprocessing.ts
491
721
  var CLIP_MEAN = [
@@ -528,11 +758,286 @@ function l2Normalize(vec) {
528
758
  if (norm > 0) for (let i = 0; i < vec.length; i++) vec[i] /= norm;
529
759
  return vec;
530
760
  }
761
+ /**
762
+ * A tower unused this long is disposed ({@link ClipTextEngines.evictIdle}) —
763
+ * EXCEPT the tower of the model the cluster row names, which stays loaded for
764
+ * the life of the process exactly as it did before D649: the steady-state
765
+ * search must not pay a Python spawn plus a model load after a quiet quarter
766
+ * hour. Only the OTHER towers (SigLIP2 after a comparison, the old model after
767
+ * a flip) are let go when idle. When the active model is unknown (the row was
768
+ * never read), nothing is evicted — the side that keeps memory as before.
769
+ * The encoder is `placement: any-node` and cannot tell locally whether it is
770
+ * the node that answers text queries, so towers are loaded ON DEMAND — never
771
+ * prefetched on a model flip (565 MB on every node). The first query after a
772
+ * flip pays the load; if it fails, the search fails loudly rather than reading
773
+ * as "no results".
774
+ */
775
+ var TEXT_ENGINE_IDLE_MS = 15 * 6e4;
776
+ var ClipTextEngines = class {
777
+ deps;
778
+ /** Insertion order is recency: a hit is re-inserted at the end. */
779
+ engines = /* @__PURE__ */ new Map();
780
+ /** Single-flight per model: two concurrent first queries spawn one engine. */
781
+ inflight = /* @__PURE__ */ new Map();
782
+ /**
783
+ * Encodes in flight per model. An engine with any is NEVER evicted: disposing
784
+ * it would fail a search that is already running. When every held engine is
785
+ * busy the set goes over `maxEngines` temporarily and shrinks when one drains.
786
+ */
787
+ busy = /* @__PURE__ */ new Map();
788
+ /** When each held tower last finished an encode (or was built). */
789
+ lastUsed = /* @__PURE__ */ new Map();
790
+ /** Bumped by {@link disposeAll}: a build landing after it is disposed on arrival. */
791
+ generation = 0;
792
+ constructor(deps) {
793
+ this.deps = deps;
794
+ }
795
+ /** Resolve a model id to its CLIP model, or throw naming the id. */
796
+ requireModel(modelId) {
797
+ const model = this.deps.registry.resolve(modelId);
798
+ if (model === null) throw new Error(`EmbeddingEncoder: "${modelId}" has no CLIP metadata — not a CLIP model`);
799
+ return model;
800
+ }
801
+ async encode(modelId, text) {
802
+ const model = this.requireModel(modelId);
803
+ this.busy.set(modelId, (this.busy.get(modelId) ?? 0) + 1);
804
+ let vector;
805
+ try {
806
+ vector = await (await this.engineFor(model)).encode(text);
807
+ } finally {
808
+ const left = (this.busy.get(modelId) ?? 1) - 1;
809
+ if (left > 0) this.busy.set(modelId, left);
810
+ else this.busy.delete(modelId);
811
+ if (this.engines.has(modelId)) this.lastUsed.set(modelId, this.now());
812
+ this.scheduleEviction();
813
+ }
814
+ if (vector.length !== model.meta.embeddingDim) throw new Error(`EmbeddingEncoder: ${model.meta.textModelId} produced ${String(vector.length)} dims, its metadata declares ${String(model.meta.embeddingDim)} — refusing the query rather than truncating it`);
815
+ return {
816
+ vector,
817
+ model
818
+ };
819
+ }
820
+ /**
821
+ * Dispose every IDLE tower unused for {@link TEXT_ENGINE_IDLE_MS} (a timer in
822
+ * the addon calls this), except the active model's. Returns the evicted
823
+ * model ids; each is logged. An unknown active model evicts nothing.
824
+ */
825
+ async evictIdle(idleMs = TEXT_ENGINE_IDLE_MS) {
826
+ const active = this.deps.activeModelId();
827
+ if (active === null) return [];
828
+ const now = this.now();
829
+ const idle = [...this.engines.keys()].filter((id) => id !== active && (this.busy.get(id) ?? 0) === 0 && now - (this.lastUsed.get(id) ?? now) >= idleMs);
830
+ const evicted = [];
831
+ for (const modelId of idle) {
832
+ if ((this.busy.get(modelId) ?? 0) > 0 || !this.engines.has(modelId)) continue;
833
+ evicted.push(modelId);
834
+ this.deps.logger.info("CLIP text tower evicted — idle", { meta: {
835
+ modelId,
836
+ idleMs: now - (this.lastUsed.get(modelId) ?? now)
837
+ } });
838
+ await this.restart(modelId);
839
+ }
840
+ return evicted;
841
+ }
842
+ /** Live engines, for the memory watchdog. */
843
+ live() {
844
+ return this.engines;
845
+ }
846
+ /** Dispose one engine; the next query rebuilds it. */
847
+ async restart(modelId) {
848
+ const engine = this.engines.get(modelId);
849
+ if (engine === void 0) return;
850
+ this.engines.delete(modelId);
851
+ this.lastUsed.delete(modelId);
852
+ await engine.dispose();
853
+ }
854
+ /** Dispose every tower AND cancel the builds in flight (disposed as they land). */
855
+ async disposeAll() {
856
+ this.generation += 1;
857
+ const all = [...this.engines.values()];
858
+ this.engines.clear();
859
+ this.lastUsed.clear();
860
+ await Promise.all(all.map((e) => e.dispose()));
861
+ await Promise.allSettled(this.inflight.values());
862
+ }
863
+ async engineFor(model) {
864
+ const held = this.engines.get(model.modelId);
865
+ if (held !== void 0 && held.isAlive()) {
866
+ this.engines.delete(model.modelId);
867
+ this.engines.set(model.modelId, held);
868
+ return held;
869
+ }
870
+ if (held !== void 0) {
871
+ this.deps.logger.warn("CLIP text tower died — dropping it and rebuilding", { meta: {
872
+ modelId: model.modelId,
873
+ textModelId: model.meta.textModelId
874
+ } });
875
+ this.engines.delete(model.modelId);
876
+ this.lastUsed.delete(model.modelId);
877
+ this.deps.budget.record(model.modelId, "process died after loading");
878
+ held.dispose().catch(() => {});
879
+ }
880
+ const pending = this.inflight.get(model.modelId);
881
+ if (pending !== void 0) return pending;
882
+ this.deps.budget.check(model.modelId);
883
+ this.deps.requirements.check(model.modelId);
884
+ const textEntry = this.deps.textCatalog.find((e) => e.id === model.meta.textModelId);
885
+ if (textEntry === void 0) throw this.deps.requirements.refuse(model.modelId, new PermanentRequirementsFailure(`EmbeddingEncoder: text model "${model.meta.textModelId}" (for ${model.modelId}) is not in the text catalog`));
886
+ const generation = this.generation;
887
+ const build = this.buildAndLoad(model, textEntry);
888
+ this.inflight.set(model.modelId, build);
889
+ try {
890
+ const engine = await build;
891
+ if (this.generation !== generation) {
892
+ await engine.dispose();
893
+ throw new Error(`text tower for "${model.modelId}" cancelled: disposed while building`);
894
+ }
895
+ this.engines.set(model.modelId, engine);
896
+ this.lastUsed.set(model.modelId, this.now());
897
+ this.deps.logger.info("CLIP text tower loaded", { meta: {
898
+ modelId: model.modelId,
899
+ textModelId: model.meta.textModelId
900
+ } });
901
+ this.scheduleEviction();
902
+ return engine;
903
+ } finally {
904
+ this.inflight.delete(model.modelId);
905
+ }
906
+ }
907
+ /**
908
+ * Two stages, two owners (`model-requirements.ts`): `build` resolves files
909
+ * and runtime and constructs the tower — a throw there is refused by name
910
+ * and spends nothing; `initialize()` spawns and loads — a throw there is a
911
+ * load failure and spends the crash budget.
912
+ */
913
+ async buildAndLoad(model, textEntry) {
914
+ let engine;
915
+ try {
916
+ engine = await this.deps.build({
917
+ model,
918
+ textEntry,
919
+ meta: model.meta
920
+ });
921
+ } catch (err) {
922
+ throw this.deps.requirements.refuse(model.modelId, err);
923
+ }
924
+ this.deps.requirements.satisfied(model.modelId);
925
+ try {
926
+ await engine.initialize();
927
+ } catch (err) {
928
+ this.deps.budget.record(model.modelId, `load failed: ${err instanceof Error ? err.message : String(err)}`);
929
+ throw err;
930
+ }
931
+ return engine;
932
+ }
933
+ now() {
934
+ return (this.deps.now ?? Date.now)();
935
+ }
936
+ /** Fire-and-log: eviction errors never reach an encode. */
937
+ scheduleEviction() {
938
+ this.evictBeyond(this.deps.maxEngines ?? 2).catch((err) => {
939
+ this.deps.logger.warn("CLIP text tower eviction failed", { meta: { error: err instanceof Error ? err.message : String(err) } });
940
+ });
941
+ }
942
+ /** Evict the coldest IDLE engines until at most `max` remain, or none is idle. */
943
+ async evictBeyond(max) {
944
+ while (this.engines.size > max) {
945
+ const coldestIdle = [...this.engines.keys()].find((id) => (this.busy.get(id) ?? 0) === 0);
946
+ if (coldestIdle === void 0) return;
947
+ await this.restart(coldestIdle);
948
+ }
949
+ }
950
+ /** Encodes in flight on a model — exposed for the eviction tests. */
951
+ inFlight(modelId) {
952
+ return this.busy.get(modelId) ?? 0;
953
+ }
954
+ };
531
955
  //#endregion
532
956
  //#region src/embedding-encoder/addon/index.ts
957
+ /** Memory-watchdog key prefix of a text tower; the suffix is its CLIP model id. */
958
+ var TEXT_ENGINE_KEY_PREFIX = "clip-text:";
959
+ /** How often idle text towers are looked for (see `TEXT_ENGINE_IDLE_MS`). */
960
+ var IDLE_SWEEP_MS = 6e4;
961
+ /** A model URL answering one of these does not exist; retrying changes nothing. */
962
+ var PERMANENT_HTTP_STATUS = new Set([404, 410]);
533
963
  var EmbeddingEncoderAddon = class extends BaseAddon {
534
- imageRawEngine = null;
535
- textEngine = null;
964
+ /**
965
+ * Everything on this node that can be terminally refused (`failed-models-report.ts`):
966
+ * the respawn budgets per tower (D6, `crash-budget.ts`) and the requirements
967
+ * gates (`model-requirements.ts` — cooldown, and terminal for what can never
968
+ * be met). All in memory: a runner respawn starts them from zero. Every row
969
+ * and refusal names this node.
970
+ */
971
+ towers = {
972
+ budgets: {
973
+ image: new CrashBudget({
974
+ tower: "image",
975
+ nodeId: () => this.localNodeId(),
976
+ logger: {
977
+ info: (message, extras) => this.ctx.logger.info(message, extras),
978
+ error: (message, extras) => this.ctx.logger.error(message, extras)
979
+ }
980
+ }),
981
+ text: new CrashBudget({
982
+ tower: "text",
983
+ nodeId: () => this.localNodeId(),
984
+ logger: {
985
+ info: (message, extras) => this.ctx.logger.info(message, extras),
986
+ error: (message, extras) => this.ctx.logger.error(message, extras)
987
+ }
988
+ })
989
+ },
990
+ requirements: {
991
+ image: new RequirementsGate({
992
+ tower: "image",
993
+ nodeId: () => this.localNodeId(),
994
+ logger: {
995
+ info: (message, extras) => this.ctx.logger.info(message, extras),
996
+ warn: (message, extras) => this.ctx.logger.warn(message, extras),
997
+ error: (message, extras) => this.ctx.logger.error(message, extras)
998
+ }
999
+ }),
1000
+ text: new RequirementsGate({
1001
+ tower: "text",
1002
+ nodeId: () => this.localNodeId(),
1003
+ logger: {
1004
+ info: (message, extras) => this.ctx.logger.info(message, extras),
1005
+ warn: (message, extras) => this.ctx.logger.warn(message, extras),
1006
+ error: (message, extras) => this.ctx.logger.error(message, extras)
1007
+ }
1008
+ })
1009
+ }
1010
+ };
1011
+ /** The image tower, one model at a time, single-flight per model (`model-engine-slot.ts`). */
1012
+ imageEngine = new ModelEngineSlot((modelId) => this.buildImageEngine(modelId), this.towers.budgets.image, this.towers.requirements.image, (modelId) => this.ctx.logger.warn("CLIP image tower died — dropping it and rebuilding", { meta: { modelId } }));
1013
+ /** One text tower per CLIP model, bounded (see `clip-text-engines.ts`). */
1014
+ textEngines = new ClipTextEngines({
1015
+ registry: BUILTIN_CLIP_MODELS,
1016
+ textCatalog: CLIP_TEXT_MODELS,
1017
+ build: (spec) => this.buildTextEngine(spec),
1018
+ budget: this.towers.budgets.text,
1019
+ requirements: this.towers.requirements.text,
1020
+ activeModelId: () => this.activeModel.known(),
1021
+ logger: {
1022
+ info: (message, extras) => this.ctx.logger.info(message, extras),
1023
+ warn: (message, extras) => this.ctx.logger.warn(message, extras)
1024
+ }
1025
+ });
1026
+ /** Disposes NON-active text towers idle for `TEXT_ENGINE_IDLE_MS` (never prefetched). */
1027
+ idleSweep = null;
1028
+ /** The cluster row's CLIP model — what an unnamed request encodes in (D649). */
1029
+ activeModel = new ActiveClipModel({
1030
+ readRow: () => this.readClusterRow(),
1031
+ fallbackModelId: () => BUILTIN_CLIP_MODELS.resolve(this.config.modelId) !== null ? this.config.modelId : DEFAULT_CLIP_MODEL,
1032
+ now: () => Date.now(),
1033
+ logger: {
1034
+ info: (message, extras) => this.ctx.logger.info(message, extras),
1035
+ warn: (message, extras) => this.ctx.logger.warn(message, extras)
1036
+ },
1037
+ onChange: (modelId) => {
1038
+ clearFailedOnModelChange(this.towers, modelId);
1039
+ }
1040
+ });
536
1041
  models = null;
537
1042
  /**
538
1043
  * RSS bound + periodic memory telemetry for the two Python engines
@@ -544,12 +1049,6 @@ var EmbeddingEncoderAddon = class extends BaseAddon {
544
1049
  * stopped in `onShutdown`.
545
1050
  */
546
1051
  memoryGuard = null;
547
- /** Single-flight guards for the lazy engine builds — a watchdog restart
548
- * followed by two concurrent encodes must not spawn the engine twice
549
- * (the same bug detection-pipeline fixed three times; see its
550
- * `engineFactoryInflight`). */
551
- imageEngineInflight = null;
552
- textEngineInflight = null;
553
1052
  constructor() {
554
1053
  super({ modelId: DEFAULT_CLIP_MODEL });
555
1054
  }
@@ -563,6 +1062,12 @@ var EmbeddingEncoderAddon = class extends BaseAddon {
563
1062
  restart: (key) => this.restartEngineForMemory(key)
564
1063
  });
565
1064
  this.memoryGuard.start();
1065
+ this.idleSweep = setInterval(() => {
1066
+ this.textEngines.evictIdle().catch((err) => {
1067
+ this.ctx.logger.warn("CLIP text tower idle sweep failed", { meta: { error: err instanceof Error ? err.message : String(err) } });
1068
+ });
1069
+ }, IDLE_SWEEP_MS);
1070
+ this.idleSweep.unref?.();
566
1071
  return [{
567
1072
  capability: embeddingEncoderCapability,
568
1073
  provider: this
@@ -573,13 +1078,15 @@ var EmbeddingEncoderAddon = class extends BaseAddon {
573
1078
  async sampleEngineMemory() {
574
1079
  const engines = [{
575
1080
  key: "clip-image",
576
- pid: this.imageRawEngine?.getPid() ?? null,
577
- requests: this.imageRawEngine?.getRequestCount() ?? 0
578
- }, {
579
- key: "clip-text",
580
- pid: this.textEngine?.getPid() ?? null,
581
- requests: this.textEngine?.getRequestCount() ?? 0
582
- }];
1081
+ modelId: this.imageEngine.held()?.modelId ?? null,
1082
+ pid: this.imageEngine.held()?.engine.getPid() ?? null,
1083
+ requests: this.imageEngine.held()?.engine.getRequestCount() ?? 0
1084
+ }, ...[...this.textEngines.live()].map(([modelId, engine]) => ({
1085
+ key: `${TEXT_ENGINE_KEY_PREFIX}${modelId}`,
1086
+ modelId,
1087
+ pid: engine.getPid(),
1088
+ requests: engine.getRequestCount()
1089
+ }))];
583
1090
  const out = [];
584
1091
  for (const engine of engines) {
585
1092
  if (engine.pid === null) continue;
@@ -597,7 +1104,7 @@ var EmbeddingEncoderAddon = class extends BaseAddon {
597
1104
  pids: [engine.pid],
598
1105
  rssBytes: mem.rssBytes,
599
1106
  meta: {
600
- modelId: this.config.modelId,
1107
+ modelId: engine.modelId,
601
1108
  vmMb: mb(mem.vmBytes),
602
1109
  peakRssMb: mb(mem.hwmBytes),
603
1110
  swapMb: mb(mem.swapBytes),
@@ -614,110 +1121,125 @@ var EmbeddingEncoderAddon = class extends BaseAddon {
614
1121
  * surface errors, never silent loss. */
615
1122
  async restartEngineForMemory(key) {
616
1123
  if (key === "clip-image") {
617
- const engine = this.imageRawEngine;
618
- if (!engine) return;
619
- this.imageRawEngine = null;
620
- await engine.dispose();
1124
+ await this.imageEngine.dispose();
621
1125
  return;
622
1126
  }
623
- const engine = this.textEngine;
624
- if (!engine) return;
625
- this.textEngine = null;
626
- await engine.dispose();
1127
+ if (key.startsWith(TEXT_ENGINE_KEY_PREFIX)) await this.textEngines.restart(key.slice(10));
1128
+ }
1129
+ /** The cluster row's `clip-embedding` model, or `null` when unreadable. */
1130
+ async readClusterRow() {
1131
+ return resolveClusterModelPin(this.ctx.api, CLIP_EMBEDDING_STEP_ID, { warn: (message, extras) => this.ctx.logger.warn(message, extras) });
627
1132
  }
628
1133
  async encode(input) {
629
1134
  const { crop, width, height } = input;
630
- await this.ensureImageEngine();
631
- const meta = getModelMeta(this.config.modelId);
1135
+ const model = this.textEngines.requireModel(await this.activeModel.get());
1136
+ const imageEngine = await this.imageEngine.ensure(model.modelId);
1137
+ const meta = model.meta;
632
1138
  const start = Date.now();
633
- const preprocessed = preprocessForClip(Buffer.isBuffer(crop) ? crop : Buffer.from(crop), width, height, meta.inputSize, meta.inputSize);
634
- const output = await this.imageRawEngine.run(preprocessed, [
1139
+ const preprocessed = preprocessForClip(Buffer.isBuffer(crop) ? crop : Buffer.from(crop), width, height, model.inputSize, model.inputSize);
1140
+ const output = await imageEngine.run(preprocessed, [
635
1141
  1,
636
1142
  3,
637
- meta.inputSize,
638
- meta.inputSize
1143
+ model.inputSize,
1144
+ model.inputSize
639
1145
  ]);
640
- const sliced = output.length > meta.embeddingDim ? output.slice(0, meta.embeddingDim) : output;
641
- const normalized = l2Normalize(new Float32Array(sliced));
1146
+ if (output.length !== meta.embeddingDim) throw new Error(`EmbeddingEncoder: ${model.modelId} image tower produced ${String(output.length)} dims, its metadata declares ${String(meta.embeddingDim)}`);
1147
+ const normalized = l2Normalize(new Float32Array(output));
642
1148
  return {
643
1149
  embedding: Array.from(normalized),
644
1150
  inferenceMs: Date.now() - start
645
1151
  };
646
1152
  }
1153
+ /**
1154
+ * Encode a query in ONE model's space: the named `modelId`, else the cluster
1155
+ * row's (D649). The token window, pad id, tokenizer and dimension all come
1156
+ * from that model's catalog metadata.
1157
+ */
647
1158
  async encodeText(input) {
648
- const { text } = input;
649
- await this.ensureTextEngine();
650
- const meta = getModelMeta(this.config.modelId);
1159
+ const modelId = input.modelId ?? await this.activeModel.get();
651
1160
  const start = Date.now();
652
- if (!this.textEngine) throw new Error("EmbeddingEncoder: text engine not loaded — ensureTextEngine() must run first");
653
- const output = await this.textEngine.encode(text);
654
- const sliced = output.length > meta.embeddingDim ? output.slice(0, meta.embeddingDim) : output;
655
- const normalized = l2Normalize(new Float32Array(sliced));
1161
+ const { vector } = await this.textEngines.encode(modelId, input.text);
656
1162
  return {
657
- embedding: Array.from(normalized),
1163
+ embedding: Array.from(l2Normalize(new Float32Array(vector))),
658
1164
  inferenceMs: Date.now() - start
659
1165
  };
660
1166
  }
1167
+ /** THIS node's answer — the cap is one provider per node (see the cap's docblock). */
661
1168
  async getInfo() {
662
- const meta = getModelMeta(this.config.modelId);
1169
+ const model = this.textEngines.requireModel(await this.activeModel.get());
1170
+ const nodeId = this.localNodeId();
663
1171
  return {
664
- modelId: this.config.modelId,
665
- embeddingDim: meta.embeddingDim,
666
- ready: this.imageRawEngine !== null
1172
+ modelId: model.modelId,
1173
+ embeddingDim: model.meta.embeddingDim,
1174
+ ready: this.imageEngine.held()?.modelId === model.modelId,
1175
+ nodeId,
1176
+ failedModels: [...reportFailedModels(this.towers.budgets)],
1177
+ failedRequirements: [...reportFailedRequirements(this.towers.requirements)]
667
1178
  };
668
1179
  }
669
- async ensureImageEngine() {
670
- if (this.imageRawEngine) return;
671
- if (this.imageEngineInflight) return this.imageEngineInflight;
672
- const meta = getModelMeta(this.config.modelId);
673
- const imageEntry = CLIP_IMAGE_MODELS.find((m) => m.id === meta.imageModelId);
674
- if (!imageEntry) throw new Error(`EmbeddingEncoderAddon: unknown image model "${meta.imageModelId}"`);
675
- const inflight = this.resolveForEntry(imageEntry, "image");
676
- this.imageEngineInflight = inflight;
677
- try {
678
- await inflight;
679
- } finally {
680
- this.imageEngineInflight = null;
681
- }
1180
+ /** The operator's explicit clear of THIS node's `failed` state (D6); names the node. */
1181
+ async resetFailedModels(input) {
1182
+ return resetFailedModelsOn(this.localNodeId(), this.towers, input.modelId);
682
1183
  }
683
- async ensureTextEngine() {
684
- if (this.textEngine) return;
685
- if (this.textEngineInflight) return this.textEngineInflight;
686
- const meta = getModelMeta(this.config.modelId);
687
- const textEntry = CLIP_TEXT_MODELS.find((m) => m.id === meta.textModelId);
688
- if (!textEntry) throw new Error(`EmbeddingEncoderAddon: unknown text model "${meta.textModelId}"`);
689
- const inflight = this.resolveForEntry(textEntry, "text");
690
- this.textEngineInflight = inflight;
691
- try {
692
- await inflight;
693
- } finally {
694
- this.textEngineInflight = null;
695
- }
1184
+ /** The logical node (a forked child's `<node>/<addon>` id stripped to the node). */
1185
+ localNodeId() {
1186
+ return normalizeNodeId(this.ctx.kernel?.localNodeId);
696
1187
  }
697
- async resolveForEntry(entry, target) {
698
- const engineLogger = this.ctx.logger.withTags({ modelId: entry.id });
699
- const modelPath = await this.models.ensure(entry.id, "onnx");
1188
+ /**
1189
+ * The REQUIREMENTS stage, shared by both towers: the embedded Python, its
1190
+ * requirements, then the model file (downloaded on first use). Anything that
1191
+ * throws here is refused as `model-requirements-unavailable` and spends no
1192
+ * crash budget (`model-requirements.ts`). The runtime is asked for first —
1193
+ * the cheap question before the 185–565 MB download (D54's rule).
1194
+ */
1195
+ async prepareOnnx(entry) {
700
1196
  const pythonPath = await this.ctx.deps.ensurePython();
701
1197
  if (!pythonPath) throw new Error("EmbeddingEncoder: embedded Python is unavailable — cannot run ONNX embeddings. ctx.deps.ensurePython() returned null (portable Python download likely failed).");
702
1198
  const pythonDir = resolveEmbeddingPythonDir();
703
1199
  await this.ctx.deps.installPythonRequirements(path$1.join(pythonDir, "requirements-embedding.txt"));
704
- if (target === "image") {
705
- const rawEngine = new PythonRawTensorEngine(pythonPath, path$1.join(pythonDir, "raw_tensor_inference.py"), modelPath, engineLogger);
706
- await rawEngine.initialize();
707
- this.imageRawEngine = rawEngine;
708
- return;
1200
+ if (entry.formats.onnx === void 0) throw new PermanentRequirementsFailure(`EmbeddingEncoder: ${entry.id} has no onnx build in its catalog entry (formats: ${Object.keys(entry.formats).join(", ") || "none"}) — no download can supply it`);
1201
+ let modelPath;
1202
+ try {
1203
+ modelPath = await this.models.ensure(entry.id, "onnx");
1204
+ } catch (err) {
1205
+ const status = downloadHttpStatusOf(err);
1206
+ if (status !== null && PERMANENT_HTTP_STATUS.has(status)) throw new PermanentRequirementsFailure(`EmbeddingEncoder: ${entry.id} is not at its catalog URL (HTTP ${String(status)}) — fix the catalog entry`);
1207
+ throw err;
709
1208
  }
710
- const tokenizerPath = path$1.join(path$1.dirname(modelPath), TOKENIZER_FILE);
711
- if (!fs.existsSync(tokenizerPath)) throw new Error(`EmbeddingEncoder: CLIP tokenizer not found at "${tokenizerPath}" — the tokenizer.json sibling download likely failed.`);
712
- const textEngine = new PythonTextEncoderEngine(pythonPath, path$1.join(pythonDir, "text_encoder_inference.py"), modelPath, tokenizerPath, engineLogger);
713
- await textEngine.initialize();
714
- this.textEngine = textEngine;
1209
+ return {
1210
+ modelPath,
1211
+ pythonPath,
1212
+ pythonDir
1213
+ };
1214
+ }
1215
+ async buildImageEngine(modelId) {
1216
+ const entry = CLIP_IMAGE_MODELS.find((m) => m.id === modelId);
1217
+ if (!entry) throw new PermanentRequirementsFailure(`EmbeddingEncoderAddon: unknown image model "${modelId}" — not in the CLIP image catalog`);
1218
+ const { modelPath, pythonPath, pythonDir } = await this.prepareOnnx(entry);
1219
+ return new PythonRawTensorEngine(pythonPath, path$1.join(pythonDir, "raw_tensor_inference.py"), modelPath, this.ctx.logger.withTags({ modelId: entry.id }));
1220
+ }
1221
+ /**
1222
+ * One text tower. Tokenization happens IN Python via the HF `tokenizers`
1223
+ * library, with the tokenizer the model's metadata names — a declared sibling
1224
+ * of the text onnx, so `ModelDownloadService.ensure()` fetched it next to the
1225
+ * model file — and the model's own token window.
1226
+ */
1227
+ async buildTextEngine(spec) {
1228
+ const { modelPath, pythonPath, pythonDir } = await this.prepareOnnx(spec.textEntry);
1229
+ const tokenizerPath = path$1.join(path$1.dirname(modelPath), spec.meta.tokenizerFile);
1230
+ if (!fs.existsSync(tokenizerPath)) throw new PermanentRequirementsFailure(`EmbeddingEncoder: tokenizer not found at "${tokenizerPath}" — the ${spec.meta.tokenizerFile} sibling download of ${spec.textEntry.id} likely failed.`);
1231
+ return new PythonTextEncoderEngine(pythonPath, path$1.join(pythonDir, "text_encoder_inference.py"), modelPath, tokenizerPath, {
1232
+ contextLength: spec.meta.contextLength,
1233
+ padId: spec.meta.padId
1234
+ }, this.ctx.logger.withTags({ modelId: spec.textEntry.id }));
715
1235
  }
716
1236
  async onShutdown() {
717
1237
  this.memoryGuard?.stop();
1238
+ if (this.idleSweep !== null) clearInterval(this.idleSweep);
1239
+ this.idleSweep = null;
718
1240
  this.memoryGuard = null;
719
- await this.imageRawEngine?.dispose();
720
- await this.textEngine?.dispose();
1241
+ await this.imageEngine.dispose();
1242
+ await this.textEngines.disposeAll();
721
1243
  }
722
1244
  globalSettingsSchema() {
723
1245
  return this.schema({ sections: [{
@@ -727,8 +1249,8 @@ var EmbeddingEncoderAddon = class extends BaseAddon {
727
1249
  fields: [{
728
1250
  type: "text",
729
1251
  key: "modelId",
730
- label: "Model ID",
731
- description: "CLIP model identifier to use for image/text embedding",
1252
+ label: "Fallback model ID",
1253
+ description: "Used only until the cluster \"Semantic search model\" row has been read once. The encoder follows that row (D649).",
732
1254
  default: DEFAULT_CLIP_MODEL
733
1255
  }]
734
1256
  }] });
@@ -750,5 +1272,17 @@ function resolveEmbeddingPythonDir() {
750
1272
  for (const c of candidates) if (fs.existsSync(path$1.join(c, "raw_tensor_inference.py"))) return c;
751
1273
  throw new Error(`EmbeddingEncoder: python/ dir (raw_tensor_inference.py) not found. Searched:\n${candidates.join("\n")}`);
752
1274
  }
1275
+ /**
1276
+ * HTTP status of a model-download failure, recognised by SHAPE (`name` +
1277
+ * numeric `status`), never by `instanceof` or a framework import.
1278
+ * `@camstack/system` is host-resolved (not bundled), so importing a guard it
1279
+ * only exports from a newer server would stop this whole entry from linking
1280
+ * on a node still running the older server — a deploy-order hazard for a
1281
+ * one-line classification. `null` = not an HTTP answer (unreachable, other).
1282
+ */
1283
+ function downloadHttpStatusOf(err) {
1284
+ if (err instanceof Error && err.name === "ModelDownloadHttpError" && "status" in err && typeof err.status === "number") return err.status;
1285
+ return null;
1286
+ }
753
1287
  //#endregion
754
1288
  export { EmbeddingEncoderAddon, EmbeddingEncoderAddon as default, __exportAll as t };