@camstack/addon-post-analysis 1.2.285 → 1.2.286
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD_PARTY_MODELS.md +8 -0
- package/dist/{dist-ITESpHou.js → clip-model-registry-CY0FGcpQ.js} +1483 -403
- package/dist/{dist-D2tXUMfE.mjs → clip-model-registry-D4mC2M7F.mjs} +1374 -390
- package/dist/embedding-encoder/index.js +962 -428
- package/dist/embedding-encoder/index.mjs +952 -418
- package/dist/pipeline-analytics/_stub.js +2 -2
- package/dist/pipeline-analytics/{_virtual_mf-localSharedImportMap___mfe_internal__addon_pipeline_analytics_widgets-7oasHfhD.mjs → _virtual_mf-localSharedImportMap___mfe_internal__addon_pipeline_analytics_widgets-CkfvSztj.mjs} +2 -2
- package/dist/pipeline-analytics/_virtual_mf___mfe_internal__addon_pipeline_analytics_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-NKbCrEkH.mjs +26 -0
- package/dist/pipeline-analytics/_virtual_mf___mfe_internal__addon_pipeline_analytics_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-BAYoLHsN.mjs +26 -0
- package/dist/pipeline-analytics/{hostInit-B6m75Rnz.mjs → hostInit-CelYj4QN.mjs} +2 -2
- package/dist/pipeline-analytics/index.js +5863 -5053
- package/dist/pipeline-analytics/index.mjs +4510 -3700
- package/dist/pipeline-analytics/remoteEntry.js +1 -1
- package/package.json +1 -1
- package/python/test_text_encoder.py +49 -2
- package/python/text_encoder_inference.py +31 -14
- package/dist/pipeline-analytics/_virtual_mf___mfe_internal__addon_pipeline_analytics_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-CjeM5Bph.mjs +0 -26
- package/dist/pipeline-analytics/_virtual_mf___mfe_internal__addon_pipeline_analytics_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-BidXbeas.mjs +0 -26
|
@@ -2,11 +2,11 @@ Object.defineProperties(exports, {
|
|
|
2
2
|
__esModule: { value: true },
|
|
3
3
|
[Symbol.toStringTag]: { value: "Module" }
|
|
4
4
|
});
|
|
5
|
-
const
|
|
5
|
+
const require_clip_model_registry = require("../clip-model-registry-CY0FGcpQ.js");
|
|
6
6
|
let node_fs = require("node:fs");
|
|
7
|
-
node_fs =
|
|
7
|
+
node_fs = require_clip_model_registry.__toESM(node_fs);
|
|
8
8
|
let node_path = require("node:path");
|
|
9
|
-
node_path =
|
|
9
|
+
node_path = require_clip_model_registry.__toESM(node_path);
|
|
10
10
|
let node_child_process = require("node:child_process");
|
|
11
11
|
let _camstack_system_addon_utils = require("@camstack/system/addon-utils");
|
|
12
12
|
//#region src/embedding-encoder/shared/process-memory.ts
|
|
@@ -28,7 +28,7 @@ let _camstack_system_addon_utils = require("@camstack/system/addon-utils");
|
|
|
28
28
|
var IS_LINUX = process.platform === "linux";
|
|
29
29
|
async function readProcessMemory(pid) {
|
|
30
30
|
if (IS_LINUX) try {
|
|
31
|
-
return
|
|
31
|
+
return require_clip_model_registry.parseProcStatus(await node_fs.promises.readFile(`/proc/${pid}/status`, "utf8"));
|
|
32
32
|
} catch {
|
|
33
33
|
return null;
|
|
34
34
|
}
|
|
@@ -56,232 +56,101 @@ function readViaPs(pid) {
|
|
|
56
56
|
});
|
|
57
57
|
}
|
|
58
58
|
//#endregion
|
|
59
|
-
//#region src/embedding-encoder/
|
|
60
|
-
var HF_REPO = "camstack/camstack-models";
|
|
61
|
-
var hf = (path) => require_dist.hfModelUrl(HF_REPO, path);
|
|
59
|
+
//#region src/embedding-encoder/shared/framed-python-process.ts
|
|
62
60
|
/**
|
|
63
|
-
*
|
|
64
|
-
*
|
|
65
|
-
*
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
*
|
|
80
|
-
*
|
|
81
|
-
*/
|
|
82
|
-
var MLPACKAGE_FILES = [
|
|
83
|
-
"Manifest.json",
|
|
84
|
-
"Data/com.apple.CoreML/model.mlmodel",
|
|
85
|
-
"Data/com.apple.CoreML/weights/weight.bin"
|
|
86
|
-
];
|
|
87
|
-
/**
|
|
88
|
-
* NO onnx vision builds, deliberately (2026-08-21). The int8 ONNX vision
|
|
89
|
-
* exports were measured misaligned with the text encoders (matched image↔text
|
|
90
|
-
* cosine ≈ 0.01 vs ≈ 0.22 for the fp16 openvino/coreml builds) — see the
|
|
91
|
-
* catalog note in `addon-pipeline/.../model-catalogs.ts`. The image-encode leg
|
|
92
|
-
* that consumed them here (`embeddingEncoder.encode` → PythonRawTensorEngine)
|
|
93
|
-
* is retired with them: its only caller discarded the vector, and its
|
|
94
|
-
* `preprocessForClip` applied OpenAI mean/std that MobileCLIP never used.
|
|
95
|
-
* S0 was retired the same day (measured worst of the family on fleet crops).
|
|
96
|
-
*/
|
|
97
|
-
var CLIP_IMAGE_MODELS = [{
|
|
98
|
-
id: "mobileclip-s1",
|
|
99
|
-
name: "MobileCLIP S1",
|
|
100
|
-
description: "Apple MobileCLIP S1 — balanced vision encoder, 512-dim, 256×256 (fp16 OpenVINO/CoreML)",
|
|
101
|
-
inputSize: {
|
|
102
|
-
width: 256,
|
|
103
|
-
height: 256
|
|
104
|
-
},
|
|
105
|
-
labels: [],
|
|
106
|
-
inputNormalization: "none",
|
|
107
|
-
formats: {
|
|
108
|
-
openvino: ovFormat(hf("clip/mobileclip-s1/openvino/camstack-mobileclip-s1-vision.xml"), 55),
|
|
109
|
-
coreml: {
|
|
110
|
-
url: hf("clip/mobileclip-s1/coreml/camstack-mobileclip-s1-vision.mlpackage"),
|
|
111
|
-
sizeMB: 65,
|
|
112
|
-
isDirectory: true,
|
|
113
|
-
files: [...MLPACKAGE_FILES],
|
|
114
|
-
runtimes: ["python"]
|
|
115
|
-
}
|
|
116
|
-
}
|
|
117
|
-
}, {
|
|
118
|
-
id: "mobileclip-s2",
|
|
119
|
-
name: "MobileCLIP S2",
|
|
120
|
-
description: "Apple MobileCLIP S2 — high-accuracy vision encoder, 512-dim, 256×256 (fp16 OpenVINO/CoreML)",
|
|
121
|
-
inputSize: {
|
|
122
|
-
width: 256,
|
|
123
|
-
height: 256
|
|
124
|
-
},
|
|
125
|
-
labels: [],
|
|
126
|
-
inputNormalization: "none",
|
|
127
|
-
formats: {
|
|
128
|
-
openvino: ovFormat(hf("clip/mobileclip-s2/openvino/camstack-mobileclip-s2-vision.xml"), 90),
|
|
129
|
-
coreml: {
|
|
130
|
-
url: hf("clip/mobileclip-s2/coreml/camstack-mobileclip-s2-vision.mlpackage"),
|
|
131
|
-
sizeMB: 110,
|
|
132
|
-
isDirectory: true,
|
|
133
|
-
files: [...MLPACKAGE_FILES],
|
|
134
|
-
runtimes: ["python"]
|
|
135
|
-
}
|
|
136
|
-
}
|
|
137
|
-
}];
|
|
138
|
-
/**
|
|
139
|
-
* The int8 TEXT onnx encoders are healthy — unlike the retired int8 vision
|
|
140
|
-
* exports. Verified 2026-08-21: the live search path (fp16 openvino vision
|
|
141
|
-
* vectors ⋅ int8 onnx text queries) ranks correctly on the real index, and the
|
|
142
|
-
* local cross-check aligns them with the fp16 vision space (cos ≈ 0.22 on
|
|
143
|
-
* matched pairs).
|
|
144
|
-
*/
|
|
145
|
-
var CLIP_TEXT_MODELS = [{
|
|
146
|
-
id: "mobileclip-s1-text",
|
|
147
|
-
name: "MobileCLIP S1 Text Encoder",
|
|
148
|
-
description: "Text encoder for MobileCLIP S1, 512-dim, int8 quantized (61 MB)",
|
|
149
|
-
inputSize: {
|
|
150
|
-
width: 0,
|
|
151
|
-
height: 0
|
|
152
|
-
},
|
|
153
|
-
labels: [],
|
|
154
|
-
formats: {
|
|
155
|
-
onnx: {
|
|
156
|
-
url: hf("clip/mobileclip-s1/onnx/camstack-mobileclip-s1-text.onnx"),
|
|
157
|
-
sizeMB: 61,
|
|
158
|
-
files: [TOKENIZER_FILE]
|
|
159
|
-
},
|
|
160
|
-
openvino: ovFormat(hf("clip/mobileclip-s1/openvino/camstack-mobileclip-s1-text.xml"), 121)
|
|
161
|
-
}
|
|
162
|
-
}, {
|
|
163
|
-
id: "mobileclip-s2-text",
|
|
164
|
-
name: "MobileCLIP S2 Text Encoder",
|
|
165
|
-
description: "Text encoder for MobileCLIP S2, 512-dim, int8 quantized (61 MB)",
|
|
166
|
-
inputSize: {
|
|
167
|
-
width: 0,
|
|
168
|
-
height: 0
|
|
169
|
-
},
|
|
170
|
-
labels: [],
|
|
171
|
-
formats: {
|
|
172
|
-
onnx: {
|
|
173
|
-
url: hf("clip/mobileclip-s2/onnx/camstack-mobileclip-s2-text.onnx"),
|
|
174
|
-
sizeMB: 61,
|
|
175
|
-
files: [TOKENIZER_FILE]
|
|
176
|
-
},
|
|
177
|
-
openvino: ovFormat(hf("clip/mobileclip-s2/openvino/camstack-mobileclip-s2-text.xml"), 121)
|
|
178
|
-
}
|
|
179
|
-
}];
|
|
180
|
-
//#endregion
|
|
181
|
-
//#region src/embedding-encoder/shared/noop-logger.ts
|
|
182
|
-
var noop = () => {};
|
|
183
|
-
function createNoopLogger() {
|
|
184
|
-
const logger = {
|
|
185
|
-
debug: noop,
|
|
186
|
-
info: noop,
|
|
187
|
-
warn: noop,
|
|
188
|
-
error: noop,
|
|
189
|
-
child: () => logger,
|
|
190
|
-
withTags: (_tags) => logger
|
|
191
|
-
};
|
|
192
|
-
return logger;
|
|
193
|
-
}
|
|
194
|
-
//#endregion
|
|
195
|
-
//#region src/embedding-encoder/shared/python-raw-tensor-engine.ts
|
|
196
|
-
/**
|
|
197
|
-
* Raw-tensor ONNX engine backed by an embedded-Python subprocess
|
|
198
|
-
* (`raw_tensor_inference.py`). Replaces the Node `onnxruntime-node` raw-tensor
|
|
199
|
-
* engine so the platform ships no Node ONNX runtime. The caller preprocesses to
|
|
200
|
-
* a Float32Array; this engine ships it to Python, which runs onnxruntime and
|
|
201
|
-
* returns the output tensor. Wire protocol = length-prefixed binary frames
|
|
202
|
-
* ([4B LE length][payload]).
|
|
61
|
+
* One embedded-Python subprocess speaking the length-prefixed frame protocol
|
|
62
|
+
* (`[4B LE len][payload]`, ready = `[0x01]`) — the lifecycle both embedding
|
|
63
|
+
* engines share, written once.
|
|
64
|
+
*
|
|
65
|
+
* ## What it guarantees (D649 review)
|
|
66
|
+
*
|
|
67
|
+
* - **Every awaited frame settles.** Waiters are a FIFO queue (the process
|
|
68
|
+
* answers in order). On ANY exit — a crash, an OOM kill, a clean 0, the
|
|
69
|
+
* SIGTERM of `dispose` (code null) — and on a stdin error, every waiter is
|
|
70
|
+
* rejected. The old guard (`code !== 0 && code !== null`) hung the encode.
|
|
71
|
+
* - **A dead process is dead, by name.** After an unexpected exit the engine
|
|
72
|
+
* is marked dead with the exit code/signal, the exit is logged, and every
|
|
73
|
+
* later request rejects immediately with that reason instead of writing to
|
|
74
|
+
* a closed pipe and waiting forever. Holders check {@link isAlive} and
|
|
75
|
+
* rebuild — so a crash costs one failed request, not a silent stop.
|
|
76
|
+
* - **EPIPE never escapes.** stdin carries an `error` listener: a write into a
|
|
77
|
+
* process that just died becomes a rejected request, never an uncaught
|
|
78
|
+
* exception in the addon.
|
|
203
79
|
*/
|
|
204
|
-
var
|
|
80
|
+
var FramedPythonProcess = class {
|
|
81
|
+
name;
|
|
205
82
|
pythonPath;
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
runtime = "onnx";
|
|
209
|
-
device = "cpu";
|
|
83
|
+
args;
|
|
84
|
+
log;
|
|
210
85
|
process = null;
|
|
211
86
|
receiveBuffer = Buffer.alloc(0);
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
constructor(
|
|
87
|
+
pending = [];
|
|
88
|
+
dead = null;
|
|
89
|
+
disposing = false;
|
|
90
|
+
constructor(name, pythonPath, args, log) {
|
|
91
|
+
this.name = name;
|
|
216
92
|
this.pythonPath = pythonPath;
|
|
217
|
-
this.
|
|
218
|
-
this.
|
|
219
|
-
this.log = logger ?? createNoopLogger();
|
|
93
|
+
this.args = args;
|
|
94
|
+
this.log = log;
|
|
220
95
|
}
|
|
221
|
-
|
|
222
|
-
|
|
96
|
+
/** Spawn and wait for the ready frame. */
|
|
97
|
+
async start() {
|
|
98
|
+
const proc = (0, node_child_process.spawn)(this.pythonPath, [...this.args], { stdio: [
|
|
223
99
|
"pipe",
|
|
224
100
|
"pipe",
|
|
225
101
|
"pipe"
|
|
226
102
|
] });
|
|
227
|
-
this.process
|
|
103
|
+
this.process = proc;
|
|
104
|
+
proc.stderr?.on("data", (chunk) => {
|
|
228
105
|
const text = chunk.toString().trim();
|
|
229
106
|
if (text) this.log.warn(text);
|
|
230
107
|
});
|
|
231
|
-
|
|
232
|
-
this.log.error(
|
|
233
|
-
this.
|
|
234
|
-
this.pendingReject = null;
|
|
235
|
-
this.pendingResolve = null;
|
|
108
|
+
proc.on("error", (err) => {
|
|
109
|
+
this.log.error(`${this.name}: process error`, { meta: { error: err.message } });
|
|
110
|
+
this.markDead(`process error: ${err.message}`);
|
|
236
111
|
});
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
}
|
|
112
|
+
proc.stdin?.on("error", (err) => {
|
|
113
|
+
this.markDead(`stdin error: ${err.message}`);
|
|
114
|
+
});
|
|
115
|
+
proc.on("exit", (code, signal) => {
|
|
116
|
+
if (this.process === proc) this.process = null;
|
|
117
|
+
const reason = `process exited (code ${String(code)}, signal ${String(signal)})`;
|
|
118
|
+
if (!this.disposing) this.log.warn(`${this.name}: process exited unexpectedly — the engine is dead`, { meta: {
|
|
119
|
+
code,
|
|
120
|
+
signal,
|
|
121
|
+
pid: proc.pid ?? null
|
|
122
|
+
} });
|
|
123
|
+
this.markDead(reason);
|
|
244
124
|
});
|
|
245
|
-
|
|
125
|
+
proc.stdout?.on("data", (chunk) => {
|
|
246
126
|
this.receiveBuffer = Buffer.concat([this.receiveBuffer, chunk]);
|
|
247
127
|
this.tryReceive();
|
|
248
128
|
});
|
|
249
129
|
const ready = await this.receiveFrame();
|
|
250
|
-
if (ready.length !== 1 || ready[0] !== 1) throw new Error(
|
|
251
|
-
|
|
130
|
+
if (ready.length !== 1 || ready[0] !== 1) throw new Error(`${this.name}: unexpected ready frame`);
|
|
131
|
+
}
|
|
132
|
+
/** False once the process died or was disposed — a holder must rebuild. */
|
|
133
|
+
isAlive() {
|
|
134
|
+
return this.process !== null && this.dead === null;
|
|
252
135
|
}
|
|
253
|
-
/** Native pid of the Python subprocess — sampled by the pool memory
|
|
254
|
-
* watchdog (`/proc/<pid>/status`). */
|
|
255
136
|
getPid() {
|
|
256
137
|
return this.process?.pid ?? null;
|
|
257
138
|
}
|
|
258
|
-
/**
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
const meta = Buffer.allocUnsafe(1 + ndims * 4);
|
|
269
|
-
meta.writeUInt8(ndims, 0);
|
|
270
|
-
for (let i = 0; i < ndims; i++) meta.writeUInt32LE(inputShape[i], 1 + i * 4);
|
|
271
|
-
const dataBuf = Buffer.from(input.buffer, input.byteOffset, input.byteLength);
|
|
272
|
-
const payload = Buffer.concat([meta, dataBuf]);
|
|
273
|
-
const lenBuf = Buffer.allocUnsafe(4);
|
|
274
|
-
lenBuf.writeUInt32LE(payload.length, 0);
|
|
275
|
-
this.process.stdin.write(Buffer.concat([lenBuf, payload]));
|
|
276
|
-
const resp = await this.receiveFrame();
|
|
277
|
-
const floatStart = 1 + resp.readUInt8(0) * 4;
|
|
278
|
-
const count = (resp.length - floatStart) / 4;
|
|
279
|
-
const out = new Float32Array(count);
|
|
280
|
-
for (let i = 0; i < count; i++) out[i] = resp.readFloatLE(floatStart + i * 4);
|
|
281
|
-
return out;
|
|
139
|
+
/** Send one request frame and await its answer. Rejects AT ONCE when dead. */
|
|
140
|
+
async request(payload) {
|
|
141
|
+
if (this.dead !== null) throw new Error(`${this.name}: engine is dead — ${this.dead}`);
|
|
142
|
+
const stdin = this.process?.stdin;
|
|
143
|
+
if (!stdin) throw new Error(`${this.name}: not initialized — call initialize() first`);
|
|
144
|
+
const answer = this.receiveFrame();
|
|
145
|
+
const len = Buffer.allocUnsafe(4);
|
|
146
|
+
len.writeUInt32LE(payload.length, 0);
|
|
147
|
+
stdin.write(Buffer.concat([len, payload]));
|
|
148
|
+
return answer;
|
|
282
149
|
}
|
|
283
150
|
async dispose() {
|
|
284
151
|
const proc = this.process;
|
|
152
|
+
this.disposing = true;
|
|
153
|
+
this.markDead("disposed");
|
|
285
154
|
if (!proc) return;
|
|
286
155
|
this.process = null;
|
|
287
156
|
proc.stdin?.end();
|
|
@@ -299,185 +168,546 @@ var PythonRawTensorEngine = class {
|
|
|
299
168
|
});
|
|
300
169
|
});
|
|
301
170
|
}
|
|
171
|
+
markDead(reason) {
|
|
172
|
+
if (this.dead === null) this.dead = reason;
|
|
173
|
+
const err = /* @__PURE__ */ new Error(`${this.name}: ${reason}`);
|
|
174
|
+
for (const waiter of this.pending.splice(0)) waiter.reject(err);
|
|
175
|
+
}
|
|
302
176
|
receiveFrame() {
|
|
303
177
|
return new Promise((resolve, reject) => {
|
|
304
|
-
this.
|
|
305
|
-
|
|
178
|
+
this.pending.push({
|
|
179
|
+
resolve,
|
|
180
|
+
reject
|
|
181
|
+
});
|
|
306
182
|
});
|
|
307
183
|
}
|
|
308
184
|
tryReceive() {
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
this.pendingReject = null;
|
|
317
|
-
resolve?.(payload);
|
|
185
|
+
while (this.receiveBuffer.length >= 4) {
|
|
186
|
+
const length = this.receiveBuffer.readUInt32LE(0);
|
|
187
|
+
if (this.receiveBuffer.length < 4 + length) return;
|
|
188
|
+
const payload = Buffer.from(this.receiveBuffer.subarray(4, 4 + length));
|
|
189
|
+
this.receiveBuffer = this.receiveBuffer.subarray(4 + length);
|
|
190
|
+
this.pending.shift()?.resolve(payload);
|
|
191
|
+
}
|
|
318
192
|
}
|
|
319
193
|
};
|
|
320
194
|
//#endregion
|
|
195
|
+
//#region src/embedding-encoder/shared/noop-logger.ts
|
|
196
|
+
var noop = () => {};
|
|
197
|
+
function createNoopLogger() {
|
|
198
|
+
const logger = {
|
|
199
|
+
debug: noop,
|
|
200
|
+
info: noop,
|
|
201
|
+
warn: noop,
|
|
202
|
+
error: noop,
|
|
203
|
+
child: () => logger,
|
|
204
|
+
withTags: (_tags) => logger
|
|
205
|
+
};
|
|
206
|
+
return logger;
|
|
207
|
+
}
|
|
208
|
+
//#endregion
|
|
321
209
|
//#region src/embedding-encoder/shared/python-text-encoder-engine.ts
|
|
322
|
-
/**
|
|
323
|
-
* CLIP text-encoder engine backed by an embedded-Python subprocess
|
|
324
|
-
* (`text_encoder_inference.py`). Tokenization happens IN Python via the HF
|
|
325
|
-
* `tokenizers` Rust BPE (exact by construction) — this replaces the former
|
|
326
|
-
* hand-rolled TypeScript CLIP BPE.
|
|
327
|
-
*
|
|
328
|
-
* The caller sends raw UTF-8 text; Python tokenizes (truncate/pad to 77), runs
|
|
329
|
-
* onnxruntime, and returns the embedding tensor. Wire protocol = length-prefixed
|
|
330
|
-
* binary frames ([4B LE length][payload]):
|
|
331
|
-
* ready (in): [0x01]
|
|
332
|
-
* request (out): UTF-8 text bytes
|
|
333
|
-
* response (in): [1B ndims][dims × 4B LE uint32][float32 LE data]
|
|
334
|
-
*/
|
|
335
210
|
var PythonTextEncoderEngine = class {
|
|
336
|
-
pythonPath;
|
|
337
211
|
scriptPath;
|
|
338
212
|
modelPath;
|
|
339
213
|
tokenizerPath;
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
pendingResolve = null;
|
|
343
|
-
pendingReject = null;
|
|
214
|
+
window;
|
|
215
|
+
proc;
|
|
344
216
|
log;
|
|
345
|
-
|
|
346
|
-
|
|
217
|
+
requestCount = 0;
|
|
218
|
+
constructor(pythonPath, scriptPath, modelPath, tokenizerPath, window, logger) {
|
|
347
219
|
this.scriptPath = scriptPath;
|
|
348
220
|
this.modelPath = modelPath;
|
|
349
221
|
this.tokenizerPath = tokenizerPath;
|
|
222
|
+
this.window = window;
|
|
350
223
|
this.log = logger ?? createNoopLogger();
|
|
224
|
+
this.proc = new FramedPythonProcess("PythonTextEncoderEngine", pythonPath, this.spawnArgs(), this.log);
|
|
351
225
|
}
|
|
352
226
|
async initialize() {
|
|
353
|
-
this.
|
|
354
|
-
this.scriptPath,
|
|
355
|
-
this.modelPath,
|
|
356
|
-
this.tokenizerPath
|
|
357
|
-
], { stdio: [
|
|
358
|
-
"pipe",
|
|
359
|
-
"pipe",
|
|
360
|
-
"pipe"
|
|
361
|
-
] });
|
|
362
|
-
this.process.stderr?.on("data", (chunk) => {
|
|
363
|
-
const text = chunk.toString().trim();
|
|
364
|
-
if (text) this.log.warn(text);
|
|
365
|
-
});
|
|
366
|
-
this.process.on("error", (err) => {
|
|
367
|
-
this.log.error("Python text-encoder process error", { meta: { error: err.message } });
|
|
368
|
-
this.pendingReject?.(err);
|
|
369
|
-
this.pendingReject = null;
|
|
370
|
-
this.pendingResolve = null;
|
|
371
|
-
});
|
|
372
|
-
this.process.on("exit", (code) => {
|
|
373
|
-
if (code !== 0 && code !== null) {
|
|
374
|
-
const err = /* @__PURE__ */ new Error(`PythonTextEncoderEngine: process exited with code ${code}`);
|
|
375
|
-
this.pendingReject?.(err);
|
|
376
|
-
this.pendingReject = null;
|
|
377
|
-
this.pendingResolve = null;
|
|
378
|
-
}
|
|
379
|
-
});
|
|
380
|
-
this.process.stdout.on("data", (chunk) => {
|
|
381
|
-
this.receiveBuffer = Buffer.concat([this.receiveBuffer, chunk]);
|
|
382
|
-
this.tryReceive();
|
|
383
|
-
});
|
|
384
|
-
const ready = await this.receiveFrame();
|
|
385
|
-
if (ready.length !== 1 || ready[0] !== 1) throw new Error("PythonTextEncoderEngine: unexpected ready frame");
|
|
227
|
+
await this.proc.start();
|
|
386
228
|
this.log.info("CLIP text-encoder engine ready (embedded Python)", { meta: {
|
|
387
229
|
modelPath: this.modelPath,
|
|
388
|
-
tokenizerPath: this.tokenizerPath
|
|
230
|
+
tokenizerPath: this.tokenizerPath,
|
|
231
|
+
contextLength: this.window.contextLength,
|
|
232
|
+
padId: this.window.padId
|
|
389
233
|
} });
|
|
390
234
|
}
|
|
235
|
+
/** The subprocess argv after the interpreter — the window rides on it. */
|
|
236
|
+
spawnArgs() {
|
|
237
|
+
return [
|
|
238
|
+
this.scriptPath,
|
|
239
|
+
this.modelPath,
|
|
240
|
+
this.tokenizerPath,
|
|
241
|
+
"--context-length",
|
|
242
|
+
String(this.window.contextLength),
|
|
243
|
+
"--pad-id",
|
|
244
|
+
String(this.window.padId)
|
|
245
|
+
];
|
|
246
|
+
}
|
|
247
|
+
/** False once the process died (crash, OOM kill) or was disposed. */
|
|
248
|
+
isAlive() {
|
|
249
|
+
return this.proc.isAlive();
|
|
250
|
+
}
|
|
391
251
|
/** Native pid of the Python subprocess — sampled by the pool memory
|
|
392
252
|
* watchdog (`/proc/<pid>/status`). */
|
|
393
253
|
getPid() {
|
|
394
|
-
return this.
|
|
254
|
+
return this.proc.getPid();
|
|
395
255
|
}
|
|
396
256
|
/** Encode requests served since spawn — the denominator that separates
|
|
397
257
|
* "RSS grew per unit of work" (leak) from "RSS grew because it worked". */
|
|
398
258
|
getRequestCount() {
|
|
399
259
|
return this.requestCount;
|
|
400
260
|
}
|
|
401
|
-
requestCount = 0;
|
|
402
261
|
/** Tokenize + encode `text` into the model embedding (float32). */
|
|
403
262
|
async encode(text) {
|
|
404
|
-
if (!this.process?.stdin) throw new Error("PythonTextEncoderEngine: not initialized — call initialize() first");
|
|
405
263
|
this.requestCount++;
|
|
406
|
-
|
|
407
|
-
const lenBuf = Buffer.allocUnsafe(4);
|
|
408
|
-
lenBuf.writeUInt32LE(payload.length, 0);
|
|
409
|
-
this.process.stdin.write(Buffer.concat([lenBuf, payload]));
|
|
410
|
-
const resp = await this.receiveFrame();
|
|
411
|
-
const floatStart = 1 + resp.readUInt8(0) * 4;
|
|
412
|
-
const count = (resp.length - floatStart) / 4;
|
|
413
|
-
const out = new Float32Array(count);
|
|
414
|
-
for (let i = 0; i < count; i++) out[i] = resp.readFloatLE(floatStart + i * 4);
|
|
415
|
-
return out;
|
|
264
|
+
return decodeTensorFrame(await this.proc.request(Buffer.from(text, "utf-8")));
|
|
416
265
|
}
|
|
417
266
|
async dispose() {
|
|
418
|
-
|
|
419
|
-
if (!proc) return;
|
|
420
|
-
this.process = null;
|
|
421
|
-
proc.stdin?.end();
|
|
422
|
-
proc.kill("SIGTERM");
|
|
423
|
-
await new Promise((resolve) => {
|
|
424
|
-
const timer = setTimeout(() => {
|
|
425
|
-
try {
|
|
426
|
-
proc.kill("SIGKILL");
|
|
427
|
-
} catch {}
|
|
428
|
-
resolve();
|
|
429
|
-
}, 5e3);
|
|
430
|
-
proc.once("exit", () => {
|
|
431
|
-
clearTimeout(timer);
|
|
432
|
-
resolve();
|
|
433
|
-
});
|
|
434
|
-
});
|
|
267
|
+
await this.proc.dispose();
|
|
435
268
|
}
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
269
|
+
};
|
|
270
|
+
/** Decode `[1B ndims][dims × 4B][float32 data]`. */
|
|
271
|
+
function decodeTensorFrame(resp) {
|
|
272
|
+
const floatStart = 1 + resp.readUInt8(0) * 4;
|
|
273
|
+
const count = (resp.length - floatStart) / 4;
|
|
274
|
+
const out = new Float32Array(count);
|
|
275
|
+
for (let i = 0; i < count; i++) out[i] = resp.readFloatLE(floatStart + i * 4);
|
|
276
|
+
return out;
|
|
277
|
+
}
|
|
278
|
+
//#endregion
|
|
279
|
+
//#region src/embedding-encoder/shared/python-raw-tensor-engine.ts
|
|
280
|
+
/**
|
|
281
|
+
* Raw-tensor ONNX engine backed by an embedded-Python subprocess
|
|
282
|
+
* (`raw_tensor_inference.py`). Replaces the Node `onnxruntime-node` raw-tensor
|
|
283
|
+
* engine so the platform ships no Node ONNX runtime. The caller preprocesses to
|
|
284
|
+
* a Float32Array; this engine ships it to Python, which runs onnxruntime and
|
|
285
|
+
* returns the output tensor. Wire protocol = length-prefixed binary frames
|
|
286
|
+
* ([4B LE length][payload]); the process lifecycle is `FramedPythonProcess`.
|
|
287
|
+
*/
|
|
288
|
+
var PythonRawTensorEngine = class {
|
|
289
|
+
modelPath;
|
|
290
|
+
runtime = "onnx";
|
|
291
|
+
device = "cpu";
|
|
292
|
+
proc;
|
|
293
|
+
log;
|
|
294
|
+
requestCount = 0;
|
|
295
|
+
constructor(pythonPath, scriptPath, modelPath, logger) {
|
|
296
|
+
this.modelPath = modelPath;
|
|
297
|
+
this.log = logger ?? createNoopLogger();
|
|
298
|
+
this.proc = new FramedPythonProcess("PythonRawTensorEngine", pythonPath, [scriptPath, modelPath], this.log);
|
|
441
299
|
}
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
300
|
+
async initialize() {
|
|
301
|
+
await this.proc.start();
|
|
302
|
+
this.log.info("ONNX raw-tensor engine ready (embedded Python)", { meta: { modelPath: this.modelPath } });
|
|
303
|
+
}
|
|
304
|
+
/** False once the process died (crash, OOM kill) or was disposed. */
|
|
305
|
+
isAlive() {
|
|
306
|
+
return this.proc.isAlive();
|
|
307
|
+
}
|
|
308
|
+
/** Native pid of the Python subprocess — sampled by the pool memory
|
|
309
|
+
* watchdog (`/proc/<pid>/status`). */
|
|
310
|
+
getPid() {
|
|
311
|
+
return this.proc.getPid();
|
|
312
|
+
}
|
|
313
|
+
/** Inference requests served since spawn — the denominator that separates
|
|
314
|
+
* "RSS grew per unit of work" (leak) from "RSS grew because it worked". */
|
|
315
|
+
getRequestCount() {
|
|
316
|
+
return this.requestCount;
|
|
317
|
+
}
|
|
318
|
+
async run(input, inputShape) {
|
|
319
|
+
this.requestCount++;
|
|
320
|
+
const ndims = inputShape.length;
|
|
321
|
+
const meta = Buffer.allocUnsafe(1 + ndims * 4);
|
|
322
|
+
meta.writeUInt8(ndims, 0);
|
|
323
|
+
for (let i = 0; i < ndims; i++) meta.writeUInt32LE(inputShape[i], 1 + i * 4);
|
|
324
|
+
const dataBuf = Buffer.from(input.buffer, input.byteOffset, input.byteLength);
|
|
325
|
+
return decodeTensorFrame(await this.proc.request(Buffer.concat([meta, dataBuf])));
|
|
326
|
+
}
|
|
327
|
+
async dispose() {
|
|
328
|
+
await this.proc.dispose();
|
|
329
|
+
}
|
|
330
|
+
};
|
|
331
|
+
var ActiveClipModel = class {
|
|
332
|
+
deps;
|
|
333
|
+
cached = null;
|
|
334
|
+
lastRead = null;
|
|
335
|
+
constructor(deps) {
|
|
336
|
+
this.deps = deps;
|
|
337
|
+
}
|
|
338
|
+
/**
|
|
339
|
+
* The last row a read RETURNED, without reading — `null` while no read has
|
|
340
|
+
* ever succeeded (the config fallback is a guess, not a known model). The
|
|
341
|
+
* idle sweep exempts this model's text tower (`clip-text-engines.ts`).
|
|
342
|
+
*/
|
|
343
|
+
known() {
|
|
344
|
+
return this.lastRead;
|
|
345
|
+
}
|
|
346
|
+
async get() {
|
|
347
|
+
const now = this.deps.now();
|
|
348
|
+
if (this.cached !== null && now - this.cached.at < 3e4) return this.cached.modelId;
|
|
349
|
+
const row = await this.deps.readRow();
|
|
350
|
+
if (row === null) {
|
|
351
|
+
const kept = this.lastRead ?? this.deps.fallbackModelId();
|
|
352
|
+
this.deps.logger.warn("CLIP encoder kept its model — the cluster row could not be read", { meta: {
|
|
353
|
+
modelId: kept,
|
|
354
|
+
source: this.lastRead !== null ? "last-read" : "addon-config"
|
|
355
|
+
} });
|
|
356
|
+
return kept;
|
|
357
|
+
}
|
|
358
|
+
const previous = this.lastRead;
|
|
359
|
+
if (row !== previous) this.deps.logger.info("CLIP encoder follows the cluster model", { meta: {
|
|
360
|
+
modelId: row,
|
|
361
|
+
previous
|
|
362
|
+
} });
|
|
363
|
+
this.lastRead = row;
|
|
364
|
+
this.cached = {
|
|
365
|
+
modelId: row,
|
|
366
|
+
at: now
|
|
367
|
+
};
|
|
368
|
+
if (previous !== null && row !== previous) this.deps.onChange?.(row, previous);
|
|
369
|
+
return row;
|
|
370
|
+
}
|
|
371
|
+
};
|
|
372
|
+
var CrashBudget = class {
|
|
373
|
+
deps;
|
|
374
|
+
recent = /* @__PURE__ */ new Map();
|
|
375
|
+
failedModels = /* @__PURE__ */ new Map();
|
|
376
|
+
constructor(deps) {
|
|
377
|
+
this.deps = deps;
|
|
378
|
+
}
|
|
379
|
+
/** Throw the named refusal when the model is in `failed`. Spawns nothing. */
|
|
380
|
+
check(modelId) {
|
|
381
|
+
const failed = this.failedModels.get(modelId);
|
|
382
|
+
if (failed === void 0) return;
|
|
383
|
+
throw require_clip_model_registry.embeddingModelFailedError(failed);
|
|
384
|
+
}
|
|
385
|
+
/** Record one crash or load failure. Returns true when it made the model `failed`. */
|
|
386
|
+
record(modelId, reason) {
|
|
387
|
+
if (this.failedModels.has(modelId)) return false;
|
|
388
|
+
const now = this.now();
|
|
389
|
+
const windowMs = this.deps.windowMs ?? 6e5;
|
|
390
|
+
const times = (this.recent.get(modelId) ?? []).filter((t) => now - t < windowMs);
|
|
391
|
+
times.push(now);
|
|
392
|
+
this.recent.set(modelId, times);
|
|
393
|
+
if (times.length < (this.deps.maxCrashes ?? 3)) return false;
|
|
394
|
+
const failed = {
|
|
395
|
+
modelId,
|
|
396
|
+
tower: this.deps.tower,
|
|
397
|
+
nodeId: this.deps.nodeId(),
|
|
398
|
+
crashes: times.length,
|
|
399
|
+
windowMs,
|
|
400
|
+
sinceMs: now,
|
|
401
|
+
lastReason: reason
|
|
402
|
+
};
|
|
403
|
+
this.failedModels.set(modelId, failed);
|
|
404
|
+
this.recent.delete(modelId);
|
|
405
|
+
this.deps.logger.error("CLIP tower FAILED — crash budget exhausted, requests refused until reset", { meta: { ...failed } });
|
|
406
|
+
return true;
|
|
407
|
+
}
|
|
408
|
+
failed() {
|
|
409
|
+
return [...this.failedModels.values()];
|
|
410
|
+
}
|
|
411
|
+
/** Clear one model (or all). Returns the ids that were failed. */
|
|
412
|
+
reset(modelId, why = "operator") {
|
|
413
|
+
const ids = modelId !== void 0 ? [modelId] : [...this.failedModels.keys()];
|
|
414
|
+
const cleared = ids.filter((id) => this.failedModels.delete(id));
|
|
415
|
+
for (const id of ids) this.recent.delete(id);
|
|
416
|
+
if (cleared.length > 0) this.deps.logger.info("CLIP tower failed state cleared", { meta: {
|
|
417
|
+
tower: this.deps.tower,
|
|
418
|
+
nodeId: this.deps.nodeId(),
|
|
419
|
+
cleared,
|
|
420
|
+
why
|
|
421
|
+
} });
|
|
422
|
+
return cleared;
|
|
423
|
+
}
|
|
424
|
+
now() {
|
|
425
|
+
return (this.deps.now ?? Date.now)();
|
|
452
426
|
}
|
|
453
427
|
};
|
|
454
428
|
//#endregion
|
|
455
|
-
//#region src/embedding-encoder/addon/
|
|
429
|
+
//#region src/embedding-encoder/addon/failed-models-report.ts
|
|
456
430
|
/**
|
|
457
|
-
*
|
|
458
|
-
*
|
|
459
|
-
*
|
|
431
|
+
* Every failed tower on this node. Each row already names the node: the budget
|
|
432
|
+
* stamps it when the model goes `failed` (it is in the refusal too), and this
|
|
433
|
+
* report does not re-derive it — one authority for the fact.
|
|
460
434
|
*/
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
435
|
+
function reportFailedModels(budgets) {
|
|
436
|
+
return [...budgets.image.failed(), ...budgets.text.failed()];
|
|
437
|
+
}
|
|
438
|
+
/** Every model whose requirements are terminally unmeetable on this node; rows stamped by the gate. */
|
|
439
|
+
function reportFailedRequirements(requirements) {
|
|
440
|
+
return [...requirements.image.failed(), ...requirements.text.failed()];
|
|
441
|
+
}
|
|
442
|
+
/** The operator's clear (one model, or all) of BOTH terminal states on this node — the result names it. */
|
|
443
|
+
function resetFailedModelsOn(nodeId, towers, modelId) {
|
|
444
|
+
const cleared = [
|
|
445
|
+
...towers.budgets.image.reset(modelId),
|
|
446
|
+
...towers.budgets.text.reset(modelId),
|
|
447
|
+
...towers.requirements.image.reset(modelId),
|
|
448
|
+
...towers.requirements.text.reset(modelId)
|
|
449
|
+
];
|
|
450
|
+
return {
|
|
451
|
+
cleared: [...new Set(cleared)],
|
|
452
|
+
nodeId
|
|
453
|
+
};
|
|
454
|
+
}
|
|
455
|
+
/**
|
|
456
|
+
* The cluster row now names `modelId`: a fresh start for THAT model's towers
|
|
457
|
+
* only (D6's second exit from `failed`). The model the row left keeps its
|
|
458
|
+
* state — a rollback to a model that crashed three times must not buy three
|
|
459
|
+
* more spawns nobody asked for, and a search that still names it
|
|
460
|
+
* (`searchObjectEvents({ modelId })`) is refused with the reset hint, as
|
|
461
|
+
* before the flip.
|
|
462
|
+
*/
|
|
463
|
+
function clearFailedOnModelChange(towers, modelId) {
|
|
464
|
+
return [...new Set([
|
|
465
|
+
...towers.budgets.image.reset(modelId, "model-change"),
|
|
466
|
+
...towers.budgets.text.reset(modelId, "model-change"),
|
|
467
|
+
...towers.requirements.image.reset(modelId, "model-change"),
|
|
468
|
+
...towers.requirements.text.reset(modelId, "model-change")
|
|
469
|
+
])];
|
|
470
|
+
}
|
|
471
|
+
//#endregion
|
|
472
|
+
//#region src/embedding-encoder/addon/model-engine-slot.ts
|
|
473
|
+
var ModelEngineSlot = class {
|
|
474
|
+
build;
|
|
475
|
+
budget;
|
|
476
|
+
requirements;
|
|
477
|
+
onDead;
|
|
478
|
+
current = null;
|
|
479
|
+
inflight = /* @__PURE__ */ new Map();
|
|
480
|
+
/** The last model asked for — a build for any other model is stale on arrival. */
|
|
481
|
+
wanted = null;
|
|
482
|
+
/** Serialises switches so two models never tear the slot down concurrently. */
|
|
483
|
+
switching = Promise.resolve();
|
|
484
|
+
/**
|
|
485
|
+
* Bumped by {@link dispose}: a build that started before it is stale when it
|
|
486
|
+
* lands and is disposed on arrival — shutdown leaves no process behind.
|
|
487
|
+
*/
|
|
488
|
+
generation = 0;
|
|
489
|
+
constructor(build, budget, requirements, onDead = () => {}) {
|
|
490
|
+
this.build = build;
|
|
491
|
+
this.budget = budget;
|
|
492
|
+
this.requirements = requirements;
|
|
493
|
+
this.onDead = onDead;
|
|
494
|
+
}
|
|
495
|
+
held() {
|
|
496
|
+
return this.current;
|
|
497
|
+
}
|
|
498
|
+
async ensure(modelId) {
|
|
499
|
+
this.wanted = modelId;
|
|
500
|
+
this.dropIfDead();
|
|
501
|
+
if (this.current?.modelId === modelId) return this.current.engine;
|
|
502
|
+
this.budget.check(modelId);
|
|
503
|
+
this.requirements.check(modelId);
|
|
504
|
+
const pending = this.inflight.get(modelId);
|
|
505
|
+
if (pending !== void 0) return pending;
|
|
506
|
+
const generation = this.generation;
|
|
507
|
+
const guarded = this.switching.then(async () => {
|
|
508
|
+
this.dropIfDead();
|
|
509
|
+
if (this.current?.modelId === modelId) return this.current.engine;
|
|
510
|
+
this.budget.check(modelId);
|
|
511
|
+
this.requirements.check(modelId);
|
|
512
|
+
await this.disposeCurrent();
|
|
513
|
+
let engine;
|
|
514
|
+
try {
|
|
515
|
+
engine = await this.build(modelId);
|
|
516
|
+
} catch (err) {
|
|
517
|
+
throw this.requirements.refuse(modelId, err);
|
|
518
|
+
}
|
|
519
|
+
this.requirements.satisfied(modelId);
|
|
520
|
+
try {
|
|
521
|
+
await engine.initialize();
|
|
522
|
+
} catch (err) {
|
|
523
|
+
this.budget.record(modelId, `load failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
524
|
+
throw err;
|
|
525
|
+
}
|
|
526
|
+
if (this.generation !== generation) {
|
|
527
|
+
await engine.dispose();
|
|
528
|
+
throw new Error(`engine for "${modelId}" cancelled: the slot was disposed while building`);
|
|
529
|
+
}
|
|
530
|
+
if (this.wanted !== modelId) {
|
|
531
|
+
await engine.dispose();
|
|
532
|
+
throw new Error(`engine for "${modelId}" superseded while building`);
|
|
533
|
+
}
|
|
534
|
+
this.current = {
|
|
535
|
+
modelId,
|
|
536
|
+
engine
|
|
537
|
+
};
|
|
538
|
+
return engine;
|
|
539
|
+
}).finally(() => {
|
|
540
|
+
this.inflight.delete(modelId);
|
|
541
|
+
});
|
|
542
|
+
this.inflight.set(modelId, guarded);
|
|
543
|
+
this.switching = guarded.then(() => void 0, () => void 0);
|
|
544
|
+
return guarded;
|
|
545
|
+
}
|
|
546
|
+
/**
|
|
547
|
+
* Dispose the held engine AND cancel every build in flight: each is disposed
|
|
548
|
+
* when it lands (see `generation`). Resolves once in-flight builds settled.
|
|
549
|
+
*/
|
|
550
|
+
async dispose() {
|
|
551
|
+
this.generation += 1;
|
|
552
|
+
await this.disposeCurrent();
|
|
553
|
+
await this.switching;
|
|
554
|
+
}
|
|
555
|
+
async disposeCurrent() {
|
|
556
|
+
const held = this.current;
|
|
557
|
+
this.current = null;
|
|
558
|
+
await held?.engine.dispose();
|
|
559
|
+
}
|
|
560
|
+
dropIfDead() {
|
|
561
|
+
const held = this.current;
|
|
562
|
+
if (held === null || held.engine.isAlive()) return;
|
|
563
|
+
this.current = null;
|
|
564
|
+
this.onDead(held.modelId);
|
|
565
|
+
this.budget.record(held.modelId, "process died after loading");
|
|
566
|
+
held.engine.dispose().catch(() => {});
|
|
567
|
+
}
|
|
568
|
+
};
|
|
569
|
+
/**
|
|
570
|
+
* Thrown by the requirements stage when a retry can never help: the catalog
|
|
571
|
+
* does not know the id, the server said the file does not exist (404/410), a
|
|
572
|
+
* declared sibling is missing. Same process as the gate, so the class is the
|
|
573
|
+
* marker; the gate never matches wording.
|
|
574
|
+
*/
|
|
575
|
+
var PermanentRequirementsFailure = class extends Error {
|
|
576
|
+
constructor(message) {
|
|
577
|
+
super(message);
|
|
578
|
+
this.name = "PermanentRequirementsFailure";
|
|
475
579
|
}
|
|
476
580
|
};
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
return CLIP_MODEL_META[modelId] ?? CLIP_MODEL_META["mobileclip-s1"];
|
|
581
|
+
function isPermanentRequirementsFailure(err) {
|
|
582
|
+
return err instanceof PermanentRequirementsFailure;
|
|
480
583
|
}
|
|
584
|
+
var RequirementsGate = class {
|
|
585
|
+
deps;
|
|
586
|
+
unavailable = /* @__PURE__ */ new Map();
|
|
587
|
+
constructor(deps) {
|
|
588
|
+
this.deps = deps;
|
|
589
|
+
}
|
|
590
|
+
/**
|
|
591
|
+
* Consulted BEFORE a build. Throws the named refusal — and downloads nothing
|
|
592
|
+
* — while the model is terminal or inside its cooldown. Returns otherwise.
|
|
593
|
+
*/
|
|
594
|
+
check(modelId) {
|
|
595
|
+
const state = this.unavailable.get(modelId);
|
|
596
|
+
if (state === void 0) return;
|
|
597
|
+
if (state.terminal !== null) {
|
|
598
|
+
this.unavailable.set(modelId, {
|
|
599
|
+
...state,
|
|
600
|
+
refused: state.refused + 1
|
|
601
|
+
});
|
|
602
|
+
throw require_clip_model_registry.embeddingRequirementsFailedError(state.terminal);
|
|
603
|
+
}
|
|
604
|
+
const now = this.now();
|
|
605
|
+
if (now >= state.cooldownUntilMs) return;
|
|
606
|
+
this.unavailable.set(modelId, {
|
|
607
|
+
...state,
|
|
608
|
+
refused: state.refused + 1
|
|
609
|
+
});
|
|
610
|
+
throw require_clip_model_registry.embeddingRequirementsUnavailableError(this.on(modelId), `retry in ${String(Math.ceil((state.cooldownUntilMs - now) / 1e3))}s — ${state.reason}`);
|
|
611
|
+
}
|
|
612
|
+
/**
|
|
613
|
+
* A build failed before any process existed. Starts (or doubles) the
|
|
614
|
+
* cooldown; a PERMANENT failure past the bound makes the model terminal.
|
|
615
|
+
* Logs at the transitions only and returns the named refusal to throw.
|
|
616
|
+
*/
|
|
617
|
+
refuse(modelId, err) {
|
|
618
|
+
const reason = err instanceof Error ? err.message : String(err);
|
|
619
|
+
const now = this.now();
|
|
620
|
+
const prior = this.unavailable.get(modelId);
|
|
621
|
+
const attempts = (prior?.attempts ?? 0) + 1;
|
|
622
|
+
const permanent = isPermanentRequirementsFailure(err);
|
|
623
|
+
const permanentAttempts = (prior?.permanentAttempts ?? 0) + (permanent ? 1 : 0);
|
|
624
|
+
const cooldownMs = Math.min(this.deps.cooldownMaxMs ?? 3e5, (this.deps.cooldownBaseMs ?? 5e3) * 2 ** (attempts - 1));
|
|
625
|
+
const base = {
|
|
626
|
+
sinceMs: prior?.sinceMs ?? now,
|
|
627
|
+
reason,
|
|
628
|
+
refused: (prior?.refused ?? 0) + 1,
|
|
629
|
+
attempts,
|
|
630
|
+
permanentAttempts,
|
|
631
|
+
cooldownUntilMs: now + cooldownMs,
|
|
632
|
+
terminal: null
|
|
633
|
+
};
|
|
634
|
+
if (prior === void 0) this.deps.logger.warn("CLIP model requirements unavailable — requests refused until they are; the crash budget is untouched", { meta: {
|
|
635
|
+
modelId,
|
|
636
|
+
tower: this.deps.tower,
|
|
637
|
+
nodeId: this.deps.nodeId(),
|
|
638
|
+
reason,
|
|
639
|
+
cooldownMs
|
|
640
|
+
} });
|
|
641
|
+
if (permanent && permanentAttempts >= (this.deps.maxAttempts ?? 3)) {
|
|
642
|
+
const terminal = {
|
|
643
|
+
...this.on(modelId),
|
|
644
|
+
attempts: permanentAttempts,
|
|
645
|
+
sinceMs: now,
|
|
646
|
+
lastReason: reason
|
|
647
|
+
};
|
|
648
|
+
this.unavailable.set(modelId, {
|
|
649
|
+
...base,
|
|
650
|
+
terminal
|
|
651
|
+
});
|
|
652
|
+
this.deps.logger.error("CLIP model requirements can never be met — terminally failed, requests refused until reset", { meta: {
|
|
653
|
+
...terminal,
|
|
654
|
+
refusedSoFar: base.refused
|
|
655
|
+
} });
|
|
656
|
+
return require_clip_model_registry.embeddingRequirementsFailedError(terminal);
|
|
657
|
+
}
|
|
658
|
+
this.unavailable.set(modelId, base);
|
|
659
|
+
return require_clip_model_registry.embeddingRequirementsUnavailableError(this.on(modelId), reason);
|
|
660
|
+
}
|
|
661
|
+
/** A build got its engine: the requirements are met again. Logs the recovery once. */
|
|
662
|
+
satisfied(modelId) {
|
|
663
|
+
const prior = this.unavailable.get(modelId);
|
|
664
|
+
if (prior === void 0) return;
|
|
665
|
+
this.unavailable.delete(modelId);
|
|
666
|
+
this.deps.logger.info("CLIP model requirements available again", { meta: {
|
|
667
|
+
modelId,
|
|
668
|
+
tower: this.deps.tower,
|
|
669
|
+
nodeId: this.deps.nodeId(),
|
|
670
|
+
unavailableMs: this.now() - prior.sinceMs,
|
|
671
|
+
refused: prior.refused,
|
|
672
|
+
attempts: prior.attempts,
|
|
673
|
+
lastReason: prior.reason
|
|
674
|
+
} });
|
|
675
|
+
}
|
|
676
|
+
/** Models terminally `requirements-failed` on this node, as `getInfo` reports them. */
|
|
677
|
+
failed() {
|
|
678
|
+
return [...this.unavailable.values()].map((s) => s.terminal).filter((t) => t !== null);
|
|
679
|
+
}
|
|
680
|
+
/**
|
|
681
|
+
* Clear one model (or all): terminal state AND cooldown. Returns the ids
|
|
682
|
+
* that were TERMINAL — a cooldown is not a state an operator is told about.
|
|
683
|
+
*/
|
|
684
|
+
reset(modelId, why = "operator") {
|
|
685
|
+
const ids = modelId !== void 0 ? [modelId] : [...this.unavailable.keys()];
|
|
686
|
+
const cleared = ids.filter((id) => this.unavailable.get(id)?.terminal !== null && this.unavailable.has(id));
|
|
687
|
+
for (const id of ids) this.unavailable.delete(id);
|
|
688
|
+
if (cleared.length > 0) this.deps.logger.info("CLIP model requirements-failed state cleared", { meta: {
|
|
689
|
+
tower: this.deps.tower,
|
|
690
|
+
nodeId: this.deps.nodeId(),
|
|
691
|
+
cleared,
|
|
692
|
+
why
|
|
693
|
+
} });
|
|
694
|
+
return cleared;
|
|
695
|
+
}
|
|
696
|
+
/** Models currently refused (cooldown or terminal) — exposed for tests. */
|
|
697
|
+
pending() {
|
|
698
|
+
return [...this.unavailable.keys()];
|
|
699
|
+
}
|
|
700
|
+
on(modelId) {
|
|
701
|
+
return {
|
|
702
|
+
modelId,
|
|
703
|
+
tower: this.deps.tower,
|
|
704
|
+
nodeId: this.deps.nodeId()
|
|
705
|
+
};
|
|
706
|
+
}
|
|
707
|
+
now() {
|
|
708
|
+
return (this.deps.now ?? Date.now)();
|
|
709
|
+
}
|
|
710
|
+
};
|
|
481
711
|
//#endregion
|
|
482
712
|
//#region src/embedding-encoder/addon/clip-preprocessing.ts
|
|
483
713
|
var CLIP_MEAN = [
|
|
@@ -520,11 +750,286 @@ function l2Normalize(vec) {
|
|
|
520
750
|
if (norm > 0) for (let i = 0; i < vec.length; i++) vec[i] /= norm;
|
|
521
751
|
return vec;
|
|
522
752
|
}
|
|
753
|
+
/**
|
|
754
|
+
* A tower unused this long is disposed ({@link ClipTextEngines.evictIdle}) —
|
|
755
|
+
* EXCEPT the tower of the model the cluster row names, which stays loaded for
|
|
756
|
+
* the life of the process exactly as it did before D649: the steady-state
|
|
757
|
+
* search must not pay a Python spawn plus a model load after a quiet quarter
|
|
758
|
+
* hour. Only the OTHER towers (SigLIP2 after a comparison, the old model after
|
|
759
|
+
* a flip) are let go when idle. When the active model is unknown (the row was
|
|
760
|
+
* never read), nothing is evicted — the side that keeps memory as before.
|
|
761
|
+
* The encoder is `placement: any-node` and cannot tell locally whether it is
|
|
762
|
+
* the node that answers text queries, so towers are loaded ON DEMAND — never
|
|
763
|
+
* prefetched on a model flip (565 MB on every node). The first query after a
|
|
764
|
+
* flip pays the load; if it fails, the search fails loudly rather than reading
|
|
765
|
+
* as "no results".
|
|
766
|
+
*/
|
|
767
|
+
var TEXT_ENGINE_IDLE_MS = 15 * 6e4;
|
|
768
|
+
var ClipTextEngines = class {
|
|
769
|
+
deps;
|
|
770
|
+
/** Insertion order is recency: a hit is re-inserted at the end. */
|
|
771
|
+
engines = /* @__PURE__ */ new Map();
|
|
772
|
+
/** Single-flight per model: two concurrent first queries spawn one engine. */
|
|
773
|
+
inflight = /* @__PURE__ */ new Map();
|
|
774
|
+
/**
|
|
775
|
+
* Encodes in flight per model. An engine with any is NEVER evicted: disposing
|
|
776
|
+
* it would fail a search that is already running. When every held engine is
|
|
777
|
+
* busy the set goes over `maxEngines` temporarily and shrinks when one drains.
|
|
778
|
+
*/
|
|
779
|
+
busy = /* @__PURE__ */ new Map();
|
|
780
|
+
/** When each held tower last finished an encode (or was built). */
|
|
781
|
+
lastUsed = /* @__PURE__ */ new Map();
|
|
782
|
+
/** Bumped by {@link disposeAll}: a build landing after it is disposed on arrival. */
|
|
783
|
+
generation = 0;
|
|
784
|
+
constructor(deps) {
|
|
785
|
+
this.deps = deps;
|
|
786
|
+
}
|
|
787
|
+
/** Resolve a model id to its CLIP model, or throw naming the id. */
|
|
788
|
+
requireModel(modelId) {
|
|
789
|
+
const model = this.deps.registry.resolve(modelId);
|
|
790
|
+
if (model === null) throw new Error(`EmbeddingEncoder: "${modelId}" has no CLIP metadata — not a CLIP model`);
|
|
791
|
+
return model;
|
|
792
|
+
}
|
|
793
|
+
async encode(modelId, text) {
|
|
794
|
+
const model = this.requireModel(modelId);
|
|
795
|
+
this.busy.set(modelId, (this.busy.get(modelId) ?? 0) + 1);
|
|
796
|
+
let vector;
|
|
797
|
+
try {
|
|
798
|
+
vector = await (await this.engineFor(model)).encode(text);
|
|
799
|
+
} finally {
|
|
800
|
+
const left = (this.busy.get(modelId) ?? 1) - 1;
|
|
801
|
+
if (left > 0) this.busy.set(modelId, left);
|
|
802
|
+
else this.busy.delete(modelId);
|
|
803
|
+
if (this.engines.has(modelId)) this.lastUsed.set(modelId, this.now());
|
|
804
|
+
this.scheduleEviction();
|
|
805
|
+
}
|
|
806
|
+
if (vector.length !== model.meta.embeddingDim) throw new Error(`EmbeddingEncoder: ${model.meta.textModelId} produced ${String(vector.length)} dims, its metadata declares ${String(model.meta.embeddingDim)} — refusing the query rather than truncating it`);
|
|
807
|
+
return {
|
|
808
|
+
vector,
|
|
809
|
+
model
|
|
810
|
+
};
|
|
811
|
+
}
|
|
812
|
+
/**
|
|
813
|
+
* Dispose every IDLE tower unused for {@link TEXT_ENGINE_IDLE_MS} (a timer in
|
|
814
|
+
* the addon calls this), except the active model's. Returns the evicted
|
|
815
|
+
* model ids; each is logged. An unknown active model evicts nothing.
|
|
816
|
+
*/
|
|
817
|
+
async evictIdle(idleMs = TEXT_ENGINE_IDLE_MS) {
|
|
818
|
+
const active = this.deps.activeModelId();
|
|
819
|
+
if (active === null) return [];
|
|
820
|
+
const now = this.now();
|
|
821
|
+
const idle = [...this.engines.keys()].filter((id) => id !== active && (this.busy.get(id) ?? 0) === 0 && now - (this.lastUsed.get(id) ?? now) >= idleMs);
|
|
822
|
+
const evicted = [];
|
|
823
|
+
for (const modelId of idle) {
|
|
824
|
+
if ((this.busy.get(modelId) ?? 0) > 0 || !this.engines.has(modelId)) continue;
|
|
825
|
+
evicted.push(modelId);
|
|
826
|
+
this.deps.logger.info("CLIP text tower evicted — idle", { meta: {
|
|
827
|
+
modelId,
|
|
828
|
+
idleMs: now - (this.lastUsed.get(modelId) ?? now)
|
|
829
|
+
} });
|
|
830
|
+
await this.restart(modelId);
|
|
831
|
+
}
|
|
832
|
+
return evicted;
|
|
833
|
+
}
|
|
834
|
+
/** Live engines, for the memory watchdog. */
|
|
835
|
+
live() {
|
|
836
|
+
return this.engines;
|
|
837
|
+
}
|
|
838
|
+
/** Dispose one engine; the next query rebuilds it. */
|
|
839
|
+
async restart(modelId) {
|
|
840
|
+
const engine = this.engines.get(modelId);
|
|
841
|
+
if (engine === void 0) return;
|
|
842
|
+
this.engines.delete(modelId);
|
|
843
|
+
this.lastUsed.delete(modelId);
|
|
844
|
+
await engine.dispose();
|
|
845
|
+
}
|
|
846
|
+
/** Dispose every tower AND cancel the builds in flight (disposed as they land). */
|
|
847
|
+
async disposeAll() {
|
|
848
|
+
this.generation += 1;
|
|
849
|
+
const all = [...this.engines.values()];
|
|
850
|
+
this.engines.clear();
|
|
851
|
+
this.lastUsed.clear();
|
|
852
|
+
await Promise.all(all.map((e) => e.dispose()));
|
|
853
|
+
await Promise.allSettled(this.inflight.values());
|
|
854
|
+
}
|
|
855
|
+
async engineFor(model) {
|
|
856
|
+
const held = this.engines.get(model.modelId);
|
|
857
|
+
if (held !== void 0 && held.isAlive()) {
|
|
858
|
+
this.engines.delete(model.modelId);
|
|
859
|
+
this.engines.set(model.modelId, held);
|
|
860
|
+
return held;
|
|
861
|
+
}
|
|
862
|
+
if (held !== void 0) {
|
|
863
|
+
this.deps.logger.warn("CLIP text tower died — dropping it and rebuilding", { meta: {
|
|
864
|
+
modelId: model.modelId,
|
|
865
|
+
textModelId: model.meta.textModelId
|
|
866
|
+
} });
|
|
867
|
+
this.engines.delete(model.modelId);
|
|
868
|
+
this.lastUsed.delete(model.modelId);
|
|
869
|
+
this.deps.budget.record(model.modelId, "process died after loading");
|
|
870
|
+
held.dispose().catch(() => {});
|
|
871
|
+
}
|
|
872
|
+
const pending = this.inflight.get(model.modelId);
|
|
873
|
+
if (pending !== void 0) return pending;
|
|
874
|
+
this.deps.budget.check(model.modelId);
|
|
875
|
+
this.deps.requirements.check(model.modelId);
|
|
876
|
+
const textEntry = this.deps.textCatalog.find((e) => e.id === model.meta.textModelId);
|
|
877
|
+
if (textEntry === void 0) throw this.deps.requirements.refuse(model.modelId, new PermanentRequirementsFailure(`EmbeddingEncoder: text model "${model.meta.textModelId}" (for ${model.modelId}) is not in the text catalog`));
|
|
878
|
+
const generation = this.generation;
|
|
879
|
+
const build = this.buildAndLoad(model, textEntry);
|
|
880
|
+
this.inflight.set(model.modelId, build);
|
|
881
|
+
try {
|
|
882
|
+
const engine = await build;
|
|
883
|
+
if (this.generation !== generation) {
|
|
884
|
+
await engine.dispose();
|
|
885
|
+
throw new Error(`text tower for "${model.modelId}" cancelled: disposed while building`);
|
|
886
|
+
}
|
|
887
|
+
this.engines.set(model.modelId, engine);
|
|
888
|
+
this.lastUsed.set(model.modelId, this.now());
|
|
889
|
+
this.deps.logger.info("CLIP text tower loaded", { meta: {
|
|
890
|
+
modelId: model.modelId,
|
|
891
|
+
textModelId: model.meta.textModelId
|
|
892
|
+
} });
|
|
893
|
+
this.scheduleEviction();
|
|
894
|
+
return engine;
|
|
895
|
+
} finally {
|
|
896
|
+
this.inflight.delete(model.modelId);
|
|
897
|
+
}
|
|
898
|
+
}
|
|
899
|
+
/**
|
|
900
|
+
* Two stages, two owners (`model-requirements.ts`): `build` resolves files
|
|
901
|
+
* and runtime and constructs the tower — a throw there is refused by name
|
|
902
|
+
* and spends nothing; `initialize()` spawns and loads — a throw there is a
|
|
903
|
+
* load failure and spends the crash budget.
|
|
904
|
+
*/
|
|
905
|
+
async buildAndLoad(model, textEntry) {
|
|
906
|
+
let engine;
|
|
907
|
+
try {
|
|
908
|
+
engine = await this.deps.build({
|
|
909
|
+
model,
|
|
910
|
+
textEntry,
|
|
911
|
+
meta: model.meta
|
|
912
|
+
});
|
|
913
|
+
} catch (err) {
|
|
914
|
+
throw this.deps.requirements.refuse(model.modelId, err);
|
|
915
|
+
}
|
|
916
|
+
this.deps.requirements.satisfied(model.modelId);
|
|
917
|
+
try {
|
|
918
|
+
await engine.initialize();
|
|
919
|
+
} catch (err) {
|
|
920
|
+
this.deps.budget.record(model.modelId, `load failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
921
|
+
throw err;
|
|
922
|
+
}
|
|
923
|
+
return engine;
|
|
924
|
+
}
|
|
925
|
+
now() {
|
|
926
|
+
return (this.deps.now ?? Date.now)();
|
|
927
|
+
}
|
|
928
|
+
/** Fire-and-log: eviction errors never reach an encode. */
|
|
929
|
+
scheduleEviction() {
|
|
930
|
+
this.evictBeyond(this.deps.maxEngines ?? 2).catch((err) => {
|
|
931
|
+
this.deps.logger.warn("CLIP text tower eviction failed", { meta: { error: err instanceof Error ? err.message : String(err) } });
|
|
932
|
+
});
|
|
933
|
+
}
|
|
934
|
+
/** Evict the coldest IDLE engines until at most `max` remain, or none is idle. */
|
|
935
|
+
async evictBeyond(max) {
|
|
936
|
+
while (this.engines.size > max) {
|
|
937
|
+
const coldestIdle = [...this.engines.keys()].find((id) => (this.busy.get(id) ?? 0) === 0);
|
|
938
|
+
if (coldestIdle === void 0) return;
|
|
939
|
+
await this.restart(coldestIdle);
|
|
940
|
+
}
|
|
941
|
+
}
|
|
942
|
+
/** Encodes in flight on a model — exposed for the eviction tests. */
|
|
943
|
+
inFlight(modelId) {
|
|
944
|
+
return this.busy.get(modelId) ?? 0;
|
|
945
|
+
}
|
|
946
|
+
};
|
|
523
947
|
//#endregion
|
|
524
948
|
//#region src/embedding-encoder/addon/index.ts
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
949
|
+
/** Memory-watchdog key prefix of a text tower; the suffix is its CLIP model id. */
|
|
950
|
+
var TEXT_ENGINE_KEY_PREFIX = "clip-text:";
|
|
951
|
+
/** How often idle text towers are looked for (see `TEXT_ENGINE_IDLE_MS`). */
|
|
952
|
+
var IDLE_SWEEP_MS = 6e4;
|
|
953
|
+
/** A model URL answering one of these does not exist; retrying changes nothing. */
|
|
954
|
+
var PERMANENT_HTTP_STATUS = new Set([404, 410]);
|
|
955
|
+
var EmbeddingEncoderAddon = class extends require_clip_model_registry.BaseAddon {
|
|
956
|
+
/**
|
|
957
|
+
* Everything on this node that can be terminally refused (`failed-models-report.ts`):
|
|
958
|
+
* the respawn budgets per tower (D6, `crash-budget.ts`) and the requirements
|
|
959
|
+
* gates (`model-requirements.ts` — cooldown, and terminal for what can never
|
|
960
|
+
* be met). All in memory: a runner respawn starts them from zero. Every row
|
|
961
|
+
* and refusal names this node.
|
|
962
|
+
*/
|
|
963
|
+
towers = {
|
|
964
|
+
budgets: {
|
|
965
|
+
image: new CrashBudget({
|
|
966
|
+
tower: "image",
|
|
967
|
+
nodeId: () => this.localNodeId(),
|
|
968
|
+
logger: {
|
|
969
|
+
info: (message, extras) => this.ctx.logger.info(message, extras),
|
|
970
|
+
error: (message, extras) => this.ctx.logger.error(message, extras)
|
|
971
|
+
}
|
|
972
|
+
}),
|
|
973
|
+
text: new CrashBudget({
|
|
974
|
+
tower: "text",
|
|
975
|
+
nodeId: () => this.localNodeId(),
|
|
976
|
+
logger: {
|
|
977
|
+
info: (message, extras) => this.ctx.logger.info(message, extras),
|
|
978
|
+
error: (message, extras) => this.ctx.logger.error(message, extras)
|
|
979
|
+
}
|
|
980
|
+
})
|
|
981
|
+
},
|
|
982
|
+
requirements: {
|
|
983
|
+
image: new RequirementsGate({
|
|
984
|
+
tower: "image",
|
|
985
|
+
nodeId: () => this.localNodeId(),
|
|
986
|
+
logger: {
|
|
987
|
+
info: (message, extras) => this.ctx.logger.info(message, extras),
|
|
988
|
+
warn: (message, extras) => this.ctx.logger.warn(message, extras),
|
|
989
|
+
error: (message, extras) => this.ctx.logger.error(message, extras)
|
|
990
|
+
}
|
|
991
|
+
}),
|
|
992
|
+
text: new RequirementsGate({
|
|
993
|
+
tower: "text",
|
|
994
|
+
nodeId: () => this.localNodeId(),
|
|
995
|
+
logger: {
|
|
996
|
+
info: (message, extras) => this.ctx.logger.info(message, extras),
|
|
997
|
+
warn: (message, extras) => this.ctx.logger.warn(message, extras),
|
|
998
|
+
error: (message, extras) => this.ctx.logger.error(message, extras)
|
|
999
|
+
}
|
|
1000
|
+
})
|
|
1001
|
+
}
|
|
1002
|
+
};
|
|
1003
|
+
/** The image tower, one model at a time, single-flight per model (`model-engine-slot.ts`). */
|
|
1004
|
+
imageEngine = new ModelEngineSlot((modelId) => this.buildImageEngine(modelId), this.towers.budgets.image, this.towers.requirements.image, (modelId) => this.ctx.logger.warn("CLIP image tower died — dropping it and rebuilding", { meta: { modelId } }));
|
|
1005
|
+
/** One text tower per CLIP model, bounded (see `clip-text-engines.ts`). */
|
|
1006
|
+
textEngines = new ClipTextEngines({
|
|
1007
|
+
registry: require_clip_model_registry.BUILTIN_CLIP_MODELS,
|
|
1008
|
+
textCatalog: require_clip_model_registry.CLIP_TEXT_MODELS,
|
|
1009
|
+
build: (spec) => this.buildTextEngine(spec),
|
|
1010
|
+
budget: this.towers.budgets.text,
|
|
1011
|
+
requirements: this.towers.requirements.text,
|
|
1012
|
+
activeModelId: () => this.activeModel.known(),
|
|
1013
|
+
logger: {
|
|
1014
|
+
info: (message, extras) => this.ctx.logger.info(message, extras),
|
|
1015
|
+
warn: (message, extras) => this.ctx.logger.warn(message, extras)
|
|
1016
|
+
}
|
|
1017
|
+
});
|
|
1018
|
+
/** Disposes NON-active text towers idle for `TEXT_ENGINE_IDLE_MS` (never prefetched). */
|
|
1019
|
+
idleSweep = null;
|
|
1020
|
+
/** The cluster row's CLIP model — what an unnamed request encodes in (D649). */
|
|
1021
|
+
activeModel = new ActiveClipModel({
|
|
1022
|
+
readRow: () => this.readClusterRow(),
|
|
1023
|
+
fallbackModelId: () => require_clip_model_registry.BUILTIN_CLIP_MODELS.resolve(this.config.modelId) !== null ? this.config.modelId : require_clip_model_registry.DEFAULT_CLIP_MODEL,
|
|
1024
|
+
now: () => Date.now(),
|
|
1025
|
+
logger: {
|
|
1026
|
+
info: (message, extras) => this.ctx.logger.info(message, extras),
|
|
1027
|
+
warn: (message, extras) => this.ctx.logger.warn(message, extras)
|
|
1028
|
+
},
|
|
1029
|
+
onChange: (modelId) => {
|
|
1030
|
+
clearFailedOnModelChange(this.towers, modelId);
|
|
1031
|
+
}
|
|
1032
|
+
});
|
|
528
1033
|
models = null;
|
|
529
1034
|
/**
|
|
530
1035
|
* RSS bound + periodic memory telemetry for the two Python engines
|
|
@@ -536,27 +1041,27 @@ var EmbeddingEncoderAddon = class extends require_dist.BaseAddon {
|
|
|
536
1041
|
* stopped in `onShutdown`.
|
|
537
1042
|
*/
|
|
538
1043
|
memoryGuard = null;
|
|
539
|
-
/** Single-flight guards for the lazy engine builds — a watchdog restart
|
|
540
|
-
* followed by two concurrent encodes must not spawn the engine twice
|
|
541
|
-
* (the same bug detection-pipeline fixed three times; see its
|
|
542
|
-
* `engineFactoryInflight`). */
|
|
543
|
-
imageEngineInflight = null;
|
|
544
|
-
textEngineInflight = null;
|
|
545
1044
|
constructor() {
|
|
546
|
-
super({ modelId: DEFAULT_CLIP_MODEL });
|
|
1045
|
+
super({ modelId: require_clip_model_registry.DEFAULT_CLIP_MODEL });
|
|
547
1046
|
}
|
|
548
1047
|
async onInitialize() {
|
|
549
1048
|
const modelsDir = await this.resolveModelsDir();
|
|
550
|
-
this.models = new _camstack_system_addon_utils.ModelDownloadService(modelsDir, [...CLIP_IMAGE_MODELS, ...CLIP_TEXT_MODELS]);
|
|
551
|
-
this.memoryGuard = new
|
|
552
|
-
policy:
|
|
1049
|
+
this.models = new _camstack_system_addon_utils.ModelDownloadService(modelsDir, [...require_clip_model_registry.CLIP_IMAGE_MODELS, ...require_clip_model_registry.CLIP_TEXT_MODELS]);
|
|
1050
|
+
this.memoryGuard = new require_clip_model_registry.PoolMemoryWatchdog({
|
|
1051
|
+
policy: require_clip_model_registry.resolvePoolMemoryPolicy(process.env),
|
|
553
1052
|
log: this.ctx.logger.child("pool-memory"),
|
|
554
1053
|
sample: () => this.sampleEngineMemory(),
|
|
555
1054
|
restart: (key) => this.restartEngineForMemory(key)
|
|
556
1055
|
});
|
|
557
1056
|
this.memoryGuard.start();
|
|
1057
|
+
this.idleSweep = setInterval(() => {
|
|
1058
|
+
this.textEngines.evictIdle().catch((err) => {
|
|
1059
|
+
this.ctx.logger.warn("CLIP text tower idle sweep failed", { meta: { error: err instanceof Error ? err.message : String(err) } });
|
|
1060
|
+
});
|
|
1061
|
+
}, IDLE_SWEEP_MS);
|
|
1062
|
+
this.idleSweep.unref?.();
|
|
558
1063
|
return [{
|
|
559
|
-
capability:
|
|
1064
|
+
capability: require_clip_model_registry.embeddingEncoderCapability,
|
|
560
1065
|
provider: this
|
|
561
1066
|
}];
|
|
562
1067
|
}
|
|
@@ -565,13 +1070,15 @@ var EmbeddingEncoderAddon = class extends require_dist.BaseAddon {
|
|
|
565
1070
|
async sampleEngineMemory() {
|
|
566
1071
|
const engines = [{
|
|
567
1072
|
key: "clip-image",
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
1073
|
+
modelId: this.imageEngine.held()?.modelId ?? null,
|
|
1074
|
+
pid: this.imageEngine.held()?.engine.getPid() ?? null,
|
|
1075
|
+
requests: this.imageEngine.held()?.engine.getRequestCount() ?? 0
|
|
1076
|
+
}, ...[...this.textEngines.live()].map(([modelId, engine]) => ({
|
|
1077
|
+
key: `${TEXT_ENGINE_KEY_PREFIX}${modelId}`,
|
|
1078
|
+
modelId,
|
|
1079
|
+
pid: engine.getPid(),
|
|
1080
|
+
requests: engine.getRequestCount()
|
|
1081
|
+
}))];
|
|
575
1082
|
const out = [];
|
|
576
1083
|
for (const engine of engines) {
|
|
577
1084
|
if (engine.pid === null) continue;
|
|
@@ -589,7 +1096,7 @@ var EmbeddingEncoderAddon = class extends require_dist.BaseAddon {
|
|
|
589
1096
|
pids: [engine.pid],
|
|
590
1097
|
rssBytes: mem.rssBytes,
|
|
591
1098
|
meta: {
|
|
592
|
-
modelId:
|
|
1099
|
+
modelId: engine.modelId,
|
|
593
1100
|
vmMb: mb(mem.vmBytes),
|
|
594
1101
|
peakRssMb: mb(mem.hwmBytes),
|
|
595
1102
|
swapMb: mb(mem.swapBytes),
|
|
@@ -606,110 +1113,125 @@ var EmbeddingEncoderAddon = class extends require_dist.BaseAddon {
|
|
|
606
1113
|
* surface errors, never silent loss. */
|
|
607
1114
|
async restartEngineForMemory(key) {
|
|
608
1115
|
if (key === "clip-image") {
|
|
609
|
-
|
|
610
|
-
if (!engine) return;
|
|
611
|
-
this.imageRawEngine = null;
|
|
612
|
-
await engine.dispose();
|
|
1116
|
+
await this.imageEngine.dispose();
|
|
613
1117
|
return;
|
|
614
1118
|
}
|
|
615
|
-
|
|
616
|
-
|
|
617
|
-
|
|
618
|
-
|
|
1119
|
+
if (key.startsWith(TEXT_ENGINE_KEY_PREFIX)) await this.textEngines.restart(key.slice(10));
|
|
1120
|
+
}
|
|
1121
|
+
/** The cluster row's `clip-embedding` model, or `null` when unreadable. */
|
|
1122
|
+
async readClusterRow() {
|
|
1123
|
+
return require_clip_model_registry.resolveClusterModelPin(this.ctx.api, require_clip_model_registry.CLIP_EMBEDDING_STEP_ID, { warn: (message, extras) => this.ctx.logger.warn(message, extras) });
|
|
619
1124
|
}
|
|
620
1125
|
async encode(input) {
|
|
621
1126
|
const { crop, width, height } = input;
|
|
622
|
-
await this.
|
|
623
|
-
const
|
|
1127
|
+
const model = this.textEngines.requireModel(await this.activeModel.get());
|
|
1128
|
+
const imageEngine = await this.imageEngine.ensure(model.modelId);
|
|
1129
|
+
const meta = model.meta;
|
|
624
1130
|
const start = Date.now();
|
|
625
|
-
const preprocessed = preprocessForClip(Buffer.isBuffer(crop) ? crop : Buffer.from(crop), width, height,
|
|
626
|
-
const output = await
|
|
1131
|
+
const preprocessed = preprocessForClip(Buffer.isBuffer(crop) ? crop : Buffer.from(crop), width, height, model.inputSize, model.inputSize);
|
|
1132
|
+
const output = await imageEngine.run(preprocessed, [
|
|
627
1133
|
1,
|
|
628
1134
|
3,
|
|
629
|
-
|
|
630
|
-
|
|
1135
|
+
model.inputSize,
|
|
1136
|
+
model.inputSize
|
|
631
1137
|
]);
|
|
632
|
-
|
|
633
|
-
const normalized = l2Normalize(new Float32Array(
|
|
1138
|
+
if (output.length !== meta.embeddingDim) throw new Error(`EmbeddingEncoder: ${model.modelId} image tower produced ${String(output.length)} dims, its metadata declares ${String(meta.embeddingDim)}`);
|
|
1139
|
+
const normalized = l2Normalize(new Float32Array(output));
|
|
634
1140
|
return {
|
|
635
1141
|
embedding: Array.from(normalized),
|
|
636
1142
|
inferenceMs: Date.now() - start
|
|
637
1143
|
};
|
|
638
1144
|
}
|
|
1145
|
+
/**
|
|
1146
|
+
* Encode a query in ONE model's space: the named `modelId`, else the cluster
|
|
1147
|
+
* row's (D649). The token window, pad id, tokenizer and dimension all come
|
|
1148
|
+
* from that model's catalog metadata.
|
|
1149
|
+
*/
|
|
639
1150
|
async encodeText(input) {
|
|
640
|
-
const
|
|
641
|
-
await this.ensureTextEngine();
|
|
642
|
-
const meta = getModelMeta(this.config.modelId);
|
|
1151
|
+
const modelId = input.modelId ?? await this.activeModel.get();
|
|
643
1152
|
const start = Date.now();
|
|
644
|
-
|
|
645
|
-
const output = await this.textEngine.encode(text);
|
|
646
|
-
const sliced = output.length > meta.embeddingDim ? output.slice(0, meta.embeddingDim) : output;
|
|
647
|
-
const normalized = l2Normalize(new Float32Array(sliced));
|
|
1153
|
+
const { vector } = await this.textEngines.encode(modelId, input.text);
|
|
648
1154
|
return {
|
|
649
|
-
embedding: Array.from(
|
|
1155
|
+
embedding: Array.from(l2Normalize(new Float32Array(vector))),
|
|
650
1156
|
inferenceMs: Date.now() - start
|
|
651
1157
|
};
|
|
652
1158
|
}
|
|
1159
|
+
/** THIS node's answer — the cap is one provider per node (see the cap's docblock). */
|
|
653
1160
|
async getInfo() {
|
|
654
|
-
const
|
|
1161
|
+
const model = this.textEngines.requireModel(await this.activeModel.get());
|
|
1162
|
+
const nodeId = this.localNodeId();
|
|
655
1163
|
return {
|
|
656
|
-
modelId:
|
|
657
|
-
embeddingDim: meta.embeddingDim,
|
|
658
|
-
ready: this.
|
|
1164
|
+
modelId: model.modelId,
|
|
1165
|
+
embeddingDim: model.meta.embeddingDim,
|
|
1166
|
+
ready: this.imageEngine.held()?.modelId === model.modelId,
|
|
1167
|
+
nodeId,
|
|
1168
|
+
failedModels: [...reportFailedModels(this.towers.budgets)],
|
|
1169
|
+
failedRequirements: [...reportFailedRequirements(this.towers.requirements)]
|
|
659
1170
|
};
|
|
660
1171
|
}
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
const meta = getModelMeta(this.config.modelId);
|
|
665
|
-
const imageEntry = CLIP_IMAGE_MODELS.find((m) => m.id === meta.imageModelId);
|
|
666
|
-
if (!imageEntry) throw new Error(`EmbeddingEncoderAddon: unknown image model "${meta.imageModelId}"`);
|
|
667
|
-
const inflight = this.resolveForEntry(imageEntry, "image");
|
|
668
|
-
this.imageEngineInflight = inflight;
|
|
669
|
-
try {
|
|
670
|
-
await inflight;
|
|
671
|
-
} finally {
|
|
672
|
-
this.imageEngineInflight = null;
|
|
673
|
-
}
|
|
1172
|
+
/** The operator's explicit clear of THIS node's `failed` state (D6); names the node. */
|
|
1173
|
+
async resetFailedModels(input) {
|
|
1174
|
+
return resetFailedModelsOn(this.localNodeId(), this.towers, input.modelId);
|
|
674
1175
|
}
|
|
675
|
-
|
|
676
|
-
|
|
677
|
-
|
|
678
|
-
const meta = getModelMeta(this.config.modelId);
|
|
679
|
-
const textEntry = CLIP_TEXT_MODELS.find((m) => m.id === meta.textModelId);
|
|
680
|
-
if (!textEntry) throw new Error(`EmbeddingEncoderAddon: unknown text model "${meta.textModelId}"`);
|
|
681
|
-
const inflight = this.resolveForEntry(textEntry, "text");
|
|
682
|
-
this.textEngineInflight = inflight;
|
|
683
|
-
try {
|
|
684
|
-
await inflight;
|
|
685
|
-
} finally {
|
|
686
|
-
this.textEngineInflight = null;
|
|
687
|
-
}
|
|
1176
|
+
/** The logical node (a forked child's `<node>/<addon>` id stripped to the node). */
|
|
1177
|
+
localNodeId() {
|
|
1178
|
+
return require_clip_model_registry.normalizeNodeId(this.ctx.kernel?.localNodeId);
|
|
688
1179
|
}
|
|
689
|
-
|
|
690
|
-
|
|
691
|
-
|
|
1180
|
+
/**
|
|
1181
|
+
* The REQUIREMENTS stage, shared by both towers: the embedded Python, its
|
|
1182
|
+
* requirements, then the model file (downloaded on first use). Anything that
|
|
1183
|
+
* throws here is refused as `model-requirements-unavailable` and spends no
|
|
1184
|
+
* crash budget (`model-requirements.ts`). The runtime is asked for first —
|
|
1185
|
+
* the cheap question before the 185–565 MB download (D54's rule).
|
|
1186
|
+
*/
|
|
1187
|
+
async prepareOnnx(entry) {
|
|
692
1188
|
const pythonPath = await this.ctx.deps.ensurePython();
|
|
693
1189
|
if (!pythonPath) throw new Error("EmbeddingEncoder: embedded Python is unavailable — cannot run ONNX embeddings. ctx.deps.ensurePython() returned null (portable Python download likely failed).");
|
|
694
1190
|
const pythonDir = resolveEmbeddingPythonDir();
|
|
695
1191
|
await this.ctx.deps.installPythonRequirements(node_path.join(pythonDir, "requirements-embedding.txt"));
|
|
696
|
-
if (
|
|
697
|
-
|
|
698
|
-
|
|
699
|
-
this.
|
|
700
|
-
|
|
1192
|
+
if (entry.formats.onnx === void 0) throw new PermanentRequirementsFailure(`EmbeddingEncoder: ${entry.id} has no onnx build in its catalog entry (formats: ${Object.keys(entry.formats).join(", ") || "none"}) — no download can supply it`);
|
|
1193
|
+
let modelPath;
|
|
1194
|
+
try {
|
|
1195
|
+
modelPath = await this.models.ensure(entry.id, "onnx");
|
|
1196
|
+
} catch (err) {
|
|
1197
|
+
const status = downloadHttpStatusOf(err);
|
|
1198
|
+
if (status !== null && PERMANENT_HTTP_STATUS.has(status)) throw new PermanentRequirementsFailure(`EmbeddingEncoder: ${entry.id} is not at its catalog URL (HTTP ${String(status)}) — fix the catalog entry`);
|
|
1199
|
+
throw err;
|
|
701
1200
|
}
|
|
702
|
-
|
|
703
|
-
|
|
704
|
-
|
|
705
|
-
|
|
706
|
-
|
|
1201
|
+
return {
|
|
1202
|
+
modelPath,
|
|
1203
|
+
pythonPath,
|
|
1204
|
+
pythonDir
|
|
1205
|
+
};
|
|
1206
|
+
}
|
|
1207
|
+
async buildImageEngine(modelId) {
|
|
1208
|
+
const entry = require_clip_model_registry.CLIP_IMAGE_MODELS.find((m) => m.id === modelId);
|
|
1209
|
+
if (!entry) throw new PermanentRequirementsFailure(`EmbeddingEncoderAddon: unknown image model "${modelId}" — not in the CLIP image catalog`);
|
|
1210
|
+
const { modelPath, pythonPath, pythonDir } = await this.prepareOnnx(entry);
|
|
1211
|
+
return new PythonRawTensorEngine(pythonPath, node_path.join(pythonDir, "raw_tensor_inference.py"), modelPath, this.ctx.logger.withTags({ modelId: entry.id }));
|
|
1212
|
+
}
|
|
1213
|
+
/**
|
|
1214
|
+
* One text tower. Tokenization happens IN Python via the HF `tokenizers`
|
|
1215
|
+
* library, with the tokenizer the model's metadata names — a declared sibling
|
|
1216
|
+
* of the text onnx, so `ModelDownloadService.ensure()` fetched it next to the
|
|
1217
|
+
* model file — and the model's own token window.
|
|
1218
|
+
*/
|
|
1219
|
+
async buildTextEngine(spec) {
|
|
1220
|
+
const { modelPath, pythonPath, pythonDir } = await this.prepareOnnx(spec.textEntry);
|
|
1221
|
+
const tokenizerPath = node_path.join(node_path.dirname(modelPath), spec.meta.tokenizerFile);
|
|
1222
|
+
if (!node_fs.existsSync(tokenizerPath)) throw new PermanentRequirementsFailure(`EmbeddingEncoder: tokenizer not found at "${tokenizerPath}" — the ${spec.meta.tokenizerFile} sibling download of ${spec.textEntry.id} likely failed.`);
|
|
1223
|
+
return new PythonTextEncoderEngine(pythonPath, node_path.join(pythonDir, "text_encoder_inference.py"), modelPath, tokenizerPath, {
|
|
1224
|
+
contextLength: spec.meta.contextLength,
|
|
1225
|
+
padId: spec.meta.padId
|
|
1226
|
+
}, this.ctx.logger.withTags({ modelId: spec.textEntry.id }));
|
|
707
1227
|
}
|
|
708
1228
|
async onShutdown() {
|
|
709
1229
|
this.memoryGuard?.stop();
|
|
1230
|
+
if (this.idleSweep !== null) clearInterval(this.idleSweep);
|
|
1231
|
+
this.idleSweep = null;
|
|
710
1232
|
this.memoryGuard = null;
|
|
711
|
-
await this.
|
|
712
|
-
await this.
|
|
1233
|
+
await this.imageEngine.dispose();
|
|
1234
|
+
await this.textEngines.disposeAll();
|
|
713
1235
|
}
|
|
714
1236
|
globalSettingsSchema() {
|
|
715
1237
|
return this.schema({ sections: [{
|
|
@@ -719,9 +1241,9 @@ var EmbeddingEncoderAddon = class extends require_dist.BaseAddon {
|
|
|
719
1241
|
fields: [{
|
|
720
1242
|
type: "text",
|
|
721
1243
|
key: "modelId",
|
|
722
|
-
label: "
|
|
723
|
-
description: "
|
|
724
|
-
default: DEFAULT_CLIP_MODEL
|
|
1244
|
+
label: "Fallback model ID",
|
|
1245
|
+
description: "Used only until the cluster \"Semantic search model\" row has been read once. The encoder follows that row (D649).",
|
|
1246
|
+
default: require_clip_model_registry.DEFAULT_CLIP_MODEL
|
|
725
1247
|
}]
|
|
726
1248
|
}] });
|
|
727
1249
|
}
|
|
@@ -742,6 +1264,18 @@ function resolveEmbeddingPythonDir() {
|
|
|
742
1264
|
for (const c of candidates) if (node_fs.existsSync(node_path.join(c, "raw_tensor_inference.py"))) return c;
|
|
743
1265
|
throw new Error(`EmbeddingEncoder: python/ dir (raw_tensor_inference.py) not found. Searched:\n${candidates.join("\n")}`);
|
|
744
1266
|
}
|
|
1267
|
+
/**
|
|
1268
|
+
* HTTP status of a model-download failure, recognised by SHAPE (`name` +
|
|
1269
|
+
* numeric `status`), never by `instanceof` or a framework import.
|
|
1270
|
+
* `@camstack/system` is host-resolved (not bundled), so importing a guard it
|
|
1271
|
+
* only exports from a newer server would stop this whole entry from linking
|
|
1272
|
+
* on a node still running the older server — a deploy-order hazard for a
|
|
1273
|
+
* one-line classification. `null` = not an HTTP answer (unreachable, other).
|
|
1274
|
+
*/
|
|
1275
|
+
function downloadHttpStatusOf(err) {
|
|
1276
|
+
if (err instanceof Error && err.name === "ModelDownloadHttpError" && "status" in err && typeof err.status === "number") return err.status;
|
|
1277
|
+
return null;
|
|
1278
|
+
}
|
|
745
1279
|
//#endregion
|
|
746
1280
|
exports.EmbeddingEncoderAddon = EmbeddingEncoderAddon;
|
|
747
1281
|
exports.default = EmbeddingEncoderAddon;
|