@camstack/addon-training 0.0.0-stage → 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/training/icon.svg +5 -0
- package/assets/training-modal/icon.svg +4 -0
- package/dist/MaskShapeCanvas-BokARvOj.mjs +13806 -0
- package/dist/MotionZonesSettings-aj085Nrn.mjs +310 -0
- package/dist/PrivacyMaskSettings-BwYjIkQR.mjs +377 -0
- package/dist/SceneMonitorEditor-BBfow3Lm.mjs +831 -0
- package/dist/_stub.js +17613 -0
- package/dist/_virtual_mf-localSharedImportMap___mfe_internal__addon_training_page-CspLU0I2.mjs +156 -0
- package/dist/_virtual_mf___mfe_internal__addon_training_page__loadShare___mf_0_camstack_mf_1_sdk__loadShare__.js-ZeT4eUYy.mjs +25 -0
- package/dist/_virtual_mf___mfe_internal__addon_training_page__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-BkKuzhOJ.mjs +26 -0
- package/dist/_virtual_mf___mfe_internal__addon_training_page__loadShare___mf_0_tanstack_mf_1_react_mf_2_query__loadShare__.js-CaB1jnkt.mjs +25 -0
- package/dist/_virtual_mf___mfe_internal__addon_training_page__loadShare___mf_0_trpc_mf_1_client__loadShare__.js-B3rryujf.mjs +25 -0
- package/dist/_virtual_mf___mfe_internal__addon_training_page__loadShare___mf_0_trpc_mf_1_react_mf_2_query__loadShare__.js-nL4MGFRP.mjs +26 -0
- package/dist/_virtual_mf___mfe_internal__addon_training_page__loadShare__react__loadShare__.js-BT_wL8s3.mjs +70 -0
- package/dist/_virtual_mf___mfe_internal__addon_training_page__loadShare__react_mf_1_jsx_mf_2_runtime__loadShare__.js-DDBEvLdo.mjs +26 -0
- package/dist/_virtual_mf___mfe_internal__addon_training_page__loadShare__react_mf_2_dom__loadShare__.js-i_pnGDxd.mjs +25 -0
- package/dist/_virtual_mf___mfe_internal__addon_training_page__loadShare__react_mf_2_dom_mf_1_client__loadShare__.js-gTIQrhKs.mjs +25 -0
- package/dist/addon-training.css +3 -0
- package/dist/dist-BnT7Wsvm.mjs +52556 -0
- package/dist/dist-WGwGDX6_.js +52708 -0
- package/dist/hostInit-Dl0wFDad.mjs +129 -0
- package/dist/index.js +205 -0
- package/dist/index.mjs +201 -0
- package/dist/orchestrator/training.addon.js +1712 -0
- package/dist/orchestrator/training.addon.mjs +1710 -0
- package/dist/player-overlays-wE5IcDiI.mjs +44 -0
- package/dist/providers/modal/training-modal.addon.js +88499 -0
- package/dist/providers/modal/training-modal.addon.mjs +88527 -0
- package/dist/remoteEntry.js +2 -0
- package/dist/remoteEntry.ssr.js +33 -0
- package/dist/responsive-D17kMsm1.mjs +122 -0
- package/dist/rolldown-runtime-HEgqtunE.mjs +20 -0
- package/dist/scene-monitor-copy-LjlvhSRM.mjs +187 -0
- package/dist/square-JGDHhLlY.mjs +45 -0
- package/dist/trash-2-zeFnLJvf.mjs +25 -0
- package/dist/virtualExposes-0m8NqEDE.mjs +27 -0
- package/dist/virtual_mf-REMOTE_ENTRY_ID___mfe_internal__addon_training_page__remoteEntry_js-iFB6fGI7.mjs +2857 -0
- package/dist/virtual_mf-exposes-ssr___mfe_internal__addon_training_page__remoteEntry_js-BbeHzLhl.mjs +10 -0
- package/package.json +116 -4
- package/README.md +0 -3
|
@@ -0,0 +1,1712 @@
|
|
|
1
|
+
Object.defineProperty(exports, Symbol.toStringTag, { value: "Module" });
|
|
2
|
+
const require_dist = require("../dist-WGwGDX6_.js");
|
|
3
|
+
let node_path = require("node:path");
|
|
4
|
+
node_path = require_dist.__toESM(node_path);
|
|
5
|
+
let node_crypto = require("node:crypto");
|
|
6
|
+
let node_fs = require("node:fs");
|
|
7
|
+
let node_fs_promises = require("node:fs/promises");
|
|
8
|
+
//#region src/orchestrator/api-ports.ts
|
|
9
|
+
var DETECTOR_STEP = "object-detection";
|
|
10
|
+
var REGISTRY_CAP = "custom-model-registry";
|
|
11
|
+
/** The provider cap, routed to ONE provider by the `addonId` every call carries. */
|
|
12
|
+
function providerPort(api) {
|
|
13
|
+
const p = api.trainingProvider;
|
|
14
|
+
return {
|
|
15
|
+
describe: () => p.describe.query({}),
|
|
16
|
+
validateCredentials: (i) => p.validateCredentials.mutate(i),
|
|
17
|
+
estimateJob: (i) => p.estimateJob.query(i),
|
|
18
|
+
uploadDataset: (i) => p.uploadDataset.mutate(i),
|
|
19
|
+
submitJob: (i) => p.submitJob.mutate(i),
|
|
20
|
+
getJob: (i) => p.getJob.query(i),
|
|
21
|
+
listJobEvents: (i) => p.listJobEvents.query(i),
|
|
22
|
+
cancelJob: (i) => p.cancelJob.mutate(i),
|
|
23
|
+
readResult: (i) => p.readResult.mutate(i),
|
|
24
|
+
offerArtifact: (i) => p.offerArtifact.mutate(i),
|
|
25
|
+
cleanupJob: (i) => p.cleanupJob.mutate(i),
|
|
26
|
+
listJobs: (i) => p.listJobs.query(i)
|
|
27
|
+
};
|
|
28
|
+
}
|
|
29
|
+
function datasetPort(api) {
|
|
30
|
+
return { offerArchive: async (input) => {
|
|
31
|
+
const offered = await api.pipelineAnalytics.offerRetrainArchive.mutate({
|
|
32
|
+
filter: { ...input.filter },
|
|
33
|
+
layout: input.layout
|
|
34
|
+
});
|
|
35
|
+
return {
|
|
36
|
+
ticket: offered.ticket,
|
|
37
|
+
frames: offered.frames
|
|
38
|
+
};
|
|
39
|
+
} };
|
|
40
|
+
}
|
|
41
|
+
/** Model Studio's registry: the hub's provider of `custom-model-registry`. */
|
|
42
|
+
function registryPort(api) {
|
|
43
|
+
return { importBundle: async (input) => {
|
|
44
|
+
const hub = (await api.addons.listCapabilityProviders.query({ capName: REGISTRY_CAP })).find((p) => !p.addonId.includes("@"));
|
|
45
|
+
if (hub === void 0) throw new Error("no custom-model registry on the hub — is Model Studio installed?");
|
|
46
|
+
return api.customModelRegistry.importModelBundle.mutate({
|
|
47
|
+
...input,
|
|
48
|
+
addonId: hub.addonId
|
|
49
|
+
});
|
|
50
|
+
} };
|
|
51
|
+
}
|
|
52
|
+
function pinPort(api) {
|
|
53
|
+
const orchestrator = api.pipelineOrchestrator;
|
|
54
|
+
return {
|
|
55
|
+
listDetectorDevices: async () => {
|
|
56
|
+
const agents = await orchestrator.listAgentSettings.query();
|
|
57
|
+
const out = [];
|
|
58
|
+
for (const agent of agents) {
|
|
59
|
+
if (agent.settings.detect === false) continue;
|
|
60
|
+
const node = await orchestrator.getNodeInferenceDevices.query({ nodeId: agent.nodeId });
|
|
61
|
+
for (const device of node.devices) {
|
|
62
|
+
if (!device.enabled || device.exclusion !== null) continue;
|
|
63
|
+
out.push({
|
|
64
|
+
nodeId: agent.nodeId,
|
|
65
|
+
deviceKey: device.key,
|
|
66
|
+
pinnedModelId: device.steps?.["object-detection"]?.modelId ?? device.defaultModelId
|
|
67
|
+
});
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
return out;
|
|
71
|
+
},
|
|
72
|
+
getOverride: async (deviceId, nodeId, deviceKey) => {
|
|
73
|
+
const patch = (await orchestrator.getCameraStepOverrides.query({ deviceId }))?.[nodeId]?.[deviceKey]?.[DETECTOR_STEP];
|
|
74
|
+
if (patch === void 0) return null;
|
|
75
|
+
return {
|
|
76
|
+
...patch.modelId !== void 0 ? { modelId: patch.modelId } : {},
|
|
77
|
+
...patch.settings !== void 0 ? { settings: { ...patch.settings } } : {}
|
|
78
|
+
};
|
|
79
|
+
},
|
|
80
|
+
setOverride: async (deviceId, nodeId, deviceKey, patch) => {
|
|
81
|
+
await orchestrator.setCameraStepOverride.mutate({
|
|
82
|
+
deviceId,
|
|
83
|
+
agentNodeId: nodeId,
|
|
84
|
+
deviceKey,
|
|
85
|
+
addonId: DETECTOR_STEP,
|
|
86
|
+
patch: patch === null ? null : {
|
|
87
|
+
...patch.modelId !== void 0 ? { modelId: patch.modelId } : {},
|
|
88
|
+
...patch.settings !== void 0 ? { settings: { ...patch.settings } } : {}
|
|
89
|
+
}
|
|
90
|
+
});
|
|
91
|
+
}
|
|
92
|
+
};
|
|
93
|
+
}
|
|
94
|
+
//#endregion
|
|
95
|
+
//#region src/orchestrator/artifact-staging.ts
|
|
96
|
+
/**
|
|
97
|
+
* The hub-side staging of a job's artefacts: `<dir>/<jobId>/bundle.tar` and
|
|
98
|
+
* `result.json`. A bundle arrives over `ctx.peerBytes`, bounded by the size
|
|
99
|
+
* result.json declared and hashed after it lands; it leaves the same way to
|
|
100
|
+
* Model Studio. The whole job directory goes when the result expires (14 days
|
|
101
|
+
* after ready / registration) or is discarded.
|
|
102
|
+
*/
|
|
103
|
+
var BUNDLE = "bundle.tar";
|
|
104
|
+
var RESULT = "result.json";
|
|
105
|
+
async function sha256File(file) {
|
|
106
|
+
const hash = (0, node_crypto.createHash)("sha256");
|
|
107
|
+
for await (const chunk of (0, node_fs.createReadStream)(file)) hash.update(chunk);
|
|
108
|
+
return hash.digest("hex");
|
|
109
|
+
}
|
|
110
|
+
/** Job ids are uuids; anything else never becomes a path. */
|
|
111
|
+
function jobDir(root, jobId) {
|
|
112
|
+
if (!/^[0-9a-f-]{36}$/.test(jobId)) throw new Error(`not a job id: ${jobId}`);
|
|
113
|
+
return node_path.default.join(root, jobId);
|
|
114
|
+
}
|
|
115
|
+
function createDiskArtifacts(root, peer) {
|
|
116
|
+
return {
|
|
117
|
+
receiveBundle: async (jobId, ticket, maxBytes) => {
|
|
118
|
+
const dir = jobDir(root, jobId);
|
|
119
|
+
await (0, node_fs_promises.mkdir)(dir, { recursive: true });
|
|
120
|
+
const target = node_path.default.join(dir, BUNDLE);
|
|
121
|
+
const opened = await peer.open(ticket);
|
|
122
|
+
if (opened.kind === "refused") throw new Error(`bundle offer refused: ${opened.code} — ${opened.detail}`);
|
|
123
|
+
const written = await opened.body.writeToFile(target, { maxBytes });
|
|
124
|
+
if (written.kind === "refused") throw new Error(`bundle transfer refused: ${written.code} — ${written.detail}`);
|
|
125
|
+
return {
|
|
126
|
+
bytes: written.bytes,
|
|
127
|
+
sha256: await sha256File(target)
|
|
128
|
+
};
|
|
129
|
+
},
|
|
130
|
+
writeResult: async (jobId, text) => {
|
|
131
|
+
const dir = jobDir(root, jobId);
|
|
132
|
+
await (0, node_fs_promises.mkdir)(dir, { recursive: true });
|
|
133
|
+
await (0, node_fs_promises.writeFile)(node_path.default.join(dir, RESULT), text, "utf8");
|
|
134
|
+
},
|
|
135
|
+
readResult: async (jobId) => {
|
|
136
|
+
try {
|
|
137
|
+
return await (0, node_fs_promises.readFile)(node_path.default.join(jobDir(root, jobId), RESULT), "utf8");
|
|
138
|
+
} catch {
|
|
139
|
+
return null;
|
|
140
|
+
}
|
|
141
|
+
},
|
|
142
|
+
offerBundle: async (jobId) => {
|
|
143
|
+
const offered = await peer.offerFile({
|
|
144
|
+
path: node_path.default.join(jobDir(root, jobId), BUNDLE),
|
|
145
|
+
contentType: "application/x-tar",
|
|
146
|
+
label: "training-bundle"
|
|
147
|
+
});
|
|
148
|
+
if (offered.kind === "refused") throw new Error(`bundle offer refused: ${offered.code} — ${offered.detail}`);
|
|
149
|
+
return offered.ticket;
|
|
150
|
+
},
|
|
151
|
+
remove: async (jobId) => {
|
|
152
|
+
await (0, node_fs_promises.rm)(jobDir(root, jobId), {
|
|
153
|
+
recursive: true,
|
|
154
|
+
force: true
|
|
155
|
+
});
|
|
156
|
+
}
|
|
157
|
+
};
|
|
158
|
+
}
|
|
159
|
+
//#endregion
|
|
160
|
+
//#region src/orchestrator/base-models.ts
|
|
161
|
+
var HF_MODELS = "https://huggingface.co/camstack/camstack-models/resolve/main";
|
|
162
|
+
var TIER_LABEL = {
|
|
163
|
+
t: "Tiny",
|
|
164
|
+
s: "Small",
|
|
165
|
+
m: "Medium",
|
|
166
|
+
c: "Compact"
|
|
167
|
+
};
|
|
168
|
+
function base(tier, res, floors, supplied = {}) {
|
|
169
|
+
return {
|
|
170
|
+
modelId: `yolov9${tier}-${String(res)}`,
|
|
171
|
+
label: `YOLOv9 ${TIER_LABEL[tier]} @${String(res)}`,
|
|
172
|
+
weights: `yolov9${tier}.pt`,
|
|
173
|
+
imgsz: res,
|
|
174
|
+
baselineOnnxUrl: `${HF_MODELS}/objectDetection/yolov9/onnx/camstack-yolov9${tier}-${String(res)}.onnx`,
|
|
175
|
+
baselineFloors: floors,
|
|
176
|
+
recommended: tier === "m" && res === 640,
|
|
177
|
+
tier,
|
|
178
|
+
suppliedLabelFloors: supplied.fp32 ?? {},
|
|
179
|
+
suppliedLabelFloorsInt8: supplied.int8 ?? {}
|
|
180
|
+
};
|
|
181
|
+
}
|
|
182
|
+
var BASE_MODELS = [
|
|
183
|
+
base("m", 640, {
|
|
184
|
+
person: .6,
|
|
185
|
+
vehicle: .4,
|
|
186
|
+
animal: .55
|
|
187
|
+
}, { int8: { "animal-classifier": .7 } }),
|
|
188
|
+
base("m", 320, {
|
|
189
|
+
person: .7,
|
|
190
|
+
vehicle: .5,
|
|
191
|
+
animal: .55
|
|
192
|
+
}, { int8: { "animal-classifier": .75 } }),
|
|
193
|
+
base("s", 640, {
|
|
194
|
+
person: .7,
|
|
195
|
+
vehicle: .4,
|
|
196
|
+
animal: .6
|
|
197
|
+
}),
|
|
198
|
+
base("s", 320, {
|
|
199
|
+
person: .7,
|
|
200
|
+
vehicle: .55,
|
|
201
|
+
animal: .4
|
|
202
|
+
}),
|
|
203
|
+
base("c", 640, {
|
|
204
|
+
person: .55,
|
|
205
|
+
vehicle: .45,
|
|
206
|
+
animal: .3
|
|
207
|
+
}, { int8: { "animal-classifier": .75 } }),
|
|
208
|
+
base("c", 320, {
|
|
209
|
+
person: .7,
|
|
210
|
+
vehicle: .45,
|
|
211
|
+
animal: .45
|
|
212
|
+
}),
|
|
213
|
+
base("t", 640, {
|
|
214
|
+
person: .65,
|
|
215
|
+
vehicle: .5,
|
|
216
|
+
animal: .5
|
|
217
|
+
}),
|
|
218
|
+
base("t", 320, {
|
|
219
|
+
person: .55,
|
|
220
|
+
vehicle: .45,
|
|
221
|
+
animal: .3
|
|
222
|
+
})
|
|
223
|
+
];
|
|
224
|
+
function findBaseModel(modelId) {
|
|
225
|
+
return BASE_MODELS.find((m) => m.modelId === modelId) ?? null;
|
|
226
|
+
}
|
|
227
|
+
/** The public shape (`listBaseModels`) — the internal fields stay here. */
|
|
228
|
+
function publicBaseModel(spec) {
|
|
229
|
+
return {
|
|
230
|
+
modelId: spec.modelId,
|
|
231
|
+
label: spec.label,
|
|
232
|
+
weights: spec.weights,
|
|
233
|
+
imgsz: spec.imgsz,
|
|
234
|
+
baselineOnnxUrl: spec.baselineOnnxUrl,
|
|
235
|
+
baselineFloors: { ...spec.baselineFloors },
|
|
236
|
+
recommended: spec.recommended
|
|
237
|
+
};
|
|
238
|
+
}
|
|
239
|
+
function clip(text) {
|
|
240
|
+
const line = text.replace(/\s+$/u, "");
|
|
241
|
+
return line.length > 500 ? `${line.slice(0, 499)}…` : line;
|
|
242
|
+
}
|
|
243
|
+
/** Provider log chunks may hold several lines, or half of one; split on newlines. */
|
|
244
|
+
function splitLogLines(lines) {
|
|
245
|
+
return lines.flatMap((l) => l.text.split(/\r?\n/u)).filter((t) => t.trim().length > 0);
|
|
246
|
+
}
|
|
247
|
+
function applyLogLines(job, lines) {
|
|
248
|
+
let stage = job.stage;
|
|
249
|
+
let epoch = job.progress.epoch;
|
|
250
|
+
let epochs = job.progress.epochs;
|
|
251
|
+
let etaSec = job.progress.etaSec;
|
|
252
|
+
const series = [...job.progress.series];
|
|
253
|
+
const tail = [...job.logTail];
|
|
254
|
+
for (const text of splitLogLines(lines)) {
|
|
255
|
+
tail.push(clip(text));
|
|
256
|
+
const event = require_dist.parseTrainingEventLine(text);
|
|
257
|
+
if (event === null) continue;
|
|
258
|
+
if (event.type === "stage") stage = event.stage;
|
|
259
|
+
if (event.type === "epoch") {
|
|
260
|
+
epoch = event.epoch;
|
|
261
|
+
epochs = event.epochs;
|
|
262
|
+
etaSec = event.etaSec ?? null;
|
|
263
|
+
const point = {
|
|
264
|
+
epoch: event.epoch,
|
|
265
|
+
map50: event.map50 ?? null,
|
|
266
|
+
map50_95: event.map50_95 ?? null
|
|
267
|
+
};
|
|
268
|
+
if (series.at(-1)?.epoch === point.epoch) series[series.length - 1] = point;
|
|
269
|
+
else series.push(point);
|
|
270
|
+
}
|
|
271
|
+
}
|
|
272
|
+
return {
|
|
273
|
+
stage,
|
|
274
|
+
progress: {
|
|
275
|
+
epoch,
|
|
276
|
+
epochs,
|
|
277
|
+
etaSec,
|
|
278
|
+
series: series.slice(-300)
|
|
279
|
+
},
|
|
280
|
+
logTail: tail.slice(-200)
|
|
281
|
+
};
|
|
282
|
+
}
|
|
283
|
+
//#endregion
|
|
284
|
+
//#region src/orchestrator/job-mutator.ts
|
|
285
|
+
var JobMutator = class {
|
|
286
|
+
jobs;
|
|
287
|
+
now;
|
|
288
|
+
chains = /* @__PURE__ */ new Map();
|
|
289
|
+
constructor(jobs, now) {
|
|
290
|
+
this.jobs = jobs;
|
|
291
|
+
this.now = now;
|
|
292
|
+
}
|
|
293
|
+
/** Apply `fn` to the CURRENT row; `null` from `fn` = write nothing. Returns the row after. */
|
|
294
|
+
async update(id, fn) {
|
|
295
|
+
return this.locked(id, async () => {
|
|
296
|
+
const job = await this.jobs.get(id);
|
|
297
|
+
if (job === null) return null;
|
|
298
|
+
const next = fn(job);
|
|
299
|
+
if (next === null) return job;
|
|
300
|
+
const stamped = {
|
|
301
|
+
...next,
|
|
302
|
+
updatedAt: this.now()
|
|
303
|
+
};
|
|
304
|
+
await this.jobs.put(stamped);
|
|
305
|
+
return stamped;
|
|
306
|
+
});
|
|
307
|
+
}
|
|
308
|
+
/** Run `body` under the job's lock (for a step that must not interleave). */
|
|
309
|
+
locked(id, body) {
|
|
310
|
+
const run = (this.chains.get(id) ?? Promise.resolve()).then(body, body);
|
|
311
|
+
const settled = run.then(() => void 0, () => void 0);
|
|
312
|
+
this.chains.set(id, settled);
|
|
313
|
+
settled.then(() => {
|
|
314
|
+
if (this.chains.get(id) === settled) this.chains.delete(id);
|
|
315
|
+
});
|
|
316
|
+
return run;
|
|
317
|
+
}
|
|
318
|
+
};
|
|
319
|
+
var DETECTOR_CLASSES = [
|
|
320
|
+
"person",
|
|
321
|
+
"vehicle",
|
|
322
|
+
"animal"
|
|
323
|
+
];
|
|
324
|
+
var PRESETS = {
|
|
325
|
+
quick: {
|
|
326
|
+
epochs: 15,
|
|
327
|
+
freeze: 10,
|
|
328
|
+
replay: 800
|
|
329
|
+
},
|
|
330
|
+
standard: {
|
|
331
|
+
epochs: 40,
|
|
332
|
+
freeze: 10,
|
|
333
|
+
replay: 1500
|
|
334
|
+
},
|
|
335
|
+
thorough: {
|
|
336
|
+
epochs: 80,
|
|
337
|
+
freeze: 0,
|
|
338
|
+
replay: 2500
|
|
339
|
+
}
|
|
340
|
+
};
|
|
341
|
+
/** Training images per second on an L4 at 640 (unmeasured first guess). */
|
|
342
|
+
var L4_IMAGES_PER_SEC_640 = {
|
|
343
|
+
t: 180,
|
|
344
|
+
s: 120,
|
|
345
|
+
m: 60,
|
|
346
|
+
c: 45
|
|
347
|
+
};
|
|
348
|
+
/** Throughput relative to an L4 (unmeasured first guess, from public benchmarks). */
|
|
349
|
+
var GPU_SPEED = {
|
|
350
|
+
T4: .6,
|
|
351
|
+
L4: 1,
|
|
352
|
+
A10G: 1.3,
|
|
353
|
+
L40S: 2.2,
|
|
354
|
+
"A100-40GB": 2.5,
|
|
355
|
+
"A100-80GB": 2.6,
|
|
356
|
+
H100: 3.5
|
|
357
|
+
};
|
|
358
|
+
/** Pull, replay download, evaluation and export — a fixed slice of every run. */
|
|
359
|
+
var OVERHEAD_SEC = 900;
|
|
360
|
+
/** A detector frame in the archive (a full-resolution JPEG), on average. */
|
|
361
|
+
var BYTES_PER_FRAME = 350 * 1024;
|
|
362
|
+
function cameraHoldout(stats) {
|
|
363
|
+
const withFrames = stats.cameras.filter((c) => c.completeFrames > 0).toSorted((a, b) => b.completeFrames - a.completeFrames || a.deviceId - b.deviceId);
|
|
364
|
+
if (withFrames.length < 6) return [];
|
|
365
|
+
return withFrames.slice(1, 3).map((c) => c.deviceId);
|
|
366
|
+
}
|
|
367
|
+
function supportOf(stats, cameras) {
|
|
368
|
+
const out = Object.fromEntries(DETECTOR_CLASSES.map((c) => [c, 0]));
|
|
369
|
+
for (const cell of stats.cells) {
|
|
370
|
+
if (!cameras.has(cell.deviceId)) continue;
|
|
371
|
+
out[cell.macroClass] = (out[cell.macroClass] ?? 0) + cell.subjects;
|
|
372
|
+
}
|
|
373
|
+
return out;
|
|
374
|
+
}
|
|
375
|
+
function planSplit(stats, requestedHoldout) {
|
|
376
|
+
const holdout = requestedHoldout ?? cameraHoldout(stats);
|
|
377
|
+
const warnings = [];
|
|
378
|
+
if (holdout.length === 0) {
|
|
379
|
+
warnings.push(`Fewer than ${String(6)} cameras: the last ${String(3)} days of every camera are held out instead of whole cameras.`);
|
|
380
|
+
return {
|
|
381
|
+
split: {
|
|
382
|
+
holdoutCameras: [],
|
|
383
|
+
holdoutLastDays: 3,
|
|
384
|
+
holdoutSupport: null,
|
|
385
|
+
trainSubjects: stats.subjects
|
|
386
|
+
},
|
|
387
|
+
warnings
|
|
388
|
+
};
|
|
389
|
+
}
|
|
390
|
+
const support = supportOf(stats, new Set(holdout));
|
|
391
|
+
const heldSubjects = Object.values(support).reduce((a, b) => a + b, 0);
|
|
392
|
+
for (const cls of DETECTOR_CLASSES) {
|
|
393
|
+
const n = support[cls] ?? 0;
|
|
394
|
+
if (n < 20) warnings.push(`${cls}: ${String(n)} held-out positives < ${String(20)} — its floor will inherit the baseline's.`);
|
|
395
|
+
}
|
|
396
|
+
return {
|
|
397
|
+
split: {
|
|
398
|
+
holdoutCameras: [...holdout],
|
|
399
|
+
holdoutLastDays: 0,
|
|
400
|
+
holdoutSupport: support,
|
|
401
|
+
trainSubjects: Math.max(0, stats.subjects - heldSubjects)
|
|
402
|
+
},
|
|
403
|
+
warnings
|
|
404
|
+
};
|
|
405
|
+
}
|
|
406
|
+
function estimateGpuSeconds(input) {
|
|
407
|
+
const preset = PRESETS[input.preset];
|
|
408
|
+
const images = input.frames + preset.replay;
|
|
409
|
+
const resFactor = (input.base.imgsz / 640) ** 2;
|
|
410
|
+
const throughput = L4_IMAGES_PER_SEC_640[input.base.tier] / resFactor * (GPU_SPEED[input.gpuType] ?? 1);
|
|
411
|
+
const nominal = images * preset.epochs / throughput + OVERHEAD_SEC;
|
|
412
|
+
return {
|
|
413
|
+
gpuSecondsLow: Math.round(nominal * .7),
|
|
414
|
+
gpuSecondsHigh: Math.round(nominal * 1.6)
|
|
415
|
+
};
|
|
416
|
+
}
|
|
417
|
+
function datasetBytesEstimate(frames) {
|
|
418
|
+
return frames * BYTES_PER_FRAME;
|
|
419
|
+
}
|
|
420
|
+
function uploadMinutes(bytes, mbps = 20) {
|
|
421
|
+
return Math.round(bytes * 8 / (mbps * 1e6) / 60 * 10) / 10;
|
|
422
|
+
}
|
|
423
|
+
/**
|
|
424
|
+
* The default cost cap: the provider's own setting (operator decision
|
|
425
|
+
* 2026-10-08 — it depends on the account), else 1.5× the high estimate.
|
|
426
|
+
*/
|
|
427
|
+
function defaultCostCap(usdHigh, providerCap) {
|
|
428
|
+
if (providerCap > 0) return providerCap;
|
|
429
|
+
return Math.max(1, Math.ceil(usdHigh * 1.5 * 100) / 100);
|
|
430
|
+
}
|
|
431
|
+
function knownGpu(gpus, gpuType) {
|
|
432
|
+
return gpus.some((g) => g.id === gpuType);
|
|
433
|
+
}
|
|
434
|
+
//#endregion
|
|
435
|
+
//#region src/orchestrator/job-spec.ts
|
|
436
|
+
/**
|
|
437
|
+
* `job.json` for the trainer, built from a job record — every field filled
|
|
438
|
+
* here, none left to a default (defaults do not run on the addon cap path,
|
|
439
|
+
* and the trainer's spec has none either). Validated against the shared
|
|
440
|
+
* contract before it leaves the hub.
|
|
441
|
+
*/
|
|
442
|
+
/** The trainer's evaluation and gate parameters — one place, one value. */
|
|
443
|
+
var JOB_EVAL_DEFAULTS = {
|
|
444
|
+
valFraction: .15,
|
|
445
|
+
seed: 1,
|
|
446
|
+
batch: 16,
|
|
447
|
+
lr0: .002,
|
|
448
|
+
patience: 10,
|
|
449
|
+
calibrationFrames: 300,
|
|
450
|
+
minSupport: 20,
|
|
451
|
+
floorClamp: .15,
|
|
452
|
+
int8MaxMap50Drop: .02
|
|
453
|
+
};
|
|
454
|
+
/** The catalog id of a job's result: `custom-<base>-<first 8 of the job key>`. */
|
|
455
|
+
function resultModelId(job) {
|
|
456
|
+
return `custom-${job.request.baseModelId}-${job.id.replaceAll("-", "").slice(0, 8)}`;
|
|
457
|
+
}
|
|
458
|
+
function buildJobSpec(input) {
|
|
459
|
+
const { job, base } = input;
|
|
460
|
+
const preset = PRESETS[job.request.preset];
|
|
461
|
+
const spec = {
|
|
462
|
+
schema: require_dist.TRAINING_JOB_SCHEMA,
|
|
463
|
+
jobKey: job.id,
|
|
464
|
+
target: {
|
|
465
|
+
kind: "detector",
|
|
466
|
+
step: "object-detection"
|
|
467
|
+
},
|
|
468
|
+
base: {
|
|
469
|
+
modelId: base.modelId,
|
|
470
|
+
weights: base.weights,
|
|
471
|
+
imgsz: base.imgsz
|
|
472
|
+
},
|
|
473
|
+
baseline: {
|
|
474
|
+
modelId: base.modelId,
|
|
475
|
+
onnxUrl: base.baselineOnnxUrl,
|
|
476
|
+
imgsz: base.imgsz,
|
|
477
|
+
floors: { ...base.baselineFloors }
|
|
478
|
+
},
|
|
479
|
+
dataset: {
|
|
480
|
+
versionId: input.versionId,
|
|
481
|
+
annotationHash: input.annotationHash,
|
|
482
|
+
sha256: input.datasetSha256
|
|
483
|
+
},
|
|
484
|
+
split: {
|
|
485
|
+
holdoutCameras: [...job.split.holdoutCameras],
|
|
486
|
+
holdoutLastDays: job.split.holdoutLastDays,
|
|
487
|
+
valFraction: JOB_EVAL_DEFAULTS.valFraction,
|
|
488
|
+
seed: JOB_EVAL_DEFAULTS.seed
|
|
489
|
+
},
|
|
490
|
+
hyper: {
|
|
491
|
+
epochs: preset.epochs,
|
|
492
|
+
batch: JOB_EVAL_DEFAULTS.batch,
|
|
493
|
+
lr0: JOB_EVAL_DEFAULTS.lr0,
|
|
494
|
+
freeze: job.request.freeze ?? preset.freeze,
|
|
495
|
+
patience: JOB_EVAL_DEFAULTS.patience,
|
|
496
|
+
seed: JOB_EVAL_DEFAULTS.seed
|
|
497
|
+
},
|
|
498
|
+
replay: {
|
|
499
|
+
source: "coco2017",
|
|
500
|
+
count: preset.replay
|
|
501
|
+
},
|
|
502
|
+
exports: [
|
|
503
|
+
"onnx",
|
|
504
|
+
"openvino-fp16",
|
|
505
|
+
"openvino-int8",
|
|
506
|
+
"coreml-fp16"
|
|
507
|
+
],
|
|
508
|
+
calibration: { maxFrames: JOB_EVAL_DEFAULTS.calibrationFrames },
|
|
509
|
+
floors: {
|
|
510
|
+
minSupport: JOB_EVAL_DEFAULTS.minSupport,
|
|
511
|
+
clamp: JOB_EVAL_DEFAULTS.floorClamp
|
|
512
|
+
},
|
|
513
|
+
int8Gate: { maxMap50Drop: JOB_EVAL_DEFAULTS.int8MaxMap50Drop },
|
|
514
|
+
catalog: {
|
|
515
|
+
id: resultModelId(job),
|
|
516
|
+
name: `${base.label} — ${job.name}`
|
|
517
|
+
}
|
|
518
|
+
};
|
|
519
|
+
return require_dist.TrainingJobSpecSchema.parse(spec);
|
|
520
|
+
}
|
|
521
|
+
function refused(reason, detail, dataNotice = "") {
|
|
522
|
+
return {
|
|
523
|
+
plan: {
|
|
524
|
+
refusal: {
|
|
525
|
+
reason,
|
|
526
|
+
detail
|
|
527
|
+
},
|
|
528
|
+
warnings: [],
|
|
529
|
+
split: null,
|
|
530
|
+
estimate: null,
|
|
531
|
+
defaults: null,
|
|
532
|
+
dataNotice
|
|
533
|
+
},
|
|
534
|
+
accepted: null
|
|
535
|
+
};
|
|
536
|
+
}
|
|
537
|
+
function dataNoticeFor(provider, region, frames) {
|
|
538
|
+
const where = region === "" ? "in the region the provider picks" : `in region "${region}"`;
|
|
539
|
+
const grace = provider.defaults.failedDatasetGraceHours;
|
|
540
|
+
return `${String(frames)} frames and their boxes are uploaded to YOUR ${provider.label} account, ${where}. ${provider.dataResidencyNote} They are deleted there as soon as the result is fetched (${String(grace)} h after a failure, so it can be inspected), and they are never put in a log, an environment variable or the job file. The result stays on this hub for ${String(14)} days; nothing is registered until you click Register.`;
|
|
541
|
+
}
|
|
542
|
+
async function planJob(request, deps) {
|
|
543
|
+
const version = await deps.versions.get(request.datasetVersionId);
|
|
544
|
+
if (version === null) return refused("unknown-dataset", `dataset version "${request.datasetVersionId}" does not exist`);
|
|
545
|
+
const provider = (await deps.provider.describe({})).find((p) => p.addonId === request.addonId);
|
|
546
|
+
if (provider === void 0) return refused("unknown-provider", `no training provider "${request.addonId}"`);
|
|
547
|
+
const notice = dataNoticeFor(provider, request.region, version.frameIds.length);
|
|
548
|
+
if (!provider.configured) return refused("provider-not-configured", `${provider.label} has no credentials yet — set them in its settings`, notice);
|
|
549
|
+
const base = findBaseModel(request.baseModelId);
|
|
550
|
+
if (base === null) return refused("unknown-base-model", `"${request.baseModelId}" is not a trainable base`, notice);
|
|
551
|
+
if (!knownGpu(provider.gpuTypes, request.gpuType)) return refused("unknown-gpu", `${provider.label} offers no "${request.gpuType}"`, notice);
|
|
552
|
+
const datasetBytes = datasetBytesEstimate(version.frameIds.length);
|
|
553
|
+
if (datasetBytes > provider.maxDatasetBytes) return refused("dataset-too-large", `about ${String(Math.round(datasetBytes / 1e6))} MB is over ${provider.label}'s limit`, notice);
|
|
554
|
+
const { split, warnings } = planSplit(version.stats, request.holdoutCameras);
|
|
555
|
+
if (split.trainSubjects <= 0) return refused("no-training-subjects", "no boxed subject is left to train on once the holdout is set aside", notice);
|
|
556
|
+
const gpu = estimateGpuSeconds({
|
|
557
|
+
frames: version.frameIds.length,
|
|
558
|
+
base,
|
|
559
|
+
preset: request.preset,
|
|
560
|
+
gpuType: request.gpuType
|
|
561
|
+
});
|
|
562
|
+
const usd = await deps.provider.estimateJob({
|
|
563
|
+
addonId: request.addonId,
|
|
564
|
+
gpuType: request.gpuType,
|
|
565
|
+
region: request.region,
|
|
566
|
+
datasetBytes,
|
|
567
|
+
gpuSecondsLow: gpu.gpuSecondsLow,
|
|
568
|
+
gpuSecondsHigh: gpu.gpuSecondsHigh
|
|
569
|
+
});
|
|
570
|
+
const maxCostUsd = defaultCostCap(usd.usdHigh, provider.defaults.costCapUsd);
|
|
571
|
+
const allWarnings = [...warnings];
|
|
572
|
+
if (usd.usdHigh > maxCostUsd) allWarnings.push(`The high estimate ($${usd.usdHigh.toFixed(2)}) is above the cost cap ($${maxCostUsd.toFixed(2)}): the job may be stopped before it finishes.`);
|
|
573
|
+
if (version.stats.completeness < .5) allWarnings.push("Fewer than half of the frames are marked complete: evaluation uses complete frames only, so the comparison rests on few frames.");
|
|
574
|
+
return {
|
|
575
|
+
plan: {
|
|
576
|
+
refusal: null,
|
|
577
|
+
warnings: allWarnings,
|
|
578
|
+
split,
|
|
579
|
+
estimate: {
|
|
580
|
+
datasetBytes,
|
|
581
|
+
uploadMinutesAt20Mbps: uploadMinutes(datasetBytes),
|
|
582
|
+
gpuSecondsLow: gpu.gpuSecondsLow,
|
|
583
|
+
gpuSecondsHigh: gpu.gpuSecondsHigh,
|
|
584
|
+
usdLow: usd.usdLow,
|
|
585
|
+
usdHigh: usd.usdHigh,
|
|
586
|
+
notes: ["GPU time is a first estimate (not yet measured on real runs); the cost cap bounds a wrong guess.", ...usd.notes]
|
|
587
|
+
},
|
|
588
|
+
defaults: {
|
|
589
|
+
maxCostUsd,
|
|
590
|
+
maxRuntimeSec: provider.defaults.maxRuntimeSec
|
|
591
|
+
},
|
|
592
|
+
dataNotice: notice
|
|
593
|
+
},
|
|
594
|
+
accepted: {
|
|
595
|
+
version,
|
|
596
|
+
provider,
|
|
597
|
+
base,
|
|
598
|
+
split
|
|
599
|
+
}
|
|
600
|
+
};
|
|
601
|
+
}
|
|
602
|
+
//#endregion
|
|
603
|
+
//#region src/orchestrator/job-engine.ts
|
|
604
|
+
/**
|
|
605
|
+
* The job state machine (D805): start → upload → submit → poll → collect →
|
|
606
|
+
* ready, with every failure path named and every terminal path cleaning the
|
|
607
|
+
* cloud up.
|
|
608
|
+
*
|
|
609
|
+
* Registry-driven, not timer-driven: the reconcile tick compares every
|
|
610
|
+
* non-terminal job with what the provider says (on boot, then every 30 s), so
|
|
611
|
+
* a hub restart mid-run loses nothing — the job record and the provider's
|
|
612
|
+
* jobKey tag are the whole truth. Background work (the upload, the collect)
|
|
613
|
+
* is tracked in memory; a job found in one of those states with no task
|
|
614
|
+
* behind it was interrupted by a restart and is handled as such.
|
|
615
|
+
*/
|
|
616
|
+
var RECONCILE_INTERVAL_MS = 3e4;
|
|
617
|
+
var ORPHAN_SWEEP_INTERVAL_MS = 1440 * 6e4;
|
|
618
|
+
var ARTIFACT_RETENTION_MS = 14 * (1440 * 6e4);
|
|
619
|
+
/** The provider enforces maxRuntimeSec; past it by this much, the hub cancels too. */
|
|
620
|
+
var RUNTIME_GRACE_MS = 10 * 6e4;
|
|
621
|
+
var EVENT_PAGE = 500;
|
|
622
|
+
var EVENT_PAGES_PER_POLL = 5;
|
|
623
|
+
/** States the reconcile tick polls the provider for. */
|
|
624
|
+
var POLLED = new Set([
|
|
625
|
+
"submitted",
|
|
626
|
+
"starting",
|
|
627
|
+
"running",
|
|
628
|
+
"cancelling"
|
|
629
|
+
]);
|
|
630
|
+
/** States from which nothing more happens in the cloud. */
|
|
631
|
+
var SETTLED = new Set([
|
|
632
|
+
"ready",
|
|
633
|
+
"registered",
|
|
634
|
+
"canary",
|
|
635
|
+
"failed",
|
|
636
|
+
"cancelled",
|
|
637
|
+
"lost",
|
|
638
|
+
"expired",
|
|
639
|
+
"discarded"
|
|
640
|
+
]);
|
|
641
|
+
/** A job in one of these holds (or is about to hold) a GPU. */
|
|
642
|
+
var ACTIVE = new Set([
|
|
643
|
+
"uploading",
|
|
644
|
+
"submitted",
|
|
645
|
+
"starting",
|
|
646
|
+
"running",
|
|
647
|
+
"collecting",
|
|
648
|
+
"cancelling"
|
|
649
|
+
]);
|
|
650
|
+
function errText(err) {
|
|
651
|
+
return err instanceof Error ? err.message : String(err);
|
|
652
|
+
}
|
|
653
|
+
var TrainingEngine = class {
|
|
654
|
+
ports;
|
|
655
|
+
trainer;
|
|
656
|
+
mutator;
|
|
657
|
+
tasks = /* @__PURE__ */ new Map();
|
|
658
|
+
constructor(ports, trainer) {
|
|
659
|
+
this.ports = ports;
|
|
660
|
+
this.trainer = trainer;
|
|
661
|
+
this.mutator = new JobMutator(ports.jobs, ports.now);
|
|
662
|
+
}
|
|
663
|
+
/** Await every background task (tests, shutdown). */
|
|
664
|
+
async settle() {
|
|
665
|
+
while (this.tasks.size > 0) await Promise.allSettled([...this.tasks.values()]);
|
|
666
|
+
}
|
|
667
|
+
async startJob(input) {
|
|
668
|
+
const existing = await this.ports.jobs.findByIdempotencyKey(input.idempotencyKey);
|
|
669
|
+
if (existing !== null) return {
|
|
670
|
+
kind: "started",
|
|
671
|
+
job: existing
|
|
672
|
+
};
|
|
673
|
+
const active = (await this.ports.jobs.list()).find((j) => ACTIVE.has(j.state));
|
|
674
|
+
if (active !== void 0) return {
|
|
675
|
+
kind: "refused",
|
|
676
|
+
reason: "job-already-running",
|
|
677
|
+
detail: `"${active.name}" is still ${active.state}`
|
|
678
|
+
};
|
|
679
|
+
const planned = await planJob(input, {
|
|
680
|
+
versions: this.ports.versions,
|
|
681
|
+
provider: this.ports.provider
|
|
682
|
+
});
|
|
683
|
+
if (planned.plan.refusal !== null || planned.accepted === null) {
|
|
684
|
+
const refusal = planned.plan.refusal ?? {
|
|
685
|
+
reason: "unknown-dataset",
|
|
686
|
+
detail: "no plan"
|
|
687
|
+
};
|
|
688
|
+
return {
|
|
689
|
+
kind: "refused",
|
|
690
|
+
reason: refusal.reason,
|
|
691
|
+
detail: refusal.detail
|
|
692
|
+
};
|
|
693
|
+
}
|
|
694
|
+
const { version, split } = planned.accepted;
|
|
695
|
+
const now = this.ports.now();
|
|
696
|
+
const { name, maxCostUsd, maxRuntimeSec, idempotencyKey, ...request } = input;
|
|
697
|
+
const job = {
|
|
698
|
+
id: this.ports.newId(),
|
|
699
|
+
idempotencyKey,
|
|
700
|
+
name,
|
|
701
|
+
createdAt: now,
|
|
702
|
+
updatedAt: now,
|
|
703
|
+
state: "uploading",
|
|
704
|
+
stage: null,
|
|
705
|
+
failure: null,
|
|
706
|
+
request,
|
|
707
|
+
maxCostUsd,
|
|
708
|
+
maxRuntimeSec,
|
|
709
|
+
trainer: this.trainer,
|
|
710
|
+
datasetLayout: version.layout,
|
|
711
|
+
dataset: null,
|
|
712
|
+
split,
|
|
713
|
+
provider: {
|
|
714
|
+
state: null,
|
|
715
|
+
providerJobId: null,
|
|
716
|
+
startedAt: null,
|
|
717
|
+
endedAt: null,
|
|
718
|
+
billedSeconds: 0,
|
|
719
|
+
costUsd: 0,
|
|
720
|
+
costIsEstimate: true,
|
|
721
|
+
lastPolledAt: null
|
|
722
|
+
},
|
|
723
|
+
progress: {
|
|
724
|
+
epoch: null,
|
|
725
|
+
epochs: null,
|
|
726
|
+
etaSec: null,
|
|
727
|
+
series: []
|
|
728
|
+
},
|
|
729
|
+
logTail: [],
|
|
730
|
+
logCursor: null,
|
|
731
|
+
artifacts: null,
|
|
732
|
+
cloudDatasetDeletedAt: null,
|
|
733
|
+
cloudCleanupDueAt: null,
|
|
734
|
+
registeredModelIds: [],
|
|
735
|
+
canary: null
|
|
736
|
+
};
|
|
737
|
+
await this.ports.jobs.put(job);
|
|
738
|
+
this.ports.logger.info("training job started", { meta: {
|
|
739
|
+
jobId: job.id,
|
|
740
|
+
provider: request.addonId,
|
|
741
|
+
gpu: request.gpuType,
|
|
742
|
+
frames: version.frameIds.length
|
|
743
|
+
} });
|
|
744
|
+
this.runTask(job.id, () => this.uploadAndSubmit(job.id));
|
|
745
|
+
return {
|
|
746
|
+
kind: "started",
|
|
747
|
+
job
|
|
748
|
+
};
|
|
749
|
+
}
|
|
750
|
+
async cancelJob(id) {
|
|
751
|
+
const job = await this.ports.jobs.get(id);
|
|
752
|
+
if (job === null) throw new Error(`training job "${id}" does not exist`);
|
|
753
|
+
if (!ACTIVE.has(job.state) || job.state === "cancelling") return job;
|
|
754
|
+
const moved = await this.mutator.update(id, (j) => ({
|
|
755
|
+
...j,
|
|
756
|
+
state: "cancelling"
|
|
757
|
+
}));
|
|
758
|
+
if (job.state !== "uploading") try {
|
|
759
|
+
await this.ports.provider.cancelJob({
|
|
760
|
+
addonId: job.request.addonId,
|
|
761
|
+
jobKey: id
|
|
762
|
+
});
|
|
763
|
+
} catch (err) {
|
|
764
|
+
this.ports.logger.warn("training cancel call failed — the next reconcile retries", { meta: {
|
|
765
|
+
jobId: id,
|
|
766
|
+
error: errText(err)
|
|
767
|
+
} });
|
|
768
|
+
}
|
|
769
|
+
return moved ?? job;
|
|
770
|
+
}
|
|
771
|
+
/** One reconcile pass over every job. Never throws; each job fails alone. */
|
|
772
|
+
async tick() {
|
|
773
|
+
const jobs = await this.ports.jobs.list();
|
|
774
|
+
const now = this.ports.now();
|
|
775
|
+
for (const job of jobs) try {
|
|
776
|
+
await this.reconcile(job, now);
|
|
777
|
+
} catch (err) {
|
|
778
|
+
this.ports.logger.warn("training reconcile failed for one job", { meta: {
|
|
779
|
+
jobId: job.id,
|
|
780
|
+
state: job.state,
|
|
781
|
+
error: errText(err)
|
|
782
|
+
} });
|
|
783
|
+
}
|
|
784
|
+
}
|
|
785
|
+
/** Cloud jobs we hold no live record of are cancelled and cleaned (they burn credit). */
|
|
786
|
+
async sweepOrphans() {
|
|
787
|
+
const known = new Map((await this.ports.jobs.list()).map((j) => [j.id, j]));
|
|
788
|
+
const swept = [];
|
|
789
|
+
for (const provider of await this.ports.provider.describe({})) {
|
|
790
|
+
if (!provider.configured) continue;
|
|
791
|
+
for (const status of await this.ports.provider.listJobs({ addonId: provider.addonId })) {
|
|
792
|
+
const record = known.get(status.jobKey);
|
|
793
|
+
const live = record !== void 0 && !SETTLED.has(record.state);
|
|
794
|
+
const stillRunning = status.state === "pending" || status.state === "starting" || status.state === "running";
|
|
795
|
+
if (live || record !== void 0 && !stillRunning) continue;
|
|
796
|
+
this.ports.logger.warn("training orphan swept — a cloud job with no live record", { meta: {
|
|
797
|
+
jobKey: status.jobKey,
|
|
798
|
+
provider: provider.addonId,
|
|
799
|
+
state: status.state
|
|
800
|
+
} });
|
|
801
|
+
await this.ports.provider.cancelJob({
|
|
802
|
+
addonId: provider.addonId,
|
|
803
|
+
jobKey: status.jobKey
|
|
804
|
+
});
|
|
805
|
+
await this.ports.provider.cleanupJob({
|
|
806
|
+
addonId: provider.addonId,
|
|
807
|
+
jobKey: status.jobKey
|
|
808
|
+
});
|
|
809
|
+
swept.push(status.jobKey);
|
|
810
|
+
}
|
|
811
|
+
}
|
|
812
|
+
return swept;
|
|
813
|
+
}
|
|
814
|
+
runTask(id, body) {
|
|
815
|
+
if (this.tasks.has(id)) return;
|
|
816
|
+
const task = body().catch((err) => {
|
|
817
|
+
this.ports.logger.warn("training background step failed", { meta: {
|
|
818
|
+
jobId: id,
|
|
819
|
+
error: errText(err)
|
|
820
|
+
} });
|
|
821
|
+
}).finally(() => this.tasks.delete(id));
|
|
822
|
+
this.tasks.set(id, task);
|
|
823
|
+
}
|
|
824
|
+
async uploadAndSubmit(id) {
|
|
825
|
+
const job = await this.ports.jobs.get(id);
|
|
826
|
+
if (job === null) return;
|
|
827
|
+
const version = await this.ports.versions.get(job.request.datasetVersionId);
|
|
828
|
+
const base = findBaseModel(job.request.baseModelId);
|
|
829
|
+
if (version === null || base === null) {
|
|
830
|
+
await this.fail(id, "upload-failed", null, "the dataset version or base model disappeared");
|
|
831
|
+
return;
|
|
832
|
+
}
|
|
833
|
+
let uploaded;
|
|
834
|
+
try {
|
|
835
|
+
const offer = await this.ports.datasets.offerArchive({
|
|
836
|
+
filter: {
|
|
837
|
+
frameIds: [...version.frameIds],
|
|
838
|
+
...version.completeOnly ? { completeOnly: true } : {}
|
|
839
|
+
},
|
|
840
|
+
layout: version.layout
|
|
841
|
+
});
|
|
842
|
+
uploaded = await this.ports.provider.uploadDataset({
|
|
843
|
+
addonId: job.request.addonId,
|
|
844
|
+
jobKey: id,
|
|
845
|
+
ticket: offer.ticket
|
|
846
|
+
});
|
|
847
|
+
} catch (err) {
|
|
848
|
+
await this.fail(id, "upload-failed", "prepare", errText(err));
|
|
849
|
+
await this.cleanupCloud(id);
|
|
850
|
+
return;
|
|
851
|
+
}
|
|
852
|
+
const afterUpload = await this.mutator.update(id, (j) => ({
|
|
853
|
+
...j,
|
|
854
|
+
dataset: {
|
|
855
|
+
bytes: uploaded.bytes,
|
|
856
|
+
sha256: uploaded.sha256,
|
|
857
|
+
uploadedAt: this.ports.now()
|
|
858
|
+
}
|
|
859
|
+
}));
|
|
860
|
+
if (afterUpload?.state === "cancelling") {
|
|
861
|
+
await this.finishCancelled(id);
|
|
862
|
+
return;
|
|
863
|
+
}
|
|
864
|
+
try {
|
|
865
|
+
const spec = buildJobSpec({
|
|
866
|
+
job: afterUpload ?? job,
|
|
867
|
+
base,
|
|
868
|
+
versionId: version.id,
|
|
869
|
+
annotationHash: version.annotationHash,
|
|
870
|
+
datasetSha256: uploaded.sha256
|
|
871
|
+
});
|
|
872
|
+
const status = await this.ports.provider.submitJob({
|
|
873
|
+
addonId: job.request.addonId,
|
|
874
|
+
jobKey: id,
|
|
875
|
+
trainer: job.trainer,
|
|
876
|
+
jobSpecJson: JSON.stringify(spec),
|
|
877
|
+
gpuType: job.request.gpuType,
|
|
878
|
+
region: job.request.region,
|
|
879
|
+
maxRuntimeSec: job.maxRuntimeSec
|
|
880
|
+
});
|
|
881
|
+
await this.mutator.update(id, (j) => this.withStatus({
|
|
882
|
+
...j,
|
|
883
|
+
state: "submitted"
|
|
884
|
+
}, status));
|
|
885
|
+
} catch (err) {
|
|
886
|
+
await this.fail(id, "submit-failed", "prepare", errText(err));
|
|
887
|
+
await this.cleanupCloud(id);
|
|
888
|
+
}
|
|
889
|
+
}
|
|
890
|
+
async reconcile(job, now) {
|
|
891
|
+
if (job.state === "uploading" && !this.tasks.has(job.id)) {
|
|
892
|
+
await this.fail(job.id, "hub-restart-during-upload", "prepare", "the hub restarted while the dataset was uploading");
|
|
893
|
+
await this.cleanupCloud(job.id);
|
|
894
|
+
return;
|
|
895
|
+
}
|
|
896
|
+
if (POLLED.has(job.state)) {
|
|
897
|
+
await this.poll(job, now);
|
|
898
|
+
return;
|
|
899
|
+
}
|
|
900
|
+
if (job.state === "collecting") {
|
|
901
|
+
this.runTask(job.id, () => this.collect(job.id));
|
|
902
|
+
return;
|
|
903
|
+
}
|
|
904
|
+
await this.housekeep(job, now);
|
|
905
|
+
}
|
|
906
|
+
async housekeep(job, now) {
|
|
907
|
+
if (job.cloudDatasetDeletedAt === null && SETTLED.has(job.state)) {
|
|
908
|
+
if (job.state !== "failed" || (job.cloudCleanupDueAt ?? 0) <= now) await this.cleanupCloud(job.id);
|
|
909
|
+
}
|
|
910
|
+
if (job.artifacts !== null && job.artifacts.expiresAt <= now) {
|
|
911
|
+
await this.ports.artifacts.remove(job.id);
|
|
912
|
+
await this.mutator.update(job.id, (j) => ({
|
|
913
|
+
...j,
|
|
914
|
+
artifacts: null,
|
|
915
|
+
state: j.state === "ready" ? "expired" : j.state
|
|
916
|
+
}));
|
|
917
|
+
this.ports.logger.info("training artefacts expired from the hub", { meta: { jobId: job.id } });
|
|
918
|
+
}
|
|
919
|
+
}
|
|
920
|
+
async poll(job, now) {
|
|
921
|
+
const ref = {
|
|
922
|
+
addonId: job.request.addonId,
|
|
923
|
+
jobKey: job.id
|
|
924
|
+
};
|
|
925
|
+
const status = await this.ports.provider.getJob(ref);
|
|
926
|
+
let cursor = job.logCursor ?? void 0;
|
|
927
|
+
let progressed = {
|
|
928
|
+
stage: job.stage,
|
|
929
|
+
progress: job.progress,
|
|
930
|
+
logTail: job.logTail
|
|
931
|
+
};
|
|
932
|
+
for (let page = 0; page < EVENT_PAGES_PER_POLL; page += 1) {
|
|
933
|
+
const batch = await this.ports.provider.listJobEvents({
|
|
934
|
+
...ref,
|
|
935
|
+
max: EVENT_PAGE,
|
|
936
|
+
...cursor !== void 0 ? { cursor } : {}
|
|
937
|
+
});
|
|
938
|
+
progressed = applyLogLines(progressed, batch.lines);
|
|
939
|
+
cursor = batch.cursor;
|
|
940
|
+
if (batch.lines.length < EVENT_PAGE) break;
|
|
941
|
+
}
|
|
942
|
+
const updated = await this.mutator.update(job.id, (j) => this.withStatus({
|
|
943
|
+
...j,
|
|
944
|
+
...progressed,
|
|
945
|
+
logCursor: cursor ?? null
|
|
946
|
+
}, status));
|
|
947
|
+
if (updated === null) return;
|
|
948
|
+
await this.afterPoll(updated, status, now);
|
|
949
|
+
}
|
|
950
|
+
async afterPoll(job, status, now) {
|
|
951
|
+
const ref = {
|
|
952
|
+
addonId: job.request.addonId,
|
|
953
|
+
jobKey: job.id
|
|
954
|
+
};
|
|
955
|
+
const running = status.state === "pending" || status.state === "starting" || status.state === "running";
|
|
956
|
+
if (job.state === "cancelling") {
|
|
957
|
+
if (!running) await this.finishCancelled(job.id);
|
|
958
|
+
return;
|
|
959
|
+
}
|
|
960
|
+
if (running && job.provider.costUsd >= job.maxCostUsd) {
|
|
961
|
+
await this.ports.provider.cancelJob(ref);
|
|
962
|
+
await this.fail(job.id, "cost-cap", job.stage, `spent $${job.provider.costUsd.toFixed(2)} of a $${job.maxCostUsd.toFixed(2)} cap`);
|
|
963
|
+
return;
|
|
964
|
+
}
|
|
965
|
+
const started = job.provider.startedAt;
|
|
966
|
+
if (running && started !== null && now - started > job.maxRuntimeSec * 1e3 + RUNTIME_GRACE_MS) {
|
|
967
|
+
await this.ports.provider.cancelJob(ref);
|
|
968
|
+
await this.fail(job.id, "runtime-cap", job.stage, `ran past ${String(job.maxRuntimeSec)} s`);
|
|
969
|
+
return;
|
|
970
|
+
}
|
|
971
|
+
if (status.state === "succeeded") {
|
|
972
|
+
await this.mutator.update(job.id, (j) => ({
|
|
973
|
+
...j,
|
|
974
|
+
state: "collecting"
|
|
975
|
+
}));
|
|
976
|
+
this.runTask(job.id, () => this.collect(job.id));
|
|
977
|
+
return;
|
|
978
|
+
}
|
|
979
|
+
if (status.state === "failed") {
|
|
980
|
+
const detail = await this.failureDetail(job);
|
|
981
|
+
await this.fail(job.id, "trainer-failed", job.stage, detail ?? status.failureReason ?? `exit ${String(status.exitCode ?? "?")}`);
|
|
982
|
+
return;
|
|
983
|
+
}
|
|
984
|
+
if (status.state === "cancelled") {
|
|
985
|
+
await this.fail(job.id, "provider-error", job.stage, "the provider stopped the job");
|
|
986
|
+
return;
|
|
987
|
+
}
|
|
988
|
+
if (status.state === "lost") await this.mutator.update(job.id, (j) => ({
|
|
989
|
+
...j,
|
|
990
|
+
state: "lost"
|
|
991
|
+
}));
|
|
992
|
+
}
|
|
993
|
+
async failureDetail(job) {
|
|
994
|
+
try {
|
|
995
|
+
const raw = await this.ports.provider.readResult({
|
|
996
|
+
addonId: job.request.addonId,
|
|
997
|
+
jobKey: job.id
|
|
998
|
+
});
|
|
999
|
+
if (!raw.found || raw.text === void 0) return null;
|
|
1000
|
+
const parsed = require_dist.TrainingResultSchema.safeParse(JSON.parse(raw.text));
|
|
1001
|
+
const error = parsed.success ? parsed.data.error : null;
|
|
1002
|
+
return error === null ? null : `${error.stage}: ${error.code} — ${error.message}`;
|
|
1003
|
+
} catch {
|
|
1004
|
+
return null;
|
|
1005
|
+
}
|
|
1006
|
+
}
|
|
1007
|
+
async collect(id) {
|
|
1008
|
+
const job = await this.ports.jobs.get(id);
|
|
1009
|
+
if (job === null || job.state !== "collecting") return;
|
|
1010
|
+
const ref = {
|
|
1011
|
+
addonId: job.request.addonId,
|
|
1012
|
+
jobKey: id
|
|
1013
|
+
};
|
|
1014
|
+
try {
|
|
1015
|
+
const raw = await this.ports.provider.readResult(ref);
|
|
1016
|
+
if (!raw.found || raw.text === void 0) {
|
|
1017
|
+
await this.fail(id, "result-invalid", "package", "the job ended without a result.json");
|
|
1018
|
+
return;
|
|
1019
|
+
}
|
|
1020
|
+
const parsed = require_dist.TrainingResultSchema.safeParse(JSON.parse(raw.text));
|
|
1021
|
+
if (!parsed.success) {
|
|
1022
|
+
await this.fail(id, "result-invalid", "package", parsed.error.issues[0]?.message ?? "unreadable");
|
|
1023
|
+
return;
|
|
1024
|
+
}
|
|
1025
|
+
const result = parsed.data;
|
|
1026
|
+
if (result.status !== "succeeded" || result.bundle == null || result.catalogDraft == null || result.files === void 0) {
|
|
1027
|
+
const why = result.error === null ? "the result names no bundle" : `${result.error.stage}: ${result.error.message}`;
|
|
1028
|
+
await this.fail(id, result.status === "failed" ? "trainer-failed" : "result-invalid", "package", why);
|
|
1029
|
+
return;
|
|
1030
|
+
}
|
|
1031
|
+
const { ticket } = await this.ports.provider.offerArtifact({
|
|
1032
|
+
...ref,
|
|
1033
|
+
path: "bundle.tar"
|
|
1034
|
+
});
|
|
1035
|
+
const got = await this.ports.artifacts.receiveBundle(id, ticket, result.bundle.bytes);
|
|
1036
|
+
if (got.sha256 !== result.bundle.sha256 || got.bytes !== result.bundle.bytes) {
|
|
1037
|
+
await this.ports.artifacts.remove(id);
|
|
1038
|
+
await this.fail(id, "checksum-mismatch", "package", "the fetched bundle does not match result.json");
|
|
1039
|
+
return;
|
|
1040
|
+
}
|
|
1041
|
+
await this.ports.artifacts.writeResult(id, raw.text);
|
|
1042
|
+
const now = this.ports.now();
|
|
1043
|
+
await this.mutator.update(id, (j) => ({
|
|
1044
|
+
...j,
|
|
1045
|
+
state: "ready",
|
|
1046
|
+
artifacts: {
|
|
1047
|
+
bundleBytes: got.bytes,
|
|
1048
|
+
bundleSha256: got.sha256,
|
|
1049
|
+
fetchedAt: now,
|
|
1050
|
+
expiresAt: now + ARTIFACT_RETENTION_MS
|
|
1051
|
+
}
|
|
1052
|
+
}));
|
|
1053
|
+
this.ports.logger.info("training result fetched and verified", { meta: {
|
|
1054
|
+
jobId: id,
|
|
1055
|
+
bytes: got.bytes
|
|
1056
|
+
} });
|
|
1057
|
+
await this.cleanupCloud(id);
|
|
1058
|
+
} catch (err) {
|
|
1059
|
+
await this.fail(id, "fetch-failed", "package", errText(err));
|
|
1060
|
+
}
|
|
1061
|
+
}
|
|
1062
|
+
async finishCancelled(id) {
|
|
1063
|
+
await this.mutator.update(id, (j) => ({
|
|
1064
|
+
...j,
|
|
1065
|
+
state: "cancelled"
|
|
1066
|
+
}));
|
|
1067
|
+
await this.cleanupCloud(id);
|
|
1068
|
+
}
|
|
1069
|
+
/** Delete the cloud copy (dataset first). A failure is retried by the next tick. */
|
|
1070
|
+
async cleanupCloud(id) {
|
|
1071
|
+
const job = await this.ports.jobs.get(id);
|
|
1072
|
+
if (job === null || job.cloudDatasetDeletedAt !== null) return;
|
|
1073
|
+
try {
|
|
1074
|
+
const { deleted } = await this.ports.provider.cleanupJob({
|
|
1075
|
+
addonId: job.request.addonId,
|
|
1076
|
+
jobKey: id
|
|
1077
|
+
});
|
|
1078
|
+
await this.mutator.update(id, (j) => ({
|
|
1079
|
+
...j,
|
|
1080
|
+
cloudDatasetDeletedAt: this.ports.now(),
|
|
1081
|
+
cloudCleanupDueAt: null
|
|
1082
|
+
}));
|
|
1083
|
+
this.ports.logger.info("training cloud copy deleted", { meta: {
|
|
1084
|
+
jobId: id,
|
|
1085
|
+
deleted: [...deleted]
|
|
1086
|
+
} });
|
|
1087
|
+
} catch (err) {
|
|
1088
|
+
this.ports.logger.warn("training cloud cleanup failed — retried on the next reconcile", { meta: {
|
|
1089
|
+
jobId: id,
|
|
1090
|
+
error: errText(err)
|
|
1091
|
+
} });
|
|
1092
|
+
}
|
|
1093
|
+
}
|
|
1094
|
+
async fail(id, reason, stage, detail) {
|
|
1095
|
+
let grace = 0;
|
|
1096
|
+
const job = await this.ports.jobs.get(id);
|
|
1097
|
+
if (job !== null && reason === "trainer-failed") grace = ((await this.ports.provider.describe({})).find((p) => p.addonId === job.request.addonId)?.defaults.failedDatasetGraceHours ?? 0) * 60 * 6e4;
|
|
1098
|
+
await this.mutator.update(id, (j) => ({
|
|
1099
|
+
...j,
|
|
1100
|
+
state: "failed",
|
|
1101
|
+
failure: {
|
|
1102
|
+
reason,
|
|
1103
|
+
stage,
|
|
1104
|
+
detail
|
|
1105
|
+
},
|
|
1106
|
+
cloudCleanupDueAt: this.ports.now() + grace
|
|
1107
|
+
}));
|
|
1108
|
+
this.ports.logger.warn("training job failed", { meta: {
|
|
1109
|
+
jobId: id,
|
|
1110
|
+
reason,
|
|
1111
|
+
stage,
|
|
1112
|
+
detail
|
|
1113
|
+
} });
|
|
1114
|
+
}
|
|
1115
|
+
withStatus(job, status) {
|
|
1116
|
+
const mapped = job.state === "cancelling" ? "cancelling" : status.state === "pending" ? "submitted" : status.state === "starting" ? "starting" : status.state === "running" ? "running" : job.state;
|
|
1117
|
+
return {
|
|
1118
|
+
...job,
|
|
1119
|
+
state: mapped,
|
|
1120
|
+
provider: {
|
|
1121
|
+
state: status.state,
|
|
1122
|
+
providerJobId: status.providerJobId ?? job.provider.providerJobId,
|
|
1123
|
+
startedAt: status.startedAt ?? job.provider.startedAt,
|
|
1124
|
+
endedAt: status.endedAt ?? job.provider.endedAt,
|
|
1125
|
+
billedSeconds: status.billedSeconds,
|
|
1126
|
+
costUsd: status.costUsd,
|
|
1127
|
+
costIsEstimate: status.costIsEstimate,
|
|
1128
|
+
lastPolledAt: this.ports.now()
|
|
1129
|
+
}
|
|
1130
|
+
};
|
|
1131
|
+
}
|
|
1132
|
+
};
|
|
1133
|
+
//#endregion
|
|
1134
|
+
//#region src/orchestrator/return-flow.ts
|
|
1135
|
+
/**
|
|
1136
|
+
* The return flow (D805, D806): a verified result → Model Studio → a canary
|
|
1137
|
+
* on chosen cameras → rollback. Every step is an operator click; nothing here
|
|
1138
|
+
* runs on its own (operator decision 2026-10-08: never auto-register).
|
|
1139
|
+
*
|
|
1140
|
+
* Register hands the staged bundle to `custom-model-registry.importModelBundle`
|
|
1141
|
+
* over `ctx.peerBytes`, with the catalog rows completed here (labels, licence,
|
|
1142
|
+
* the base's D741 floors) and every format URL rewritten to
|
|
1143
|
+
* `camstack-local://`. A canary pins the camera's root detector, on every
|
|
1144
|
+
* accelerator it can land on, to the variant that matches what that
|
|
1145
|
+
* accelerator ran (INT8 where the base pin was INT8), and remembers EXACTLY
|
|
1146
|
+
* what it replaced so a rollback restores it.
|
|
1147
|
+
*/
|
|
1148
|
+
var DETECTOR_STEP_ID = "object-detection";
|
|
1149
|
+
var TRAINER_SOURCE_URL = "https://huggingface.co/camstack/camstack-trainer";
|
|
1150
|
+
function localFormat(id, fmt) {
|
|
1151
|
+
if (fmt === void 0) return void 0;
|
|
1152
|
+
return {
|
|
1153
|
+
url: `camstack-local://${id}/${fmt.url}`,
|
|
1154
|
+
sizeMB: fmt.sizeMB,
|
|
1155
|
+
revision: fmt.revision,
|
|
1156
|
+
...fmt.isDirectory !== void 0 ? { isDirectory: fmt.isDirectory } : {},
|
|
1157
|
+
...fmt.files !== void 0 ? { files: [...fmt.files] } : {},
|
|
1158
|
+
...fmt.runtimes !== void 0 ? { runtimes: [...fmt.runtimes] } : {}
|
|
1159
|
+
};
|
|
1160
|
+
}
|
|
1161
|
+
/** A trainer draft completed into a registry row. */
|
|
1162
|
+
function catalogEntryFromDraft(input) {
|
|
1163
|
+
const { draft, job, result, base } = input;
|
|
1164
|
+
const supplied = input.int8 ? base.suppliedLabelFloorsInt8 : base.suppliedLabelFloors;
|
|
1165
|
+
const formats = {};
|
|
1166
|
+
const onnx = localFormat(draft.id, draft.formats.onnx);
|
|
1167
|
+
const openvino = localFormat(draft.id, draft.formats.openvino);
|
|
1168
|
+
const coreml = localFormat(draft.id, draft.formats.coreml);
|
|
1169
|
+
if (onnx !== void 0) formats.onnx = onnx;
|
|
1170
|
+
if (openvino !== void 0) formats.openvino = openvino;
|
|
1171
|
+
if (coreml !== void 0) formats.coreml = coreml;
|
|
1172
|
+
return {
|
|
1173
|
+
id: draft.id,
|
|
1174
|
+
name: draft.name,
|
|
1175
|
+
description: `${draft.description} Fine-tuned from ${base.modelId} by CamStack Training (job ${job.id}, dataset ${job.request.datasetVersionId}).`,
|
|
1176
|
+
inputSize: { ...draft.inputSize },
|
|
1177
|
+
labels: [],
|
|
1178
|
+
preprocessMode: "letterbox",
|
|
1179
|
+
defaultClassFloors: { ...draft.defaultClassFloors },
|
|
1180
|
+
...Object.keys(supplied).length > 0 ? { suppliedLabelFloors: { ...supplied } } : {},
|
|
1181
|
+
license: {
|
|
1182
|
+
weights: "AGPL-3.0-only",
|
|
1183
|
+
code: "AGPL-3.0-only",
|
|
1184
|
+
upstream: {
|
|
1185
|
+
name: "Ultralytics YOLOv9",
|
|
1186
|
+
url: "https://github.com/ultralytics/ultralytics"
|
|
1187
|
+
},
|
|
1188
|
+
attribution: `© Ultralytics, AGPL-3.0 (YOLOv9 weights); ${result.licence}. Corresponding source: ${TRAINER_SOURCE_URL} at revision ${result.trainer.revision}.`,
|
|
1189
|
+
modifications: `fine-tuned on the operator's own frames (dataset ${job.request.datasetVersionId}) and exported by CamStack Training`
|
|
1190
|
+
},
|
|
1191
|
+
formats
|
|
1192
|
+
};
|
|
1193
|
+
}
|
|
1194
|
+
/** Which registered variant each (camera, accelerator) gets: INT8 where the base pin was INT8. */
|
|
1195
|
+
function planCanary(input) {
|
|
1196
|
+
return input.deviceIds.flatMap((deviceId) => input.devices.map((d) => ({
|
|
1197
|
+
deviceId,
|
|
1198
|
+
nodeId: d.nodeId,
|
|
1199
|
+
deviceKey: d.deviceKey,
|
|
1200
|
+
modelId: input.int8Id !== null && d.pinnedModelId?.endsWith("-int8") === true ? input.int8Id : input.baseId
|
|
1201
|
+
})));
|
|
1202
|
+
}
|
|
1203
|
+
var ReturnFlow = class {
|
|
1204
|
+
ports;
|
|
1205
|
+
mutator;
|
|
1206
|
+
constructor(ports, mutator) {
|
|
1207
|
+
this.ports = ports;
|
|
1208
|
+
this.mutator = mutator;
|
|
1209
|
+
}
|
|
1210
|
+
async readResult(job) {
|
|
1211
|
+
if (job.artifacts === null) return null;
|
|
1212
|
+
const text = await this.ports.artifacts.readResult(job.id);
|
|
1213
|
+
if (text === null) return null;
|
|
1214
|
+
const parsed = require_dist.TrainingResultSchema.safeParse(JSON.parse(text));
|
|
1215
|
+
return parsed.success ? parsed.data : null;
|
|
1216
|
+
}
|
|
1217
|
+
async register(id, canaryDeviceIds) {
|
|
1218
|
+
const job = await this.requireJob(id);
|
|
1219
|
+
if (job.state !== "ready") throw new Error(`only a ready result can be registered (this one is ${job.state})`);
|
|
1220
|
+
const result = await this.readResult(job);
|
|
1221
|
+
const base = findBaseModel(job.request.baseModelId);
|
|
1222
|
+
if (result === null || result.catalogDraft == null || result.files === void 0 || job.artifacts === null || base === null) throw new Error("the staged result is missing or unreadable — fetch it again or discard the job");
|
|
1223
|
+
const entries = [catalogEntryFromDraft({
|
|
1224
|
+
draft: result.catalogDraft.base,
|
|
1225
|
+
job,
|
|
1226
|
+
result,
|
|
1227
|
+
base,
|
|
1228
|
+
int8: false
|
|
1229
|
+
})];
|
|
1230
|
+
if (result.catalogDraft.int8 !== null) entries.push(catalogEntryFromDraft({
|
|
1231
|
+
draft: result.catalogDraft.int8,
|
|
1232
|
+
job,
|
|
1233
|
+
result,
|
|
1234
|
+
base,
|
|
1235
|
+
int8: true
|
|
1236
|
+
}));
|
|
1237
|
+
const ticket = await this.ports.artifacts.offerBundle(id);
|
|
1238
|
+
const answer = await this.ports.registry.importBundle({
|
|
1239
|
+
stepId: DETECTOR_STEP_ID,
|
|
1240
|
+
entries,
|
|
1241
|
+
bundle: {
|
|
1242
|
+
ticket,
|
|
1243
|
+
bytes: job.artifacts.bundleBytes,
|
|
1244
|
+
sha256: job.artifacts.bundleSha256
|
|
1245
|
+
},
|
|
1246
|
+
files: result.files.map((f) => ({
|
|
1247
|
+
path: f.path,
|
|
1248
|
+
bytes: f.bytes,
|
|
1249
|
+
sha256: f.sha256
|
|
1250
|
+
})),
|
|
1251
|
+
distribute: true
|
|
1252
|
+
});
|
|
1253
|
+
if (answer.kind === "refused") throw new Error(`Model Studio refused the bundle: ${answer.reason} — ${answer.detail}`);
|
|
1254
|
+
const now = this.ports.now();
|
|
1255
|
+
const registered = await this.mutator.update(id, (j) => ({
|
|
1256
|
+
...j,
|
|
1257
|
+
state: "registered",
|
|
1258
|
+
registeredModelIds: [...answer.modelIds],
|
|
1259
|
+
artifacts: j.artifacts === null ? null : {
|
|
1260
|
+
...j.artifacts,
|
|
1261
|
+
expiresAt: now + ARTIFACT_RETENTION_MS
|
|
1262
|
+
}
|
|
1263
|
+
}));
|
|
1264
|
+
this.ports.logger.info("training result registered", { meta: {
|
|
1265
|
+
jobId: id,
|
|
1266
|
+
modelIds: [...answer.modelIds],
|
|
1267
|
+
distributed: answer.distributed.length
|
|
1268
|
+
} });
|
|
1269
|
+
if (canaryDeviceIds.length > 0) return this.canary(id, canaryDeviceIds);
|
|
1270
|
+
return registered ?? job;
|
|
1271
|
+
}
|
|
1272
|
+
async canary(id, deviceIds) {
|
|
1273
|
+
const job = await this.requireJob(id);
|
|
1274
|
+
if (job.state !== "registered" && job.state !== "canary") throw new Error("register the result before a canary");
|
|
1275
|
+
const [baseId, int8Id] = job.registeredModelIds;
|
|
1276
|
+
if (baseId === void 0) throw new Error("no registered model to canary");
|
|
1277
|
+
const already = new Set(job.canary?.deviceIds ?? []);
|
|
1278
|
+
const fresh = deviceIds.filter((d) => !already.has(d));
|
|
1279
|
+
const writes = planCanary({
|
|
1280
|
+
deviceIds: fresh,
|
|
1281
|
+
devices: await this.ports.pins.listDetectorDevices(),
|
|
1282
|
+
baseId,
|
|
1283
|
+
int8Id: int8Id ?? null
|
|
1284
|
+
});
|
|
1285
|
+
const previous = [];
|
|
1286
|
+
for (const w of writes) {
|
|
1287
|
+
const before = await this.ports.pins.getOverride(w.deviceId, w.nodeId, w.deviceKey);
|
|
1288
|
+
previous.push({
|
|
1289
|
+
deviceId: w.deviceId,
|
|
1290
|
+
nodeId: w.nodeId,
|
|
1291
|
+
deviceKey: w.deviceKey,
|
|
1292
|
+
patch: before === null ? null : { ...before }
|
|
1293
|
+
});
|
|
1294
|
+
await this.ports.pins.setOverride(w.deviceId, w.nodeId, w.deviceKey, { modelId: w.modelId });
|
|
1295
|
+
}
|
|
1296
|
+
const updated = await this.mutator.update(id, (j) => ({
|
|
1297
|
+
...j,
|
|
1298
|
+
state: "canary",
|
|
1299
|
+
canary: {
|
|
1300
|
+
deviceIds: [...already, ...fresh],
|
|
1301
|
+
since: j.canary?.since ?? this.ports.now(),
|
|
1302
|
+
previous: [...j.canary?.previous ?? [], ...previous]
|
|
1303
|
+
}
|
|
1304
|
+
}));
|
|
1305
|
+
this.ports.logger.info("training canary pinned", { meta: {
|
|
1306
|
+
jobId: id,
|
|
1307
|
+
cameras: fresh,
|
|
1308
|
+
writes: writes.length
|
|
1309
|
+
} });
|
|
1310
|
+
return updated ?? job;
|
|
1311
|
+
}
|
|
1312
|
+
async rollback(id) {
|
|
1313
|
+
const job = await this.requireJob(id);
|
|
1314
|
+
if (job.canary === null) return job;
|
|
1315
|
+
for (const p of job.canary.previous) await this.ports.pins.setOverride(p.deviceId, p.nodeId, p.deviceKey, p.patch);
|
|
1316
|
+
const updated = await this.mutator.update(id, (j) => ({
|
|
1317
|
+
...j,
|
|
1318
|
+
state: "registered",
|
|
1319
|
+
canary: null
|
|
1320
|
+
}));
|
|
1321
|
+
this.ports.logger.info("training canary rolled back", { meta: {
|
|
1322
|
+
jobId: id,
|
|
1323
|
+
restored: job.canary.previous.length
|
|
1324
|
+
} });
|
|
1325
|
+
return updated ?? job;
|
|
1326
|
+
}
|
|
1327
|
+
async discard(id) {
|
|
1328
|
+
const job = await this.requireJob(id);
|
|
1329
|
+
if (job.state !== "ready") throw new Error(`only an unregistered result can be discarded (this one is ${job.state})`);
|
|
1330
|
+
await this.ports.artifacts.remove(id);
|
|
1331
|
+
return await this.mutator.update(id, (j) => ({
|
|
1332
|
+
...j,
|
|
1333
|
+
state: "discarded",
|
|
1334
|
+
artifacts: null
|
|
1335
|
+
})) ?? job;
|
|
1336
|
+
}
|
|
1337
|
+
async requireJob(id) {
|
|
1338
|
+
const job = await this.ports.jobs.get(id);
|
|
1339
|
+
if (job === null) throw new Error(`training job "${id}" does not exist`);
|
|
1340
|
+
return job;
|
|
1341
|
+
}
|
|
1342
|
+
};
|
|
1343
|
+
//#endregion
|
|
1344
|
+
//#region src/orchestrator/stores.ts
|
|
1345
|
+
/**
|
|
1346
|
+
* The orchestrator's two durable tables, both in the settings store.
|
|
1347
|
+
*
|
|
1348
|
+
* Rows are validated on the way OUT (a row a newer build wrote, or a hand
|
|
1349
|
+
* edit, never reaches the state machine as a half-shaped job) and dropped
|
|
1350
|
+
* with a log line when they do not parse — a job that cannot be read cannot
|
|
1351
|
+
* be reconciled, and saying so beats acting on a guess.
|
|
1352
|
+
*/
|
|
1353
|
+
/**
|
|
1354
|
+
* @durable class=ledger owner=training
|
|
1355
|
+
* write="one row per training job, rewritten on every state change by the job engine
|
|
1356
|
+
* (upload, submit, each reconcile poll, collect, register, canary, rollback)."
|
|
1357
|
+
* retention="none — a job row is the record of what ran, what it cost and what was
|
|
1358
|
+
* registered; it leaves only when the operator deletes it. Its staged artefacts expire
|
|
1359
|
+
* separately (14 days after ready/registration) and the row says so."
|
|
1360
|
+
*/
|
|
1361
|
+
var TRAINING_JOBS_COLLECTION = "training:jobs";
|
|
1362
|
+
/**
|
|
1363
|
+
* @durable class=ledger owner=training
|
|
1364
|
+
* write="one row per dataset version, inserted once (cut by pipeline-analytics, or moved
|
|
1365
|
+
* from its local store on first sight of this capability) and never updated."
|
|
1366
|
+
* retention="none — rows leave only when the operator deletes the version."
|
|
1367
|
+
*/
|
|
1368
|
+
var TRAINING_DATASET_VERSIONS_COLLECTION = "training:dataset-versions";
|
|
1369
|
+
var JOB_COLUMNS = [
|
|
1370
|
+
{
|
|
1371
|
+
name: "id",
|
|
1372
|
+
type: "TEXT",
|
|
1373
|
+
primaryKey: true,
|
|
1374
|
+
notNull: true
|
|
1375
|
+
},
|
|
1376
|
+
{
|
|
1377
|
+
name: "state",
|
|
1378
|
+
type: "TEXT",
|
|
1379
|
+
notNull: true
|
|
1380
|
+
},
|
|
1381
|
+
{
|
|
1382
|
+
name: "createdAt",
|
|
1383
|
+
type: "INTEGER",
|
|
1384
|
+
notNull: true
|
|
1385
|
+
},
|
|
1386
|
+
{
|
|
1387
|
+
name: "idempotencyKey",
|
|
1388
|
+
type: "TEXT",
|
|
1389
|
+
notNull: true
|
|
1390
|
+
},
|
|
1391
|
+
{
|
|
1392
|
+
name: "job",
|
|
1393
|
+
type: "JSON",
|
|
1394
|
+
notNull: true
|
|
1395
|
+
}
|
|
1396
|
+
];
|
|
1397
|
+
var JOB_INDEXES = [{
|
|
1398
|
+
name: "idx_training_jobs_idem",
|
|
1399
|
+
columns: ["idempotencyKey"],
|
|
1400
|
+
unique: true
|
|
1401
|
+
}, {
|
|
1402
|
+
name: "idx_training_jobs_created",
|
|
1403
|
+
columns: ["createdAt"]
|
|
1404
|
+
}];
|
|
1405
|
+
var VERSION_COLUMNS = [
|
|
1406
|
+
{
|
|
1407
|
+
name: "id",
|
|
1408
|
+
type: "TEXT",
|
|
1409
|
+
primaryKey: true,
|
|
1410
|
+
notNull: true
|
|
1411
|
+
},
|
|
1412
|
+
{
|
|
1413
|
+
name: "nameKey",
|
|
1414
|
+
type: "TEXT",
|
|
1415
|
+
notNull: true
|
|
1416
|
+
},
|
|
1417
|
+
{
|
|
1418
|
+
name: "createdAt",
|
|
1419
|
+
type: "INTEGER",
|
|
1420
|
+
notNull: true
|
|
1421
|
+
},
|
|
1422
|
+
{
|
|
1423
|
+
name: "row",
|
|
1424
|
+
type: "JSON",
|
|
1425
|
+
notNull: true
|
|
1426
|
+
}
|
|
1427
|
+
];
|
|
1428
|
+
var VERSION_INDEXES = [{
|
|
1429
|
+
name: "idx_training_versions_name_key",
|
|
1430
|
+
columns: ["nameKey"],
|
|
1431
|
+
unique: true
|
|
1432
|
+
}, {
|
|
1433
|
+
name: "idx_training_versions_created",
|
|
1434
|
+
columns: ["createdAt"]
|
|
1435
|
+
}];
|
|
1436
|
+
function parseJson(value) {
|
|
1437
|
+
if (typeof value !== "string") return value;
|
|
1438
|
+
try {
|
|
1439
|
+
return JSON.parse(value);
|
|
1440
|
+
} catch {
|
|
1441
|
+
return;
|
|
1442
|
+
}
|
|
1443
|
+
}
|
|
1444
|
+
function recordOf(value) {
|
|
1445
|
+
return value !== null && typeof value === "object" && !Array.isArray(value) ? { ...value } : {};
|
|
1446
|
+
}
|
|
1447
|
+
async function declareTrainingCollections(store) {
|
|
1448
|
+
await store.declareCollection.mutate({
|
|
1449
|
+
collection: TRAINING_JOBS_COLLECTION,
|
|
1450
|
+
columns: [...JOB_COLUMNS],
|
|
1451
|
+
indexes: [...JOB_INDEXES]
|
|
1452
|
+
});
|
|
1453
|
+
await store.declareCollection.mutate({
|
|
1454
|
+
collection: TRAINING_DATASET_VERSIONS_COLLECTION,
|
|
1455
|
+
columns: [...VERSION_COLUMNS],
|
|
1456
|
+
indexes: [...VERSION_INDEXES]
|
|
1457
|
+
});
|
|
1458
|
+
}
|
|
1459
|
+
var SettingsJobStore = class {
|
|
1460
|
+
store;
|
|
1461
|
+
logger;
|
|
1462
|
+
constructor(store, logger) {
|
|
1463
|
+
this.store = store;
|
|
1464
|
+
this.logger = logger;
|
|
1465
|
+
}
|
|
1466
|
+
put = async (job) => {
|
|
1467
|
+
await this.store.set.mutate({
|
|
1468
|
+
collection: TRAINING_JOBS_COLLECTION,
|
|
1469
|
+
key: job.id,
|
|
1470
|
+
value: {
|
|
1471
|
+
state: job.state,
|
|
1472
|
+
createdAt: job.createdAt,
|
|
1473
|
+
idempotencyKey: job.idempotencyKey,
|
|
1474
|
+
job
|
|
1475
|
+
}
|
|
1476
|
+
});
|
|
1477
|
+
};
|
|
1478
|
+
get = async (id) => {
|
|
1479
|
+
const row = await this.store.get.query({
|
|
1480
|
+
collection: TRAINING_JOBS_COLLECTION,
|
|
1481
|
+
key: id
|
|
1482
|
+
});
|
|
1483
|
+
return row === null || row === void 0 ? null : this.parse(id, recordOf(row));
|
|
1484
|
+
};
|
|
1485
|
+
list = async () => {
|
|
1486
|
+
return (await this.store.query.query({
|
|
1487
|
+
collection: TRAINING_JOBS_COLLECTION,
|
|
1488
|
+
filter: { orderBy: {
|
|
1489
|
+
field: "createdAt",
|
|
1490
|
+
direction: "desc"
|
|
1491
|
+
} }
|
|
1492
|
+
})).flatMap((r) => {
|
|
1493
|
+
const job = this.parse(r.id, recordOf(r.data));
|
|
1494
|
+
return job === null ? [] : [job];
|
|
1495
|
+
});
|
|
1496
|
+
};
|
|
1497
|
+
findByIdempotencyKey = async (key) => {
|
|
1498
|
+
const row = (await this.store.query.query({
|
|
1499
|
+
collection: TRAINING_JOBS_COLLECTION,
|
|
1500
|
+
filter: {
|
|
1501
|
+
where: { idempotencyKey: key },
|
|
1502
|
+
limit: 1
|
|
1503
|
+
}
|
|
1504
|
+
}))[0];
|
|
1505
|
+
return row === void 0 ? null : this.parse(row.id, recordOf(row.data));
|
|
1506
|
+
};
|
|
1507
|
+
parse(id, data) {
|
|
1508
|
+
const parsed = require_dist.TrainingJobSchema.safeParse(parseJson(data["job"]));
|
|
1509
|
+
if (parsed.success) return parsed.data;
|
|
1510
|
+
this.logger.warn("training job row does not parse — skipped", { meta: {
|
|
1511
|
+
jobId: id,
|
|
1512
|
+
issue: parsed.error.issues[0]?.message ?? "unknown"
|
|
1513
|
+
} });
|
|
1514
|
+
return null;
|
|
1515
|
+
}
|
|
1516
|
+
};
|
|
1517
|
+
var nameKeyOf = (name) => name.trim().toLowerCase();
|
|
1518
|
+
var SettingsVersionStore = class {
|
|
1519
|
+
store;
|
|
1520
|
+
logger;
|
|
1521
|
+
constructor(store, logger) {
|
|
1522
|
+
this.store = store;
|
|
1523
|
+
this.logger = logger;
|
|
1524
|
+
}
|
|
1525
|
+
/** Insert once; an existing id is left as it is (a version is immutable). */
|
|
1526
|
+
put = async (row) => {
|
|
1527
|
+
if (await this.get(row.id) !== null) return false;
|
|
1528
|
+
await this.store.set.mutate({
|
|
1529
|
+
collection: TRAINING_DATASET_VERSIONS_COLLECTION,
|
|
1530
|
+
key: row.id,
|
|
1531
|
+
value: {
|
|
1532
|
+
nameKey: nameKeyOf(row.name),
|
|
1533
|
+
createdAt: row.createdAt,
|
|
1534
|
+
row
|
|
1535
|
+
}
|
|
1536
|
+
});
|
|
1537
|
+
return true;
|
|
1538
|
+
};
|
|
1539
|
+
get = async (id) => {
|
|
1540
|
+
const row = await this.store.get.query({
|
|
1541
|
+
collection: TRAINING_DATASET_VERSIONS_COLLECTION,
|
|
1542
|
+
key: id
|
|
1543
|
+
});
|
|
1544
|
+
return row === null || row === void 0 ? null : this.parse(id, recordOf(row));
|
|
1545
|
+
};
|
|
1546
|
+
list = async () => {
|
|
1547
|
+
return (await this.store.query.query({
|
|
1548
|
+
collection: TRAINING_DATASET_VERSIONS_COLLECTION,
|
|
1549
|
+
filter: { orderBy: {
|
|
1550
|
+
field: "createdAt",
|
|
1551
|
+
direction: "desc"
|
|
1552
|
+
} }
|
|
1553
|
+
})).flatMap((r) => {
|
|
1554
|
+
const parsed = this.parse(r.id, recordOf(r.data));
|
|
1555
|
+
return parsed === null ? [] : [parsed];
|
|
1556
|
+
});
|
|
1557
|
+
};
|
|
1558
|
+
findIdByName = async (name) => {
|
|
1559
|
+
return (await this.store.query.query({
|
|
1560
|
+
collection: "training:dataset-versions",
|
|
1561
|
+
filter: {
|
|
1562
|
+
where: { nameKey: nameKeyOf(name) },
|
|
1563
|
+
limit: 1
|
|
1564
|
+
}
|
|
1565
|
+
}))[0]?.id ?? null;
|
|
1566
|
+
};
|
|
1567
|
+
remove = async (id) => {
|
|
1568
|
+
if (await this.get(id) === null) return false;
|
|
1569
|
+
await this.store.delete.mutate({
|
|
1570
|
+
collection: TRAINING_DATASET_VERSIONS_COLLECTION,
|
|
1571
|
+
key: id
|
|
1572
|
+
});
|
|
1573
|
+
return true;
|
|
1574
|
+
};
|
|
1575
|
+
parse(id, data) {
|
|
1576
|
+
const parsed = require_dist.TrainingDatasetVersionRowSchema.safeParse(parseJson(data["row"]));
|
|
1577
|
+
if (parsed.success) return parsed.data;
|
|
1578
|
+
this.logger.warn("training dataset version row does not parse — skipped", { meta: {
|
|
1579
|
+
versionId: id,
|
|
1580
|
+
issue: parsed.error.issues[0]?.message ?? "unknown"
|
|
1581
|
+
} });
|
|
1582
|
+
return null;
|
|
1583
|
+
}
|
|
1584
|
+
};
|
|
1585
|
+
//#endregion
|
|
1586
|
+
//#region src/orchestrator/trainer-source.ts
|
|
1587
|
+
var TRAINER_SOURCE = {
|
|
1588
|
+
repo: "camstack/camstack-trainer",
|
|
1589
|
+
revision: "ebeed67a296241fcb5b0d97cf7c85dd12a743db8"
|
|
1590
|
+
};
|
|
1591
|
+
//#endregion
|
|
1592
|
+
//#region src/orchestrator/training.addon.ts
|
|
1593
|
+
/**
|
|
1594
|
+
* `training` — the cloud-training orchestrator (D805). Hub-only.
|
|
1595
|
+
*
|
|
1596
|
+
* Wires the engine to `ctx.api` and the hub's disk, provides the `training`
|
|
1597
|
+
* capability and the Training page, and runs the reconcile loop: once at boot
|
|
1598
|
+
* (a restart loses nothing — the job rows and the provider's jobKey tags are
|
|
1599
|
+
* the whole truth), then every 30 s, plus a daily orphan sweep.
|
|
1600
|
+
*/
|
|
1601
|
+
/** The first orphan sweep waits for the providers to register after a boot. */
|
|
1602
|
+
var FIRST_SWEEP_DELAY_MS = 5 * 6e4;
|
|
1603
|
+
var TrainingAddon = class extends require_dist.BaseAddon {
|
|
1604
|
+
pages = [{
|
|
1605
|
+
id: "training",
|
|
1606
|
+
label: "Training",
|
|
1607
|
+
icon: "graduation-cap",
|
|
1608
|
+
path: "/addon/training",
|
|
1609
|
+
remoteName: "addon_training_page",
|
|
1610
|
+
bundle: "remoteEntry.js",
|
|
1611
|
+
section: "detection"
|
|
1612
|
+
}];
|
|
1613
|
+
engine = null;
|
|
1614
|
+
timers = [];
|
|
1615
|
+
ticking = false;
|
|
1616
|
+
constructor() {
|
|
1617
|
+
super({});
|
|
1618
|
+
}
|
|
1619
|
+
async onInitialize() {
|
|
1620
|
+
const api = this.ctx.api;
|
|
1621
|
+
const peer = this.ctx.peerBytes;
|
|
1622
|
+
if (peer === void 0) throw new Error("training needs the peer-bytes facility");
|
|
1623
|
+
await declareTrainingCollections(api.settingsStore);
|
|
1624
|
+
const logger = this.ctx.logger;
|
|
1625
|
+
const versions = new SettingsVersionStore(api.settingsStore, logger);
|
|
1626
|
+
const ports = {
|
|
1627
|
+
provider: providerPort(api),
|
|
1628
|
+
datasets: datasetPort(api),
|
|
1629
|
+
versions,
|
|
1630
|
+
jobs: new SettingsJobStore(api.settingsStore, logger),
|
|
1631
|
+
artifacts: createDiskArtifacts(node_path.default.join(this.ctx.dataDir, "training", "jobs"), peer),
|
|
1632
|
+
registry: registryPort(api),
|
|
1633
|
+
pins: pinPort(api),
|
|
1634
|
+
now: () => Date.now(),
|
|
1635
|
+
newId: () => crypto.randomUUID(),
|
|
1636
|
+
logger
|
|
1637
|
+
};
|
|
1638
|
+
const engine = new TrainingEngine(ports, TRAINER_SOURCE);
|
|
1639
|
+
const flow = new ReturnFlow(ports, engine.mutator);
|
|
1640
|
+
this.engine = engine;
|
|
1641
|
+
const provider = {
|
|
1642
|
+
putDatasetVersion: async (row) => ({ stored: await versions.put(row) }),
|
|
1643
|
+
getDatasetVersion: async ({ id }) => versions.get(id),
|
|
1644
|
+
listDatasetVersions: async () => (await versions.list()).map(({ frameIds, ...rest }) => ({
|
|
1645
|
+
...rest,
|
|
1646
|
+
frameCount: frameIds.length
|
|
1647
|
+
})),
|
|
1648
|
+
findDatasetVersionIdByName: async ({ name }) => ({ id: await versions.findIdByName(name) }),
|
|
1649
|
+
removeDatasetVersion: async ({ id }) => ({ removed: await versions.remove(id) }),
|
|
1650
|
+
listBaseModels: async () => BASE_MODELS.map(publicBaseModel),
|
|
1651
|
+
planJob: async (request) => (await planJob(request, {
|
|
1652
|
+
versions,
|
|
1653
|
+
provider: ports.provider
|
|
1654
|
+
})).plan,
|
|
1655
|
+
startJob: async (input) => engine.startJob(input),
|
|
1656
|
+
listJobs: async () => ports.jobs.list(),
|
|
1657
|
+
getJob: async ({ id }) => ports.jobs.get(id),
|
|
1658
|
+
cancelJob: async ({ id }) => engine.cancelJob(id),
|
|
1659
|
+
getJobResult: async ({ id }) => {
|
|
1660
|
+
const job = await ports.jobs.get(id);
|
|
1661
|
+
return job === null ? null : flow.readResult(job);
|
|
1662
|
+
},
|
|
1663
|
+
registerResult: async ({ id, canaryDeviceIds }) => flow.register(id, canaryDeviceIds),
|
|
1664
|
+
canaryResult: async ({ id, deviceIds }) => flow.canary(id, deviceIds),
|
|
1665
|
+
rollbackCanary: async ({ id }) => flow.rollback(id),
|
|
1666
|
+
discardResult: async ({ id }) => flow.discard(id)
|
|
1667
|
+
};
|
|
1668
|
+
const pagesProvider = {
|
|
1669
|
+
id: this.ctx.id,
|
|
1670
|
+
listPages: () => this.pages
|
|
1671
|
+
};
|
|
1672
|
+
this.schedule(0, RECONCILE_INTERVAL_MS, () => this.reconcile());
|
|
1673
|
+
this.schedule(FIRST_SWEEP_DELAY_MS, ORPHAN_SWEEP_INTERVAL_MS, async () => {
|
|
1674
|
+
const swept = await engine.sweepOrphans();
|
|
1675
|
+
if (swept.length > 0) logger.warn("training orphan sweep stopped cloud jobs", { meta: { swept: [...swept] } });
|
|
1676
|
+
});
|
|
1677
|
+
return { providers: [{
|
|
1678
|
+
capability: require_dist.trainingCapability,
|
|
1679
|
+
provider
|
|
1680
|
+
}, {
|
|
1681
|
+
capability: require_dist.addonPagesSourceCapability,
|
|
1682
|
+
provider: pagesProvider
|
|
1683
|
+
}] };
|
|
1684
|
+
}
|
|
1685
|
+
async onShutdown() {
|
|
1686
|
+
for (const t of this.timers) clearTimeout(t);
|
|
1687
|
+
this.timers = [];
|
|
1688
|
+
this.engine = null;
|
|
1689
|
+
}
|
|
1690
|
+
async reconcile() {
|
|
1691
|
+
if (this.ticking || this.engine === null) return;
|
|
1692
|
+
this.ticking = true;
|
|
1693
|
+
try {
|
|
1694
|
+
await this.engine.tick();
|
|
1695
|
+
} finally {
|
|
1696
|
+
this.ticking = false;
|
|
1697
|
+
}
|
|
1698
|
+
}
|
|
1699
|
+
/** A self-rescheduling timer that never overlaps itself and never throws out. */
|
|
1700
|
+
schedule(firstDelayMs, everyMs, body) {
|
|
1701
|
+
const run = () => {
|
|
1702
|
+
body().catch((err) => {
|
|
1703
|
+
this.ctx.logger.warn("training scheduled pass failed", { meta: { error: err instanceof Error ? err.message : String(err) } });
|
|
1704
|
+
}).finally(() => {
|
|
1705
|
+
if (this.engine !== null) this.timers.push(setTimeout(run, everyMs));
|
|
1706
|
+
});
|
|
1707
|
+
};
|
|
1708
|
+
this.timers.push(setTimeout(run, firstDelayMs));
|
|
1709
|
+
}
|
|
1710
|
+
};
|
|
1711
|
+
//#endregion
|
|
1712
|
+
exports.TrainingAddon = TrainingAddon;
|