@camstack/addon-pipeline 1.2.177 → 1.2.178
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/detection-pipeline/index.js +125 -17
- package/dist/detection-pipeline/index.mjs +125 -17
- package/package.json +1 -1
- package/python/__pycache__/inference_pool.cpython-314.pyc +0 -0
- package/python/__pycache__/test_inference_pool_deadline.cpython-314.pyc +0 -0
- package/python/inference_pool.py +225 -28
- package/python/test_inference_pool_deadline.py +253 -0
|
@@ -679,6 +679,8 @@ var MSG_INFER_CACHED = 5;
|
|
|
679
679
|
*/
|
|
680
680
|
var MSG_INFER_BATCH = 3;
|
|
681
681
|
var PREFIX_LEN = 9;
|
|
682
|
+
/** Bytes of the absolute-deadline prefix a v2 sheddable payload carries. */
|
|
683
|
+
var DEADLINE_PREFIX_LEN = 8;
|
|
682
684
|
/**
|
|
683
685
|
* Wire-level enum for the raw-frame fast path. Values are append-only:
|
|
684
686
|
* the Python pool reads the byte directly off the IPC frame; reordering
|
|
@@ -867,6 +869,20 @@ var PoolWorker = class {
|
|
|
867
869
|
* Surfaced so "the camera is quiet" and "the camera is saturated" can be
|
|
868
870
|
* told apart without reading the logs. */
|
|
869
871
|
shedCount = 0;
|
|
872
|
+
/**
|
|
873
|
+
* Live requests whose deadline fired IN FLIGHT and were answered dropped by
|
|
874
|
+
* the local watchdog (the worker never replied in time). With a v2 worker
|
|
875
|
+
* this is the rare backstop — the worker sheds expired work itself and
|
|
876
|
+
* replies fast; against a v1 worker it is the only shed there is.
|
|
877
|
+
*/
|
|
878
|
+
deadlineShedCount = 0;
|
|
879
|
+
/**
|
|
880
|
+
* Negotiated wire version for THIS worker: `min(POOL_PROTOCOL_VERSION,
|
|
881
|
+
* what the ready handshake reported)`. A worker that reports nothing is a
|
|
882
|
+
* v1 worker and keeps receiving the byte-identical old layout — the deploy
|
|
883
|
+
* skew case that must always work (D350).
|
|
884
|
+
*/
|
|
885
|
+
wireVersion = 1;
|
|
870
886
|
nextRequestId = 1;
|
|
871
887
|
ready = false;
|
|
872
888
|
log;
|
|
@@ -936,6 +952,7 @@ var PoolWorker = class {
|
|
|
936
952
|
const config = {
|
|
937
953
|
runtime: this.opts.poolRuntime,
|
|
938
954
|
concurrency: this.opts.concurrency,
|
|
955
|
+
protocolVersion: 2,
|
|
939
956
|
models: initialModels.map((m) => serializeModelConfig(m))
|
|
940
957
|
};
|
|
941
958
|
if (this.opts.device) config["device"] = this.opts.device;
|
|
@@ -960,10 +977,13 @@ var PoolWorker = class {
|
|
|
960
977
|
this.ready = true;
|
|
961
978
|
const loadedCount = result["models"];
|
|
962
979
|
const startupMs = result["startupMs"];
|
|
980
|
+
const workers = result["workers"] ?? 1;
|
|
981
|
+
const reported = result["protocolVersion"];
|
|
982
|
+
this.wireVersion = Math.min(2, typeof reported === "number" && Number.isFinite(reported) ? reported : 1);
|
|
963
983
|
resolve({
|
|
964
984
|
startupMs,
|
|
965
985
|
loadedCount,
|
|
966
|
-
workers
|
|
986
|
+
workers
|
|
967
987
|
});
|
|
968
988
|
} else reject(/* @__PURE__ */ new Error(`Unexpected pool status: ${JSON.stringify(result)}`));
|
|
969
989
|
},
|
|
@@ -1095,19 +1115,35 @@ var PoolWorker = class {
|
|
|
1095
1115
|
deadlineFor(msgType) {
|
|
1096
1116
|
return SHEDDABLE_MSG_TYPES.has(msgType) ? POOL_LIVE_INFER_TIMEOUT_MS : POOL_INFER_TIMEOUT_MS;
|
|
1097
1117
|
}
|
|
1118
|
+
/**
|
|
1119
|
+
* The request's ABSOLUTE deadline for the wire (D350), or `null` when this
|
|
1120
|
+
* request must not carry one: commands / model loads / cacheFrame at any
|
|
1121
|
+
* version, and EVERYTHING against a v1 worker — the old layout has no room
|
|
1122
|
+
* for the prefix, and a worker at a different version is the normal state
|
|
1123
|
+
* during a rolling deploy. The stamp is the same instant the local watchdog
|
|
1124
|
+
* arms, so "the TS side has abandoned this" and "the worker refuses to run
|
|
1125
|
+
* it" are one moment, not two clocks drifting apart.
|
|
1126
|
+
*/
|
|
1127
|
+
deadlineStamp(msgType) {
|
|
1128
|
+
if (this.wireVersion < 2 || !SHEDDABLE_MSG_TYPES.has(msgType)) return null;
|
|
1129
|
+
const stamp = Buffer.allocUnsafe(DEADLINE_PREFIX_LEN);
|
|
1130
|
+
stamp.writeBigUInt64LE(BigInt(Date.now() + this.deadlineFor(msgType)), 0);
|
|
1131
|
+
return stamp;
|
|
1132
|
+
}
|
|
1098
1133
|
dispatch(msgType, payload, deviceId) {
|
|
1099
1134
|
const shed = this.shedIfSaturated(msgType, deviceId);
|
|
1100
1135
|
if (shed) return Promise.resolve(shed);
|
|
1101
1136
|
const reqId = this.allocRequestId();
|
|
1102
1137
|
return new Promise((resolve, reject) => {
|
|
1103
|
-
const timer = this.armRequestTimeout(reqId,
|
|
1138
|
+
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, deviceId);
|
|
1104
1139
|
this.pending.set(reqId, {
|
|
1105
1140
|
resolve,
|
|
1106
1141
|
reject,
|
|
1107
1142
|
timer
|
|
1108
1143
|
});
|
|
1109
1144
|
try {
|
|
1110
|
-
this.
|
|
1145
|
+
const stamp = this.deadlineStamp(msgType);
|
|
1146
|
+
this.writeFrame(reqId, msgType, stamp ? Buffer.concat([stamp, payload]) : payload);
|
|
1111
1147
|
} catch (err) {
|
|
1112
1148
|
clearTimeout(timer);
|
|
1113
1149
|
this.pending.delete(reqId);
|
|
@@ -1116,13 +1152,51 @@ var PoolWorker = class {
|
|
|
1116
1152
|
});
|
|
1117
1153
|
}
|
|
1118
1154
|
/**
|
|
1119
|
-
* Arm a watchdog
|
|
1120
|
-
*
|
|
1121
|
-
*
|
|
1122
|
-
|
|
1123
|
-
|
|
1155
|
+
* Arm a watchdog for a still-pending request. `unref` so it never keeps the
|
|
1156
|
+
* event loop alive. The normal response + `rejectAll` paths clear it via
|
|
1157
|
+
* `PendingRequest.timer`.
|
|
1158
|
+
*
|
|
1159
|
+
* The two request classes settle DIFFERENTLY on expiry (D350):
|
|
1160
|
+
*
|
|
1161
|
+
* - A SHEDDABLE (live inference) request RESOLVES as a shed —
|
|
1162
|
+
* `{dropped: true, shedReason: 'deadline-expired'}` — because at this
|
|
1163
|
+
* point the answer is worthless whether or not it eventually arrives, and
|
|
1164
|
+
* a shed is flow control the executor already handles. It fires only when
|
|
1165
|
+
* the worker never replied at all: a v2 worker sheds expired work itself
|
|
1166
|
+
* (fast reply, this timer is cleared), so this is the wedged-worker
|
|
1167
|
+
* backstop and the v1-worker compatibility path. The 2026-09-04 storm
|
|
1168
|
+
* logged this state 16 146 times at ERROR; the shed is logged sampled at
|
|
1169
|
+
* WARN and counted (`getDeadlineExpiredShedCount`) instead.
|
|
1170
|
+
* - Anything else (commands, model loads) still REJECTS with the error the
|
|
1171
|
+
* dashboards grep for — a lost command is a fault, not flow control.
|
|
1172
|
+
*/
|
|
1173
|
+
armRequestTimeout(reqId, msgType, resolve, reject, deviceId) {
|
|
1174
|
+
const timeoutMs = this.deadlineFor(msgType);
|
|
1124
1175
|
const timer = setTimeout(() => {
|
|
1125
1176
|
if (this.pending.delete(reqId)) {
|
|
1177
|
+
if (SHEDDABLE_MSG_TYPES.has(msgType)) {
|
|
1178
|
+
this.deadlineShedCount++;
|
|
1179
|
+
const total = this.deadlineShedCount;
|
|
1180
|
+
if (total === 1 || total % SHED_LOG_SAMPLE_EVERY === 0) this.log.warn("live inference deadline expired in flight — answered dropped", {
|
|
1181
|
+
...deviceId !== void 0 ? { tags: { deviceId } } : {},
|
|
1182
|
+
meta: {
|
|
1183
|
+
worker: this.opts.workerLabel,
|
|
1184
|
+
pid: this.getPid(),
|
|
1185
|
+
runtime: this.opts.poolRuntime,
|
|
1186
|
+
device: this.opts.device ?? "default",
|
|
1187
|
+
inFlight: this.pending.size,
|
|
1188
|
+
reqId,
|
|
1189
|
+
timeoutMs,
|
|
1190
|
+
deadlineShedTotal: total,
|
|
1191
|
+
sampledEvery: SHED_LOG_SAMPLE_EVERY
|
|
1192
|
+
}
|
|
1193
|
+
});
|
|
1194
|
+
resolve({
|
|
1195
|
+
dropped: true,
|
|
1196
|
+
shedReason: "deadline-expired"
|
|
1197
|
+
});
|
|
1198
|
+
return;
|
|
1199
|
+
}
|
|
1126
1200
|
this.log.error("inference request timed out", {
|
|
1127
1201
|
...deviceId !== void 0 ? { tags: { deviceId } } : {},
|
|
1128
1202
|
meta: {
|
|
@@ -1146,7 +1220,7 @@ var PoolWorker = class {
|
|
|
1146
1220
|
if (shed) return Promise.resolve(shed);
|
|
1147
1221
|
const reqId = this.allocRequestId();
|
|
1148
1222
|
return new Promise((resolve, reject) => {
|
|
1149
|
-
const timer = this.armRequestTimeout(reqId,
|
|
1223
|
+
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, deviceId);
|
|
1150
1224
|
this.pending.set(reqId, {
|
|
1151
1225
|
resolve,
|
|
1152
1226
|
reject,
|
|
@@ -1154,11 +1228,14 @@ var PoolWorker = class {
|
|
|
1154
1228
|
});
|
|
1155
1229
|
try {
|
|
1156
1230
|
if (!this.process?.stdin) throw new Error("PoolWorker: not initialized");
|
|
1231
|
+
const stamp = this.deadlineStamp(msgType);
|
|
1232
|
+
const wireLen = payloadLen + (stamp ? DEADLINE_PREFIX_LEN : 0);
|
|
1157
1233
|
const prefix = Buffer.allocUnsafe(PREFIX_LEN);
|
|
1158
|
-
prefix.writeUInt32LE(5 +
|
|
1234
|
+
prefix.writeUInt32LE(5 + wireLen, 0);
|
|
1159
1235
|
prefix.writeUInt32LE(reqId, 4);
|
|
1160
1236
|
prefix[8] = msgType;
|
|
1161
1237
|
this.process.stdin.write(prefix);
|
|
1238
|
+
if (stamp) this.process.stdin.write(stamp);
|
|
1162
1239
|
for (const part of parts) this.process.stdin.write(part);
|
|
1163
1240
|
} catch (err) {
|
|
1164
1241
|
clearTimeout(timer);
|
|
@@ -1261,12 +1338,22 @@ var SharedInferencePool = class {
|
|
|
1261
1338
|
nextFreeIndex = 0;
|
|
1262
1339
|
nextFrameId = 1;
|
|
1263
1340
|
/**
|
|
1264
|
-
* Cumulative count of
|
|
1265
|
-
*
|
|
1266
|
-
*
|
|
1267
|
-
*
|
|
1341
|
+
* Cumulative count of dropped (shed) responses on the single-frame and
|
|
1342
|
+
* batch inference paths: the Python per-model bound, the Python
|
|
1343
|
+
* deadline-expired shed, and the TS in-flight watchdog's own
|
|
1344
|
+
* deadline-expired resolution all land here. Without this a shed response
|
|
1345
|
+
* is indistinguishable from a genuine "no detections" result.
|
|
1268
1346
|
*/
|
|
1269
1347
|
droppedResponseCount = 0;
|
|
1348
|
+
/**
|
|
1349
|
+
* The `shedReason: 'deadline-expired'` subset of {@link
|
|
1350
|
+
* droppedResponseCount} — work that was ALREADY DEAD when it would have
|
|
1351
|
+
* run (D350). This is the number that was zero-by-construction while the
|
|
1352
|
+
* 2026-09-04 storm burned the accelerator on abandoned requests for 4h15m;
|
|
1353
|
+
* a rising value here under load is the pool refusing dead work, which is
|
|
1354
|
+
* the fix working, not a fault.
|
|
1355
|
+
*/
|
|
1356
|
+
deadlineExpiredShedCount = 0;
|
|
1270
1357
|
log;
|
|
1271
1358
|
concurrency;
|
|
1272
1359
|
tuning;
|
|
@@ -1352,7 +1439,12 @@ var SharedInferencePool = class {
|
|
|
1352
1439
|
}
|
|
1353
1440
|
async inferBatch(modelIndex, items, frameId = 0) {
|
|
1354
1441
|
if (items.length > 255) throw new Error(`SharedInferencePool.inferBatch: max 255 items per call, got ${items.length}`);
|
|
1355
|
-
|
|
1442
|
+
const results = await this.pickWorker().inferBatch(this.encodeModelByte(modelIndex), items, frameId);
|
|
1443
|
+
for (const item of results) if (item["dropped"] === true) {
|
|
1444
|
+
this.droppedResponseCount++;
|
|
1445
|
+
if (item["shedReason"] === "deadline-expired") this.deadlineExpiredShedCount++;
|
|
1446
|
+
}
|
|
1447
|
+
return results;
|
|
1356
1448
|
}
|
|
1357
1449
|
async inferCached(modelIndex, frameId, deviceId) {
|
|
1358
1450
|
const w = this.pickWorker();
|
|
@@ -1366,6 +1458,16 @@ var SharedInferencePool = class {
|
|
|
1366
1458
|
getDroppedResponseCount() {
|
|
1367
1459
|
return this.droppedResponseCount;
|
|
1368
1460
|
}
|
|
1461
|
+
/**
|
|
1462
|
+
* The `deadline-expired` subset of {@link getDroppedResponseCount}:
|
|
1463
|
+
* requests refused (by the Python worker, or by the local in-flight
|
|
1464
|
+
* watchdog as its backstop) because their absolute deadline had already
|
|
1465
|
+
* passed — dead work NOT executed (D350). Monotonic for the pool's
|
|
1466
|
+
* lifetime; surfaced on the memory watchdog's `pool memory` line.
|
|
1467
|
+
*/
|
|
1468
|
+
getDeadlineExpiredShedCount() {
|
|
1469
|
+
return this.deadlineExpiredShedCount;
|
|
1470
|
+
}
|
|
1369
1471
|
getHandle(modelIndex) {
|
|
1370
1472
|
return new PoolHandle(this, modelIndex);
|
|
1371
1473
|
}
|
|
@@ -1504,10 +1606,13 @@ var SharedInferencePool = class {
|
|
|
1504
1606
|
trackDroppedResponse(result, modelIndex) {
|
|
1505
1607
|
if (result["dropped"] === true) {
|
|
1506
1608
|
this.droppedResponseCount++;
|
|
1609
|
+
if (result["shedReason"] === "deadline-expired") this.deadlineExpiredShedCount++;
|
|
1507
1610
|
const total = this.droppedResponseCount;
|
|
1508
1611
|
if (total === 1 || total % SHED_LOG_SAMPLE_EVERY === 0) this.log.debug("Python pool shed frame under overload", { meta: {
|
|
1509
1612
|
modelIndex,
|
|
1510
1613
|
droppedTotal: total,
|
|
1614
|
+
deadlineExpiredTotal: this.deadlineExpiredShedCount,
|
|
1615
|
+
...typeof result["shedReason"] === "string" ? { shedReason: result["shedReason"] } : {},
|
|
1511
1616
|
sampledEvery: SHED_LOG_SAMPLE_EVERY,
|
|
1512
1617
|
runtime: this.poolRuntime,
|
|
1513
1618
|
device: this.device ?? "default"
|
|
@@ -2396,11 +2501,13 @@ var EngineFactory = class {
|
|
|
2396
2501
|
if (!this.pool) return {
|
|
2397
2502
|
inFlight: 0,
|
|
2398
2503
|
shed: 0,
|
|
2399
|
-
dropped: 0
|
|
2504
|
+
dropped: 0,
|
|
2505
|
+
deadlineExpired: 0
|
|
2400
2506
|
};
|
|
2401
2507
|
return {
|
|
2402
2508
|
...this.pool.getBacklog(),
|
|
2403
|
-
dropped: this.pool.getDroppedResponseCount()
|
|
2509
|
+
dropped: this.pool.getDroppedResponseCount(),
|
|
2510
|
+
deadlineExpired: this.pool.getDeadlineExpiredShedCount()
|
|
2404
2511
|
};
|
|
2405
2512
|
}
|
|
2406
2513
|
/** Python-side per-worker memory diagnostics (see SharedInferencePool). */
|
|
@@ -2887,6 +2994,7 @@ async function samplePool(pool, log, reportedDead) {
|
|
|
2887
2994
|
inFlight: backlog.inFlight,
|
|
2888
2995
|
shed: backlog.shed,
|
|
2889
2996
|
dropped: backlog.dropped,
|
|
2997
|
+
deadlineExpired: backlog.deadlineExpired,
|
|
2890
2998
|
modelsLoaded: pool.factory.listLoaded().length,
|
|
2891
2999
|
memStats
|
|
2892
3000
|
}
|
|
@@ -672,6 +672,8 @@ var MSG_INFER_CACHED = 5;
|
|
|
672
672
|
*/
|
|
673
673
|
var MSG_INFER_BATCH = 3;
|
|
674
674
|
var PREFIX_LEN = 9;
|
|
675
|
+
/** Bytes of the absolute-deadline prefix a v2 sheddable payload carries. */
|
|
676
|
+
var DEADLINE_PREFIX_LEN = 8;
|
|
675
677
|
/**
|
|
676
678
|
* Wire-level enum for the raw-frame fast path. Values are append-only:
|
|
677
679
|
* the Python pool reads the byte directly off the IPC frame; reordering
|
|
@@ -860,6 +862,20 @@ var PoolWorker = class {
|
|
|
860
862
|
* Surfaced so "the camera is quiet" and "the camera is saturated" can be
|
|
861
863
|
* told apart without reading the logs. */
|
|
862
864
|
shedCount = 0;
|
|
865
|
+
/**
|
|
866
|
+
* Live requests whose deadline fired IN FLIGHT and were answered dropped by
|
|
867
|
+
* the local watchdog (the worker never replied in time). With a v2 worker
|
|
868
|
+
* this is the rare backstop — the worker sheds expired work itself and
|
|
869
|
+
* replies fast; against a v1 worker it is the only shed there is.
|
|
870
|
+
*/
|
|
871
|
+
deadlineShedCount = 0;
|
|
872
|
+
/**
|
|
873
|
+
* Negotiated wire version for THIS worker: `min(POOL_PROTOCOL_VERSION,
|
|
874
|
+
* what the ready handshake reported)`. A worker that reports nothing is a
|
|
875
|
+
* v1 worker and keeps receiving the byte-identical old layout — the deploy
|
|
876
|
+
* skew case that must always work (D350).
|
|
877
|
+
*/
|
|
878
|
+
wireVersion = 1;
|
|
863
879
|
nextRequestId = 1;
|
|
864
880
|
ready = false;
|
|
865
881
|
log;
|
|
@@ -929,6 +945,7 @@ var PoolWorker = class {
|
|
|
929
945
|
const config = {
|
|
930
946
|
runtime: this.opts.poolRuntime,
|
|
931
947
|
concurrency: this.opts.concurrency,
|
|
948
|
+
protocolVersion: 2,
|
|
932
949
|
models: initialModels.map((m) => serializeModelConfig(m))
|
|
933
950
|
};
|
|
934
951
|
if (this.opts.device) config["device"] = this.opts.device;
|
|
@@ -953,10 +970,13 @@ var PoolWorker = class {
|
|
|
953
970
|
this.ready = true;
|
|
954
971
|
const loadedCount = result["models"];
|
|
955
972
|
const startupMs = result["startupMs"];
|
|
973
|
+
const workers = result["workers"] ?? 1;
|
|
974
|
+
const reported = result["protocolVersion"];
|
|
975
|
+
this.wireVersion = Math.min(2, typeof reported === "number" && Number.isFinite(reported) ? reported : 1);
|
|
956
976
|
resolve({
|
|
957
977
|
startupMs,
|
|
958
978
|
loadedCount,
|
|
959
|
-
workers
|
|
979
|
+
workers
|
|
960
980
|
});
|
|
961
981
|
} else reject(/* @__PURE__ */ new Error(`Unexpected pool status: ${JSON.stringify(result)}`));
|
|
962
982
|
},
|
|
@@ -1088,19 +1108,35 @@ var PoolWorker = class {
|
|
|
1088
1108
|
deadlineFor(msgType) {
|
|
1089
1109
|
return SHEDDABLE_MSG_TYPES.has(msgType) ? POOL_LIVE_INFER_TIMEOUT_MS : POOL_INFER_TIMEOUT_MS;
|
|
1090
1110
|
}
|
|
1111
|
+
/**
|
|
1112
|
+
* The request's ABSOLUTE deadline for the wire (D350), or `null` when this
|
|
1113
|
+
* request must not carry one: commands / model loads / cacheFrame at any
|
|
1114
|
+
* version, and EVERYTHING against a v1 worker — the old layout has no room
|
|
1115
|
+
* for the prefix, and a worker at a different version is the normal state
|
|
1116
|
+
* during a rolling deploy. The stamp is the same instant the local watchdog
|
|
1117
|
+
* arms, so "the TS side has abandoned this" and "the worker refuses to run
|
|
1118
|
+
* it" are one moment, not two clocks drifting apart.
|
|
1119
|
+
*/
|
|
1120
|
+
deadlineStamp(msgType) {
|
|
1121
|
+
if (this.wireVersion < 2 || !SHEDDABLE_MSG_TYPES.has(msgType)) return null;
|
|
1122
|
+
const stamp = Buffer.allocUnsafe(DEADLINE_PREFIX_LEN);
|
|
1123
|
+
stamp.writeBigUInt64LE(BigInt(Date.now() + this.deadlineFor(msgType)), 0);
|
|
1124
|
+
return stamp;
|
|
1125
|
+
}
|
|
1091
1126
|
dispatch(msgType, payload, deviceId) {
|
|
1092
1127
|
const shed = this.shedIfSaturated(msgType, deviceId);
|
|
1093
1128
|
if (shed) return Promise.resolve(shed);
|
|
1094
1129
|
const reqId = this.allocRequestId();
|
|
1095
1130
|
return new Promise((resolve, reject) => {
|
|
1096
|
-
const timer = this.armRequestTimeout(reqId,
|
|
1131
|
+
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, deviceId);
|
|
1097
1132
|
this.pending.set(reqId, {
|
|
1098
1133
|
resolve,
|
|
1099
1134
|
reject,
|
|
1100
1135
|
timer
|
|
1101
1136
|
});
|
|
1102
1137
|
try {
|
|
1103
|
-
this.
|
|
1138
|
+
const stamp = this.deadlineStamp(msgType);
|
|
1139
|
+
this.writeFrame(reqId, msgType, stamp ? Buffer.concat([stamp, payload]) : payload);
|
|
1104
1140
|
} catch (err) {
|
|
1105
1141
|
clearTimeout(timer);
|
|
1106
1142
|
this.pending.delete(reqId);
|
|
@@ -1109,13 +1145,51 @@ var PoolWorker = class {
|
|
|
1109
1145
|
});
|
|
1110
1146
|
}
|
|
1111
1147
|
/**
|
|
1112
|
-
* Arm a watchdog
|
|
1113
|
-
*
|
|
1114
|
-
*
|
|
1115
|
-
|
|
1116
|
-
|
|
1148
|
+
* Arm a watchdog for a still-pending request. `unref` so it never keeps the
|
|
1149
|
+
* event loop alive. The normal response + `rejectAll` paths clear it via
|
|
1150
|
+
* `PendingRequest.timer`.
|
|
1151
|
+
*
|
|
1152
|
+
* The two request classes settle DIFFERENTLY on expiry (D350):
|
|
1153
|
+
*
|
|
1154
|
+
* - A SHEDDABLE (live inference) request RESOLVES as a shed —
|
|
1155
|
+
* `{dropped: true, shedReason: 'deadline-expired'}` — because at this
|
|
1156
|
+
* point the answer is worthless whether or not it eventually arrives, and
|
|
1157
|
+
* a shed is flow control the executor already handles. It fires only when
|
|
1158
|
+
* the worker never replied at all: a v2 worker sheds expired work itself
|
|
1159
|
+
* (fast reply, this timer is cleared), so this is the wedged-worker
|
|
1160
|
+
* backstop and the v1-worker compatibility path. The 2026-09-04 storm
|
|
1161
|
+
* logged this state 16 146 times at ERROR; the shed is logged sampled at
|
|
1162
|
+
* WARN and counted (`getDeadlineExpiredShedCount`) instead.
|
|
1163
|
+
* - Anything else (commands, model loads) still REJECTS with the error the
|
|
1164
|
+
* dashboards grep for — a lost command is a fault, not flow control.
|
|
1165
|
+
*/
|
|
1166
|
+
armRequestTimeout(reqId, msgType, resolve, reject, deviceId) {
|
|
1167
|
+
const timeoutMs = this.deadlineFor(msgType);
|
|
1117
1168
|
const timer = setTimeout(() => {
|
|
1118
1169
|
if (this.pending.delete(reqId)) {
|
|
1170
|
+
if (SHEDDABLE_MSG_TYPES.has(msgType)) {
|
|
1171
|
+
this.deadlineShedCount++;
|
|
1172
|
+
const total = this.deadlineShedCount;
|
|
1173
|
+
if (total === 1 || total % SHED_LOG_SAMPLE_EVERY === 0) this.log.warn("live inference deadline expired in flight — answered dropped", {
|
|
1174
|
+
...deviceId !== void 0 ? { tags: { deviceId } } : {},
|
|
1175
|
+
meta: {
|
|
1176
|
+
worker: this.opts.workerLabel,
|
|
1177
|
+
pid: this.getPid(),
|
|
1178
|
+
runtime: this.opts.poolRuntime,
|
|
1179
|
+
device: this.opts.device ?? "default",
|
|
1180
|
+
inFlight: this.pending.size,
|
|
1181
|
+
reqId,
|
|
1182
|
+
timeoutMs,
|
|
1183
|
+
deadlineShedTotal: total,
|
|
1184
|
+
sampledEvery: SHED_LOG_SAMPLE_EVERY
|
|
1185
|
+
}
|
|
1186
|
+
});
|
|
1187
|
+
resolve({
|
|
1188
|
+
dropped: true,
|
|
1189
|
+
shedReason: "deadline-expired"
|
|
1190
|
+
});
|
|
1191
|
+
return;
|
|
1192
|
+
}
|
|
1119
1193
|
this.log.error("inference request timed out", {
|
|
1120
1194
|
...deviceId !== void 0 ? { tags: { deviceId } } : {},
|
|
1121
1195
|
meta: {
|
|
@@ -1139,7 +1213,7 @@ var PoolWorker = class {
|
|
|
1139
1213
|
if (shed) return Promise.resolve(shed);
|
|
1140
1214
|
const reqId = this.allocRequestId();
|
|
1141
1215
|
return new Promise((resolve, reject) => {
|
|
1142
|
-
const timer = this.armRequestTimeout(reqId,
|
|
1216
|
+
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, deviceId);
|
|
1143
1217
|
this.pending.set(reqId, {
|
|
1144
1218
|
resolve,
|
|
1145
1219
|
reject,
|
|
@@ -1147,11 +1221,14 @@ var PoolWorker = class {
|
|
|
1147
1221
|
});
|
|
1148
1222
|
try {
|
|
1149
1223
|
if (!this.process?.stdin) throw new Error("PoolWorker: not initialized");
|
|
1224
|
+
const stamp = this.deadlineStamp(msgType);
|
|
1225
|
+
const wireLen = payloadLen + (stamp ? DEADLINE_PREFIX_LEN : 0);
|
|
1150
1226
|
const prefix = Buffer.allocUnsafe(PREFIX_LEN);
|
|
1151
|
-
prefix.writeUInt32LE(5 +
|
|
1227
|
+
prefix.writeUInt32LE(5 + wireLen, 0);
|
|
1152
1228
|
prefix.writeUInt32LE(reqId, 4);
|
|
1153
1229
|
prefix[8] = msgType;
|
|
1154
1230
|
this.process.stdin.write(prefix);
|
|
1231
|
+
if (stamp) this.process.stdin.write(stamp);
|
|
1155
1232
|
for (const part of parts) this.process.stdin.write(part);
|
|
1156
1233
|
} catch (err) {
|
|
1157
1234
|
clearTimeout(timer);
|
|
@@ -1254,12 +1331,22 @@ var SharedInferencePool = class {
|
|
|
1254
1331
|
nextFreeIndex = 0;
|
|
1255
1332
|
nextFrameId = 1;
|
|
1256
1333
|
/**
|
|
1257
|
-
* Cumulative count of
|
|
1258
|
-
*
|
|
1259
|
-
*
|
|
1260
|
-
*
|
|
1334
|
+
* Cumulative count of dropped (shed) responses on the single-frame and
|
|
1335
|
+
* batch inference paths: the Python per-model bound, the Python
|
|
1336
|
+
* deadline-expired shed, and the TS in-flight watchdog's own
|
|
1337
|
+
* deadline-expired resolution all land here. Without this a shed response
|
|
1338
|
+
* is indistinguishable from a genuine "no detections" result.
|
|
1261
1339
|
*/
|
|
1262
1340
|
droppedResponseCount = 0;
|
|
1341
|
+
/**
|
|
1342
|
+
* The `shedReason: 'deadline-expired'` subset of {@link
|
|
1343
|
+
* droppedResponseCount} — work that was ALREADY DEAD when it would have
|
|
1344
|
+
* run (D350). This is the number that was zero-by-construction while the
|
|
1345
|
+
* 2026-09-04 storm burned the accelerator on abandoned requests for 4h15m;
|
|
1346
|
+
* a rising value here under load is the pool refusing dead work, which is
|
|
1347
|
+
* the fix working, not a fault.
|
|
1348
|
+
*/
|
|
1349
|
+
deadlineExpiredShedCount = 0;
|
|
1263
1350
|
log;
|
|
1264
1351
|
concurrency;
|
|
1265
1352
|
tuning;
|
|
@@ -1345,7 +1432,12 @@ var SharedInferencePool = class {
|
|
|
1345
1432
|
}
|
|
1346
1433
|
async inferBatch(modelIndex, items, frameId = 0) {
|
|
1347
1434
|
if (items.length > 255) throw new Error(`SharedInferencePool.inferBatch: max 255 items per call, got ${items.length}`);
|
|
1348
|
-
|
|
1435
|
+
const results = await this.pickWorker().inferBatch(this.encodeModelByte(modelIndex), items, frameId);
|
|
1436
|
+
for (const item of results) if (item["dropped"] === true) {
|
|
1437
|
+
this.droppedResponseCount++;
|
|
1438
|
+
if (item["shedReason"] === "deadline-expired") this.deadlineExpiredShedCount++;
|
|
1439
|
+
}
|
|
1440
|
+
return results;
|
|
1349
1441
|
}
|
|
1350
1442
|
async inferCached(modelIndex, frameId, deviceId) {
|
|
1351
1443
|
const w = this.pickWorker();
|
|
@@ -1359,6 +1451,16 @@ var SharedInferencePool = class {
|
|
|
1359
1451
|
getDroppedResponseCount() {
|
|
1360
1452
|
return this.droppedResponseCount;
|
|
1361
1453
|
}
|
|
1454
|
+
/**
|
|
1455
|
+
* The `deadline-expired` subset of {@link getDroppedResponseCount}:
|
|
1456
|
+
* requests refused (by the Python worker, or by the local in-flight
|
|
1457
|
+
* watchdog as its backstop) because their absolute deadline had already
|
|
1458
|
+
* passed — dead work NOT executed (D350). Monotonic for the pool's
|
|
1459
|
+
* lifetime; surfaced on the memory watchdog's `pool memory` line.
|
|
1460
|
+
*/
|
|
1461
|
+
getDeadlineExpiredShedCount() {
|
|
1462
|
+
return this.deadlineExpiredShedCount;
|
|
1463
|
+
}
|
|
1362
1464
|
getHandle(modelIndex) {
|
|
1363
1465
|
return new PoolHandle(this, modelIndex);
|
|
1364
1466
|
}
|
|
@@ -1497,10 +1599,13 @@ var SharedInferencePool = class {
|
|
|
1497
1599
|
trackDroppedResponse(result, modelIndex) {
|
|
1498
1600
|
if (result["dropped"] === true) {
|
|
1499
1601
|
this.droppedResponseCount++;
|
|
1602
|
+
if (result["shedReason"] === "deadline-expired") this.deadlineExpiredShedCount++;
|
|
1500
1603
|
const total = this.droppedResponseCount;
|
|
1501
1604
|
if (total === 1 || total % SHED_LOG_SAMPLE_EVERY === 0) this.log.debug("Python pool shed frame under overload", { meta: {
|
|
1502
1605
|
modelIndex,
|
|
1503
1606
|
droppedTotal: total,
|
|
1607
|
+
deadlineExpiredTotal: this.deadlineExpiredShedCount,
|
|
1608
|
+
...typeof result["shedReason"] === "string" ? { shedReason: result["shedReason"] } : {},
|
|
1504
1609
|
sampledEvery: SHED_LOG_SAMPLE_EVERY,
|
|
1505
1610
|
runtime: this.poolRuntime,
|
|
1506
1611
|
device: this.device ?? "default"
|
|
@@ -2389,11 +2494,13 @@ var EngineFactory = class {
|
|
|
2389
2494
|
if (!this.pool) return {
|
|
2390
2495
|
inFlight: 0,
|
|
2391
2496
|
shed: 0,
|
|
2392
|
-
dropped: 0
|
|
2497
|
+
dropped: 0,
|
|
2498
|
+
deadlineExpired: 0
|
|
2393
2499
|
};
|
|
2394
2500
|
return {
|
|
2395
2501
|
...this.pool.getBacklog(),
|
|
2396
|
-
dropped: this.pool.getDroppedResponseCount()
|
|
2502
|
+
dropped: this.pool.getDroppedResponseCount(),
|
|
2503
|
+
deadlineExpired: this.pool.getDeadlineExpiredShedCount()
|
|
2397
2504
|
};
|
|
2398
2505
|
}
|
|
2399
2506
|
/** Python-side per-worker memory diagnostics (see SharedInferencePool). */
|
|
@@ -2880,6 +2987,7 @@ async function samplePool(pool, log, reportedDead) {
|
|
|
2880
2987
|
inFlight: backlog.inFlight,
|
|
2881
2988
|
shed: backlog.shed,
|
|
2882
2989
|
dropped: backlog.dropped,
|
|
2990
|
+
deadlineExpired: backlog.deadlineExpired,
|
|
2883
2991
|
modelsLoaded: pool.factory.listLoaded().length,
|
|
2884
2992
|
memStats
|
|
2885
2993
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@camstack/addon-pipeline",
|
|
3
|
-
"version": "1.2.
|
|
3
|
+
"version": "1.2.178",
|
|
4
4
|
"description": "Pipeline bundle — runner, detection, motion, audio + stream broker. Multi-entry npm package shipping pipeline addons under a single bundle.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"camstack",
|
|
Binary file
|
|
Binary file
|
package/python/inference_pool.py
CHANGED
|
@@ -36,6 +36,21 @@ Runtime protocol:
|
|
|
36
36
|
0x02 — infer_raw payload = [1B model_idx][4B width][4B height]
|
|
37
37
|
[1B fmt 0=RGB,1=BGR,2=GRAY][pixels]
|
|
38
38
|
|
|
39
|
+
Wire-protocol versioning (D350): the startup config may declare
|
|
40
|
+
`protocolVersion`; the effective wire version is `min(PROTOCOL_VERSION,
|
|
41
|
+
declared)` and the ready reply reports this worker's own PROTOCOL_VERSION so
|
|
42
|
+
the Node side can negotiate the same minimum. At v2, every SHEDDABLE request
|
|
43
|
+
(infer_jpeg / infer_raw / infer_batch / infer_cached) prefixes its payload
|
|
44
|
+
with the request's ABSOLUTE deadline — [8B epoch-ms uint64 LE] — and this
|
|
45
|
+
worker checks it immediately before predict, answering
|
|
46
|
+
`{"dropped": true, "shedReason": "deadline-expired"}` instead of executing
|
|
47
|
+
work whose caller has already abandoned it (the 2026-09-04 storm: 4h15m of
|
|
48
|
+
the accelerator at 100% on replies discarded as `unknown request id`).
|
|
49
|
+
Commands and model loads NEVER carry a deadline at any version — a shed
|
|
50
|
+
command desynchronizes model state. A host that declares nothing is an OLD
|
|
51
|
+
host and gets byte-identical v1 behaviour; skew in either direction is the
|
|
52
|
+
NORMAL state during a rolling deploy.
|
|
53
|
+
|
|
39
54
|
Commands: load, unload, replace, reconfigure, status, uncache_frame, mem_stats.
|
|
40
55
|
"""
|
|
41
56
|
from __future__ import annotations
|
|
@@ -87,6 +102,93 @@ RAW_FMT_RGB = 0x00
|
|
|
87
102
|
RAW_FMT_BGR = 0x01
|
|
88
103
|
RAW_FMT_GRAY = 0x02
|
|
89
104
|
|
|
105
|
+
# ---------------------------------------------------------------------------
|
|
106
|
+
# Wire-protocol versioning + the deadline on the wire (D350)
|
|
107
|
+
# ---------------------------------------------------------------------------
|
|
108
|
+
|
|
109
|
+
# Highest wire version THIS worker implements. Reported in the ready reply;
|
|
110
|
+
# the effective version per connection is min() of both sides' declarations,
|
|
111
|
+
# so a worker and a host at different versions (normal during a rolling
|
|
112
|
+
# deploy) always agree on the layout.
|
|
113
|
+
PROTOCOL_VERSION = 2
|
|
114
|
+
|
|
115
|
+
# Opcodes that may carry (and be shed by) a deadline. Mirrors
|
|
116
|
+
# SHEDDABLE_MSG_TYPES in shared-inference-pool.ts — commands and cacheFrame
|
|
117
|
+
# are absent on BOTH sides: a shed command desynchronizes model state, a shed
|
|
118
|
+
# cacheFrame strands the inferCached that follows it.
|
|
119
|
+
SHEDDABLE_MSG_TYPES = frozenset({MSG_INFER_JPEG, MSG_INFER_RAW, MSG_INFER_BATCH, MSG_INFER_CACHED})
|
|
120
|
+
|
|
121
|
+
# Bytes of the absolute-deadline prefix a v2 sheddable payload carries.
|
|
122
|
+
DEADLINE_PREFIX_LEN = 8
|
|
123
|
+
|
|
124
|
+
# Sampling stride for the deadline-expired shed stderr line. Under a storm
|
|
125
|
+
# this fires at frame rate; one line per shed buried every other signal the
|
|
126
|
+
# last time a shed path logged unsampled (45 741 rows/hour, 2026-08-01). The
|
|
127
|
+
# counter (PoolCounters.shed_deadline_expired) keeps the exact total.
|
|
128
|
+
DEADLINE_SHED_LOG_EVERY = 500
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _negotiate_wire_version(config: dict) -> int:
|
|
132
|
+
"""Effective wire version for this connection: min(ours, host declared).
|
|
133
|
+
|
|
134
|
+
No declaration (an OLD host) or a garbage value negotiates v1 — the
|
|
135
|
+
byte-identical historical layout. Pure; unit-tested.
|
|
136
|
+
"""
|
|
137
|
+
try:
|
|
138
|
+
declared = int(config.get("protocolVersion", 1) or 1)
|
|
139
|
+
except (TypeError, ValueError):
|
|
140
|
+
declared = 1
|
|
141
|
+
return max(1, min(PROTOCOL_VERSION, declared))
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _split_deadline(msg_type: int, payload: bytes, wire_version: int) -> "tuple[int, bytes]":
|
|
145
|
+
"""Split a request payload into (deadline_ms, rest) by the NEGOTIATED
|
|
146
|
+
version. v1, and every non-sheddable opcode at any version, passes the
|
|
147
|
+
payload through untouched with deadline 0 (= no deadline, never sheds).
|
|
148
|
+
Raises ValueError on a v2 sheddable payload too short to carry its
|
|
149
|
+
prefix. Pure; unit-tested.
|
|
150
|
+
"""
|
|
151
|
+
if wire_version < 2 or msg_type not in SHEDDABLE_MSG_TYPES:
|
|
152
|
+
return 0, payload
|
|
153
|
+
if len(payload) < DEADLINE_PREFIX_LEN:
|
|
154
|
+
raise ValueError(
|
|
155
|
+
f"truncated deadline prefix on msg_type={msg_type} "
|
|
156
|
+
f"({len(payload)} bytes, need {DEADLINE_PREFIX_LEN})"
|
|
157
|
+
)
|
|
158
|
+
(deadline_ms,) = struct.unpack("<Q", payload[:DEADLINE_PREFIX_LEN])
|
|
159
|
+
return deadline_ms, payload[DEADLINE_PREFIX_LEN:]
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _deadline_expired(deadline_ms: int, now_ms: float) -> bool:
|
|
163
|
+
"""0 means "no deadline" and never expires (v1 host / unstamped)."""
|
|
164
|
+
return deadline_ms > 0 and now_ms > deadline_ms
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _dropped_deadline_payload() -> dict:
|
|
168
|
+
"""The shed reply for an expired request — same shape the TS side already
|
|
169
|
+
recognises as flow control (`poolResultToEngineOutput`)."""
|
|
170
|
+
return {
|
|
171
|
+
"kind": "detections",
|
|
172
|
+
"detections": [],
|
|
173
|
+
"inferenceMs": 0,
|
|
174
|
+
"dropped": True,
|
|
175
|
+
"shedReason": "deadline-expired",
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _log_deadline_shed(total: int, model_idx: int, late_ms: float) -> None:
|
|
180
|
+
"""Sampled stderr line for a deadline shed — first occurrence and every
|
|
181
|
+
DEADLINE_SHED_LOG_EVERY-th, carrying the running total so nothing is
|
|
182
|
+
lost, only repeated. A branch that drops work silently is how the
|
|
183
|
+
2026-09-04 collapse produced not one line in four hours."""
|
|
184
|
+
if total != 1 and total % DEADLINE_SHED_LOG_EVERY != 0:
|
|
185
|
+
return
|
|
186
|
+
sys.stderr.write(
|
|
187
|
+
f"deadline-expired shed: model {model_idx}, {late_ms:.0f}ms past deadline "
|
|
188
|
+
f"(total {total}, sampled every {DEADLINE_SHED_LOG_EVERY})\n"
|
|
189
|
+
)
|
|
190
|
+
sys.stderr.flush()
|
|
191
|
+
|
|
90
192
|
|
|
91
193
|
# ---------------------------------------------------------------------------
|
|
92
194
|
# Preprocessing (unchanged from v1 — same shapes/math, moved into helpers)
|
|
@@ -1891,6 +1993,10 @@ class PoolCounters:
|
|
|
1891
1993
|
infer_cached: int = 0
|
|
1892
1994
|
commands: int = 0
|
|
1893
1995
|
frames_cached: int = 0
|
|
1996
|
+
# Requests answered `dropped` because their wire deadline had already
|
|
1997
|
+
# passed when they would have run — dead work NOT executed (D350). For a
|
|
1998
|
+
# batch, every item counts.
|
|
1999
|
+
shed_deadline_expired: int = 0
|
|
1894
2000
|
|
|
1895
2001
|
def as_dict(self) -> dict:
|
|
1896
2002
|
return {
|
|
@@ -1900,6 +2006,7 @@ class PoolCounters:
|
|
|
1900
2006
|
"inferCached": self.infer_cached,
|
|
1901
2007
|
"commands": self.commands,
|
|
1902
2008
|
"framesCached": self.frames_cached,
|
|
2009
|
+
"shedDeadlineExpired": self.shed_deadline_expired,
|
|
1903
2010
|
"totalInference": (
|
|
1904
2011
|
self.infer_jpeg + self.infer_raw + self.infer_batch + self.infer_cached
|
|
1905
2012
|
),
|
|
@@ -2008,6 +2115,58 @@ def _mem_stats_payload(
|
|
|
2008
2115
|
}
|
|
2009
2116
|
|
|
2010
2117
|
|
|
2118
|
+
# ---------------------------------------------------------------------------
|
|
2119
|
+
# Single-frame inference execution — module-level so the deadline gate is
|
|
2120
|
+
# unit-testable (test_inference_pool_deadline.py) without driving the whole
|
|
2121
|
+
# event loop. `handle_inference` in _run() delegates here; the caller's
|
|
2122
|
+
# `finally` still owns the backpressure slot release.
|
|
2123
|
+
# ---------------------------------------------------------------------------
|
|
2124
|
+
|
|
2125
|
+
|
|
2126
|
+
async def _execute_inference(
|
|
2127
|
+
req_id: int,
|
|
2128
|
+
img: Any,
|
|
2129
|
+
model_idx: int,
|
|
2130
|
+
deadline_ms: int,
|
|
2131
|
+
*,
|
|
2132
|
+
models: "list[ModelSlot]",
|
|
2133
|
+
dispatcher: Any,
|
|
2134
|
+
writer: Any,
|
|
2135
|
+
batch_mode: str,
|
|
2136
|
+
get_accumulator: "Optional[Callable[[int], Any]]",
|
|
2137
|
+
counters: "PoolCounters",
|
|
2138
|
+
) -> None:
|
|
2139
|
+
"""Deadline gate + predict for ONE admitted single-frame request.
|
|
2140
|
+
|
|
2141
|
+
The deadline is checked HERE — immediately before predict — rather than
|
|
2142
|
+
only at arrival, because this coroutine is entered both on admission and
|
|
2143
|
+
when the backpressure queue promotes a waiting frame: the wait is exactly
|
|
2144
|
+
where a request goes stale. The gate comes first (before the model-loaded
|
|
2145
|
+
check) on purpose — the cheapest question first, and an abandoned request
|
|
2146
|
+
deserves a shed, not a diagnosis. `dispatcher` and `writer` are
|
|
2147
|
+
duck-typed (RuntimeDispatcher / ResponseWriter in production).
|
|
2148
|
+
"""
|
|
2149
|
+
now_ms = time.time() * 1000.0
|
|
2150
|
+
if _deadline_expired(deadline_ms, now_ms):
|
|
2151
|
+
counters.shed_deadline_expired += 1
|
|
2152
|
+
_log_deadline_shed(counters.shed_deadline_expired, model_idx, now_ms - deadline_ms)
|
|
2153
|
+
await writer.send(req_id, _dropped_deadline_payload())
|
|
2154
|
+
return
|
|
2155
|
+
if model_idx >= len(models) or not models[model_idx].loaded:
|
|
2156
|
+
await writer.send(req_id, {
|
|
2157
|
+
"error": f"Model {model_idx} not loaded",
|
|
2158
|
+
"kind": "detections",
|
|
2159
|
+
"detections": [],
|
|
2160
|
+
"inferenceMs": 0,
|
|
2161
|
+
})
|
|
2162
|
+
return
|
|
2163
|
+
if batch_mode == "window" and get_accumulator is not None:
|
|
2164
|
+
await get_accumulator(model_idx).submit(req_id, img)
|
|
2165
|
+
return
|
|
2166
|
+
result = await dispatcher.run(models[model_idx], img)
|
|
2167
|
+
await writer.send(req_id, result)
|
|
2168
|
+
|
|
2169
|
+
|
|
2011
2170
|
# ---------------------------------------------------------------------------
|
|
2012
2171
|
# IPC — binary framing with request_id multiplexing
|
|
2013
2172
|
# ---------------------------------------------------------------------------
|
|
@@ -2192,6 +2351,10 @@ async def _run() -> None:
|
|
|
2192
2351
|
config = json.loads(payload)
|
|
2193
2352
|
runtime = config.get("runtime", "coreml")
|
|
2194
2353
|
pool_device = str(config.get("device", "") or "")
|
|
2354
|
+
# Wire version for THIS connection (D350): an old host declares nothing
|
|
2355
|
+
# and negotiates v1 — byte-identical historical behaviour, no deadline
|
|
2356
|
+
# prefix expected on any payload.
|
|
2357
|
+
wire_version = _negotiate_wire_version(config)
|
|
2195
2358
|
concurrency = int(config.get("concurrency", 1) or 1)
|
|
2196
2359
|
batch_mode = str(config.get("batch_mode", "none"))
|
|
2197
2360
|
if batch_mode not in ("none", "list", "window"):
|
|
@@ -2201,7 +2364,8 @@ async def _run() -> None:
|
|
|
2201
2364
|
|
|
2202
2365
|
sys.stderr.write(
|
|
2203
2366
|
f"Initializing runtime: {runtime} (concurrency={concurrency}, "
|
|
2204
|
-
f"batch_mode={batch_mode}, window_ms={window_ms}, max_batch={max_batch_size}
|
|
2367
|
+
f"batch_mode={batch_mode}, window_ms={window_ms}, max_batch={max_batch_size}, "
|
|
2368
|
+
f"wire_version={wire_version})\n"
|
|
2205
2369
|
)
|
|
2206
2370
|
sys.stderr.flush()
|
|
2207
2371
|
_init_runtime(runtime, pool_device)
|
|
@@ -2266,6 +2430,10 @@ async def _run() -> None:
|
|
|
2266
2430
|
"startupMs": startup_ms,
|
|
2267
2431
|
"runtime": runtime,
|
|
2268
2432
|
"workers": dispatcher.workers,
|
|
2433
|
+
# Our own capability, NOT the negotiated minimum: the Node side runs
|
|
2434
|
+
# the same min() against what it declared, so both ends derive the
|
|
2435
|
+
# identical effective version (D350).
|
|
2436
|
+
"protocolVersion": PROTOCOL_VERSION,
|
|
2269
2437
|
})
|
|
2270
2438
|
|
|
2271
2439
|
# ── Window accumulator (per-model) ──────────────────────────────
|
|
@@ -2358,21 +2526,23 @@ async def _run() -> None:
|
|
|
2358
2526
|
return acc
|
|
2359
2527
|
|
|
2360
2528
|
# ── Main loop ───────────────────────────────────────────────────
|
|
2361
|
-
async def handle_inference(
|
|
2529
|
+
async def handle_inference(
|
|
2530
|
+
req_id: int, img: Image.Image, model_idx: int, deadline_ms: int = 0,
|
|
2531
|
+
) -> None:
|
|
2362
2532
|
try:
|
|
2363
|
-
|
|
2364
|
-
|
|
2365
|
-
|
|
2366
|
-
|
|
2367
|
-
|
|
2368
|
-
|
|
2369
|
-
|
|
2370
|
-
|
|
2371
|
-
|
|
2372
|
-
|
|
2373
|
-
|
|
2374
|
-
|
|
2375
|
-
|
|
2533
|
+
# The deadline gate lives in _execute_inference, immediately
|
|
2534
|
+
# before predict — it runs here on admission AND again when a
|
|
2535
|
+
# queued frame is promoted below, which is where a request goes
|
|
2536
|
+
# stale (D350).
|
|
2537
|
+
await _execute_inference(
|
|
2538
|
+
req_id, img, model_idx, deadline_ms,
|
|
2539
|
+
models=models,
|
|
2540
|
+
dispatcher=dispatcher,
|
|
2541
|
+
writer=writer,
|
|
2542
|
+
batch_mode=batch_mode,
|
|
2543
|
+
get_accumulator=get_accumulator,
|
|
2544
|
+
counters=counters,
|
|
2545
|
+
)
|
|
2376
2546
|
except Exception as exc:
|
|
2377
2547
|
sys.stderr.write(f"Inference error (model {model_idx}, req {req_id}): {exc}\n")
|
|
2378
2548
|
sys.stderr.flush()
|
|
@@ -2385,8 +2555,10 @@ async def _run() -> None:
|
|
|
2385
2555
|
finally:
|
|
2386
2556
|
# Release this model's running slot; promote the next queued
|
|
2387
2557
|
# frame (if any) into the freed slot immediately.
|
|
2388
|
-
for next_req_id, next_img in backpressure.complete(model_idx):
|
|
2389
|
-
asyncio.create_task(
|
|
2558
|
+
for next_req_id, next_img, next_deadline_ms in backpressure.complete(model_idx):
|
|
2559
|
+
asyncio.create_task(
|
|
2560
|
+
handle_inference(next_req_id, next_img, model_idx, next_deadline_ms)
|
|
2561
|
+
)
|
|
2390
2562
|
|
|
2391
2563
|
def _send_dropped(req_id: int) -> None:
|
|
2392
2564
|
# Fast shed response — returned in microseconds instead of queuing the
|
|
@@ -2399,19 +2571,36 @@ async def _run() -> None:
|
|
|
2399
2571
|
"dropped": True,
|
|
2400
2572
|
}))
|
|
2401
2573
|
|
|
2402
|
-
def _dispatch_inference(
|
|
2574
|
+
def _dispatch_inference(
|
|
2575
|
+
req_id: int, img: Image.Image, model_idx: int, deadline_ms: int = 0,
|
|
2576
|
+
) -> None:
|
|
2403
2577
|
# Gate every single-frame inference through the per-model in-flight
|
|
2404
2578
|
# bound. Slot accounting is synchronous (admit here, complete in the
|
|
2405
2579
|
# handle_inference `finally`), so a slot can never leak past a task's
|
|
2406
2580
|
# lifetime. Under overload the OLDEST queued frame is shed and answered
|
|
2407
2581
|
# immediately so the caller never hangs on a frame we chose not to run.
|
|
2408
|
-
|
|
2409
|
-
|
|
2582
|
+
# The wire deadline rides the queued tuple so the gate in
|
|
2583
|
+
# _execute_inference sees it AFTER the wait, not just at arrival.
|
|
2584
|
+
to_run, to_drop = backpressure.admit(model_idx, (req_id, img, deadline_ms))
|
|
2585
|
+
for dropped_req_id, _dropped_img, _dropped_deadline in to_drop:
|
|
2410
2586
|
_send_dropped(dropped_req_id)
|
|
2411
|
-
for run_req_id, run_img in to_run:
|
|
2412
|
-
asyncio.create_task(handle_inference(run_req_id, run_img, model_idx))
|
|
2413
|
-
|
|
2414
|
-
async def handle_batch(
|
|
2587
|
+
for run_req_id, run_img, run_deadline_ms in to_run:
|
|
2588
|
+
asyncio.create_task(handle_inference(run_req_id, run_img, model_idx, run_deadline_ms))
|
|
2589
|
+
|
|
2590
|
+
async def handle_batch(
|
|
2591
|
+
req_id: int, model_idx: int, items: list[Image.Image], deadline_ms: int = 0,
|
|
2592
|
+
) -> None:
|
|
2593
|
+
# Deadline gate first — the whole batch is one abandoned caller, so
|
|
2594
|
+
# every item is answered as a well-formed dropped record and none of
|
|
2595
|
+
# the work runs (D350).
|
|
2596
|
+
now_ms = time.time() * 1000.0
|
|
2597
|
+
if _deadline_expired(deadline_ms, now_ms):
|
|
2598
|
+
counters.shed_deadline_expired += len(items)
|
|
2599
|
+
_log_deadline_shed(counters.shed_deadline_expired, model_idx, now_ms - deadline_ms)
|
|
2600
|
+
await writer.send(req_id, {
|
|
2601
|
+
"results": [_dropped_deadline_payload() for _ in items],
|
|
2602
|
+
})
|
|
2603
|
+
return
|
|
2415
2604
|
if model_idx >= len(models) or not models[model_idx].loaded:
|
|
2416
2605
|
await writer.send(req_id, {
|
|
2417
2606
|
"error": f"Model {model_idx} not loaded",
|
|
@@ -2459,6 +2648,14 @@ async def _run() -> None:
|
|
|
2459
2648
|
except asyncio.IncompleteReadError:
|
|
2460
2649
|
return
|
|
2461
2650
|
|
|
2651
|
+
# v2 sheddable payloads carry [8B absolute deadline] first; v1 (and
|
|
2652
|
+
# every command) passes through untouched with deadline 0 (D350).
|
|
2653
|
+
try:
|
|
2654
|
+
deadline_ms, payload = _split_deadline(msg_type, payload, wire_version)
|
|
2655
|
+
except ValueError as exc:
|
|
2656
|
+
await writer.send(req_id, {"error": str(exc)})
|
|
2657
|
+
continue
|
|
2658
|
+
|
|
2462
2659
|
if msg_type == MSG_COMMAND:
|
|
2463
2660
|
counters.commands += 1
|
|
2464
2661
|
try:
|
|
@@ -2505,7 +2702,7 @@ async def _run() -> None:
|
|
|
2505
2702
|
except Exception as exc:
|
|
2506
2703
|
await writer.send(req_id, {"error": f"jpeg decode failed: {exc}"})
|
|
2507
2704
|
continue
|
|
2508
|
-
_dispatch_inference(req_id, img, model_idx)
|
|
2705
|
+
_dispatch_inference(req_id, img, model_idx, deadline_ms)
|
|
2509
2706
|
|
|
2510
2707
|
elif msg_type == MSG_INFER_RAW:
|
|
2511
2708
|
counters.infer_raw += 1
|
|
@@ -2521,7 +2718,7 @@ async def _run() -> None:
|
|
|
2521
2718
|
except Exception as exc:
|
|
2522
2719
|
await writer.send(req_id, {"error": f"raw wrap failed: {exc}"})
|
|
2523
2720
|
continue
|
|
2524
|
-
_dispatch_inference(req_id, img, model_idx)
|
|
2721
|
+
_dispatch_inference(req_id, img, model_idx, deadline_ms)
|
|
2525
2722
|
|
|
2526
2723
|
elif msg_type == MSG_INFER_BATCH:
|
|
2527
2724
|
# Header: [1B model_idx][1B count][4B frame_id]
|
|
@@ -2567,7 +2764,7 @@ async def _run() -> None:
|
|
|
2567
2764
|
if parse_err is not None:
|
|
2568
2765
|
await writer.send(req_id, {"error": parse_err, "results": []})
|
|
2569
2766
|
continue
|
|
2570
|
-
asyncio.create_task(handle_batch(req_id, model_idx, items))
|
|
2767
|
+
asyncio.create_task(handle_batch(req_id, model_idx, items, deadline_ms))
|
|
2571
2768
|
|
|
2572
2769
|
elif msg_type == MSG_CACHE_FRAME:
|
|
2573
2770
|
counters.frames_cached += 1
|
|
@@ -2605,7 +2802,7 @@ async def _run() -> None:
|
|
|
2605
2802
|
if model_idx >= len(models) or not models[model_idx].loaded:
|
|
2606
2803
|
await writer.send(req_id, {"error": f"model {model_idx} not loaded"})
|
|
2607
2804
|
continue
|
|
2608
|
-
_dispatch_inference(req_id, img, model_idx)
|
|
2805
|
+
_dispatch_inference(req_id, img, model_idx, deadline_ms)
|
|
2609
2806
|
|
|
2610
2807
|
else:
|
|
2611
2808
|
await writer.send(req_id, {"error": f"unknown msg_type: {msg_type}"})
|
|
@@ -0,0 +1,253 @@
|
|
|
1
|
+
"""Unit tests for the deadline-on-the-wire shed (D350).
|
|
2
|
+
|
|
3
|
+
On a pool timeout the TS runner drops the frame and returns null — but the
|
|
4
|
+
Python worker still executed the abandoned request and its late reply was
|
|
5
|
+
discarded as `Response for unknown request id`. Measured 2026-09-04: a 4h15m
|
|
6
|
+
storm, 16 146 timeouts, the accelerator at 100% on results nobody read.
|
|
7
|
+
|
|
8
|
+
Each SHEDDABLE request (infer_jpeg / infer_raw / infer_batch / infer_cached)
|
|
9
|
+
now carries its ABSOLUTE deadline (epoch ms, uint64 LE) on the wire when both
|
|
10
|
+
sides negotiated protocol v2; the worker checks it immediately before predict
|
|
11
|
+
and answers `{dropped: true, shedReason: "deadline-expired"}` instead of
|
|
12
|
+
running dead work. Commands and model loads NEVER carry a deadline — a shed
|
|
13
|
+
command desynchronizes model state.
|
|
14
|
+
|
|
15
|
+
Pure-python: stubs numpy / PIL / postprocessors in sys.modules so the module
|
|
16
|
+
imports without the ML runtime installed (inference_pool uses
|
|
17
|
+
`from __future__ import annotations`, so annotations never touch the stubs).
|
|
18
|
+
|
|
19
|
+
Run: python3 -m unittest test_inference_pool_deadline -v
|
|
20
|
+
"""
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import struct
|
|
24
|
+
import sys
|
|
25
|
+
import time
|
|
26
|
+
import types
|
|
27
|
+
import unittest
|
|
28
|
+
|
|
29
|
+
# ---------------------------------------------------------------------------
|
|
30
|
+
# Stub the heavy third-party imports BEFORE importing inference_pool.
|
|
31
|
+
# ---------------------------------------------------------------------------
|
|
32
|
+
|
|
33
|
+
try: # real numpy when the sandbox has it — a blind stub POISONS the whole
|
|
34
|
+
import numpy # noqa: F401 # pytest session for every sibling that needs it
|
|
35
|
+
except ImportError:
|
|
36
|
+
_np = types.ModuleType("numpy")
|
|
37
|
+
_np.float32 = float
|
|
38
|
+
_np.array = lambda values, dtype=None: values
|
|
39
|
+
sys.modules["numpy"] = _np
|
|
40
|
+
|
|
41
|
+
try: # real PIL when present, for the same reason
|
|
42
|
+
import PIL.Image # noqa: F401
|
|
43
|
+
except ImportError:
|
|
44
|
+
_pil = types.ModuleType("PIL")
|
|
45
|
+
_pil_image = types.ModuleType("PIL.Image")
|
|
46
|
+
_pil.Image = _pil_image
|
|
47
|
+
sys.modules["PIL"] = _pil
|
|
48
|
+
sys.modules["PIL.Image"] = _pil_image
|
|
49
|
+
|
|
50
|
+
if "postprocessors" not in sys.modules:
|
|
51
|
+
_pp = types.ModuleType("postprocessors")
|
|
52
|
+
_pp.POSTPROCESSORS = {}
|
|
53
|
+
sys.modules["postprocessors"] = _pp
|
|
54
|
+
|
|
55
|
+
from inference_pool import ( # noqa: E402
|
|
56
|
+
MSG_COMMAND,
|
|
57
|
+
MSG_INFER_BATCH,
|
|
58
|
+
MSG_INFER_CACHED,
|
|
59
|
+
MSG_INFER_JPEG,
|
|
60
|
+
MSG_INFER_RAW,
|
|
61
|
+
PROTOCOL_VERSION,
|
|
62
|
+
SHEDDABLE_MSG_TYPES,
|
|
63
|
+
ModelSlot,
|
|
64
|
+
PoolCounters,
|
|
65
|
+
_deadline_expired,
|
|
66
|
+
_execute_inference,
|
|
67
|
+
_negotiate_wire_version,
|
|
68
|
+
_split_deadline,
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
# ---------------------------------------------------------------------------
|
|
73
|
+
# Fakes — duck-typed stand-ins for the loop's writer / dispatcher.
|
|
74
|
+
# ---------------------------------------------------------------------------
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class _FakeWriter:
|
|
78
|
+
"""Collects (req_id, payload) pairs the loop would write to stdout."""
|
|
79
|
+
|
|
80
|
+
def __init__(self) -> None:
|
|
81
|
+
self.sent: list[tuple[int, dict]] = []
|
|
82
|
+
|
|
83
|
+
async def send(self, req_id: int, payload: dict) -> None:
|
|
84
|
+
self.sent.append((req_id, payload))
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class _FakeDispatcher:
|
|
88
|
+
"""Counts predict dispatches — the thing an expired request must NOT do."""
|
|
89
|
+
|
|
90
|
+
def __init__(self) -> None:
|
|
91
|
+
self.run_calls = 0
|
|
92
|
+
|
|
93
|
+
async def run(self, slot: object, img: object) -> dict:
|
|
94
|
+
self.run_calls += 1
|
|
95
|
+
return {"kind": "detections", "detections": [], "inferenceMs": 1.0}
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _loaded_models() -> list:
|
|
99
|
+
slot = ModelSlot()
|
|
100
|
+
slot.loaded = True
|
|
101
|
+
return [slot]
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _now_ms() -> int:
|
|
105
|
+
return int(time.time() * 1000)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
# ---------------------------------------------------------------------------
|
|
109
|
+
# The gate: expired ⇒ dropped, NO predict. Fresh ⇒ predict runs.
|
|
110
|
+
# ---------------------------------------------------------------------------
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
class DeadlineGateTest(unittest.IsolatedAsyncioTestCase):
|
|
114
|
+
async def test_expired_request_is_answered_dropped_and_no_predict_ran(self) -> None:
|
|
115
|
+
writer = _FakeWriter()
|
|
116
|
+
dispatcher = _FakeDispatcher()
|
|
117
|
+
counters = PoolCounters()
|
|
118
|
+
await _execute_inference(
|
|
119
|
+
7,
|
|
120
|
+
object(),
|
|
121
|
+
0,
|
|
122
|
+
_now_ms() - 5_000, # expired 5s ago — the storm's steady state
|
|
123
|
+
models=_loaded_models(),
|
|
124
|
+
dispatcher=dispatcher,
|
|
125
|
+
writer=writer,
|
|
126
|
+
batch_mode="none",
|
|
127
|
+
get_accumulator=None,
|
|
128
|
+
counters=counters,
|
|
129
|
+
)
|
|
130
|
+
# THE fix: dead work is not executed.
|
|
131
|
+
self.assertEqual(dispatcher.run_calls, 0)
|
|
132
|
+
self.assertEqual(len(writer.sent), 1)
|
|
133
|
+
req_id, payload = writer.sent[0]
|
|
134
|
+
self.assertEqual(req_id, 7)
|
|
135
|
+
self.assertIs(payload.get("dropped"), True)
|
|
136
|
+
self.assertEqual(payload.get("shedReason"), "deadline-expired")
|
|
137
|
+
self.assertEqual(payload.get("kind"), "detections")
|
|
138
|
+
self.assertEqual(payload.get("detections"), [])
|
|
139
|
+
# The shed is COUNTABLE — a silent drop is what let the original
|
|
140
|
+
# collapse run for four hours.
|
|
141
|
+
self.assertEqual(counters.shed_deadline_expired, 1)
|
|
142
|
+
self.assertEqual(counters.as_dict().get("shedDeadlineExpired"), 1)
|
|
143
|
+
|
|
144
|
+
async def test_future_deadline_runs_normally(self) -> None:
|
|
145
|
+
writer = _FakeWriter()
|
|
146
|
+
dispatcher = _FakeDispatcher()
|
|
147
|
+
counters = PoolCounters()
|
|
148
|
+
await _execute_inference(
|
|
149
|
+
8,
|
|
150
|
+
object(),
|
|
151
|
+
0,
|
|
152
|
+
_now_ms() + 60_000,
|
|
153
|
+
models=_loaded_models(),
|
|
154
|
+
dispatcher=dispatcher,
|
|
155
|
+
writer=writer,
|
|
156
|
+
batch_mode="none",
|
|
157
|
+
get_accumulator=None,
|
|
158
|
+
counters=counters,
|
|
159
|
+
)
|
|
160
|
+
self.assertEqual(dispatcher.run_calls, 1)
|
|
161
|
+
req_id, payload = writer.sent[0]
|
|
162
|
+
self.assertEqual(req_id, 8)
|
|
163
|
+
self.assertNotIn("dropped", payload)
|
|
164
|
+
self.assertEqual(counters.shed_deadline_expired, 0)
|
|
165
|
+
|
|
166
|
+
async def test_no_deadline_runs_normally(self) -> None:
|
|
167
|
+
"""deadline 0 = old host / no deadline — never sheds (skew safety)."""
|
|
168
|
+
writer = _FakeWriter()
|
|
169
|
+
dispatcher = _FakeDispatcher()
|
|
170
|
+
counters = PoolCounters()
|
|
171
|
+
await _execute_inference(
|
|
172
|
+
9,
|
|
173
|
+
object(),
|
|
174
|
+
0,
|
|
175
|
+
0,
|
|
176
|
+
models=_loaded_models(),
|
|
177
|
+
dispatcher=dispatcher,
|
|
178
|
+
writer=writer,
|
|
179
|
+
batch_mode="none",
|
|
180
|
+
get_accumulator=None,
|
|
181
|
+
counters=counters,
|
|
182
|
+
)
|
|
183
|
+
self.assertEqual(dispatcher.run_calls, 1)
|
|
184
|
+
self.assertEqual(counters.shed_deadline_expired, 0)
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
# ---------------------------------------------------------------------------
|
|
188
|
+
# Wire parsing — both layouts, chosen by the negotiated version.
|
|
189
|
+
# ---------------------------------------------------------------------------
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
class SplitDeadlineTest(unittest.TestCase):
|
|
193
|
+
def test_v1_payload_passes_through_untouched(self) -> None:
|
|
194
|
+
"""An OLD host never stamps a deadline — the payload is the old layout
|
|
195
|
+
and must not lose its first 8 bytes to a prefix that is not there."""
|
|
196
|
+
payload = bytes([3]) + b"jpegdata"
|
|
197
|
+
self.assertEqual(_split_deadline(MSG_INFER_JPEG, payload, 1), (0, payload))
|
|
198
|
+
|
|
199
|
+
def test_v2_sheddable_payload_strips_the_deadline_prefix(self) -> None:
|
|
200
|
+
deadline = 1_757_000_000_123
|
|
201
|
+
rest = bytes([3]) + b"jpegdata"
|
|
202
|
+
payload = struct.pack("<Q", deadline) + rest
|
|
203
|
+
self.assertEqual(_split_deadline(MSG_INFER_JPEG, payload, 2), (deadline, rest))
|
|
204
|
+
|
|
205
|
+
def test_v2_applies_to_every_sheddable_opcode(self) -> None:
|
|
206
|
+
deadline = 42
|
|
207
|
+
rest = b"\x01payload"
|
|
208
|
+
for msg_type in (MSG_INFER_JPEG, MSG_INFER_RAW, MSG_INFER_BATCH, MSG_INFER_CACHED):
|
|
209
|
+
payload = struct.pack("<Q", deadline) + rest
|
|
210
|
+
self.assertEqual(_split_deadline(msg_type, payload, 2), (deadline, rest))
|
|
211
|
+
self.assertIn(msg_type, SHEDDABLE_MSG_TYPES)
|
|
212
|
+
|
|
213
|
+
def test_commands_never_carry_a_deadline_even_at_v2(self) -> None:
|
|
214
|
+
"""A shed command desynchronizes model state — commands keep the old
|
|
215
|
+
layout at every version."""
|
|
216
|
+
payload = b'{"cmd": "status"}'
|
|
217
|
+
self.assertEqual(_split_deadline(MSG_COMMAND, payload, 2), (0, payload))
|
|
218
|
+
self.assertNotIn(MSG_COMMAND, SHEDDABLE_MSG_TYPES)
|
|
219
|
+
|
|
220
|
+
def test_truncated_deadline_prefix_raises(self) -> None:
|
|
221
|
+
with self.assertRaises(ValueError):
|
|
222
|
+
_split_deadline(MSG_INFER_JPEG, b"\x00\x01\x02", 2)
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
class DeadlineExpiredTest(unittest.TestCase):
|
|
226
|
+
def test_zero_deadline_never_expires(self) -> None:
|
|
227
|
+
self.assertFalse(_deadline_expired(0, 1e15))
|
|
228
|
+
|
|
229
|
+
def test_future_deadline_is_not_expired(self) -> None:
|
|
230
|
+
self.assertFalse(_deadline_expired(2_000, 1_999.0))
|
|
231
|
+
|
|
232
|
+
def test_past_deadline_is_expired(self) -> None:
|
|
233
|
+
self.assertTrue(_deadline_expired(2_000, 2_001.0))
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
class NegotiateWireVersionTest(unittest.TestCase):
|
|
237
|
+
def test_old_host_without_declaration_negotiates_v1(self) -> None:
|
|
238
|
+
"""A host at a different version is the NORMAL state during a deploy —
|
|
239
|
+
no declaration means the old layout, always."""
|
|
240
|
+
self.assertEqual(_negotiate_wire_version({}), 1)
|
|
241
|
+
|
|
242
|
+
def test_matching_host_negotiates_v2(self) -> None:
|
|
243
|
+
self.assertEqual(_negotiate_wire_version({"protocolVersion": 2}), 2)
|
|
244
|
+
|
|
245
|
+
def test_newer_host_is_capped_at_our_version(self) -> None:
|
|
246
|
+
self.assertEqual(_negotiate_wire_version({"protocolVersion": 99}), PROTOCOL_VERSION)
|
|
247
|
+
|
|
248
|
+
def test_garbage_declaration_falls_back_to_v1(self) -> None:
|
|
249
|
+
self.assertEqual(_negotiate_wire_version({"protocolVersion": "banana"}), 1)
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
if __name__ == "__main__":
|
|
253
|
+
unittest.main()
|