@xenon-device-management/xenon 1.11.0 → 1.11.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@xenon-device-management/xenon",
3
- "version": "1.11.0",
3
+ "version": "1.11.2",
4
4
  "description": "Xenon - Intelligent Mobile Infrastructure. A self-healing device orchestration platform for Appium.",
5
5
  "main": "./lib/src/index.js",
6
6
  "exports": {
@@ -832,20 +832,18 @@ router.get('/:udid/stream', (req, res) => __awaiter(void 0, void 0, void 0, func
832
832
  }
833
833
  }
834
834
  else {
835
- // Android auto-start
836
- const androidStreamService = typedi_1.Container.get(AndroidStreamService_1.default);
837
- const session = androidStreamService.getStreamStatus(udid);
838
- if (session && session.status === 'running') {
839
- mjpegPort = session.mjpegPort;
835
+ // Android auto-start. Always go through startStream: it health-checks an
836
+ // existing session before reusing it (decideAndroidStreamReuse) and dedupes
837
+ // concurrent starts. The short-circuit that used to live here — hand back
838
+ // any session marked 'running' — bypassed both, re-introducing the very
839
+ // stale-port bug the service now guards against. Reuse stays just as cheap;
840
+ // the health check is an in-process property read, not a request.
841
+ try {
842
+ const result = yield typedi_1.Container.get(AndroidStreamService_1.default).startStream(udid);
843
+ mjpegPort = result.mjpegPort;
840
844
  }
841
- else {
842
- try {
843
- const result = yield androidStreamService.startStream(udid);
844
- mjpegPort = result.mjpegPort;
845
- }
846
- catch (err) {
847
- return res.status(503).send({ error: 'Android stream failed', message: err.message });
848
- }
845
+ catch (err) {
846
+ return res.status(503).send({ error: 'Android stream failed', message: err.message });
849
847
  }
850
848
  }
851
849
  if (!mjpegPort) {
@@ -66,6 +66,7 @@ const ResourceIsolationService_1 = require("../../services/ResourceIsolationServ
66
66
  const PortAllocator_1 = require("../../services/PortAllocator");
67
67
  const ffmpegPath_1 = require("../../helpers/ffmpegPath");
68
68
  const singleFlight_1 = require("../../helpers/singleFlight");
69
+ const androidStreamReuse_1 = require("./androidStreamReuse");
69
70
  // JPEG quality for the in-process sharp encoder (0-100). ~78 approximates the
70
71
  // old ffmpeg `-q:v 8` (mjpeg quantizer scale) — the low-lag/quality knob.
71
72
  const JPEG_QUALITY = 78;
@@ -127,6 +128,29 @@ let AndroidStreamService = class AndroidStreamService {
127
128
  shouldIdleCapture(session) {
128
129
  return session.status === 'running' && session.viewerCount === 0;
129
130
  }
131
+ /**
132
+ * Log a stall once per episode, whichever path notices it first.
133
+ *
134
+ * Both the capture loop's catch and the MJPEG writer call this. The writer
135
+ * matters most: ending responses on a stall drops the viewer count to zero,
136
+ * which makes `shouldIdleCapture` true, which stops the loop attempting
137
+ * further captures — so the catch may never run again. A live cable-pull run
138
+ * hit exactly that and logged nothing at all.
139
+ *
140
+ * Returns whether this call did the logging, so the caller can decide what to
141
+ * say instead.
142
+ */
143
+ announceStallOnce(session, now, lastError) {
144
+ if (!session.captureHealth.takeStallAnnouncement(now))
145
+ return false;
146
+ const seconds = Math.round(session.captureHealth.failingForMs(now) / 1000);
147
+ logger_1.default.warn(`[${session.udid}] Device has not answered screencap for ${seconds}s ` +
148
+ `(${session.captureHealth.consecutiveFailures} consecutive failures` +
149
+ `${lastError ? `, last: ${lastError}` : ''}). Stream clients are being disconnected ` +
150
+ 'rather than served a frozen frame; the stream restarts on the next request once the ' +
151
+ 'device recovers.');
152
+ return true;
153
+ }
130
154
  captureLoop(udid, session) {
131
155
  return __awaiter(this, void 0, void 0, function* () {
132
156
  logger_1.default.info(`[${udid}] Background capture loop started.`);
@@ -142,8 +166,11 @@ let AndroidStreamService = class AndroidStreamService {
142
166
  yield new Promise((r) => setTimeout(r, 150));
143
167
  continue;
144
168
  }
169
+ // Hoisted out of the try so the catch can date a failure from when the
170
+ // attempt began: a 15s ADB timeout means the device had already been
171
+ // silent for 15s by the time we reach the catch.
172
+ const startTime = Date.now();
145
173
  try {
146
- const startTime = Date.now();
147
174
  // High-Speed Binary Snapshot
148
175
  const screenshot = yield DeviceLockManager_1.deviceLock.acquire(udid, () => __awaiter(this, void 0, void 0, function* () {
149
176
  return yield new Promise((resolve, reject) => {
@@ -196,6 +223,7 @@ let AndroidStreamService = class AndroidStreamService {
196
223
  logger_1.default.info(`[${udid}] First Android frame successfully captured and converted.`);
197
224
  }
198
225
  session.latestFrameTimestamp = Date.now();
226
+ session.captureHealth.recordSuccess();
199
227
  }
200
228
  }
201
229
  else {
@@ -206,7 +234,11 @@ let AndroidStreamService = class AndroidStreamService {
206
234
  yield new Promise((r) => setTimeout(r, delay));
207
235
  }
208
236
  catch (e) {
209
- logger_1.default.debug(`[${udid}] Capture failure: ${e.message}`);
237
+ const now = Date.now();
238
+ const { consecutive } = session.captureHealth.recordFailure(startTime, now);
239
+ if (!this.announceStallOnce(session, now, e.message)) {
240
+ logger_1.default.debug(`[${udid}] Capture failure (${consecutive} consecutive): ${e.message}`);
241
+ }
210
242
  yield new Promise((r) => setTimeout(r, 1000));
211
243
  }
212
244
  }
@@ -216,11 +248,27 @@ let AndroidStreamService = class AndroidStreamService {
216
248
  startStream(udid) {
217
249
  return __awaiter(this, void 0, void 0, function* () {
218
250
  const performStartup = () => __awaiter(this, void 0, void 0, function* () {
219
- var _a, _b;
251
+ var _a, _b, _c;
220
252
  try {
253
+ // A session is handed back only when it is actually serving. A closed
254
+ // server, or one that went live without ever capturing a frame, was
255
+ // otherwise reused indefinitely and every consumer — recording included
256
+ // — got a port that yields nothing. See androidStreamReuse / issue #196.
221
257
  const existing = this.sessions.get(udid);
222
- if (existing && (existing.status === 'running' || existing.status === 'starting')) {
223
- return { mjpegPort: existing.mjpegPort };
258
+ const decision = (0, androidStreamReuse_1.decideAndroidStreamReuse)(existing && {
259
+ status: existing.status,
260
+ serverListening: ((_a = existing.server) === null || _a === void 0 ? void 0 : _a.listening) === true,
261
+ hasFrame: existing.latestFrame !== undefined,
262
+ captureStalled: existing.captureHealth.isStalled(Date.now()),
263
+ });
264
+ if (decision.reuse) {
265
+ if (existing)
266
+ return { mjpegPort: existing.mjpegPort };
267
+ }
268
+ else if (existing) {
269
+ logger_1.default.warn(`[${udid}] Stream session on port ${existing.mjpegPort} is not usable ` +
270
+ `(${decision.reason}); discarding it and starting fresh.`);
271
+ yield this.disposeSession(udid, existing);
224
272
  }
225
273
  const device = yield device_store_1.DeviceStoreFactory.getStore().findDevice({ udid });
226
274
  if (!device)
@@ -247,6 +295,7 @@ let AndroidStreamService = class AndroidStreamService {
247
295
  viewerCount: 0,
248
296
  adbPath,
249
297
  adbHostArgs,
298
+ captureHealth: new androidStreamReuse_1.CaptureHealth(),
250
299
  };
251
300
  this.sessions.set(udid, candidate);
252
301
  try {
@@ -256,12 +305,12 @@ let AndroidStreamService = class AndroidStreamService {
256
305
  break;
257
306
  }
258
307
  catch (bindErr) {
259
- logger_1.default.warn(`[${udid}] MJPEG bind failed on ${mjpegPort} (attempt ${attempt + 1}): ${(_a = bindErr === null || bindErr === void 0 ? void 0 : bindErr.message) !== null && _a !== void 0 ? _a : bindErr}`);
308
+ logger_1.default.warn(`[${udid}] MJPEG bind failed on ${mjpegPort} (attempt ${attempt + 1}): ${(_b = bindErr === null || bindErr === void 0 ? void 0 : bindErr.message) !== null && _b !== void 0 ? _b : bindErr}`);
260
309
  this.sessions.delete(udid);
261
310
  try {
262
- (_b = candidate.server) === null || _b === void 0 ? void 0 : _b.close();
311
+ (_c = candidate.server) === null || _c === void 0 ? void 0 : _c.close();
263
312
  }
264
- catch (_c) {
313
+ catch (_d) {
265
314
  /* ignore */
266
315
  }
267
316
  yield portAllocator.release(mjpegPort).catch(() => undefined);
@@ -318,6 +367,37 @@ let AndroidStreamService = class AndroidStreamService {
318
367
  return this.startFlight.run(udid, performStartup);
319
368
  });
320
369
  }
370
+ /**
371
+ * Release a dead session's resources so a fresh one can replace it.
372
+ *
373
+ * Deliberately *not* stopStream(): that also unblocks the device when the
374
+ * lock is a manual one, which would drop the caller's own hold on the device
375
+ * halfway through a restart. Only stream-owned resources are touched here.
376
+ */
377
+ disposeSession(udid, session) {
378
+ return __awaiter(this, void 0, void 0, function* () {
379
+ var _a;
380
+ // The capture loop and every writeFrame loop poll this, so flipping it
381
+ // first stops them before the server goes away.
382
+ session.status = 'stopped';
383
+ try {
384
+ (_a = session.server) === null || _a === void 0 ? void 0 : _a.close();
385
+ }
386
+ catch (_b) {
387
+ /* ignore */
388
+ }
389
+ session.server = null;
390
+ // Only drop our own entry — never evict a newer session for this udid.
391
+ if (this.sessions.get(udid) === session)
392
+ this.sessions.delete(udid);
393
+ try {
394
+ yield typedi_1.Container.get(PortAllocator_1.PortAllocator).release(session.mjpegPort);
395
+ }
396
+ catch (e) {
397
+ logger_1.default.warn(`[${udid}] Failed to release mjpeg port lease while restarting stream: ${e}`);
398
+ }
399
+ });
400
+ }
321
401
  stopStream(udid) {
322
402
  return __awaiter(this, void 0, void 0, function* () {
323
403
  var _a;
@@ -386,6 +466,26 @@ let AndroidStreamService = class AndroidStreamService {
386
466
  !res.writable) {
387
467
  return;
388
468
  }
469
+ // The device has gone silent, so `latestFrame` is now a still image.
470
+ // End the response instead of rewriting it forever: ffmpeg finalises
471
+ // the mp4 with the real footage it captured, and browser clients fall
472
+ // into their normal stream-retry — which re-enters startStream, whose
473
+ // health check restarts the stream. Serving on would have produced a
474
+ // recording that looks valid and is a photograph (issue #200).
475
+ const nowMs = Date.now();
476
+ if (session.captureHealth.isStalled(nowMs)) {
477
+ // Announce from here too: this path is often the only one that gets
478
+ // to notice, because the disconnect below is what idles the capture
479
+ // loop in the first place.
480
+ this.announceStallOnce(session, nowMs);
481
+ try {
482
+ res.end();
483
+ }
484
+ catch (_a) {
485
+ /* ignore */
486
+ }
487
+ return;
488
+ }
389
489
  if (session.latestFrame) {
390
490
  try {
391
491
  res.write('--BoundaryString\r\n');
@@ -394,7 +494,7 @@ let AndroidStreamService = class AndroidStreamService {
394
494
  res.write(session.latestFrame);
395
495
  res.write('\r\n');
396
496
  }
397
- catch (_a) {
497
+ catch (_b) {
398
498
  return;
399
499
  }
400
500
  }
@@ -0,0 +1,128 @@
1
+ "use strict";
2
+ /**
3
+ * Decide whether an existing Android MJPEG stream session can be handed back as
4
+ * is, or has to be torn down and restarted.
5
+ *
6
+ * Why this exists: `startStream` used to return `existing.mjpegPort` for any
7
+ * session marked 'running' *or* 'starting', with nothing verifying that a
8
+ * server was still bound to it. A session whose HTTP server had closed, or one
9
+ * that went live without ever capturing a frame (startStream warns and
10
+ * continues after a 5s first-frame wait), was therefore reused indefinitely —
11
+ * every caller, `ensureMjpegForRecording` included, got a port that serves
12
+ * nothing, and ffmpeg left a 0-byte mp4 behind. Same silent symptom as #194,
13
+ * reached by a different route. See issue #196.
14
+ *
15
+ * This is the Android analogue of `resolveIosMjpegPort` (#191), minus the async
16
+ * probe: the MJPEG server is ours and in-process, so its liveness is a property
17
+ * read rather than a loopback request.
18
+ *
19
+ * Kept free of TypeDI and `http` so the decision is unit-testable; the service
20
+ * maps its live session onto the view below.
21
+ */
22
+ Object.defineProperty(exports, "__esModule", { value: true });
23
+ exports.CaptureHealth = exports.CAPTURE_STALL_MS = void 0;
24
+ exports.decideAndroidStreamReuse = decideAndroidStreamReuse;
25
+ function decideAndroidStreamReuse(session) {
26
+ if (!session)
27
+ return { reuse: false, reason: 'no-session' };
28
+ // 'starting' is deliberately excluded: the bind loop registers the candidate
29
+ // before awaiting the listen, so that status means "a port exists" and not
30
+ // "a port serves". Concurrent callers join the in-flight start via
31
+ // SingleFlight (#195) instead of being handed a half-bound port here.
32
+ if (session.status !== 'running')
33
+ return { reuse: false, reason: 'not-running' };
34
+ if (!session.serverListening)
35
+ return { reuse: false, reason: 'server-closed' };
36
+ if (!session.hasFrame)
37
+ return { reuse: false, reason: 'no-frame' };
38
+ // Ordered after hasFrame: "never captured anything" and "captured, then went
39
+ // silent" are different problems, and the reason ends up in a log line.
40
+ if (session.captureStalled)
41
+ return { reuse: false, reason: 'capture-stalled' };
42
+ return { reuse: true };
43
+ }
44
+ /**
45
+ * How long the device may go without producing a frame before anything we still
46
+ * hold is treated as a still image rather than a live feed.
47
+ *
48
+ * Deliberately generous. A single failed `screencap`, a device briefly held by
49
+ * the interaction lock, or one 15s ADB timeout must not tear down a live
50
+ * preview or an in-flight recording — the cure for that would be the
51
+ * over-eager-restart failure mode #198 was careful to avoid.
52
+ */
53
+ exports.CAPTURE_STALL_MS = 10000;
54
+ /**
55
+ * Tracks whether the device is still answering `screencap`.
56
+ *
57
+ * Why this exists (issue #200): the capture loop swallows capture errors and
58
+ * loops on `while (status === 'running' || status === 'starting')`, so a device
59
+ * that goes away — unplugged, adb killed, reboot — never changes the session's
60
+ * status. `latestFrame` is only ever assigned, never cleared, so the MJPEG
61
+ * server happily rewrote the last good JPEG every 60ms: a frozen preview, and a
62
+ * recording that ffprobe calls healthy but which is a still photograph. That is
63
+ * worse than the 0-byte failure of #194/#196, because it looks valid.
64
+ *
65
+ * The measure is time-since-the-device-last-answered, not a failure count: a
66
+ * fast `ADB Exit 1` and a 15s `ADB Timeout` are wildly different amounts of
67
+ * frozen video for the same count. It is also immune to the idle path —
68
+ * `shouldIdleCapture` skips capture entirely when nobody is watching, so an
69
+ * idle-but-healthy session records neither successes nor failures and simply
70
+ * stays in whatever state it was left in.
71
+ */
72
+ class CaptureHealth {
73
+ constructor() {
74
+ this.consecutive = 0;
75
+ this.announced = false;
76
+ }
77
+ /** A frame was captured — the device is answering. */
78
+ recordSuccess() {
79
+ this.failingSince = undefined;
80
+ this.consecutive = 0;
81
+ this.announced = false;
82
+ }
83
+ /**
84
+ * A capture attempt threw.
85
+ *
86
+ * `attemptStartedAt` (not `now`) dates the stall, because a 15s ADB timeout
87
+ * means the device had already been silent for 15s by the time we got here.
88
+ */
89
+ recordFailure(attemptStartedAt, now) {
90
+ if (this.failingSince === undefined)
91
+ this.failingSince = attemptStartedAt;
92
+ this.consecutive += 1;
93
+ return { consecutive: this.consecutive, stalled: this.isStalled(now) };
94
+ }
95
+ /**
96
+ * Claim the right to announce this stall episode: true for the first caller
97
+ * that observes the session stalled, false for every caller after it until a
98
+ * successful frame resets the episode.
99
+ *
100
+ * Why a claim rather than a flag computed inside `recordFailure`: ending
101
+ * client responses on a stall drops `viewerCount` to 0, `shouldIdleCapture`
102
+ * then stops the capture loop attempting anything more, and so the failure
103
+ * that would have crossed the threshold never happens — the catch never gets
104
+ * to warn. A live cable-pull run showed exactly that: 10 failures spanning
105
+ * 9.214s against a 10s threshold, then silence, so a device that had vanished
106
+ * produced no warning at all. Either path can claim it now, and neither
107
+ * double-logs.
108
+ */
109
+ takeStallAnnouncement(now) {
110
+ if (!this.isStalled(now) || this.announced)
111
+ return false;
112
+ this.announced = true;
113
+ return true;
114
+ }
115
+ /** Consecutive failed capture attempts in the current episode, for logging. */
116
+ get consecutiveFailures() {
117
+ return this.consecutive;
118
+ }
119
+ /** Capture has been failing long enough that any cached frame is a still. */
120
+ isStalled(now) {
121
+ return this.failingSince !== undefined && now - this.failingSince >= exports.CAPTURE_STALL_MS;
122
+ }
123
+ /** How long the device has been silent, for logging. 0 when healthy. */
124
+ failingForMs(now) {
125
+ return this.failingSince === undefined ? 0 : now - this.failingSince;
126
+ }
127
+ }
128
+ exports.CaptureHealth = CaptureHealth;
@@ -380,26 +380,60 @@ let VideoPipelineService = class VideoPipelineService {
380
380
  logger_1.default.error(`[VideoPipeline] FFMPEG Error [${sessionId}]: ${msg}`);
381
381
  }
382
382
  });
383
- ffmpegProc.on('exit', (code) => {
384
- if (code !== 0 && code !== null) {
385
- logger_1.default.warn(`[VideoPipeline] FFMPEG for ${sessionId} exited with code ${code}`);
386
- }
387
- this.activeRecordings.delete(sessionId);
388
- });
383
+ ffmpegProc.on('exit', (code) => this.handleProcessExit(sessionId, code, options.onExit));
389
384
  this.activeRecordings.set(sessionId, ffmpegProc);
390
385
  this.recordingPaths.set(sessionId, outputPath);
391
386
  });
392
387
  }
388
+ /**
389
+ * A recording's ffmpeg exited. Reports it to `onExit` only when nobody asked
390
+ * for it.
391
+ *
392
+ * `stopRecording` drops the handle *before* awaiting the exit, so a still
393
+ * present entry is the signal that this exit was not requested — the source
394
+ * ended or ffmpeg died. Getting that backwards would run the caller's
395
+ * source-ended path on every ordinary Stop as well.
396
+ *
397
+ * Split out from the listener so the decision is testable without spawning a
398
+ * process.
399
+ */
400
+ handleProcessExit(sessionId, code, onExit) {
401
+ var _a;
402
+ if (code !== 0 && code !== null) {
403
+ logger_1.default.warn(`[VideoPipeline] FFMPEG for ${sessionId} exited with code ${code}`);
404
+ }
405
+ const unrequested = this.activeRecordings.delete(sessionId);
406
+ if (!unrequested || !onExit)
407
+ return;
408
+ logger_1.default.info(`[VideoPipeline] FFMPEG for ${sessionId} exited on its own (${code === null ? 'signal' : `code ${code}`})`);
409
+ try {
410
+ onExit(code);
411
+ }
412
+ catch (err) {
413
+ logger_1.default.warn(`[VideoPipeline] onExit handler threw for ${sessionId}: ${(_a = err === null || err === void 0 ? void 0 : err.message) !== null && _a !== void 0 ? _a : err}`);
414
+ }
415
+ }
393
416
  /**
394
417
  * Stop recording and return the relative asset path
395
418
  */
396
419
  stopRecording(sessionId) {
397
420
  return __awaiter(this, void 0, void 0, function* () {
398
- var _a;
421
+ var _a, _b;
399
422
  const proc = this.activeRecordings.get(sessionId);
400
423
  const recordedPath = this.recordingPaths.get(sessionId);
401
424
  if (!proc) {
402
425
  logger_1.default.info(`[VideoPipeline] No active recording process for ${sessionId}, returning stored path if any.`);
426
+ // ffmpeg already exited on its own. The file still needs the faststart
427
+ // remux the normal stop path performs, or it stays a fragmented mp4 that
428
+ // some players will not scrub.
429
+ if (recordedPath) {
430
+ try {
431
+ yield remuxToStandardMp4(recordedPath);
432
+ }
433
+ catch (err) {
434
+ logger_1.default.warn(`[VideoPipeline] Remux failed for ${sessionId} (leaving fMP4): ${(_a = err === null || err === void 0 ? void 0 : err.message) !== null && _a !== void 0 ? _a : err}`);
435
+ }
436
+ }
403
437
  const relativePath = recordedPath
404
438
  ? path_1.default.relative(config_1.config.sessionAssetsPath, recordedPath)
405
439
  : null;
@@ -407,15 +441,17 @@ let VideoPipelineService = class VideoPipelineService {
407
441
  return relativePath;
408
442
  }
409
443
  logger_1.default.info(`[VideoPipeline] Stopping recording for ${sessionId}`);
410
- yield waitForFfmpegExit(proc, `recording ${sessionId}`);
444
+ // Drop the handle before awaiting: the exit that follows is one we asked
445
+ // for, and the exit listener must not report it as a source-initiated end.
411
446
  this.activeRecordings.delete(sessionId);
447
+ yield waitForFfmpegExit(proc, `recording ${sessionId}`);
412
448
  this.recordingPaths.delete(sessionId);
413
449
  if (recordedPath) {
414
450
  try {
415
451
  yield remuxToStandardMp4(recordedPath);
416
452
  }
417
453
  catch (err) {
418
- logger_1.default.warn(`[VideoPipeline] Remux failed for ${sessionId} (leaving fMP4): ${(_a = err === null || err === void 0 ? void 0 : err.message) !== null && _a !== void 0 ? _a : err}`);
454
+ logger_1.default.warn(`[VideoPipeline] Remux failed for ${sessionId} (leaving fMP4): ${(_b = err === null || err === void 0 ? void 0 : err.message) !== null && _b !== void 0 ? _b : err}`);
419
455
  }
420
456
  }
421
457
  return recordedPath ? path_1.default.relative(config_1.config.sessionAssetsPath, recordedPath) : null;