@appium/coresim 1.5.0 → 1.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/README.md +14 -0
- package/binding.gyp +7 -0
- package/lib/src/commands/video-recording.d.ts +26 -10
- package/lib/src/commands/video-recording.d.ts.map +1 -1
- package/lib/src/commands/video-recording.js +55 -16
- package/lib/src/commands/video-recording.js.map +1 -1
- package/lib/src/commands/video-stream.d.ts +55 -7
- package/lib/src/commands/video-stream.d.ts.map +1 -1
- package/lib/src/commands/video-stream.js +37 -35
- package/lib/src/commands/video-stream.js.map +1 -1
- package/lib/src/index.d.ts +1 -1
- package/lib/src/index.d.ts.map +1 -1
- package/lib/src/index.js.map +1 -1
- package/lib/src/native-simctl.js +12 -1
- package/lib/src/native-simctl.js.map +1 -1
- package/lib/src/types.d.ts +82 -10
- package/lib/src/types.d.ts.map +1 -1
- package/lib/src/utils/index.d.ts +1 -1
- package/lib/src/utils/index.d.ts.map +1 -1
- package/lib/src/utils/index.js +1 -1
- package/lib/src/utils/index.js.map +1 -1
- package/lib/src/utils/run-catching.d.ts +4 -0
- package/lib/src/utils/run-catching.d.ts.map +1 -1
- package/lib/src/utils/run-catching.js +11 -0
- package/lib/src/utils/run-catching.js.map +1 -1
- package/package.json +1 -1
- package/prebuilds/darwin-arm64/@appium+coresim.node +0 -0
- package/src/commands/video-recording.ts +75 -25
- package/src/commands/video-stream.ts +48 -39
- package/src/coresim.mm +547 -118
- package/src/index.ts +1 -0
- package/src/native/audio_encoder.h +65 -0
- package/src/native/audio_encoder.mm +270 -0
- package/src/native/av_recording.h +52 -0
- package/src/native/av_recording.mm +424 -0
- package/src/native/av_stream.h +70 -0
- package/src/native/av_stream.mm +217 -0
- package/src/native/monotonic_clock.h +20 -0
- package/src/native/sim_audio_tap.h +67 -0
- package/src/native/sim_audio_tap.mm +350 -0
- package/src/native/sim_process.h +10 -0
- package/src/native/sim_process.mm +81 -18
- package/src/native/sim_video_stream.h +14 -11
- package/src/native/sim_video_stream.mm +27 -383
- package/src/native/video_encoder.h +76 -0
- package/src/native/video_encoder.mm +417 -0
- package/src/native-simctl.ts +12 -1
- package/src/types.ts +95 -11
- package/src/utils/index.ts +1 -1
- package/src/utils/run-catching.ts +11 -0
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
#pragma once
|
|
2
|
+
|
|
3
|
+
#import <CoreMedia/CoreMedia.h>
|
|
4
|
+
#import <Foundation/Foundation.h>
|
|
5
|
+
|
|
6
|
+
#include <cstdint>
|
|
7
|
+
#include <functional>
|
|
8
|
+
#include <memory>
|
|
9
|
+
#include <vector>
|
|
10
|
+
|
|
11
|
+
namespace coresim {
|
|
12
|
+
|
|
13
|
+
enum class VideoStreamCodec { kH264, kHEVC };
|
|
14
|
+
|
|
15
|
+
struct VideoEncoderOptions {
|
|
16
|
+
VideoStreamCodec codec = VideoStreamCodec::kH264;
|
|
17
|
+
NSString* displayId = nil;
|
|
18
|
+
double fps = 15.0;
|
|
19
|
+
int bitrate = 2000000;
|
|
20
|
+
};
|
|
21
|
+
|
|
22
|
+
// Polls the live display IOSurface (sim_screenshot.h) on a serial queue and encodes changed
|
|
23
|
+
// frames via the public VideoToolbox API, delivering each encoded `CMSampleBufferRef` live.
|
|
24
|
+
// Extracted from VideoStreamSession (video_stream.h) so both it and the combined AV
|
|
25
|
+
// recording/streaming sessions (av_recording.h/av_stream.h) share one VTCompressionSession driver
|
|
26
|
+
// instead of each re-deriving IOSurface-poll + compression-session setup from scratch. Callers
|
|
27
|
+
// that need Annex-B-framed access units (VideoAccessUnit's wire format) repack the delivered
|
|
28
|
+
// sample buffers themselves (see RepackAsAnnexB below) — this class only drives the encoder.
|
|
29
|
+
class VideoFrameEncoder {
|
|
30
|
+
public:
|
|
31
|
+
// `onSample`'s CMSampleBufferRef is only valid for the duration of the call (VideoToolbox owns
|
|
32
|
+
// it) — a caller that needs the data beyond that must copy it out (or CFRetain it) before
|
|
33
|
+
// returning, never stash the raw pointer.
|
|
34
|
+
//
|
|
35
|
+
// `sharedClockOrigin`, if non-null, is used as this encoder's PTS-zero instant (from
|
|
36
|
+
// monotonic_clock.h) instead of capturing its own at Start() — pass the same pointer to an
|
|
37
|
+
// AudioEncoder started around the same time so both tracks' presentation timestamps measure
|
|
38
|
+
// elapsed time from one common reference (av_recording.h/av_stream.h do this; the standalone
|
|
39
|
+
// VideoStreamSession, video-only, has no such peer and leaves this null).
|
|
40
|
+
VideoFrameEncoder(id device, VideoEncoderOptions options, std::function<void(CMSampleBufferRef)> onSample,
|
|
41
|
+
std::function<void(NSError*)> onError, std::function<void()> onEnd,
|
|
42
|
+
const double* sharedClockOrigin = nullptr);
|
|
43
|
+
~VideoFrameEncoder();
|
|
44
|
+
|
|
45
|
+
VideoFrameEncoder(const VideoFrameEncoder&) = delete;
|
|
46
|
+
VideoFrameEncoder& operator=(const VideoFrameEncoder&) = delete;
|
|
47
|
+
|
|
48
|
+
// Resolves the display and starts the polling loop; throws synchronously on resolution/setup
|
|
49
|
+
// failure (`onEnd` never called then). Later failures go to `onError`, then `onEnd`.
|
|
50
|
+
void Start();
|
|
51
|
+
|
|
52
|
+
// Idempotent; blocks until the loop has fully stopped. Never call from inside onSample/onError/
|
|
53
|
+
// onEnd — same queue this blocks on, so it would deadlock.
|
|
54
|
+
void Stop();
|
|
55
|
+
|
|
56
|
+
// Forces the next encoded frame to be a keyframe (self-decodable, parameter sets included) —
|
|
57
|
+
// e.g. so a consumer that just resynced after dropping frames can resume cleanly instead of
|
|
58
|
+
// waiting for the next periodic one. Safe from any thread; just sets a flag.
|
|
59
|
+
void RequestKeyFrame();
|
|
60
|
+
|
|
61
|
+
private:
|
|
62
|
+
class Impl;
|
|
63
|
+
std::unique_ptr<Impl> impl_;
|
|
64
|
+
};
|
|
65
|
+
|
|
66
|
+
// Whether `sampleBuffer` is a sync (key) frame, from its sample attachments.
|
|
67
|
+
bool IsKeyFrame(CMSampleBufferRef sampleBuffer);
|
|
68
|
+
|
|
69
|
+
// VideoToolbox's compressed output is AVCC-framed (a 4-byte big-endian length prefix per NAL, no
|
|
70
|
+
// start codes) — rewrites `sampleBuffer` into Annex-B, appending to `out`. A keyframe's format
|
|
71
|
+
// description (SPS/PPS, or VPS/SPS/PPS for HEVC) is prepended first when `isKeyFrame`, so every
|
|
72
|
+
// keyframe is self-decodable alone. Shared by VideoStreamSession and the AV streaming path so both
|
|
73
|
+
// produce byte-identical framing for the same encoder output.
|
|
74
|
+
void RepackAsAnnexB(std::vector<uint8_t>& out, CMSampleBufferRef sampleBuffer, bool isKeyFrame, VideoStreamCodec codec);
|
|
75
|
+
|
|
76
|
+
} // namespace coresim
|
|
@@ -0,0 +1,417 @@
|
|
|
1
|
+
#include "video_encoder.h"
|
|
2
|
+
|
|
3
|
+
#import <CoreVideo/CoreVideo.h>
|
|
4
|
+
#import <IOSurface/IOSurface.h>
|
|
5
|
+
#import <VideoToolbox/VideoToolbox.h>
|
|
6
|
+
|
|
7
|
+
#include <algorithm>
|
|
8
|
+
#include <atomic>
|
|
9
|
+
|
|
10
|
+
#include "monotonic_clock.h"
|
|
11
|
+
#include "nserror_bridge.h"
|
|
12
|
+
#include "safe_dispatch.h"
|
|
13
|
+
#include "sim_screenshot.h"
|
|
14
|
+
|
|
15
|
+
namespace coresim {
|
|
16
|
+
|
|
17
|
+
namespace {
|
|
18
|
+
|
|
19
|
+
NSString* const kVideoEncoderErrorDomain = @"com.appium.coresim.VideoEncoder";
|
|
20
|
+
|
|
21
|
+
NSError* MakeError(NSInteger code, NSString* message) {
|
|
22
|
+
return [NSError errorWithDomain:kVideoEncoderErrorDomain code:code userInfo:@{NSLocalizedDescriptionKey : message}];
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
NSError* MakeStatusError(NSInteger code, NSString* what, OSStatus status) {
|
|
26
|
+
return MakeError(code, [NSString stringWithFormat:@"%@ (OSStatus %d)", what, static_cast<int>(status)]);
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
void AppendAnnexB(std::vector<uint8_t>& out, const uint8_t* nal, size_t length) {
|
|
30
|
+
static const uint8_t kStartCode[4] = {0, 0, 0, 1};
|
|
31
|
+
out.insert(out.end(), kStartCode, kStartCode + 4);
|
|
32
|
+
out.insert(out.end(), nal, nal + length);
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
// CMBlockBufferGetDataPointer's pointer only covers the contiguous region starting at the given
|
|
36
|
+
// offset, which for a segmented buffer (multiple backing memory blocks — CoreMedia's documented
|
|
37
|
+
// contract, not just a VideoToolbox implementation detail) can be far shorter than totalLength;
|
|
38
|
+
// reading up to totalLength through it would run past that region. CopyDataBytes stitches
|
|
39
|
+
// segments together into a caller-owned, guaranteed-contiguous copy instead.
|
|
40
|
+
void AppendSampleBufferNALs(std::vector<uint8_t>& out, CMSampleBufferRef sampleBuffer) {
|
|
41
|
+
CMBlockBufferRef block = CMSampleBufferGetDataBuffer(sampleBuffer);
|
|
42
|
+
if (block == nullptr) {
|
|
43
|
+
return;
|
|
44
|
+
}
|
|
45
|
+
size_t totalLength = CMBlockBufferGetDataLength(block);
|
|
46
|
+
if (totalLength == 0) {
|
|
47
|
+
return;
|
|
48
|
+
}
|
|
49
|
+
std::vector<uint8_t> data(totalLength);
|
|
50
|
+
if (CMBlockBufferCopyDataBytes(block, 0, totalLength, data.data()) != kCMBlockBufferNoErr) {
|
|
51
|
+
return;
|
|
52
|
+
}
|
|
53
|
+
const uint8_t* dataPointer = data.data();
|
|
54
|
+
size_t offset = 0;
|
|
55
|
+
while (offset + 4 <= totalLength) {
|
|
56
|
+
uint32_t nalLength = (static_cast<uint32_t>(dataPointer[offset]) << 24) |
|
|
57
|
+
(static_cast<uint32_t>(dataPointer[offset + 1]) << 16) |
|
|
58
|
+
(static_cast<uint32_t>(dataPointer[offset + 2]) << 8) | dataPointer[offset + 3];
|
|
59
|
+
offset += 4;
|
|
60
|
+
if (nalLength == 0 || offset + nalLength > totalLength) {
|
|
61
|
+
break;
|
|
62
|
+
}
|
|
63
|
+
AppendAnnexB(out, dataPointer + offset, nalLength);
|
|
64
|
+
offset += nalLength;
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
using ParameterSetAtIndexFn = OSStatus (*)(CMFormatDescriptionRef, size_t, const uint8_t**, size_t*, size_t*, int*);
|
|
69
|
+
|
|
70
|
+
void AppendParameterSets(std::vector<uint8_t>& out, CMFormatDescriptionRef format, ParameterSetAtIndexFn getAtIndex) {
|
|
71
|
+
size_t count = 0;
|
|
72
|
+
if (getAtIndex(format, 0, nullptr, nullptr, &count, nullptr) != noErr) {
|
|
73
|
+
return;
|
|
74
|
+
}
|
|
75
|
+
for (size_t i = 0; i < count; i++) {
|
|
76
|
+
const uint8_t* bytes = nullptr;
|
|
77
|
+
size_t size = 0;
|
|
78
|
+
if (getAtIndex(format, i, &bytes, &size, nullptr, nullptr) == noErr) {
|
|
79
|
+
AppendAnnexB(out, bytes, size);
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
} // namespace
|
|
85
|
+
|
|
86
|
+
bool IsKeyFrame(CMSampleBufferRef sampleBuffer) {
|
|
87
|
+
CFArrayRef attachments = CMSampleBufferGetSampleAttachmentsArray(sampleBuffer, false);
|
|
88
|
+
if (attachments == nullptr || CFArrayGetCount(attachments) == 0) {
|
|
89
|
+
return true;
|
|
90
|
+
}
|
|
91
|
+
CFDictionaryRef attachment = static_cast<CFDictionaryRef>(CFArrayGetValueAtIndex(attachments, 0));
|
|
92
|
+
return !CFDictionaryContainsKey(attachment, kCMSampleAttachmentKey_NotSync);
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
void RepackAsAnnexB(std::vector<uint8_t>& out, CMSampleBufferRef sampleBuffer, bool isKeyFrame,
|
|
96
|
+
VideoStreamCodec codec) {
|
|
97
|
+
if (isKeyFrame) {
|
|
98
|
+
CMFormatDescriptionRef format = CMSampleBufferGetFormatDescription(sampleBuffer);
|
|
99
|
+
if (format != nullptr) {
|
|
100
|
+
if (codec == VideoStreamCodec::kHEVC) {
|
|
101
|
+
AppendParameterSets(out, format, CMVideoFormatDescriptionGetHEVCParameterSetAtIndex);
|
|
102
|
+
} else {
|
|
103
|
+
AppendParameterSets(out, format, CMVideoFormatDescriptionGetH264ParameterSetAtIndex);
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
AppendSampleBufferNALs(out, sampleBuffer);
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
class VideoFrameEncoder::Impl {
|
|
111
|
+
public:
|
|
112
|
+
Impl(id device, VideoEncoderOptions options, std::function<void(CMSampleBufferRef)> onSample,
|
|
113
|
+
std::function<void(NSError*)> onError, std::function<void()> onEnd, const double* sharedClockOrigin)
|
|
114
|
+
: device_(device),
|
|
115
|
+
options_(options),
|
|
116
|
+
onSample_(std::move(onSample)),
|
|
117
|
+
onError_(std::move(onError)),
|
|
118
|
+
onEnd_(std::move(onEnd)),
|
|
119
|
+
sharedClockOrigin_(sharedClockOrigin) {
|
|
120
|
+
queue_ = dispatch_queue_create("com.appium.coresim.videoEncoder", DISPATCH_QUEUE_SERIAL);
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
~Impl() { Stop(); }
|
|
124
|
+
|
|
125
|
+
void Start() {
|
|
126
|
+
NSError* error = nil;
|
|
127
|
+
id descriptor = ResolveCaptureDisplay(device_, options_.displayId, &error);
|
|
128
|
+
if (descriptor == nil) {
|
|
129
|
+
throw NSErrorException(error);
|
|
130
|
+
}
|
|
131
|
+
id surfaceObj = CurrentDisplaySurface(descriptor);
|
|
132
|
+
if (surfaceObj == nil) {
|
|
133
|
+
throw NSErrorException(MakeError(4, @"The device's display surface is not available yet"));
|
|
134
|
+
}
|
|
135
|
+
IOSurfaceRef surface = (__bridge IOSurfaceRef)surfaceObj;
|
|
136
|
+
// Set up synchronously (not lazily on the first Tick()) so a setup failure rejects Start()
|
|
137
|
+
// directly rather than only reaching onError, which the caller may not be listening for yet.
|
|
138
|
+
NSError* setupError = nil;
|
|
139
|
+
if (!SetUpSession(surface, &setupError)) {
|
|
140
|
+
throw NSErrorException(setupError);
|
|
141
|
+
}
|
|
142
|
+
// Must be set before EncodeSurface below — both it and HandleEncodedSample measure elapsed
|
|
143
|
+
// time from this. A caller-supplied origin (see the constructor) is used as-is, not offset
|
|
144
|
+
// further — the gap between it being captured and this line running is itself the correct,
|
|
145
|
+
// meaningful startup latency to bake into this encoder's PTS zero, for sync with a peer
|
|
146
|
+
// AudioEncoder given the same origin (av_recording.h/av_stream.h).
|
|
147
|
+
startTime_ = sharedClockOrigin_ != nullptr ? *sharedClockOrigin_ : MonotonicSeconds();
|
|
148
|
+
// Encode immediately rather than waiting for a *changed* seed on the first tick, or the
|
|
149
|
+
// stream would stay silent until the display changes again. running_ is set true before this
|
|
150
|
+
// call (not after), since VTCompressionSessionEncodeFrame's output callback can in principle
|
|
151
|
+
// fire on another thread before this one returns — HandleEncodedSample discards samples while
|
|
152
|
+
// running_ is false, which would otherwise silently drop the stream's very first (keyframe)
|
|
153
|
+
// sample. A failure resets it and tears session_ down itself here, rather than going through
|
|
154
|
+
// Stop()/onEnd_ (see coresim.mm — onEnd_ firing this early would double-release its
|
|
155
|
+
// ThreadSafeFunctions).
|
|
156
|
+
running_ = true;
|
|
157
|
+
bool encoded = false;
|
|
158
|
+
try {
|
|
159
|
+
encoded = EncodeSurface(surface);
|
|
160
|
+
} catch (...) {
|
|
161
|
+
running_ = false;
|
|
162
|
+
VTCompressionSessionInvalidate(session_);
|
|
163
|
+
CFRelease(session_);
|
|
164
|
+
session_ = nullptr;
|
|
165
|
+
throw;
|
|
166
|
+
}
|
|
167
|
+
// Only commit the seed once a frame was actually submitted — a transient pixel-buffer
|
|
168
|
+
// creation failure (EncodeSurface returning false) otherwise leaves lastSeed_ at its default
|
|
169
|
+
// 0, so the first Tick() sees the real seed as "changed" and retries automatically instead of
|
|
170
|
+
// the stream going silent forever on a display that never changes again.
|
|
171
|
+
if (encoded) {
|
|
172
|
+
lastSeed_ = IOSurfaceGetSeed(surface);
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
double interval = 1.0 / std::max(options_.fps, 1.0);
|
|
176
|
+
dispatch_source_t timer = dispatch_source_create(DISPATCH_SOURCE_TYPE_TIMER, 0, 0, queue_);
|
|
177
|
+
dispatch_source_set_timer(timer, dispatch_time(DISPATCH_TIME_NOW, 0),
|
|
178
|
+
static_cast<uint64_t>(interval * NSEC_PER_SEC),
|
|
179
|
+
static_cast<uint64_t>(interval * NSEC_PER_SEC / 10));
|
|
180
|
+
// `this` outlives the timer: Stop()/StopFromQueue() always drain or outrun it before `this`
|
|
181
|
+
// can be destroyed (see their comments below).
|
|
182
|
+
dispatch_source_set_event_handler(timer, ^{
|
|
183
|
+
Tick();
|
|
184
|
+
});
|
|
185
|
+
timer_ = timer;
|
|
186
|
+
dispatch_resume(timer_);
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
// Callable from any thread except `queue_` itself (would deadlock on the dispatch_sync below).
|
|
190
|
+
void Stop() {
|
|
191
|
+
if (!running_.exchange(false)) {
|
|
192
|
+
return; // idempotent
|
|
193
|
+
}
|
|
194
|
+
if (timer_ != nullptr) {
|
|
195
|
+
dispatch_source_cancel(timer_);
|
|
196
|
+
// Blocks until any in-flight Tick() finishes — by then running_ is already false, so it
|
|
197
|
+
// won't touch session_ again.
|
|
198
|
+
dispatch_sync(queue_, ^{
|
|
199
|
+
});
|
|
200
|
+
timer_ = nullptr;
|
|
201
|
+
}
|
|
202
|
+
TearDownSessionAndFireEnd();
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
void RequestKeyFrame() { forceKeyFrame_ = true; }
|
|
206
|
+
|
|
207
|
+
private:
|
|
208
|
+
// Same as Stop() minus the dispatch_sync barrier — only safe from within Tick() itself, already
|
|
209
|
+
// serialized on `queue_`; would race a concurrent Tick() from any other thread.
|
|
210
|
+
void StopFromQueue() {
|
|
211
|
+
if (!running_.exchange(false)) {
|
|
212
|
+
return; // idempotent — e.g. an external Stop() already won this race
|
|
213
|
+
}
|
|
214
|
+
if (timer_ != nullptr) {
|
|
215
|
+
dispatch_source_cancel(timer_);
|
|
216
|
+
timer_ = nullptr;
|
|
217
|
+
}
|
|
218
|
+
TearDownSessionAndFireEnd();
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
void TearDownSessionAndFireEnd() {
|
|
222
|
+
if (session_ != nullptr) {
|
|
223
|
+
// Flushes and blocks until every already-submitted frame's callback has returned — without
|
|
224
|
+
// this, a frame submitted just before Stop() could fire after onEnd_ releases whatever
|
|
225
|
+
// resources the caller tied to it (e.g. ThreadSafeFunctions — see CLAUDE.md).
|
|
226
|
+
VTCompressionSessionCompleteFrames(session_, kCMTimeInvalid);
|
|
227
|
+
VTCompressionSessionInvalidate(session_);
|
|
228
|
+
CFRelease(session_);
|
|
229
|
+
session_ = nullptr;
|
|
230
|
+
}
|
|
231
|
+
if (onEnd_) {
|
|
232
|
+
onEnd_();
|
|
233
|
+
}
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
void Tick() {
|
|
237
|
+
if (!running_) {
|
|
238
|
+
return;
|
|
239
|
+
}
|
|
240
|
+
if (pendingErrorTeardown_) {
|
|
241
|
+
// HandleEncodedSample (below) can run on VideoToolbox's own callback thread, where it's
|
|
242
|
+
// unsafe to tear down directly — StopFromQueue()/CompleteFrames() are only safe already
|
|
243
|
+
// serialized on `queue_` (here), and CompleteFrames specifically would deadlock waiting on
|
|
244
|
+
// its own still-executing callback. It sets this flag instead; picked up on the very next
|
|
245
|
+
// tick (queue_-serialized, safe) rather than via a raw cross-thread dispatch, since nothing
|
|
246
|
+
// here can outlive `this` the way a block captured on another thread otherwise could.
|
|
247
|
+
StopFromQueue();
|
|
248
|
+
return;
|
|
249
|
+
}
|
|
250
|
+
@autoreleasepool {
|
|
251
|
+
try {
|
|
252
|
+
// Re-resolved every tick (like CaptureScreenshot does), not cached once in Start(), so a
|
|
253
|
+
// deleted device or disconnected display surfaces a real error instead of Tick() quietly
|
|
254
|
+
// doing nothing forever.
|
|
255
|
+
NSError* resolveError = nil;
|
|
256
|
+
id descriptor = ResolveCaptureDisplay(device_, options_.displayId, &resolveError);
|
|
257
|
+
if (descriptor == nil) {
|
|
258
|
+
if (onError_) {
|
|
259
|
+
onError_(resolveError);
|
|
260
|
+
}
|
|
261
|
+
StopFromQueue();
|
|
262
|
+
return;
|
|
263
|
+
}
|
|
264
|
+
id surfaceObj = CurrentDisplaySurface(descriptor);
|
|
265
|
+
if (surfaceObj == nil) {
|
|
266
|
+
return; // transient — the connection may not have a frame ready yet, try again next tick
|
|
267
|
+
}
|
|
268
|
+
IOSurfaceRef surface = (__bridge IOSurfaceRef)surfaceObj;
|
|
269
|
+
uint32_t seed = IOSurfaceGetSeed(surface);
|
|
270
|
+
if (seed == lastSeed_) {
|
|
271
|
+
return; // unchanged since the last tick — mirrors CoreSimulator's own recorder, which
|
|
272
|
+
// only encodes a frame when the display actually changes (see CLAUDE.md)
|
|
273
|
+
}
|
|
274
|
+
// Only commit the new seed once EncodeSurface actually submits it — a transient failure
|
|
275
|
+
// (pixel-buffer creation) must leave lastSeed_ stale so the next tick retries this same
|
|
276
|
+
// frame instead of silently going quiet until the display changes again.
|
|
277
|
+
if (EncodeSurface(surface)) {
|
|
278
|
+
lastSeed_ = seed;
|
|
279
|
+
}
|
|
280
|
+
} catch (const std::exception& e) {
|
|
281
|
+
// Without this, an exception here (e.g. a dropped display-proxy connection) would escape
|
|
282
|
+
// this bare GCD timer handler uncaught and crash the whole process (see CLAUDE.md).
|
|
283
|
+
if (onError_) {
|
|
284
|
+
onError_(MakeError(3, [NSString stringWithFormat:@"Video encoding failed: %s", e.what()]));
|
|
285
|
+
}
|
|
286
|
+
StopFromQueue();
|
|
287
|
+
}
|
|
288
|
+
}
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
bool SetUpSession(IOSurfaceRef surface, NSError** error) {
|
|
292
|
+
int32_t width = static_cast<int32_t>(IOSurfaceGetWidth(surface));
|
|
293
|
+
int32_t height = static_cast<int32_t>(IOSurfaceGetHeight(surface));
|
|
294
|
+
CMVideoCodecType codecType =
|
|
295
|
+
options_.codec == VideoStreamCodec::kHEVC ? kCMVideoCodecType_HEVC : kCMVideoCodecType_H264;
|
|
296
|
+
OSStatus status = VTCompressionSessionCreate(kCFAllocatorDefault, width, height, codecType, nullptr, nullptr,
|
|
297
|
+
kCFAllocatorDefault, OutputCallback, this, &session_);
|
|
298
|
+
if (status != noErr) {
|
|
299
|
+
*error = MakeStatusError(1, @"Failed to create a VTCompressionSession", status);
|
|
300
|
+
return false;
|
|
301
|
+
}
|
|
302
|
+
// Clamped like Start()'s timer interval — an unvalidated 0 here would set MaxKeyFrameInterval
|
|
303
|
+
// to an out-of-spec value.
|
|
304
|
+
double fps = std::max(options_.fps, 1.0);
|
|
305
|
+
status = VTSessionSetProperty(session_, kVTCompressionPropertyKey_RealTime, kCFBooleanTrue);
|
|
306
|
+
if (status == noErr) {
|
|
307
|
+
status = VTSessionSetProperty(session_, kVTCompressionPropertyKey_AllowFrameReordering, kCFBooleanFalse);
|
|
308
|
+
}
|
|
309
|
+
if (status == noErr) {
|
|
310
|
+
status = VTSessionSetProperty(session_, kVTCompressionPropertyKey_AverageBitRate,
|
|
311
|
+
(__bridge CFNumberRef) @(options_.bitrate));
|
|
312
|
+
}
|
|
313
|
+
if (status == noErr) {
|
|
314
|
+
status =
|
|
315
|
+
VTSessionSetProperty(session_, kVTCompressionPropertyKey_ExpectedFrameRate, (__bridge CFNumberRef) @(fps));
|
|
316
|
+
}
|
|
317
|
+
if (status == noErr) {
|
|
318
|
+
status = VTSessionSetProperty(session_, kVTCompressionPropertyKey_MaxKeyFrameInterval,
|
|
319
|
+
(__bridge CFNumberRef) @(static_cast<int>(fps * 2)));
|
|
320
|
+
}
|
|
321
|
+
if (status != noErr) {
|
|
322
|
+
*error = MakeStatusError(2, @"Failed to configure the VTCompressionSession", status);
|
|
323
|
+
VTCompressionSessionInvalidate(session_);
|
|
324
|
+
CFRelease(session_);
|
|
325
|
+
session_ = nullptr;
|
|
326
|
+
return false;
|
|
327
|
+
}
|
|
328
|
+
VTCompressionSessionPrepareToEncodeFrames(session_);
|
|
329
|
+
return true;
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
// Returns whether a frame was actually submitted to the encoder — false for a transient
|
|
333
|
+
// pixel-buffer creation failure the caller should retry, as opposed to a real encode failure
|
|
334
|
+
// (thrown, not returned, since that tears down the whole session).
|
|
335
|
+
bool EncodeSurface(IOSurfaceRef surface) {
|
|
336
|
+
CVPixelBufferRef pixelBuffer = nullptr;
|
|
337
|
+
CVReturn cvStatus = CVPixelBufferCreateWithIOSurface(kCFAllocatorDefault, surface, nullptr, &pixelBuffer);
|
|
338
|
+
if (cvStatus != kCVReturnSuccess || pixelBuffer == nullptr) {
|
|
339
|
+
return false; // transient — try again next tick rather than tearing down the whole session
|
|
340
|
+
}
|
|
341
|
+
CMTime pts = CMTimeMake(static_cast<int64_t>((MonotonicSeconds() - startTime_) * 1000000), 1000000);
|
|
342
|
+
NSDictionary* frameProperties = nil;
|
|
343
|
+
if (forceKeyFrame_.exchange(false)) {
|
|
344
|
+
frameProperties = @{(__bridge NSString*)kVTEncodeFrameOptionKey_ForceKeyFrame : @YES};
|
|
345
|
+
}
|
|
346
|
+
OSStatus status = VTCompressionSessionEncodeFrame(session_, pixelBuffer, pts, kCMTimeInvalid,
|
|
347
|
+
(__bridge CFDictionaryRef)frameProperties, nullptr, nullptr);
|
|
348
|
+
CVPixelBufferRelease(pixelBuffer);
|
|
349
|
+
if (status != noErr) {
|
|
350
|
+
throw std::runtime_error([[NSString stringWithFormat:@"VTCompressionSessionEncodeFrame failed (OSStatus %d)",
|
|
351
|
+
static_cast<int>(status)] UTF8String]);
|
|
352
|
+
}
|
|
353
|
+
return true;
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
static void OutputCallback(void* outputCallbackRefCon, void* /*sourceFrameRefCon*/, OSStatus status,
|
|
357
|
+
VTEncodeInfoFlags /*infoFlags*/, CMSampleBufferRef sampleBuffer) {
|
|
358
|
+
static_cast<Impl*>(outputCallbackRefCon)->HandleEncodedSample(status, sampleBuffer);
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
void HandleEncodedSample(OSStatus status, CMSampleBufferRef sampleBuffer) {
|
|
362
|
+
if (!running_) {
|
|
363
|
+
return;
|
|
364
|
+
}
|
|
365
|
+
if (status != noErr) {
|
|
366
|
+
// Only the first failure is reported — pendingErrorTeardown_ doubles as the report-once
|
|
367
|
+
// gate, since Tick() (the only place that consumes it) only ever needs to see it once too.
|
|
368
|
+
if (!pendingErrorTeardown_.exchange(true)) {
|
|
369
|
+
if (onError_) {
|
|
370
|
+
onError_(MakeStatusError(3, @"VideoToolbox reported an encoding failure", status));
|
|
371
|
+
}
|
|
372
|
+
}
|
|
373
|
+
return;
|
|
374
|
+
}
|
|
375
|
+
if (sampleBuffer == nullptr) {
|
|
376
|
+
return;
|
|
377
|
+
}
|
|
378
|
+
if (onSample_) {
|
|
379
|
+
onSample_(sampleBuffer);
|
|
380
|
+
}
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
id device_;
|
|
384
|
+
VideoEncoderOptions options_;
|
|
385
|
+
std::function<void(CMSampleBufferRef)> onSample_;
|
|
386
|
+
std::function<void(NSError*)> onError_;
|
|
387
|
+
std::function<void()> onEnd_;
|
|
388
|
+
const double* sharedClockOrigin_;
|
|
389
|
+
|
|
390
|
+
dispatch_queue_t queue_ = nullptr;
|
|
391
|
+
dispatch_source_t timer_ = nullptr;
|
|
392
|
+
VTCompressionSessionRef session_ = nullptr;
|
|
393
|
+
uint32_t lastSeed_ = 0;
|
|
394
|
+
double startTime_ = 0;
|
|
395
|
+
std::atomic<bool> running_{false};
|
|
396
|
+
// Set by HandleEncodedSample (possibly off queue_) on an encoder failure, consumed by the next
|
|
397
|
+
// Tick() (on queue_) — see both for why teardown can't just happen inline there.
|
|
398
|
+
std::atomic<bool> pendingErrorTeardown_{false};
|
|
399
|
+
std::atomic<bool> forceKeyFrame_{false};
|
|
400
|
+
};
|
|
401
|
+
|
|
402
|
+
VideoFrameEncoder::VideoFrameEncoder(id device, VideoEncoderOptions options,
|
|
403
|
+
std::function<void(CMSampleBufferRef)> onSample,
|
|
404
|
+
std::function<void(NSError*)> onError, std::function<void()> onEnd,
|
|
405
|
+
const double* sharedClockOrigin)
|
|
406
|
+
: impl_(std::make_unique<Impl>(device, options, std::move(onSample), std::move(onError), std::move(onEnd),
|
|
407
|
+
sharedClockOrigin)) {}
|
|
408
|
+
|
|
409
|
+
VideoFrameEncoder::~VideoFrameEncoder() = default;
|
|
410
|
+
|
|
411
|
+
void VideoFrameEncoder::Start() { impl_->Start(); }
|
|
412
|
+
|
|
413
|
+
void VideoFrameEncoder::Stop() { impl_->Stop(); }
|
|
414
|
+
|
|
415
|
+
void VideoFrameEncoder::RequestKeyFrame() { impl_->RequestKeyFrame(); }
|
|
416
|
+
|
|
417
|
+
} // namespace coresim
|
package/src/native-simctl.ts
CHANGED
|
@@ -281,7 +281,18 @@ const loadNative = util.memoize(function loadNative(): NativeCoreSimModule {
|
|
|
281
281
|
'n/a',
|
|
282
282
|
);
|
|
283
283
|
}
|
|
284
|
-
|
|
284
|
+
const native = require('node-gyp-build')(getPkgRoot()) as NativeCoreSimModule;
|
|
285
|
+
// A forgotten (never explicitly stopped) AV recording/stream otherwise silently loses data —
|
|
286
|
+
// or just leaks a live encoder — the instant a caller force-exits via `process.exit()`, since
|
|
287
|
+
// Node's own cleanup hooks (coresim.mm's CleanupActiveSessions, registered against the exact
|
|
288
|
+
// same condition) are confirmed to NOT run in that path, only on a natural empty-event-loop
|
|
289
|
+
// exit or a Worker's own termination. `process.on('exit', ...)` does fire for `process.exit()`
|
|
290
|
+
// too, and (per Node's own contract) may run synchronous code — flushActiveSessions() qualifies:
|
|
291
|
+
// it's a single blocking native call, not new async JS work. Registered once, lazily, here
|
|
292
|
+
// rather than at module import time, so merely importing this package on a non-macOS platform
|
|
293
|
+
// never touches `process` for something it'll never need.
|
|
294
|
+
process.on('exit', () => native.flushActiveSessions());
|
|
295
|
+
return native;
|
|
285
296
|
});
|
|
286
297
|
|
|
287
298
|
const DEFAULT_DEVELOPER_DIR_TIMEOUT_MS = 15_000;
|
package/src/types.ts
CHANGED
|
@@ -140,9 +140,44 @@ export interface VideoRecordingOptions {
|
|
|
140
140
|
/**
|
|
141
141
|
* For a non-rectangular display (e.g. a Dynamic Island cutout): `'ignored'` (default) saves the
|
|
142
142
|
* unmasked framebuffer, `'black'` renders the mask black, `'alpha'` is not supported and
|
|
143
|
-
* behaves like `'black'`.
|
|
143
|
+
* behaves like `'black'`. Only applies when neither `audio` nor `fps` is set — see their doc
|
|
144
|
+
* comments.
|
|
144
145
|
*/
|
|
145
146
|
mask?: 'ignored' | 'alpha' | 'black';
|
|
147
|
+
/**
|
|
148
|
+
* Also capture the device's audio into the same file, muxed as a second track. Defaults to
|
|
149
|
+
* `false`. Requires macOS 14.2+ (Core Audio process taps), the host's "System Audio Recording
|
|
150
|
+
* Only" privacy permission (System Settings > Privacy & Security — cannot be granted
|
|
151
|
+
* programmatically; a denial isn't a thrown error, it surfaces as a silent, audio-less/near-
|
|
152
|
+
* silent recording), a default audio output device on the host, and a booted device that has
|
|
153
|
+
* produced audio at least once. See the README's "Screen capture" section for the full
|
|
154
|
+
* requirements list and known failure modes.
|
|
155
|
+
*
|
|
156
|
+
* Like an explicit `fps`, this switches the implementation to this addon's own VideoToolbox +
|
|
157
|
+
* Core Audio encoders instead of CoreSimulator's private recorder, which can't mux audio.
|
|
158
|
+
*/
|
|
159
|
+
audio?: boolean;
|
|
160
|
+
/**
|
|
161
|
+
* Max frames/sec to poll the framebuffer at — see {@link VideoStreamOptions} `fps` for the
|
|
162
|
+
* identical semantics. Meaningless against CoreSimulator's private recorder (it captures on its
|
|
163
|
+
* own cadence, not one we poll), so setting `fps` — even without `audio` — switches this
|
|
164
|
+
* recording to the same own-encoder implementation `audio` does. That switch costs `mask`
|
|
165
|
+
* support, which only the private recorder implements.
|
|
166
|
+
*/
|
|
167
|
+
fps?: number;
|
|
168
|
+
/** Target average bitrate, in bits/sec. Respected on either implementation. */
|
|
169
|
+
bitrate?: number;
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/** Options for `NativeSimctl.stopVideoRecording`. */
|
|
173
|
+
export interface StopVideoRecordingOptions {
|
|
174
|
+
/**
|
|
175
|
+
* Best-effort: still attempts the native stop, but releases this device's tracked-active-
|
|
176
|
+
* recording bookkeeping regardless of whether that attempt succeeds, instead of leaving it
|
|
177
|
+
* retryable — see the doc comment above `NativeSimctl.stopVideoRecording` for when a caller
|
|
178
|
+
* needs this over a plain retry.
|
|
179
|
+
*/
|
|
180
|
+
force?: boolean;
|
|
146
181
|
}
|
|
147
182
|
|
|
148
183
|
/** Options for `NativeSimctl.startVideoStream`. */
|
|
@@ -161,18 +196,30 @@ export interface VideoStreamOptions {
|
|
|
161
196
|
fps?: number;
|
|
162
197
|
/** Target average bitrate, in bits/sec. Defaults to 2,000,000 (2 Mbps). */
|
|
163
198
|
bitrate?: number;
|
|
199
|
+
/**
|
|
200
|
+
* Also stream the device's audio, interleaved into the same `accessUnits()` sequence. Defaults
|
|
201
|
+
* to `false`. Same requirements and failure modes as {@link VideoRecordingOptions.audio} — see
|
|
202
|
+
* its doc comment and the README's "Screen capture" section.
|
|
203
|
+
*/
|
|
204
|
+
audio?: boolean;
|
|
164
205
|
}
|
|
165
206
|
|
|
166
207
|
/**
|
|
167
|
-
* One encoded
|
|
168
|
-
* parameter sets
|
|
208
|
+
* One encoded unit from `VideoStream.accessUnits()`, discriminated by `track`: a video unit
|
|
209
|
+
* (Annex-B NAL units — a keyframe's `data` has parameter sets, SPS/PPS or VPS/SPS/PPS for HEVC,
|
|
210
|
+
* prepended, so it's self-decodable alone) or, when {@link VideoStreamOptions.audio} was set, an
|
|
211
|
+
* interleaved audio unit (an ADTS-framed AAC-LC packet — the 7-byte ADTS header carries sample
|
|
212
|
+
* rate/channel count itself, so no separate decoder-config exchange is needed; always
|
|
213
|
+
* independently decodable, so `isKeyFrame` is always `true`). Without `audio`, every unit has
|
|
214
|
+
* `track: 'video'`.
|
|
169
215
|
*/
|
|
170
216
|
export interface VideoAccessUnit {
|
|
217
|
+
track: 'video' | 'audio';
|
|
171
218
|
data: Buffer;
|
|
172
219
|
isKeyFrame: boolean;
|
|
173
|
-
/** Monotonically increasing per
|
|
220
|
+
/** Monotonically increasing per track, starting at 0 — independent between `'video'` and `'audio'`. */
|
|
174
221
|
sequence: number;
|
|
175
|
-
/** Microseconds since the stream started. */
|
|
222
|
+
/** Microseconds since the stream started, on one shared clock across both tracks. */
|
|
176
223
|
timestampMicros: number;
|
|
177
224
|
}
|
|
178
225
|
|
|
@@ -280,6 +327,7 @@ export type NativeSpawnExitCallback = (code: number | null, signal: number | nul
|
|
|
280
327
|
|
|
281
328
|
/** Raw shape of an access unit as the native addon delivers it — see {@link VideoAccessUnit}. */
|
|
282
329
|
export interface NativeVideoAccessUnit {
|
|
330
|
+
track: 'video' | 'audio';
|
|
283
331
|
data: Buffer;
|
|
284
332
|
isKeyFrame: boolean;
|
|
285
333
|
sequence: number;
|
|
@@ -289,13 +337,29 @@ export interface NativeVideoAccessUnit {
|
|
|
289
337
|
export type NativeVideoAccessUnitCallback = (unit: NativeVideoAccessUnit) => void;
|
|
290
338
|
export type NativeVideoErrorCallback = (err: Error) => void;
|
|
291
339
|
|
|
292
|
-
/**
|
|
340
|
+
/**
|
|
341
|
+
* A live encoder session, wrapped by `coresim.mm`'s `NativeVideoStream` (video only) or
|
|
342
|
+
* `NativeAVStream` (`audio: true` — see {@link VideoStreamOptions}) — either way, what
|
|
343
|
+
* `NativeDeviceHandle.startVideoStream()` resolves to; the two native wrapper classes expose the
|
|
344
|
+
* identical shape below, so callers never need to know which one they got.
|
|
345
|
+
*/
|
|
293
346
|
export interface NativeVideoStreamHandle {
|
|
294
347
|
stop(): Promise<void>;
|
|
295
|
-
/** Forces the next encoded frame to be a keyframe — trivial in-memory flag, so synchronous. */
|
|
348
|
+
/** Forces the next encoded video frame to be a keyframe — trivial in-memory flag, so synchronous. */
|
|
296
349
|
requestKeyFrame(): void;
|
|
297
350
|
}
|
|
298
351
|
|
|
352
|
+
/**
|
|
353
|
+
* A live recording, wrapped by `coresim.mm`'s `NativePrivateRecordingHandle` (video only,
|
|
354
|
+
* addressing CoreSimulator's own internally-tracked private recorder) or `NativeAVRecording`
|
|
355
|
+
* (`audio: true` — see {@link VideoRecordingOptions}, a real local resource with no server-side
|
|
356
|
+
* counterpart) — either way, what `NativeDeviceHandle.startVideoRecording()` resolves to.
|
|
357
|
+
*/
|
|
358
|
+
export interface NativeVideoRecordingHandle {
|
|
359
|
+
/** Resolves once the output file has been finalized on disk and is safe to read. */
|
|
360
|
+
stop(): Promise<void>;
|
|
361
|
+
}
|
|
362
|
+
|
|
299
363
|
/** A `SimDevice`, wrapped by `coresim.mm`'s `NativeDevice` — what `NativeSimctl`'s `_findDevice()` resolves to. */
|
|
300
364
|
export interface NativeDeviceHandle {
|
|
301
365
|
// Trivial in-memory accessors — kept synchronous on the native side (see coresim.mm), never a
|
|
@@ -344,13 +408,26 @@ export interface NativeDeviceHandle {
|
|
|
344
408
|
getWebInspectorSocket(): Promise<string>;
|
|
345
409
|
screenshot(options?: {format?: 'png' | 'jpeg'; displayId?: string; quality?: number}): Promise<Buffer>;
|
|
346
410
|
getDisplays(): Promise<SimDisplayInfo[]>;
|
|
411
|
+
// `mask` only applies without `audio`/`fps`; `bitrate` applies either way — see
|
|
412
|
+
// VideoRecordingOptions's own doc comments for why. `onError` is only ever invoked on the
|
|
413
|
+
// `audio`/`fps` (own-encoder) path — a live mid-recording failure, which the private recorder
|
|
414
|
+
// has no channel to report.
|
|
347
415
|
startVideoRecording(
|
|
348
416
|
outputFile: string,
|
|
349
|
-
options
|
|
350
|
-
|
|
351
|
-
|
|
417
|
+
options:
|
|
418
|
+
| {
|
|
419
|
+
displayId?: string;
|
|
420
|
+
codec?: 'h264' | 'hevc';
|
|
421
|
+
mask?: 'ignored' | 'alpha' | 'black';
|
|
422
|
+
audio?: boolean;
|
|
423
|
+
fps?: number;
|
|
424
|
+
bitrate?: number;
|
|
425
|
+
}
|
|
426
|
+
| undefined,
|
|
427
|
+
onError: NativeVideoErrorCallback,
|
|
428
|
+
): Promise<NativeVideoRecordingHandle>;
|
|
352
429
|
startVideoStream(
|
|
353
|
-
options: {displayId?: string; codec?: 'h264' | 'hevc'; fps?: number; bitrate?: number} | undefined,
|
|
430
|
+
options: {displayId?: string; codec?: 'h264' | 'hevc'; fps?: number; bitrate?: number; audio?: boolean} | undefined,
|
|
354
431
|
onAccessUnit: NativeVideoAccessUnitCallback,
|
|
355
432
|
onError: NativeVideoErrorCallback,
|
|
356
433
|
): Promise<NativeVideoStreamHandle>;
|
|
@@ -376,4 +453,11 @@ export interface NativeServiceContextHandle {
|
|
|
376
453
|
export interface NativeCoreSimModule {
|
|
377
454
|
sharedServiceContext(developerDir: string): Promise<NativeServiceContextHandle>;
|
|
378
455
|
frameworkVersion(): Promise<string>;
|
|
456
|
+
/**
|
|
457
|
+
* Synchronously (not a Promise) stops every still-live video/AV stream or AV recording, blocking
|
|
458
|
+
* until each has released its resources. Meant to be called from a `process.on('exit', ...)`
|
|
459
|
+
* listener (see native-simctl.ts) — cleanup hooks alone don't run under `process.exit()` on the
|
|
460
|
+
* main process/thread, only on a natural empty-event-loop exit or a Worker's own termination.
|
|
461
|
+
*/
|
|
462
|
+
flushActiveSessions(): void;
|
|
379
463
|
}
|