simframe 0.17.0 → 0.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +126 -1058
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +12 -6
- package/package.json +1 -1
- package/scripts/article-md.mjs +111 -45
- package/scripts/bench-hpi.mjs +53 -1
- package/scripts/ci-device-guard.mjs +27 -0
- package/scripts/ci-memory.mjs +33 -6
- package/scripts/demo-gif/README.md +36 -0
- package/scripts/demo-gif/compose.swift +106 -0
- package/scripts/demo-gif/events.example.json +74 -0
- package/scripts/demo-gif/flow.json +6 -0
- package/scripts/device-state.mjs +5 -62
- package/scripts/smithery/icon.png +0 -0
- package/scripts/smithery-bundle.mjs +77 -0
- package/src/actions.js +309 -32
- package/src/cli.js +111 -29
- package/src/device-state.js +101 -0
- package/src/index.js +139 -11
- package/src/metrics.js +144 -3
- package/src/navigate.js +10 -0
- package/src/platform/cdp.js +242 -0
- package/src/platform/index.js +28 -2
- package/src/store.js +39 -0
- package/src/wedge.js +370 -0
package/src/metrics.js
CHANGED
|
@@ -12,6 +12,7 @@
|
|
|
12
12
|
import fs from 'node:fs';
|
|
13
13
|
import path from 'node:path';
|
|
14
14
|
import * as store from './store.js';
|
|
15
|
+
import { deviceCause } from './device-state.js';
|
|
15
16
|
|
|
16
17
|
/** The five reasons, from docs/research/03-human-parity.md §8. Nothing else is a reason. */
|
|
17
18
|
export const REASONS = [
|
|
@@ -66,6 +67,52 @@ export const FACULTY = {
|
|
|
66
67
|
*/
|
|
67
68
|
export const BUILT_FACULTIES = new Set(['sense of time (Phase 11)']);
|
|
68
69
|
|
|
70
|
+
/**
|
|
71
|
+
* The faculty a `verification_failed` record points at, by **verdict**.
|
|
72
|
+
*
|
|
73
|
+
* `FACULTY` is keyed on the reason, and for this class the reason is too coarse
|
|
74
|
+
* to steer by. Measured on the bench device's 1022 records: of the
|
|
75
|
+
* `verification_failed` ones, 162 are `no-visible-change`, 26 are
|
|
76
|
+
* `unexpected-screen`, and about 172 are a wait that timed out — three
|
|
77
|
+
* different faculties, all of which the report named as "sense of time
|
|
78
|
+
* (Phase 11)" because that is what the reason maps to.
|
|
79
|
+
*
|
|
80
|
+
* `unexpected-screen` is the clearest case: it is item 174, screen identity
|
|
81
|
+
* fragmenting on content-driven screens, and reporting it as a timing problem
|
|
82
|
+
* is how a tester was once told their unlabeled-control problem was a timing
|
|
83
|
+
* problem.
|
|
84
|
+
*/
|
|
85
|
+
/**
|
|
86
|
+
* The verdict a legacy record carries in the first token of its `detail`.
|
|
87
|
+
*
|
|
88
|
+
* Written as `${verdict}: ${detail}` since the field existed, so the prefix is
|
|
89
|
+
* reliable — but only for records whose detail came from a verdict at all,
|
|
90
|
+
* which is why this returns null rather than guessing on anything else.
|
|
91
|
+
*/
|
|
92
|
+
export function verdictFromDetail(detail) {
|
|
93
|
+
const m = /^([a-z][a-z-]{3,30}):\s/.exec(String(detail ?? ''));
|
|
94
|
+
// Only a verdict that exists. The first draft returned any lowercase prefix
|
|
95
|
+
// and duly reported a verdict called **"capture"** with a count of 8, from
|
|
96
|
+
// details reading `capture: ...`. A parser that invents a category gets it
|
|
97
|
+
// counted, named in a report, and eventually used to choose a phase — which
|
|
98
|
+
// is the whole failure this grouping was added to fix, reproduced inside the
|
|
99
|
+
// fix.
|
|
100
|
+
return m && KNOWN_VERDICTS.has(m[1]) ? m[1] : null;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
export const VERDICT_FACULTY = {
|
|
104
|
+
'unexpected-screen': 'screen identity (item 174)',
|
|
105
|
+
'still-filling-in': 'sense of time (Phase 11)',
|
|
106
|
+
// **`no-visible-change` is deliberately absent, and it is the largest verdict
|
|
107
|
+
// in the log (162 of 1022).** The escalation site has a measured argument for
|
|
108
|
+
// leaving it unclassified and it is right: in the field it came
|
|
109
|
+
// overwhelmingly from tapping an inert text label whose real hit target was
|
|
110
|
+
// an invisible chevron — icon semantics, Phase 15 — but it also covers a
|
|
111
|
+
// switch moving 0.1% of the screen, which is neither faculty. Two causes, one
|
|
112
|
+
// verdict, and no way to tell them apart from here. Putting it in this map
|
|
113
|
+
// would name a faculty for 162 records on a coin flip.
|
|
114
|
+
};
|
|
115
|
+
|
|
69
116
|
function metricPaths(udid) {
|
|
70
117
|
const dir = store.deviceDir(udid);
|
|
71
118
|
return {
|
|
@@ -323,6 +370,17 @@ export function reasonForStepError(step, err) {
|
|
|
323
370
|
// admits "other" collects a pile of "other". What changes is that the record
|
|
324
371
|
// carries whether the reason was *read off the failure* or *assumed*, and
|
|
325
372
|
// the report declines to recommend a faculty for the assumed ones.
|
|
373
|
+
// The device, not a faculty.
|
|
374
|
+
//
|
|
375
|
+
// 78 of this class on the bench device are `xcrun simctl openurl` failing, an
|
|
376
|
+
// app that would not launch, or capture stopping. Those are item 173, and
|
|
377
|
+
// filing them under a code faculty is how the log came to offer "sense of
|
|
378
|
+
// time (Phase 11)" as the remedy for a simulator that had stopped answering.
|
|
379
|
+
// `classified` stays false because no *faculty* was read; `device` says what
|
|
380
|
+
// was, so the report can take these out of the faculty count instead of
|
|
381
|
+
// counting them towards a phase.
|
|
382
|
+
const device = deviceCause(err?.message);
|
|
383
|
+
if (device) return { reason: 'verification_failed', candidates: [], tried: [], classified: false, device };
|
|
326
384
|
return { reason: 'verification_failed', candidates: [], tried: [], classified: false };
|
|
327
385
|
}
|
|
328
386
|
|
|
@@ -336,6 +394,17 @@ export function reasonForStepError(step, err) {
|
|
|
336
394
|
*/
|
|
337
395
|
export const ESCALATING_VERDICTS = new Set(['unexpected-screen', 'no-visible-change']);
|
|
338
396
|
|
|
397
|
+
/**
|
|
398
|
+
* Every verdict the engine can report, for reading legacy details back.
|
|
399
|
+
*
|
|
400
|
+
* Wider than `ESCALATING_VERDICTS` — which is the two that hand back to a model
|
|
401
|
+
* — because a record's detail may carry any of them, and narrower than "any
|
|
402
|
+
* lowercase word", which is what a prefix parser accepts if nobody bounds it.
|
|
403
|
+
*/
|
|
404
|
+
export const KNOWN_VERDICTS = new Set([
|
|
405
|
+
'unexpected-screen', 'no-visible-change', 'still-filling-in', 'unverified', 'ok',
|
|
406
|
+
]);
|
|
407
|
+
|
|
339
408
|
/** How `goto`/`flow run` refusals map. They refuse rather than guess, and the refusal is the hand-back. */
|
|
340
409
|
export const PLAN_REASONS = {
|
|
341
410
|
'unknown-screen': 'unknown_screen',
|
|
@@ -463,6 +532,14 @@ export function recordEscalation(udid, {
|
|
|
463
532
|
// Default `false`, so a caller that does not think about it cannot
|
|
464
533
|
// accidentally claim precision it does not have.
|
|
465
534
|
classified = false,
|
|
535
|
+
// Which device-state condition this failure shows, when it shows one. Kept
|
|
536
|
+
// separate from `reason` because the five-reason vocabulary is fixed and a
|
|
537
|
+
// sixth reason collects a pile of "other" — see REASONS.
|
|
538
|
+
device = null,
|
|
539
|
+
// Which local verdict fired, for the records that have one. Derivable from
|
|
540
|
+
// `detail` today by parsing a prefix, which is exactly the fragility the log
|
|
541
|
+
// should not depend on.
|
|
542
|
+
verdict = null,
|
|
466
543
|
} = {}) {
|
|
467
544
|
if (!REASONS.includes(reason)) throw new Error(`not an escalation reason: ${reason}`);
|
|
468
545
|
if (!OUTCOMES.includes(outcome)) throw new Error(`not an escalation outcome: ${outcome}`);
|
|
@@ -479,6 +556,10 @@ export function recordEscalation(udid, {
|
|
|
479
556
|
// go/no-go, and on its own it answers "what kind of decision is costing us".
|
|
480
557
|
intent: intent ? String(intent).slice(0, 120) : null,
|
|
481
558
|
classified: Boolean(classified),
|
|
559
|
+
// Null unless read. Both of these separate "we know" from "we assumed" for
|
|
560
|
+
// a class that is otherwise one coarse bucket.
|
|
561
|
+
device_cause: device ? String(device).slice(0, 120) : null,
|
|
562
|
+
verdict: verdict ? String(verdict).slice(0, 60) : null,
|
|
482
563
|
step_index: stepIndex,
|
|
483
564
|
screen_fingerprint: fingerprint,
|
|
484
565
|
reason,
|
|
@@ -538,7 +619,9 @@ export function flowRecordFrom({
|
|
|
538
619
|
images_sent: imagesSent,
|
|
539
620
|
input_tokens: null,
|
|
540
621
|
output_tokens: null,
|
|
541
|
-
escalations: escalations.map((e) => ({
|
|
622
|
+
escalations: escalations.map((e) => ({
|
|
623
|
+
reason: e.reason, step_index: e.step_index, outcome: e.outcome, device_cause: e.device_cause ?? null,
|
|
624
|
+
})),
|
|
542
625
|
escalation_count: escalations.length,
|
|
543
626
|
mis_taps: misTaps,
|
|
544
627
|
verdict_histogram: histogram,
|
|
@@ -546,6 +629,22 @@ export function flowRecordFrom({
|
|
|
546
629
|
exploration_events: [],
|
|
547
630
|
completed: Boolean(completed),
|
|
548
631
|
wrong_action_taken: verdicts.includes('unexpected-screen'),
|
|
632
|
+
// Did this run fail because the *simulator* failed?
|
|
633
|
+
//
|
|
634
|
+
// Measured, on a full local suite: of 17 runs, 5 failed because the guest's
|
|
635
|
+
// SpringBoard crashed or `simctl` stopped answering for 90 s. `hpi`
|
|
636
|
+
// counted every one against `HPI_accuracy`, so the gate CI reads was partly
|
|
637
|
+
// measuring SpringBoard's stability.
|
|
638
|
+
//
|
|
639
|
+
// This is item 172 in the other column. There, `HPI_time` took every run's
|
|
640
|
+
// wall clock regardless of completion, so breaking a flow registered as the
|
|
641
|
+
// agent getting quicker; the fix was to time only completed runs. Accuracy
|
|
642
|
+
// had the mirror-image fault and kept it.
|
|
643
|
+
//
|
|
644
|
+
// Derived from the escalations this run already wrote — `device_cause` is
|
|
645
|
+
// set at the step that suffered it — so nothing new has to be plumbed and a
|
|
646
|
+
// run cannot claim a device fault that its own log does not show.
|
|
647
|
+
device_cause: escalations.map((e) => e.device_cause).find(Boolean) ?? null,
|
|
549
648
|
};
|
|
550
649
|
}
|
|
551
650
|
|
|
@@ -691,8 +790,14 @@ export function hpi({ flows, baselines = {} }) {
|
|
|
691
790
|
};
|
|
692
791
|
}).sort((a, b) => a.flow.localeCompare(b.flow));
|
|
693
792
|
|
|
694
|
-
|
|
695
|
-
|
|
793
|
+
// A run the simulator broke is not a run the code got wrong. It is an
|
|
794
|
+
// **unmeasured** run, and it leaves the denominator rather than lowering the
|
|
795
|
+
// score — the same discipline as timing only completed runs (172), and the
|
|
796
|
+
// same discipline as the suite refusing to publish a partial HPI at all.
|
|
797
|
+
const lostToDevice = flows.filter((f) => f.device_cause && !f.completed);
|
|
798
|
+
const measurable = flows.filter((f) => !(f.device_cause && !f.completed));
|
|
799
|
+
const total = measurable.length;
|
|
800
|
+
const clean = measurable.filter((f) => f.completed && !f.wrong_action_taken).length;
|
|
696
801
|
const accuracy = total ? Number((clean / total).toFixed(3)) : null;
|
|
697
802
|
const times = perFlow.map((f) => f.hpi_time).filter((x) => Number.isFinite(x));
|
|
698
803
|
const hpiTime = harmonicMean(times);
|
|
@@ -700,6 +805,11 @@ export function hpi({ flows, baselines = {} }) {
|
|
|
700
805
|
flows: perFlow,
|
|
701
806
|
overall: {
|
|
702
807
|
runs: total,
|
|
808
|
+
// Said out loud, because a denominator that quietly shrinks is worse than
|
|
809
|
+
// one that is wrong: an accuracy of 1.0 over two measurable runs is not
|
|
810
|
+
// the same claim as 1.0 over eighteen.
|
|
811
|
+
runs_lost_to_device: lostToDevice.length,
|
|
812
|
+
device_causes: [...new Set(lostToDevice.map((f) => f.device_cause))],
|
|
703
813
|
flows_measured: perFlow.length,
|
|
704
814
|
flows_with_human_baseline: times.length,
|
|
705
815
|
hpi_accuracy: accuracy,
|
|
@@ -728,12 +838,20 @@ export function breakdown(records, { session = null, flow = null } = {}) {
|
|
|
728
838
|
const classifiedByReason = {};
|
|
729
839
|
const assumedByReason = {};
|
|
730
840
|
for (const r of REASONS) { byReason[r] = 0; classifiedByReason[r] = 0; assumedByReason[r] = 0; }
|
|
841
|
+
// Two more groupings, because the reason alone could not steer. `byVerdict`
|
|
842
|
+
// splits the largest class into the three different things it holds, and
|
|
843
|
+
// `byDevice` takes out the records that are the simulator rather than the
|
|
844
|
+
// code — those were being counted towards a perception phase.
|
|
845
|
+
const byVerdict = new Map();
|
|
846
|
+
const byDevice = new Map();
|
|
731
847
|
const byScreen = new Map();
|
|
732
848
|
const byOutcome = {};
|
|
733
849
|
const bySession = new Map();
|
|
734
850
|
const byFlow = new Map();
|
|
735
851
|
let avoidable = 0;
|
|
736
852
|
let unattributed = 0;
|
|
853
|
+
let derivedVerdicts = 0;
|
|
854
|
+
let derivedDevice = 0;
|
|
737
855
|
// Filtering happens here rather than at the call site so `total` and every
|
|
738
856
|
// rate below it describe the same set of records.
|
|
739
857
|
const kept = records.filter((r) => (session ? r?.session_id === session : true))
|
|
@@ -757,6 +875,18 @@ export function breakdown(records, { session = null, flow = null } = {}) {
|
|
|
757
875
|
// real one gets ignored, which this file already knows in another place.
|
|
758
876
|
if (r.classified === true) classifiedByReason[r.reason] += 1;
|
|
759
877
|
else if (r.classified === false) assumedByReason[r.reason] += 1;
|
|
878
|
+
// Derived for the records that predate the fields, rather than waiting for
|
|
879
|
+
// a fresh corpus. `detail` already carries both facts — the verdict as its
|
|
880
|
+
// prefix, the device signature inside the message — so a read-time
|
|
881
|
+
// derivation turns 1022 existing records into signal without rewriting a
|
|
882
|
+
// single line of the log. Marked in the output as derived, because a
|
|
883
|
+
// recorded fact and a parsed one are not the same evidence.
|
|
884
|
+
const verdict = r.verdict ?? verdictFromDetail(r.detail);
|
|
885
|
+
const device = r.device_cause ?? deviceCause(r.detail);
|
|
886
|
+
if (verdict) byVerdict.set(verdict, (byVerdict.get(verdict) ?? 0) + 1);
|
|
887
|
+
if (device) byDevice.set(device, (byDevice.get(device) ?? 0) + 1);
|
|
888
|
+
if (!r.verdict && verdict) derivedVerdicts += 1;
|
|
889
|
+
if (!r.device_cause && device) derivedDevice += 1;
|
|
760
890
|
byOutcome[r.outcome] = (byOutcome[r.outcome] ?? 0) + 1;
|
|
761
891
|
// Already avoided locally, so not avoidable by anything unbuilt.
|
|
762
892
|
if (r.outcome !== 'resolved_locally') avoidable += 1;
|
|
@@ -770,8 +900,19 @@ export function breakdown(records, { session = null, flow = null } = {}) {
|
|
|
770
900
|
return { session_id: id, client, count };
|
|
771
901
|
})
|
|
772
902
|
.sort((a, b) => b.count - a.count);
|
|
903
|
+
const sorted = (m) => [...m.entries()].sort((a, b) => b[1] - a[1]).map(([name, count]) => ({ name, count }));
|
|
773
904
|
return {
|
|
774
905
|
total,
|
|
906
|
+
// What the reason could not say. `verdicts` is the largest class split into
|
|
907
|
+
// the faculties it actually implies; `device` is the part that is item 173
|
|
908
|
+
// wearing a code reason.
|
|
909
|
+
verdicts: sorted(byVerdict),
|
|
910
|
+
device: sorted(byDevice),
|
|
911
|
+
device_total: [...byDevice.values()].reduce((a, b) => a + b, 0),
|
|
912
|
+
// How much of the two groupings above was parsed out of `detail` rather
|
|
913
|
+
// than recorded at the time. A reader deciding a phase order should know
|
|
914
|
+
// which half they are looking at.
|
|
915
|
+
derived: { verdicts: derivedVerdicts, device: derivedDevice },
|
|
775
916
|
// The log is per-device and shared: two agents on one booted simulator
|
|
776
917
|
// write one interleaved file. More than one session here means the counts
|
|
777
918
|
// below are a pool, and CLAUDE.md uses those counts to choose a phase.
|
package/src/navigate.js
CHANGED
|
@@ -47,6 +47,16 @@ function refuse(udid, result, { detail = null, flowName = null } = {}) {
|
|
|
47
47
|
if (reason) {
|
|
48
48
|
metrics.recordEscalation(udid, {
|
|
49
49
|
reason,
|
|
50
|
+
// Read, not assumed. `recordEscalation` defaults `classified` to false
|
|
51
|
+
// so a careless caller cannot claim precision it does not have, which
|
|
52
|
+
// is right — and this caller is not careless: `result.reason` is a
|
|
53
|
+
// named refusal (`no-route`, `unreplayable-edge`, `unknown-flow`,
|
|
54
|
+
// `arrived-elsewhere`…) and PLAN_REASONS maps it deterministically.
|
|
55
|
+
// Saying nothing filed 82 records on the bench device as "reason
|
|
56
|
+
// assumed" when the reason was known exactly, which makes the log
|
|
57
|
+
// understate its own knowledge and the report decline to name a
|
|
58
|
+
// faculty it was entitled to name.
|
|
59
|
+
classified: true,
|
|
50
60
|
// A refusal by `goto` is about a destination and one by `flow run` is
|
|
51
61
|
// about a named flow. Either is what a breakdown wants to group by.
|
|
52
62
|
flowName,
|
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
// A dependency-free Chrome DevTools Protocol client.
|
|
2
|
+
//
|
|
3
|
+
// Hand-rolled rather than `ws`, and not because of ideology: CLAUDE.md keeps
|
|
4
|
+
// the runtime dependency list at exactly one (the MCP SDK), and the alternative
|
|
5
|
+
// does not exist anyway. The global `WebSocket` is **absent on Node 18 and
|
|
6
|
+
// flag-only on Node 20** — verified here on v20.20.0, where `typeof WebSocket`
|
|
7
|
+
// is `undefined` — and this package supports 18, 20 and 22. So the choice is
|
|
8
|
+
// between a dependency and forty lines of RFC 6455, and forty lines is smaller
|
|
9
|
+
// than the surface a dependency brings.
|
|
10
|
+
//
|
|
11
|
+
// The other thing a probe on 2026-09-11 established, recorded because it costs
|
|
12
|
+
// an afternoon to rediscover: Chrome refuses the upgrade with **401
|
|
13
|
+
// Unauthorized** when an `Origin` header is present and does not match the
|
|
14
|
+
// inspector's own host. A browser's own WebSocket API always sends one and
|
|
15
|
+
// cannot change it, which is why a page cannot drive CDP — and why a
|
|
16
|
+
// hand-rolled client, which simply omits the header, can.
|
|
17
|
+
//
|
|
18
|
+
// Scope: enough CDP to be the transport under `src/platform/web.js`. It speaks
|
|
19
|
+
// one target at a time, it does not multiplex sessions, and it has no
|
|
20
|
+
// reconnection policy. Anything cleverer belongs above the boundary or nowhere.
|
|
21
|
+
import net from 'node:net';
|
|
22
|
+
import crypto from 'node:crypto';
|
|
23
|
+
import http from 'node:http';
|
|
24
|
+
|
|
25
|
+
/** RFC 6455's fixed GUID, concatenated with the client key to prove the handshake. */
|
|
26
|
+
export const WS_GUID = '258EAFA5-E914-47DA-95CA-C5AB0DC85B11';
|
|
27
|
+
|
|
28
|
+
const OPCODE = { text: 0x1, binary: 0x2, close: 0x8, ping: 0x9, pong: 0xa };
|
|
29
|
+
|
|
30
|
+
/** Ask a browser what it has open. `targetId` is the udid: a tab is a device. */
|
|
31
|
+
export function listTargets(port, host = '127.0.0.1', { timeoutMs = 2000 } = {}) {
|
|
32
|
+
return new Promise((resolve, reject) => {
|
|
33
|
+
const req = http.get({ host, port, path: '/json/list', timeout: timeoutMs }, (res) => {
|
|
34
|
+
let body = '';
|
|
35
|
+
res.on('data', (d) => { body += d; });
|
|
36
|
+
res.on('end', () => {
|
|
37
|
+
if (res.statusCode !== 200) {
|
|
38
|
+
reject(new Error(`the browser's debugging endpoint answered ${res.statusCode} — is it running with --remote-debugging-port=${port}?`));
|
|
39
|
+
return;
|
|
40
|
+
}
|
|
41
|
+
try { resolve(JSON.parse(body)); } catch (err) { reject(new Error(`the debugging endpoint returned something that is not JSON: ${err.message}`)); }
|
|
42
|
+
});
|
|
43
|
+
});
|
|
44
|
+
req.on('timeout', () => { req.destroy(new Error(`no answer from ${host}:${port} within ${timeoutMs}ms`)); });
|
|
45
|
+
req.on('error', (err) => reject(new Error(
|
|
46
|
+
`cannot reach a browser on ${host}:${port} — ${err.message}.`
|
|
47
|
+
+ ' Start one with --remote-debugging-port, or pass --port.',
|
|
48
|
+
)));
|
|
49
|
+
});
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* One frame, encoded.
|
|
54
|
+
*
|
|
55
|
+
* Client frames must be masked — an unmasked client frame is a protocol error
|
|
56
|
+
* and Chrome closes the socket rather than answering, which reads as a hang.
|
|
57
|
+
*/
|
|
58
|
+
function encodeFrame(payload, opcode = OPCODE.text) {
|
|
59
|
+
const data = Buffer.from(payload, 'utf8');
|
|
60
|
+
const mask = crypto.randomBytes(4);
|
|
61
|
+
const len = data.length;
|
|
62
|
+
const header = len < 126
|
|
63
|
+
? Buffer.from([0x80 | opcode, 0x80 | len])
|
|
64
|
+
: len < 65536
|
|
65
|
+
? Buffer.concat([Buffer.from([0x80 | opcode, 0x80 | 126]), (() => { const b = Buffer.alloc(2); b.writeUInt16BE(len); return b; })()])
|
|
66
|
+
: Buffer.concat([Buffer.from([0x80 | opcode, 0x80 | 127]), (() => { const b = Buffer.alloc(8); b.writeBigUInt64BE(BigInt(len)); return b; })()]);
|
|
67
|
+
const masked = Buffer.allocUnsafe(len);
|
|
68
|
+
for (let i = 0; i < len; i += 1) masked[i] = data[i] ^ mask[i % 4];
|
|
69
|
+
return Buffer.concat([header, mask, masked]);
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/**
|
|
73
|
+
* Pull whole frames out of a growing buffer.
|
|
74
|
+
*
|
|
75
|
+
* Returns the frames it could complete and what is left over. CDP replies
|
|
76
|
+
* routinely exceed one TCP segment — a full accessibility tree is tens of
|
|
77
|
+
* kilobytes — so a reader that assumes one frame per `data` event works on a
|
|
78
|
+
* hello-world and fails on the first real payload.
|
|
79
|
+
*/
|
|
80
|
+
export function decodeFrames(buffer) {
|
|
81
|
+
const frames = [];
|
|
82
|
+
let rest = buffer;
|
|
83
|
+
for (;;) {
|
|
84
|
+
if (rest.length < 2) break;
|
|
85
|
+
const opcode = rest[0] & 0x0f;
|
|
86
|
+
const fin = Boolean(rest[0] & 0x80);
|
|
87
|
+
const masked = Boolean(rest[1] & 0x80);
|
|
88
|
+
let len = rest[1] & 0x7f;
|
|
89
|
+
let offset = 2;
|
|
90
|
+
if (len === 126) {
|
|
91
|
+
if (rest.length < 4) break;
|
|
92
|
+
len = rest.readUInt16BE(2);
|
|
93
|
+
offset = 4;
|
|
94
|
+
} else if (len === 127) {
|
|
95
|
+
if (rest.length < 10) break;
|
|
96
|
+
len = Number(rest.readBigUInt64BE(2));
|
|
97
|
+
offset = 10;
|
|
98
|
+
}
|
|
99
|
+
const maskKey = masked ? rest.subarray(offset, offset + 4) : null;
|
|
100
|
+
if (masked) offset += 4;
|
|
101
|
+
if (rest.length < offset + len) break;
|
|
102
|
+
let payload = rest.subarray(offset, offset + len);
|
|
103
|
+
if (maskKey) {
|
|
104
|
+
const out = Buffer.allocUnsafe(len);
|
|
105
|
+
for (let i = 0; i < len; i += 1) out[i] = payload[i] ^ maskKey[i % 4];
|
|
106
|
+
payload = out;
|
|
107
|
+
}
|
|
108
|
+
frames.push({ opcode, fin, payload });
|
|
109
|
+
rest = rest.subarray(offset + len);
|
|
110
|
+
}
|
|
111
|
+
return { frames, rest };
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/**
|
|
115
|
+
* Open a CDP session against a websocket URL from `listTargets`.
|
|
116
|
+
*
|
|
117
|
+
* Resolves once the upgrade is accepted, so a caller that gets a connection has
|
|
118
|
+
* a usable one — the failure mode this avoids is a `send` that queues silently
|
|
119
|
+
* against a socket the browser refused.
|
|
120
|
+
*/
|
|
121
|
+
export function connect(wsUrl, { timeoutMs = 5000 } = {}) {
|
|
122
|
+
const url = new URL(wsUrl);
|
|
123
|
+
return new Promise((resolve, reject) => {
|
|
124
|
+
const key = crypto.randomBytes(16).toString('base64');
|
|
125
|
+
const socket = net.connect({ host: url.hostname, port: Number(url.port || 80) });
|
|
126
|
+
let settled = false;
|
|
127
|
+
const fail = (err) => {
|
|
128
|
+
if (settled) return;
|
|
129
|
+
settled = true;
|
|
130
|
+
socket.destroy();
|
|
131
|
+
reject(err);
|
|
132
|
+
};
|
|
133
|
+
const timer = setTimeout(() => fail(new Error(`the browser did not complete the websocket upgrade within ${timeoutMs}ms`)), timeoutMs);
|
|
134
|
+
|
|
135
|
+
socket.on('error', (err) => fail(new Error(`cannot open a debugging socket: ${err.message}`)));
|
|
136
|
+
socket.on('connect', () => {
|
|
137
|
+
// No `Origin` header, deliberately. Chrome answers 401 when one is
|
|
138
|
+
// present and does not match the inspector's host, and nothing here needs
|
|
139
|
+
// to claim an origin.
|
|
140
|
+
socket.write(
|
|
141
|
+
`GET ${url.pathname}${url.search} HTTP/1.1\r\n`
|
|
142
|
+
+ `Host: ${url.host}\r\n`
|
|
143
|
+
+ 'Upgrade: websocket\r\nConnection: Upgrade\r\n'
|
|
144
|
+
+ `Sec-WebSocket-Key: ${key}\r\n`
|
|
145
|
+
+ 'Sec-WebSocket-Version: 13\r\n\r\n',
|
|
146
|
+
);
|
|
147
|
+
});
|
|
148
|
+
|
|
149
|
+
let handshake = Buffer.alloc(0);
|
|
150
|
+
const onHandshake = (chunk) => {
|
|
151
|
+
handshake = Buffer.concat([handshake, chunk]);
|
|
152
|
+
const end = handshake.indexOf('\r\n\r\n');
|
|
153
|
+
if (end === -1) return;
|
|
154
|
+
const head = handshake.subarray(0, end).toString('latin1');
|
|
155
|
+
const status = /^HTTP\/1\.1 (\d+)/.exec(head)?.[1];
|
|
156
|
+
if (status !== '101') {
|
|
157
|
+
fail(new Error(
|
|
158
|
+
`the browser refused the debugging socket with HTTP ${status ?? '(no status)'}.`
|
|
159
|
+
+ (status === '401'
|
|
160
|
+
? ' That is the Origin check: Chrome rejects an upgrade whose Origin header does not'
|
|
161
|
+
+ ' match the inspector host. This client sends none, so something else set one.'
|
|
162
|
+
: ' Check the target still exists — a closed tab keeps its id but not its socket.'),
|
|
163
|
+
));
|
|
164
|
+
return;
|
|
165
|
+
}
|
|
166
|
+
const accept = /sec-websocket-accept:\s*(\S+)/i.exec(head)?.[1];
|
|
167
|
+
const expected = crypto.createHash('sha1').update(key + WS_GUID).digest('base64');
|
|
168
|
+
if (accept !== expected) {
|
|
169
|
+
fail(new Error('the browser accepted the upgrade with a key that does not match — this is not a websocket peer'));
|
|
170
|
+
return;
|
|
171
|
+
}
|
|
172
|
+
clearTimeout(timer);
|
|
173
|
+
settled = true;
|
|
174
|
+
socket.removeListener('data', onHandshake);
|
|
175
|
+
resolve(session(socket, handshake.subarray(end + 4)));
|
|
176
|
+
};
|
|
177
|
+
socket.on('data', onHandshake);
|
|
178
|
+
});
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
/** A connected session: one id space, one pending map, text frames only. */
|
|
182
|
+
function session(socket, leftover) {
|
|
183
|
+
let buffer = leftover ?? Buffer.alloc(0);
|
|
184
|
+
let nextId = 1;
|
|
185
|
+
let closedWith = null;
|
|
186
|
+
const pending = new Map();
|
|
187
|
+
const listeners = new Set();
|
|
188
|
+
|
|
189
|
+
const rejectAll = (err) => {
|
|
190
|
+
for (const { reject } of pending.values()) reject(err);
|
|
191
|
+
pending.clear();
|
|
192
|
+
};
|
|
193
|
+
|
|
194
|
+
socket.on('data', (chunk) => {
|
|
195
|
+
buffer = Buffer.concat([buffer, chunk]);
|
|
196
|
+
const { frames, rest } = decodeFrames(buffer);
|
|
197
|
+
buffer = rest;
|
|
198
|
+
for (const frame of frames) {
|
|
199
|
+
if (frame.opcode === OPCODE.ping) { socket.write(encodeFrame(frame.payload.toString('utf8'), OPCODE.pong)); continue; }
|
|
200
|
+
if (frame.opcode === OPCODE.close) { closedWith = 'the browser closed the debugging socket'; socket.end(); continue; }
|
|
201
|
+
if (frame.opcode !== OPCODE.text) continue;
|
|
202
|
+
let message;
|
|
203
|
+
try { message = JSON.parse(frame.payload.toString('utf8')); } catch { continue; }
|
|
204
|
+
if (message.id != null && pending.has(message.id)) {
|
|
205
|
+
const { resolve, reject } = pending.get(message.id);
|
|
206
|
+
pending.delete(message.id);
|
|
207
|
+
// A CDP error is a reply, not a transport failure, and it carries the
|
|
208
|
+
// only sentence that says what was wrong with the call.
|
|
209
|
+
if (message.error) reject(new Error(`${message.error.message ?? 'CDP error'}${message.error.data ? ` — ${message.error.data}` : ''}`));
|
|
210
|
+
else resolve(message.result ?? {});
|
|
211
|
+
} else if (message.method) {
|
|
212
|
+
for (const fn of listeners) { try { fn(message); } catch { /* a listener must not break the socket */ } }
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
});
|
|
216
|
+
|
|
217
|
+
socket.on('close', () => { rejectAll(new Error(closedWith ?? 'the debugging socket closed')); });
|
|
218
|
+
socket.on('error', (err) => { rejectAll(new Error(`debugging socket error: ${err.message}`)); });
|
|
219
|
+
|
|
220
|
+
return {
|
|
221
|
+
/** One CDP call. Rejects with the browser's own message on a protocol error. */
|
|
222
|
+
send(method, params = {}, { timeoutMs = 10000 } = {}) {
|
|
223
|
+
if (socket.destroyed) return Promise.reject(new Error(closedWith ?? 'the debugging socket is closed'));
|
|
224
|
+
const id = nextId++;
|
|
225
|
+
return new Promise((resolve, reject) => {
|
|
226
|
+
const timer = setTimeout(() => {
|
|
227
|
+
pending.delete(id);
|
|
228
|
+
reject(new Error(`${method} did not answer within ${timeoutMs}ms`));
|
|
229
|
+
}, timeoutMs);
|
|
230
|
+
pending.set(id, {
|
|
231
|
+
resolve: (v) => { clearTimeout(timer); resolve(v); },
|
|
232
|
+
reject: (e) => { clearTimeout(timer); reject(e); },
|
|
233
|
+
});
|
|
234
|
+
socket.write(encodeFrame(JSON.stringify({ id, method, params })));
|
|
235
|
+
});
|
|
236
|
+
},
|
|
237
|
+
/** Subscribe to CDP events. Returns an unsubscribe. */
|
|
238
|
+
on(fn) { listeners.add(fn); return () => listeners.delete(fn); },
|
|
239
|
+
close() { try { socket.end(encodeFrame('', OPCODE.close)); } catch { socket.destroy(); } },
|
|
240
|
+
get closed() { return socket.destroyed; },
|
|
241
|
+
};
|
|
242
|
+
}
|
package/src/platform/index.js
CHANGED
|
@@ -20,6 +20,7 @@
|
|
|
20
20
|
// function that takes a udid routes on that udid — `ownsUdid` answers that from
|
|
21
21
|
// the id's own shape, because the question is asked from inside a capture loop
|
|
22
22
|
// in a process that never listed anything.
|
|
23
|
+
import * as store from '../store.js';
|
|
23
24
|
import { platform as android } from './android.js';
|
|
24
25
|
import { platform as ios } from './ios.js';
|
|
25
26
|
|
|
@@ -204,9 +205,34 @@ export async function resolveAcross(query, opts, all) {
|
|
|
204
205
|
// backend a call reaches, never what the call means.
|
|
205
206
|
export const isBootedSync = (udid, ...args) => platformFor(udid).isBootedSync(udid, ...args);
|
|
206
207
|
export const screenshot = (udid, ...args) => platformFor(udid).screenshot(udid, ...args);
|
|
207
|
-
|
|
208
|
+
/**
|
|
209
|
+
* Launch and openUrl stamp the action clock; nothing else here does.
|
|
210
|
+
*
|
|
211
|
+
* These two change the screen without touching the digitizer, so they are the
|
|
212
|
+
* only actions a gesture log cannot see — and a settle that cannot see them
|
|
213
|
+
* measures how long the screen being *left* has been sitting still. Measured:
|
|
214
|
+
* 283ms after a Settings launch, `settled: true` with `stableForMs: 4427` over
|
|
215
|
+
* a screen holding zero elements.
|
|
216
|
+
*
|
|
217
|
+
* **At the seam rather than in the `launch` step, and that placement is the
|
|
218
|
+
* point.** It was in the step first, which covered flows and missed everything
|
|
219
|
+
* else that launches an app — `baseline.resetFor`, `wedge.revive`, the bench,
|
|
220
|
+
* and the probe that found this. A rule enforced where you noticed it covers a
|
|
221
|
+
* symptom; this project's handoff has three worked examples of exactly that and
|
|
222
|
+
* this was very nearly a fourth.
|
|
223
|
+
*
|
|
224
|
+
* `terminateApp` is deliberately not stamped: it removes an app rather than
|
|
225
|
+
* presenting one, and the screen it leaves behind is whatever was underneath.
|
|
226
|
+
*/
|
|
227
|
+
export const launchApp = (udid, ...args) => {
|
|
228
|
+
store.noteAction(udid);
|
|
229
|
+
return platformFor(udid).launchApp(udid, ...args);
|
|
230
|
+
};
|
|
208
231
|
export const terminateApp = (udid, ...args) => platformFor(udid).terminateApp(udid, ...args);
|
|
209
|
-
export const openUrl = (udid, ...args) =>
|
|
232
|
+
export const openUrl = (udid, ...args) => {
|
|
233
|
+
store.noteAction(udid);
|
|
234
|
+
return platformFor(udid).openUrl(udid, ...args);
|
|
235
|
+
};
|
|
210
236
|
export const setPermission = (udid, ...args) => platformFor(udid).setPermission(udid, ...args);
|
|
211
237
|
export const setPasteboard = (udid, ...args) => platformFor(udid).setPasteboard(udid, ...args);
|
|
212
238
|
|
package/src/store.js
CHANGED
|
@@ -35,6 +35,7 @@ export function paths(udid) {
|
|
|
35
35
|
// recorded, and a stall is the absence of frames.
|
|
36
36
|
captureHealth: path.join(dir, 'capture-health.json'),
|
|
37
37
|
lastInput: path.join(dir, 'last-input'),
|
|
38
|
+
lastAction: path.join(dir, 'last-action'),
|
|
38
39
|
};
|
|
39
40
|
}
|
|
40
41
|
|
|
@@ -58,6 +59,44 @@ export function noteInput(udid, at = Date.now()) {
|
|
|
58
59
|
} catch {
|
|
59
60
|
/* a timestamp nothing depends on for correctness must not fail an action */
|
|
60
61
|
}
|
|
62
|
+
noteAction(udid, at);
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* When we last did something that ought to change the screen.
|
|
67
|
+
*
|
|
68
|
+
* Deliberately separate from the input log above, which answers a different
|
|
69
|
+
* question — "have several *gestures* landed with no pixel moving" — and would
|
|
70
|
+
* be wrong to answer it about a launch, since a launch that paints nothing is
|
|
71
|
+
* not evidence of a dead digitizer.
|
|
72
|
+
*
|
|
73
|
+
* This one exists so a settle can tell its own stillness from the *previous*
|
|
74
|
+
* screen's. Measured on 2026-09-18: 283ms after a Settings launch the state
|
|
75
|
+
* read `settled: true` with `stableForMs: 4427` and **zero elements** — 4.4
|
|
76
|
+
* seconds of quiet that began before the launch was issued. 408ms after a
|
|
77
|
+
* `tap General`, `stableForMs: 7753` and the 23 elements of the screen being
|
|
78
|
+
* left. In both, the settle detector was honestly reporting how long the screen
|
|
79
|
+
* we had already abandoned had been sitting still.
|
|
80
|
+
*
|
|
81
|
+
* Every gesture writes it (via `noteInput`), and so does every launch and
|
|
82
|
+
* `openUrl`, because those change the screen without touching the digitizer.
|
|
83
|
+
*/
|
|
84
|
+
export function noteAction(udid, at = Date.now()) {
|
|
85
|
+
try {
|
|
86
|
+
writeAtomic(paths(udid).lastAction, String(at));
|
|
87
|
+
} catch {
|
|
88
|
+
/* as above: a timestamp may not fail the action it describes */
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/** When the last screen-changing action was issued, or null if none is recorded. */
|
|
93
|
+
export function lastActionAt(udid) {
|
|
94
|
+
try {
|
|
95
|
+
const n = Number(fs.readFileSync(paths(udid).lastAction, 'utf8').trim());
|
|
96
|
+
return Number.isFinite(n) ? n : null;
|
|
97
|
+
} catch {
|
|
98
|
+
return null;
|
|
99
|
+
}
|
|
61
100
|
}
|
|
62
101
|
|
|
63
102
|
/**
|