simframe 0.14.1 → 0.14.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/native/simframed/Sources/PrivateAPI/AccessibilityBridge.swift +49 -2
- package/native/simframed/Sources/SimframeCore/Element.swift +19 -2
- package/package.json +1 -1
- package/scripts/analyse-escalations.mjs +89 -0
- package/scripts/analyse-routes.mjs +96 -0
- package/scripts/check-published.mjs +121 -0
- package/scripts/ci-device-guard.mjs +18 -0
- package/scripts/classify-stray.mjs +63 -0
- package/scripts/eval-fingerprint.mjs +135 -10
- package/src/actions.js +256 -14
- package/src/cli.js +39 -2
- package/src/index.js +15 -4
- package/src/input.js +33 -1
- package/src/matching.js +11 -1
- package/src/mcp.js +43 -2
- package/src/storage.js +55 -6
- package/src/view.js +1 -1
|
@@ -20,6 +20,7 @@
|
|
|
20
20
|
import fs from 'node:fs';
|
|
21
21
|
import * as actions from '../src/actions.js';
|
|
22
22
|
import * as fingerprint from '../src/fingerprint.js';
|
|
23
|
+
import { classifyStray } from './classify-stray.mjs';
|
|
23
24
|
import * as graph from '../src/graph.js';
|
|
24
25
|
import * as api from '../src/index.js';
|
|
25
26
|
|
|
@@ -152,12 +153,24 @@ const STALE_READ_RETRIES = 3;
|
|
|
152
153
|
* Backed off rather than fixed, because the thing being waited for is a screen
|
|
153
154
|
* finishing its draw, and the runner that needs this is the slow one.
|
|
154
155
|
*/
|
|
156
|
+
/**
|
|
157
|
+
* How old a frame may be and still be a reading of NOW.
|
|
158
|
+
*
|
|
159
|
+
* `FRAME_IS_CURRENT_MS` in src/index.js is 2500 for the same question on the
|
|
160
|
+
* settle path. Doubled here because this harness deliberately reads cold and a
|
|
161
|
+
* hosted runner's capture p50 is 2.5x this laptop's — but bounded, because a
|
|
162
|
+
* 38-second-old frame is a different screen, not a slow one.
|
|
163
|
+
*/
|
|
164
|
+
const FRAME_TOO_OLD_MS = 5000;
|
|
165
|
+
|
|
155
166
|
const SPARSE_READ_RETRIES = 3;
|
|
156
167
|
const SPARSE_READ_WAIT_MS = 1500;
|
|
157
168
|
const STALE_READ_WAIT_MS = 1000;
|
|
158
169
|
|
|
159
170
|
/** When the steps that were supposed to change the screen finished. */
|
|
160
171
|
let navigatedAt = 0;
|
|
172
|
+
/** What the tour's own steps said, kept for the arrival report. */
|
|
173
|
+
let lastSteps = [];
|
|
161
174
|
/** Readings that never got a frame newer than their own navigation. */
|
|
162
175
|
const staleReadings = [];
|
|
163
176
|
|
|
@@ -222,7 +235,37 @@ for (let round = 1; round <= rounds; round += 1) {
|
|
|
222
235
|
// written out first, because a failing run is the one whose evidence
|
|
223
236
|
// matters.
|
|
224
237
|
try {
|
|
225
|
-
|
|
238
|
+
// Keep what the tour's own steps reported. On an arrival failure the
|
|
239
|
+
// decisive question is whether the last wait PASSED and on what — and
|
|
240
|
+
// that answer was being thrown away, so three CI failures in a row
|
|
241
|
+
// could only be reasoned about rather than read. A local reproduction
|
|
242
|
+
// ran the same sequence 14 times and landed correctly every time, so
|
|
243
|
+
// the runner is the only place this can be observed.
|
|
244
|
+
const run = await actions.runScript(device, { steps: screen.steps, verify: false });
|
|
245
|
+
lastSteps = (run?.results ?? []).map((r) => ({
|
|
246
|
+
action: r.action,
|
|
247
|
+
ok: r.ok !== false,
|
|
248
|
+
ms: r.ms,
|
|
249
|
+
detail: typeof r.detail === 'string' ? r.detail.slice(0, 200) : null,
|
|
250
|
+
}));
|
|
251
|
+
// **`runScript` does not throw on a failed step** — it returns
|
|
252
|
+
// `ok: !failed` — and this only ever caught throws. So a tour whose
|
|
253
|
+
// `waitFor` timed out was treated as having arrived: `navigatedAt` was
|
|
254
|
+
// set, the previous screen was read, and the arrival check then reported
|
|
255
|
+
// it as a fingerprint result. Three CI failures were that, and the
|
|
256
|
+
// diagnostic added one commit earlier is what showed it —
|
|
257
|
+
// `ok launch … / FAIL waitFor` sitting under a reading the harness had
|
|
258
|
+
// accepted.
|
|
259
|
+
//
|
|
260
|
+
// A tour that did not arrive is a precondition, not a measurement. It
|
|
261
|
+
// throws into the same handler as a step that threw, so the report is
|
|
262
|
+
// identical and the readings so far are still written out.
|
|
263
|
+
if (run && run.ok === false) {
|
|
264
|
+
const bad = (run.results ?? []).find((r) => r.ok === false);
|
|
265
|
+
throw new Error(
|
|
266
|
+
`step ${bad?.action ?? '?'} did not succeed: ${bad?.error ?? 'no reason given'}`,
|
|
267
|
+
);
|
|
268
|
+
}
|
|
226
269
|
navigatedAt = Date.now();
|
|
227
270
|
} catch (err) {
|
|
228
271
|
save({ abandonedAt: { screen: screen.name, round, error: err.message } });
|
|
@@ -275,18 +318,38 @@ for (let round = 1; round <= rounds; round += 1) {
|
|
|
275
318
|
// stale — an eval that scores a stale frame is measuring the runner, which
|
|
276
319
|
// is the one thing this harness says it is not doing.
|
|
277
320
|
if (navigatedAt) {
|
|
321
|
+
// TWO conditions, and the second was missing.
|
|
322
|
+
//
|
|
323
|
+
// Newer than the navigation is not the same as recent. Measured on a
|
|
324
|
+
// runner: a reading came off a frame **38,266 ms old** and passed this
|
|
325
|
+
// guard, because the navigation had also been more than 38 s earlier — so
|
|
326
|
+
// `capturedAt >= navigatedAt` held while the frame was half a minute
|
|
327
|
+
// stale and showed a screen two steps back. `settled: true` alongside it,
|
|
328
|
+
// because a starved capture looks exactly like a still one.
|
|
329
|
+
//
|
|
330
|
+
// A frame this old is capture starvation, not a slow screen, and no
|
|
331
|
+
// number of re-reads invents a frame the daemon is not producing. So it
|
|
332
|
+
// retries and then says which of the two it is, because "the frame
|
|
333
|
+
// predates the navigation" and "capture stopped producing frames" point
|
|
334
|
+
// at completely different things.
|
|
335
|
+
const tooOld = (r) => Date.now() - (r.state?.capturedAt ?? 0) > FRAME_TOO_OLD_MS;
|
|
336
|
+
const predatesNav = (r) => (r.state?.capturedAt ?? 0) < navigatedAt;
|
|
278
337
|
for (let attempt = 0; attempt < STALE_READ_RETRIES; attempt += 1) {
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
338
|
+
if (!predatesNav(id) && !tooOld(id)) break;
|
|
339
|
+
const age = Date.now() - (id.state?.capturedAt ?? 0);
|
|
340
|
+
console.log(` ("${screen.name}" read a frame from ${age}ms ago`
|
|
341
|
+
+ `${predatesNav(id) ? ', older than the navigation' : `, over the ${FRAME_TOO_OLD_MS}ms freshness bound`}`
|
|
283
342
|
+ ` — reading again ${attempt + 1}/${STALE_READ_RETRIES})`);
|
|
284
343
|
await new Promise((r) => setTimeout(r, STALE_READ_WAIT_MS));
|
|
285
344
|
id = await api.screenIdentity(device, { fresh: true, confirmNovel: false });
|
|
286
345
|
}
|
|
287
|
-
if ((id
|
|
346
|
+
if (predatesNav(id)) {
|
|
288
347
|
staleReadings.push(`${screen.name} round ${round}: frame still predates the navigation`
|
|
289
348
|
+ ` by ${navigatedAt - (id.state?.capturedAt ?? 0)}ms after ${STALE_READ_RETRIES} re-reads`);
|
|
349
|
+
} else if (tooOld(id)) {
|
|
350
|
+
staleReadings.push(`${screen.name} round ${round}: newest frame is`
|
|
351
|
+
+ ` ${Date.now() - (id.state?.capturedAt ?? 0)}ms old after ${STALE_READ_RETRIES} re-reads`
|
|
352
|
+
+ ' — capture is starved, not the screen still');
|
|
290
353
|
}
|
|
291
354
|
}
|
|
292
355
|
// A reading too sparse to be a screen is not a reading.
|
|
@@ -343,6 +406,7 @@ for (let round = 1; round <= rounds; round += 1) {
|
|
|
343
406
|
// taken, and every distribution below it is then measuring the tour rather
|
|
344
407
|
// than the fingerprint. The first version of this harness did exactly that
|
|
345
408
|
// and reported that the distributions overlapped completely.
|
|
409
|
+
id.steps = lastSteps;
|
|
346
410
|
const previous = readings[readings.length - 1];
|
|
347
411
|
if (previous && previous.name !== screen.name) {
|
|
348
412
|
// Not just an identical hash. Two readings of the same screen can differ
|
|
@@ -367,10 +431,20 @@ for (let round = 1; round <= rounds; round += 1) {
|
|
|
367
431
|
`${previous.name} -> ${screen.name}: similarity ${s.toFixed(2)} — the screen did not change`
|
|
368
432
|
+ `\n sensors: ${(id.entry?.sources ?? []).join('+') || 'none'}`
|
|
369
433
|
+ `, settled: ${id.settled}, frame ${Math.round(Date.now() - (id.state?.capturedAt ?? Date.now()))}ms old`
|
|
370
|
-
+ `\n on screen: ${labels.length ? labels.join(' · ') : '(nothing readable)'}`
|
|
434
|
+
+ `\n on screen: ${labels.length ? labels.join(' · ') : '(nothing readable)'}`
|
|
435
|
+
// The decisive line, and it was missing. Whether the tour's last wait
|
|
436
|
+
// PASSED, and what it said, separates "the wait matched something it
|
|
437
|
+
// should not have" from "the screen changed after the wait was
|
|
438
|
+
// satisfied" — the two causes this message has always named as having
|
|
439
|
+
// opposite fixes, while giving the reader nothing to tell them apart.
|
|
440
|
+
+ `\n the tour's own steps: ${
|
|
441
|
+
(lastSteps ?? []).length
|
|
442
|
+
? lastSteps.map((st) => `${st.ok ? 'ok' : 'FAIL'} ${st.action}${st.detail ? ` — ${st.detail}` : ''}`).join('\n ')
|
|
443
|
+
: '(not recorded)'}`);
|
|
371
444
|
}
|
|
372
445
|
}
|
|
373
446
|
readings.push({
|
|
447
|
+
steps: lastSteps,
|
|
374
448
|
name: screen.name,
|
|
375
449
|
round,
|
|
376
450
|
hash: id.hash,
|
|
@@ -506,6 +580,17 @@ function findStrays(all) {
|
|
|
506
580
|
}
|
|
507
581
|
|
|
508
582
|
const strays = findStrays(readings);
|
|
583
|
+
// Readings grouped by screen, for the classification below.
|
|
584
|
+
//
|
|
585
|
+
// `findStrays` builds this too and it is LOCAL to that function — so reaching
|
|
586
|
+
// for it here threw `byName is not defined` on a runner, in the middle of the
|
|
587
|
+
// report, after the strays had already been printed. I had validated the
|
|
588
|
+
// classification against a saved artifact in a standalone script, where I had
|
|
589
|
+
// declared it myself, and never once in place. The logic was right and the
|
|
590
|
+
// integration was never run: the unit tests do not reach this file at all.
|
|
591
|
+
const readingsByName = new Map();
|
|
592
|
+
for (const r of readings) readingsByName.set(r.name, [...(readingsByName.get(r.name) ?? []), r]);
|
|
593
|
+
|
|
509
594
|
if (strays.length) {
|
|
510
595
|
console.error(`\nFAIL ${strays.length} reading(s) do not resemble their own screen:`);
|
|
511
596
|
// Two causes wear the same symptom, and until now the report asserted the
|
|
@@ -516,10 +601,25 @@ if (strays.length) {
|
|
|
516
601
|
// label while matching one particular other screen almost exactly, and a
|
|
517
602
|
// wrong turn is one whose tokens name a screen the tour did not ask for.
|
|
518
603
|
let collisions = 0;
|
|
604
|
+
// Kept so the SUMMARY agrees with the lines above it. It did not: the summary
|
|
605
|
+
// tested `sparseAt` directly while each line used the classifier, so a stray
|
|
606
|
+
// correctly labelled WRONG SCREEN was followed by "every stray above was
|
|
607
|
+
// UNDER-READ". A report that contradicts itself in consecutive paragraphs is
|
|
608
|
+
// worse than either sentence alone.
|
|
609
|
+
const verdicts = [];
|
|
519
610
|
for (const { reading, bestSelf, bestOther, match } of strays) {
|
|
520
611
|
const named = namedTokens(reading.tokens);
|
|
521
612
|
const matchNamed = match ? namedTokens(match.tokens) : [];
|
|
522
|
-
|
|
613
|
+
// One classifier, one source of truth. `collided` was computed here as well
|
|
614
|
+
// and the two could drift — which is how this block came to have a third
|
|
615
|
+
// cause landing in the wrong bucket in the first place.
|
|
616
|
+
const verdict = classifyStray({
|
|
617
|
+
reading, match, bestOther, named, matchNamed,
|
|
618
|
+
siblings: readingsByName.get(reading.name) ?? [],
|
|
619
|
+
wasSparse: sparseAt.has(`${reading.name}|${reading.round}`),
|
|
620
|
+
});
|
|
621
|
+
const { collided, underRead, wrongScreen } = verdict;
|
|
622
|
+
verdicts.push(verdict);
|
|
523
623
|
if (collided) collisions += 1;
|
|
524
624
|
// The third cause, and the one that produced this report on 2026-09-15.
|
|
525
625
|
//
|
|
@@ -534,7 +634,21 @@ if (strays.length) {
|
|
|
534
634
|
// nothing to work with, which is this harness's subject. This means we did
|
|
535
635
|
// not look long enough, which is the harness's own fault and a different
|
|
536
636
|
// remedy.
|
|
537
|
-
|
|
637
|
+
//
|
|
638
|
+
// **And sparseness alone is not enough to claim it**, which this
|
|
639
|
+
// misdiagnosed once. A wrong turn that lands on a screen which legitimately
|
|
640
|
+
// reads sparse gets flagged sparse too, and was then reported as our
|
|
641
|
+
// instrument's fault rather than the tour's. Measured: `settings-general`
|
|
642
|
+
// r3 came back with tokens byte-identical to the Settings root, including
|
|
643
|
+
// `heading:nav-bar:"settings"` — a *heading*, which is the root's title,
|
|
644
|
+
// where General publishes a *button* with the same word. It was read on the
|
|
645
|
+
// wrong screen, and this called it an under-read.
|
|
646
|
+
//
|
|
647
|
+
// The discriminator is the screen's own other rounds: if `settings-general`
|
|
648
|
+
// read differently and richly in rounds 1 and 2, then the screen IS
|
|
649
|
+
// distinguishable and a round matching another screen exactly went
|
|
650
|
+
// somewhere else. Only when no round of this screen can tell itself apart
|
|
651
|
+
// is "we did not look long enough" the honest reading.
|
|
538
652
|
console.error(` ${reading.name} r${reading.round}: own screen ${bestSelf.toFixed(2)}, `
|
|
539
653
|
+ `${match ? `${match.name} r${match.round}` : 'another screen'} ${bestOther.toFixed(2)} `
|
|
540
654
|
+ `(${reading.count} tokens, ${named.length} named, sources ${reading.sources.join('+') || 'none'})`);
|
|
@@ -542,6 +656,11 @@ if (strays.length) {
|
|
|
542
656
|
console.error(' ^ a COLLISION, not a wrong turn: neither reading carries a chrome');
|
|
543
657
|
console.error(' label, so both are structure with no name and the fingerprint has');
|
|
544
658
|
console.error(' nothing left to tell two list screens apart.');
|
|
659
|
+
} else if (wrongScreen) {
|
|
660
|
+
console.error(` ^ WRONG SCREEN: these tokens are ${match ? `identical to ${match.name}` : 'another screen'},`);
|
|
661
|
+
console.error(' and this screen reads differently in its other rounds — so it is');
|
|
662
|
+
console.error(' distinguishable and the tour was simply somewhere else. A tap that');
|
|
663
|
+
console.error(' missed, or a screen that went back before it was read.');
|
|
545
664
|
} else if (underRead) {
|
|
546
665
|
console.error(' ^ UNDER-READ, not a wrong turn: this reading stayed below the token');
|
|
547
666
|
console.error(' floor after every retry, so it never saw enough of its screen to');
|
|
@@ -553,10 +672,16 @@ if (strays.length) {
|
|
|
553
672
|
console.error(`\n${collisions} of ${strays.length} are fingerprint collisions. That is this harness's own subject,`);
|
|
554
673
|
console.error('not a tour fault: a reading whose chrome label went missing cannot establish');
|
|
555
674
|
console.error('identity, and comparing it as though it could is what produced the verdict above.');
|
|
556
|
-
} else if (
|
|
675
|
+
} else if (verdicts.length && verdicts.every((v) => v.underRead)) {
|
|
557
676
|
console.error('\nEvery stray above was UNDER-READ, so this says nothing about the tour or the');
|
|
558
677
|
console.error('fingerprint — the harness scored a look that was too short. The retries are in');
|
|
559
678
|
console.error('SPARSE_READ_RETRIES; a runner that needs more than they allow is the finding.');
|
|
679
|
+
} else if (verdicts.some((v) => v.wrongScreen)) {
|
|
680
|
+
const n = verdicts.filter((v) => v.wrongScreen).length;
|
|
681
|
+
console.error(`\n${n} stray(s) were read on the WRONG SCREEN — the tokens match another screen`);
|
|
682
|
+
console.error('exactly while this one reads differently in its other rounds, so it is');
|
|
683
|
+
console.error('distinguishable and the tour was somewhere else. Not a fingerprint result:');
|
|
684
|
+
console.error('a tap that missed, or a screen that went back before it was read.');
|
|
560
685
|
} else {
|
|
561
686
|
console.error('\nThat is the tour going somewhere unintended, not the fingerprint drifting, and');
|
|
562
687
|
console.error('measuring it as either distribution poisons both ends. Fix the tour — a tap that');
|
package/src/actions.js
CHANGED
|
@@ -90,16 +90,49 @@ const ACTION_STEPS = new Set([
|
|
|
90
90
|
'launch', 'terminate', 'openUrl', 'confirm', 'chooseAny', 'permission',
|
|
91
91
|
]);
|
|
92
92
|
|
|
93
|
+
/**
|
|
94
|
+
* Step keys that are really the MCP tool names, accepted as aliases.
|
|
95
|
+
*
|
|
96
|
+
* Two field reports guessed these independently and each wrong guess cost a
|
|
97
|
+
* round trip and aborted the rest of the batch. There is a `sim_wait` tool and
|
|
98
|
+
* a `sim_type_into` tool, so `wait` and `type_into` are what a caller reaches
|
|
99
|
+
* for inside `sim_do` — and the vocabulary answered "unknown step" without
|
|
100
|
+
* saying what the words are. A tool surface that names an action one way and
|
|
101
|
+
* accepts it another is charging the caller for our inconsistency.
|
|
102
|
+
*/
|
|
103
|
+
const STEP_ALIASES = Object.freeze({
|
|
104
|
+
wait: 'settle',
|
|
105
|
+
type_into: 'type',
|
|
106
|
+
typeInto: 'type',
|
|
107
|
+
scroll_to: 'scrollTo',
|
|
108
|
+
wait_for: 'waitFor',
|
|
109
|
+
waitfor: 'waitFor',
|
|
110
|
+
tap_at: 'tapAt',
|
|
111
|
+
open_url: 'openUrl',
|
|
112
|
+
assert_gone: 'assertGone',
|
|
113
|
+
assert_text: 'assertText',
|
|
114
|
+
});
|
|
115
|
+
|
|
116
|
+
/** Every step key `runStep` understands, for an error that can be acted on. */
|
|
117
|
+
export const STEP_KEYS = Object.freeze([
|
|
118
|
+
'tap', 'tapAt', 'type', 'paste', 'clear', 'swipe', 'scroll', 'scrollTo',
|
|
119
|
+
'button', 'key', 'launch', 'terminate', 'openUrl', 'confirm', 'chooseAny',
|
|
120
|
+
'permission', 'settle', 'waitFor', 'waitText', 'assert', 'assertGone',
|
|
121
|
+
'assertText', 'visible', 'gone', 'enabled', 'disabled', 'value', 'look',
|
|
122
|
+
'seek', 'sweep', 'pause',
|
|
123
|
+
]);
|
|
124
|
+
|
|
93
125
|
/** Accept both `{tap: "Save"}` shorthand and `{action: "tap", target: "Save"}`. */
|
|
94
126
|
export function normalizeStep(raw) {
|
|
95
|
-
|
|
96
|
-
if (raw
|
|
127
|
+
const canonical = (a) => STEP_ALIASES[a] ?? a;
|
|
128
|
+
if (typeof raw === 'string') return { action: canonical(raw) };
|
|
129
|
+
if (raw.action) return { ...raw, action: canonical(raw.action) };
|
|
97
130
|
const [key] = Object.keys(raw);
|
|
98
131
|
if (!key) throw new Error('empty step');
|
|
99
132
|
// Siblings like timeoutMs sit alongside the shorthand key and must survive.
|
|
100
133
|
const { [key]: value, ...rest } = raw;
|
|
101
134
|
const inline = value && typeof value === 'object' && !Array.isArray(value) ? value : { value };
|
|
102
|
-
const step = { ...rest, ...inline, action: key };
|
|
135
|
+
const step = { ...rest, ...inline, action: canonical(key) };
|
|
103
136
|
// Drop keys that are present but undefined. `simframe tap X` used to pass
|
|
104
137
|
// `index: undefined`, which survived here and then crashed the signature
|
|
105
138
|
// builder before the tap was ever sent.
|
|
@@ -699,7 +732,10 @@ export async function runScript(
|
|
|
699
732
|
// eight times below its threshold, so it reads as no-visible-change and
|
|
700
733
|
// its edge is no longer recorded. That is the right trade while the
|
|
701
734
|
// detector cannot see it, and it comes back on its own once it can.
|
|
702
|
-
|
|
735
|
+
// A changed control IS evidence, whatever the frame hash says. Without
|
|
736
|
+
// this the edge is not recorded either, so a toggle that worked teaches
|
|
737
|
+
// the graph nothing and stays unpredictable forever.
|
|
738
|
+
const noEvidence = Boolean(settled?.noVisibleChange) && !stateDelta(beforeScreen?.entry, afterReading?.entry);
|
|
703
739
|
if (afterScreen.confirmed && afterScreen.hash && !noEvidence) {
|
|
704
740
|
// The observed cost of this transition, which is what makes the next
|
|
705
741
|
// one adaptive. Only from a settle that was actually satisfied: a
|
|
@@ -755,7 +791,22 @@ export async function runScript(
|
|
|
755
791
|
// first — and six swipes reporting "no visible change" is what item 120
|
|
756
792
|
// was reported against. Keying on one of them would have shipped a fix
|
|
757
793
|
// that did not fire on its own bug report.
|
|
758
|
-
|
|
794
|
+
// Declared BEFORE `wentNowhere`, which reads it. It sat below and the
|
|
795
|
+
// unit tests were green because nothing without a device reaches this
|
|
796
|
+
// line — a `const` in the temporal dead zone throws on first use, which
|
|
797
|
+
// is the same trap the `afterReading` comment above records.
|
|
798
|
+
//
|
|
799
|
+
// Computed on EITHER signal, because they are two different sensors and
|
|
800
|
+
// the reported case came through the one this first version missed: the
|
|
801
|
+
// radio move settled in 2714ms with `sawChange` true, so
|
|
802
|
+
// `settled.noVisibleChange` was false and the verdict came from the
|
|
803
|
+
// fingerprint instead.
|
|
804
|
+
const changedState = (settled?.noVisibleChange || verification?.verdict === 'no-visible-change')
|
|
805
|
+
? stateDelta(beforeScreen?.entry, afterReading?.entry)
|
|
806
|
+
: null;
|
|
807
|
+
if (changedState) verification = stateChanged(verification, changedState);
|
|
808
|
+
const wentNowhere = !changedState
|
|
809
|
+
&& (Boolean(settled?.noVisibleChange) || verification?.verdict === 'no-visible-change');
|
|
759
810
|
const aimNote = wentNowhere && aim.at
|
|
760
811
|
? screenmap.describePoint(
|
|
761
812
|
// `beforeScreen` is null with verification off, and that is exactly
|
|
@@ -769,7 +820,9 @@ export async function runScript(
|
|
|
769
820
|
const note = launchNote
|
|
770
821
|
+ (filling ? ` [settled, but ${filling} — waitFor content, do not act on this yet]` : '')
|
|
771
822
|
+ (settled?.smallChange ? ' [a small change, in one region only]' : '')
|
|
772
|
-
+ (
|
|
823
|
+
+ (changedState
|
|
824
|
+
? ` [the screen did not move, but ${changedState.detail} — this worked, do not retry]`
|
|
825
|
+
: (settled?.noVisibleChange ? ' [no visible change]' : ''))
|
|
773
826
|
// After the symptom, because it is the explanation of it.
|
|
774
827
|
+ (aimNote ? ` [${aimNote}]` : '')
|
|
775
828
|
+ (settled?.staleBaseline ? ' [baseline had already settled; re-taken from the live screen]' : '')
|
|
@@ -1561,6 +1614,37 @@ async function confirmNoChange(deviceQuery, verification, { beforeScreen, option
|
|
|
1561
1614
|
};
|
|
1562
1615
|
}
|
|
1563
1616
|
|
|
1617
|
+
/**
|
|
1618
|
+
* A control that changed state is not a step that did nothing.
|
|
1619
|
+
*
|
|
1620
|
+
* The sibling of `belowThreshold` on the same verdict, for the other kind of
|
|
1621
|
+
* evidence. `no-visible-change` arrives by two routes — the pixel detector and
|
|
1622
|
+
* the fingerprint agreeing we are on the same screen — and a switch or a radio
|
|
1623
|
+
* satisfies both while having plainly worked. Measured on a device: a radio
|
|
1624
|
+
* moving from "Top" to "Inline" reported `(settled 2714ms)
|
|
1625
|
+
* [no-visible-change]` with `selected` moving between the two rows in the
|
|
1626
|
+
* element list either side of it.
|
|
1627
|
+
*
|
|
1628
|
+
* This is the verdict an agent reads to decide whether to retry, and after item
|
|
1629
|
+
* 148 made the tap actually land, a retry undoes the thing that worked. So the
|
|
1630
|
+
* verdict carries the evidence rather than contradicting it.
|
|
1631
|
+
*
|
|
1632
|
+
* Still not `ok`: the screen genuinely did not become a different screen, and
|
|
1633
|
+
* claiming a transition that did not happen would be the opposite error. It
|
|
1634
|
+
* becomes `unverified` — the verdict this project uses for "something happened
|
|
1635
|
+
* and nobody can vouch for what" — with the change named.
|
|
1636
|
+
*/
|
|
1637
|
+
export function stateChanged(verification, delta) {
|
|
1638
|
+
if (verification?.verdict !== 'no-visible-change' || !delta) return verification;
|
|
1639
|
+
return {
|
|
1640
|
+
...verification,
|
|
1641
|
+
verdict: 'unverified',
|
|
1642
|
+
detail: `the screen did not change, but ${delta.detail}`
|
|
1643
|
+
+ ' — the action worked; do not retry it, or you will undo it',
|
|
1644
|
+
stateDelta: delta,
|
|
1645
|
+
};
|
|
1646
|
+
}
|
|
1647
|
+
|
|
1564
1648
|
export function belowThreshold(verification, settled) {
|
|
1565
1649
|
if (verification?.verdict !== 'no-visible-change' || !settled?.smallChange) return verification;
|
|
1566
1650
|
return {
|
|
@@ -1608,6 +1692,37 @@ async function focusHint(deviceQuery, target, ctx) {
|
|
|
1608
1692
|
}
|
|
1609
1693
|
}
|
|
1610
1694
|
|
|
1695
|
+
/**
|
|
1696
|
+
* Is anything focused? Asked before typing blind.
|
|
1697
|
+
*
|
|
1698
|
+
* `{"type": "text"}` with no `into` types into whatever has focus, and reported
|
|
1699
|
+
* `ok … [unconfirmed — no field named, so nothing was read back]`. The note was
|
|
1700
|
+
* honest and it was attached to a **green step**, so a chained `sim_do`
|
|
1701
|
+
* continued on a premise that had never been checked — a field report lost text
|
|
1702
|
+
* that went nowhere exactly this way.
|
|
1703
|
+
*
|
|
1704
|
+
* If nothing has focus, the keystrokes are lost and there is no version of that
|
|
1705
|
+
* which is an `ok`. The tree already answers it: one accessibility read, no
|
|
1706
|
+
* OCR, which is the same cost `focusHint` pays on the `into` path.
|
|
1707
|
+
*
|
|
1708
|
+
* Advisory in one direction only. A tree that cannot answer — no targets, a
|
|
1709
|
+
* read that threw — returns `unknown`, and the step proceeds with its
|
|
1710
|
+
* unconfirmed note rather than failing on our inability to look. Refusing to
|
|
1711
|
+
* type because we could not read the screen would be a worse trade than the
|
|
1712
|
+
* bug.
|
|
1713
|
+
*/
|
|
1714
|
+
async function focusedSomewhere(deviceQuery, ctx) {
|
|
1715
|
+
try {
|
|
1716
|
+
const { entry } = await api.readScreenWith(deviceQuery, { useOcr: false, options: ctx.options });
|
|
1717
|
+
const targets = entry?.targets ?? [];
|
|
1718
|
+
if (!targets.length) return { known: false };
|
|
1719
|
+
const hit = targets.find((t) => t.focused === true);
|
|
1720
|
+
return { known: true, focused: Boolean(hit), where: hit?.label ?? hit?.identifier ?? null };
|
|
1721
|
+
} catch {
|
|
1722
|
+
return { known: false };
|
|
1723
|
+
}
|
|
1724
|
+
}
|
|
1725
|
+
|
|
1611
1726
|
/**
|
|
1612
1727
|
* How long to wait for the tap to become focus before inserting text.
|
|
1613
1728
|
*
|
|
@@ -2286,11 +2401,29 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
2286
2401
|
journalWrite(udid, step, sent, back, ctx);
|
|
2287
2402
|
return `typed into ${field.where}${back.note}${back.landed ? '' : field.quiet}`;
|
|
2288
2403
|
}
|
|
2404
|
+
// **Refusing here was tried and reverted — see DEFERRED 161.** The idea
|
|
2405
|
+
// was that "nothing focused" is knowable before typing, so typing into
|
|
2406
|
+
// nothing should fail rather than report `ok`. On a device it produced a
|
|
2407
|
+
// FALSE REFUSAL on the legitimate case: after tapping Settings' own
|
|
2408
|
+
// search field, every element in the tree read `focused: false`. iOS does
|
|
2409
|
+
// not reliably publish AXFocused here, which `focusHint` above already
|
|
2410
|
+
// says in its own comment — *"with a hardware keyboard attached to the
|
|
2411
|
+
// simulator no software keyboard appears, so there is often nothing to
|
|
2412
|
+
// see"*. A gate on an unpublished signal blocks real work, which is worse
|
|
2413
|
+
// than the verdict it was fixing.
|
|
2414
|
+
//
|
|
2415
|
+
// So the focus reading stays advisory and goes into the note. What the
|
|
2416
|
+
// field report actually objected to was an honest caveat attached to a
|
|
2417
|
+
// GREEN step, and the note is where that has to be fixed until there is a
|
|
2418
|
+
// signal worth gating on.
|
|
2419
|
+
const focus = await focusedSomewhere(deviceQuery, ctx);
|
|
2289
2420
|
await input.typeText(udid, step.text ?? step.value);
|
|
2290
|
-
|
|
2291
|
-
|
|
2292
|
-
|
|
2293
|
-
|
|
2421
|
+
return 'typed text [NOT CONFIRMED: no field named, so nothing was read back'
|
|
2422
|
+
+ (focus.known && !focus.focused
|
|
2423
|
+
? ' and nothing on this screen reports keyboard focus — the text may have gone nowhere'
|
|
2424
|
+
: '')
|
|
2425
|
+
+ '. Name the field — {"type": {"into": "<field>", "text": "..."}} — to have it tapped'
|
|
2426
|
+
+ ' first and read back]';
|
|
2294
2427
|
}
|
|
2295
2428
|
case 'paste': {
|
|
2296
2429
|
// Long strings are much faster on the pasteboard than through the
|
|
@@ -2486,11 +2619,47 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
2486
2619
|
return null;
|
|
2487
2620
|
}
|
|
2488
2621
|
};
|
|
2622
|
+
/**
|
|
2623
|
+
* Is this target actually on the screen?
|
|
2624
|
+
*
|
|
2625
|
+
* Resolving is not the same as being in view, and this returned `"X" is
|
|
2626
|
+
* in view at 201,-35 after 1 scroll down` — 35pt ABOVE the viewport, so
|
|
2627
|
+
* the tap that followed missed. Reported from the field. The tree carries
|
|
2628
|
+
* scrolled-away rows with out-of-bounds coordinates, which is exactly the
|
|
2629
|
+
* case `scrollTo` exists to resolve, so claiming success on one is the
|
|
2630
|
+
* one answer it must never give.
|
|
2631
|
+
*
|
|
2632
|
+
* Generous rather than strict: any overlap with the viewport counts, so a
|
|
2633
|
+
* row half over an edge is still reachable and still reported. Only a
|
|
2634
|
+
* target entirely outside keeps the search going.
|
|
2635
|
+
*/
|
|
2636
|
+
const inViewport = (t) => {
|
|
2637
|
+
const w = points?.width;
|
|
2638
|
+
const h = points?.height;
|
|
2639
|
+
if (!Number.isFinite(w) || !Number.isFinite(h)) return true;
|
|
2640
|
+
const f = t?.frame;
|
|
2641
|
+
if (f && Number.isFinite(f.y) && Number.isFinite(f.height)) {
|
|
2642
|
+
return f.y + f.height > 0 && f.y < h && f.x + (f.width ?? 0) > 0 && f.x < w;
|
|
2643
|
+
}
|
|
2644
|
+
return Number.isFinite(t?.y) && t.y >= 0 && t.y <= h
|
|
2645
|
+
&& Number.isFinite(t?.x) && t.x >= 0 && t.x <= w;
|
|
2646
|
+
};
|
|
2647
|
+
// The direction actually scrolled, which is not always the one asked for:
|
|
2648
|
+
// `offsetSays()` may reverse it, and the message used to name the request
|
|
2649
|
+
// rather than the act — reported as "after 1 scroll down" on a request
|
|
2650
|
+
// for "up".
|
|
2651
|
+
const scrolled = [];
|
|
2489
2652
|
for (let i = 0; i <= max; i += 1) {
|
|
2490
2653
|
try {
|
|
2491
2654
|
const found = await api.locate(deviceQuery, query, { index: step.index, refresh: i > 0 });
|
|
2492
|
-
|
|
2493
|
-
|
|
2655
|
+
if (!inViewport(found.target)) throw new Error(
|
|
2656
|
+
`"${query}" is in the tree but not in view (at ${found.target.x},${found.target.y}`
|
|
2657
|
+
+ ` on a ${Math.round(points?.width ?? 0)}x${Math.round(points?.height ?? 0)}pt screen)`);
|
|
2658
|
+
const how = scrolled.length
|
|
2659
|
+
? ` after ${scrolled.length} scroll${scrolled.length === 1 ? '' : 's'} `
|
|
2660
|
+
+ (new Set(scrolled).size === 1 ? scrolled[0] : scrolled.join(' then '))
|
|
2661
|
+
: ' already';
|
|
2662
|
+
return `"${found.target.label ?? query}" is in view at ${found.target.x},${found.target.y}${how}`;
|
|
2494
2663
|
} catch (err) {
|
|
2495
2664
|
if (i === max) {
|
|
2496
2665
|
throw new Error(`scrolled ${dir} ${max}x without finding ${query}: ${err.message}`);
|
|
@@ -2499,6 +2668,7 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
2499
2668
|
const evidence = await offsetSays();
|
|
2500
2669
|
if (evidence) dir = evidence;
|
|
2501
2670
|
const wasAt = await hashNow(deviceQuery, ctx.options);
|
|
2671
|
+
scrolled.push(dir);
|
|
2502
2672
|
await runStep(deviceQuery, udid, { action: 'scroll', value: dir }, ctx);
|
|
2503
2673
|
// A scroll either moves immediately or not at all, so it does not need a
|
|
2504
2674
|
// transition's budget. Six iterations at 2,500ms was most of why this
|
|
@@ -2763,11 +2933,83 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
2763
2933
|
await sleep(ms);
|
|
2764
2934
|
return `paused ${ms}ms`;
|
|
2765
2935
|
}
|
|
2766
|
-
default:
|
|
2767
|
-
|
|
2936
|
+
default: {
|
|
2937
|
+
// Say what the words are. "unknown step" on its own sent two testers
|
|
2938
|
+
// guessing, and a guess costs a round trip and the rest of the batch.
|
|
2939
|
+
const near = STEP_KEYS.filter((k) => {
|
|
2940
|
+
const a = String(step.action ?? '').toLowerCase().replace(/[_-]/g, '');
|
|
2941
|
+
const b = k.toLowerCase();
|
|
2942
|
+
return a && (b.startsWith(a) || a.startsWith(b) || b.includes(a));
|
|
2943
|
+
}).slice(0, 3);
|
|
2944
|
+
throw new Error(
|
|
2945
|
+
`unknown step "${step.action}"`
|
|
2946
|
+
+ (near.length ? ` — did you mean ${near.map((k) => `"${k}"`).join(' or ')}?` : '')
|
|
2947
|
+
+ ` Valid steps: ${STEP_KEYS.join(', ')}.`,
|
|
2948
|
+
);
|
|
2949
|
+
}
|
|
2768
2950
|
}
|
|
2769
2951
|
}
|
|
2770
2952
|
|
|
2953
|
+
/**
|
|
2954
|
+
* What changed in the controls, when the pixels say nothing did.
|
|
2955
|
+
*
|
|
2956
|
+
* `settle()` decides `noVisibleChange` from frame hashes alone, and a control
|
|
2957
|
+
* that flips is eight times below the change threshold (DEFERRED 4, measured on
|
|
2958
|
+
* a switch: 0.00049 peak against 0.004). So a switch that genuinely flipped is
|
|
2959
|
+
* reported as an action that did nothing — with the changed value printed in the
|
|
2960
|
+
* same response.
|
|
2961
|
+
*
|
|
2962
|
+
* **That verdict became dangerous the moment item 148 made the tap land.** While
|
|
2963
|
+
* the tap missed the frame centre, `no-visible-change` was true and harmless.
|
|
2964
|
+
* Now the tap actuates, and `no-visible-change` is what an agent reads to decide
|
|
2965
|
+
* whether to retry — so a retry silently un-flips the toggle. An external
|
|
2966
|
+
* tester made exactly that point about my own commit message, which had claimed
|
|
2967
|
+
* the verdict "was telling the truth".
|
|
2968
|
+
*
|
|
2969
|
+
* This is wiring, not calibration. The element values are already in hand on
|
|
2970
|
+
* both sides of the action — `beforeScreen.entry` and the after-reading are read
|
|
2971
|
+
* for verification anyway — so the comparison costs nothing on the device. A
|
|
2972
|
+
* control whose value or selected state changed did something, whatever the
|
|
2973
|
+
* frame hash says.
|
|
2974
|
+
*
|
|
2975
|
+
* Identity across the two readings goes by identifier, then label, then
|
|
2976
|
+
* type-and-place: a row that moved is not the same reading of the same control,
|
|
2977
|
+
* and pretending otherwise would invent changes.
|
|
2978
|
+
*/
|
|
2979
|
+
export function stateDelta(beforeEntry, afterEntry) {
|
|
2980
|
+
const keyOf = (t) => t.identifier || t.label || `${t.type ?? '?'}@${Math.round(t.x ?? 0)},${Math.round(t.y ?? 0)}`;
|
|
2981
|
+
const stateOf = (t) => `${t.value ?? ''}|${t.selected ?? ''}`;
|
|
2982
|
+
const before = new Map();
|
|
2983
|
+
for (const t of beforeEntry?.targets ?? []) {
|
|
2984
|
+
if (t.value == null && t.selected == null) continue;
|
|
2985
|
+
before.set(keyOf(t), { state: stateOf(t), target: t });
|
|
2986
|
+
}
|
|
2987
|
+
if (!before.size) return null;
|
|
2988
|
+
const changes = [];
|
|
2989
|
+
for (const t of afterEntry?.targets ?? []) {
|
|
2990
|
+
if (t.value == null && t.selected == null) continue;
|
|
2991
|
+
const was = before.get(keyOf(t));
|
|
2992
|
+
if (!was || was.state === stateOf(t)) continue;
|
|
2993
|
+
changes.push({
|
|
2994
|
+
name: t.label || t.identifier || keyOf(t),
|
|
2995
|
+
from: was.target.value ?? was.target.selected,
|
|
2996
|
+
to: t.value ?? t.selected,
|
|
2997
|
+
});
|
|
2998
|
+
}
|
|
2999
|
+
if (!changes.length) return null;
|
|
3000
|
+
// Name the control that turned ON, not the one that turned off. A radio move
|
|
3001
|
+
// changes two rows and reporting "Inline changed from true to false" is
|
|
3002
|
+
// accurate and reads like a loss — the useful half is what is selected now.
|
|
3003
|
+
const on = (v) => v === true || v === 'true' || v === '1';
|
|
3004
|
+
const first = changes.find((c) => on(c.to) && !on(c.from)) ?? changes[0];
|
|
3005
|
+
return {
|
|
3006
|
+
count: changes.length,
|
|
3007
|
+
detail: `${JSON.stringify(String(first.name).slice(0, 32))} changed from `
|
|
3008
|
+
+ `${JSON.stringify(String(first.from))} to ${JSON.stringify(String(first.to))}`
|
|
3009
|
+
+ (changes.length > 1 ? ` (and ${changes.length - 1} other control(s))` : ''),
|
|
3010
|
+
};
|
|
3011
|
+
}
|
|
3012
|
+
|
|
2771
3013
|
/**
|
|
2772
3014
|
* Why a settle gave up, in the numbers that decided it.
|
|
2773
3015
|
*
|
package/src/cli.js
CHANGED
|
@@ -34,7 +34,7 @@ const USAGE = `simframe — always-warm iOS Simulator frames
|
|
|
34
34
|
simframe tap <selector> tap #3, "Save", or @120,400
|
|
35
35
|
simframe do <script.json> run a scripted flow (see below)
|
|
36
36
|
simframe screens [device] list screens this device has learned
|
|
37
|
-
simframe storage [bundle-id]
|
|
37
|
+
simframe storage [bundle-id] [--device=<name|udid>] what the app saved (no boot needed)
|
|
38
38
|
simframe goto <screen> walk to a known screen through known steps
|
|
39
39
|
simframe flow save <name> <script.json> run a flow and save it if every step verifies
|
|
40
40
|
simframe flow run <name> replay a saved flow
|
|
@@ -867,7 +867,44 @@ async function main() {
|
|
|
867
867
|
// whole value of this command is that it answers on a device that is not
|
|
868
868
|
// running — measured: simctl itself cannot, on this Xcode. Routing it
|
|
869
869
|
// through the daemon would throw that away for no gain.
|
|
870
|
-
|
|
870
|
+
// Resolving a device here must NOT require one to be running, and it did.
|
|
871
|
+
//
|
|
872
|
+
// The whole claim of this command is that it answers before a boot, and
|
|
873
|
+
// `resolveDevice(undefined)` resolves against *booted* devices — so the
|
|
874
|
+
// documented form, `simframe storage <bundle-id>`, failed with "no booted
|
|
875
|
+
// simulator" at exactly the feature's headline use case. Reported by an
|
|
876
|
+
// external tester, who found it worked only when a device was named,
|
|
877
|
+
// which the help line did not mention either.
|
|
878
|
+
//
|
|
879
|
+
// Named: resolve as usual, booted or not. Unnamed: prefer a booted device
|
|
880
|
+
// when there is one, otherwise the only device that has app containers —
|
|
881
|
+
// and when that is ambiguous, say so and name the candidates rather than
|
|
882
|
+
// complaining about a boot nobody needs.
|
|
883
|
+
const named = flags.device ?? flags.udid ?? process.env.SIMFRAME_DEVICE;
|
|
884
|
+
let device;
|
|
885
|
+
if (named) {
|
|
886
|
+
device = await resolveDevice(String(named));
|
|
887
|
+
} else {
|
|
888
|
+
device = await resolveDevice(undefined).catch(() => null);
|
|
889
|
+
if (!device) {
|
|
890
|
+
const all = await listDevices({ all: true });
|
|
891
|
+
const withApps = [];
|
|
892
|
+
for (const d of all) {
|
|
893
|
+
const apps = await storage.apps(d.udid).catch(() => []);
|
|
894
|
+
if (apps.length) withApps.push({ device: d, apps: apps.length });
|
|
895
|
+
}
|
|
896
|
+
if (withApps.length === 1) {
|
|
897
|
+
device = withApps[0].device;
|
|
898
|
+
} else {
|
|
899
|
+
throw new Error(
|
|
900
|
+
withApps.length
|
|
901
|
+
? 'storage reads a device that does not have to be running, so name which one:\n'
|
|
902
|
+
+ withApps.map((w) => ` --device=${w.device.udid} ${w.device.name} (${w.apps} app(s))`).join('\n')
|
|
903
|
+
: 'no device on this host has any app data to read',
|
|
904
|
+
);
|
|
905
|
+
}
|
|
906
|
+
}
|
|
907
|
+
}
|
|
871
908
|
const [bundleId] = positional;
|
|
872
909
|
if (!bundleId) {
|
|
873
910
|
const list = await storage.apps(device.udid);
|