simframe 0.9.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +165 -5
- package/data/vocabulary/en.json +148 -0
- package/native/ocr.swift +13 -1
- package/native/rank.swift +87 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +43 -3
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
- package/native/simframed/Sources/simframed/main.swift +13 -1
- package/native/supervise.swift +181 -0
- package/package.json +4 -1
- package/scripts/check-package.mjs +22 -2
- package/scripts/check-private.mjs +143 -0
- package/scripts/ci-memory.mjs +104 -20
- package/scripts/eval-perception.mjs +281 -0
- package/scripts/phase17-corpus.mjs +176 -0
- package/skills/simframe/SKILL.md +237 -5
- package/src/actions.js +1825 -38
- package/src/analyze.js +70 -0
- package/src/cli.js +214 -15
- package/src/control.js +1 -0
- package/src/fingerprint.js +19 -1
- package/src/graph.js +193 -11
- package/src/index.js +428 -16
- package/src/input.js +115 -8
- package/src/localhelper.js +155 -0
- package/src/matching.js +119 -3
- package/src/mcp.js +319 -27
- package/src/metrics.js +134 -8
- package/src/navigate.js +10 -7
- package/src/ocr.js +18 -1
- package/src/planner.js +195 -0
- package/src/platform/android.js +3 -2
- package/src/platform/ios.js +2 -1
- package/src/png.js +26 -0
- package/src/refs.js +51 -8
- package/src/regions.js +110 -1
- package/src/screenmap.js +109 -10
- package/src/supervisor.js +117 -0
- package/src/view.js +396 -7
- package/src/vocabulary.js +134 -0
- package/src/wrote.js +136 -0
package/src/graph.js
CHANGED
|
@@ -295,6 +295,38 @@ function replayable(step) {
|
|
|
295
295
|
return step.launch != null || step.openUrl != null ? null : step;
|
|
296
296
|
}
|
|
297
297
|
|
|
298
|
+
/**
|
|
299
|
+
* What has worked from this screen before, in the caller's own vocabulary.
|
|
300
|
+
*
|
|
301
|
+
* The graph has always known this and never said it. A map reported
|
|
302
|
+
* `(known, 3 known exits)` — the *count* — so an agent on a screen simframe had
|
|
303
|
+
* driven successfully six times still had to read it to learn what was tappable.
|
|
304
|
+
* Measured across two peer rounds: a flow whose steps were known in advance ran
|
|
305
|
+
* 16 steps in **one** call, and the same agent on a screen the graph also knew
|
|
306
|
+
* but whose labels it did not spent 25 calls on 31 steps. The difference was not
|
|
307
|
+
* perception. It was whether a plan existed before execution started.
|
|
308
|
+
*
|
|
309
|
+
* Ordered by how often each has worked, because that is the order an agent
|
|
310
|
+
* should try them in.
|
|
311
|
+
*/
|
|
312
|
+
export function exitsOf(node, { limit = 8 } = {}) {
|
|
313
|
+
return (node?.edges ?? [])
|
|
314
|
+
.map((e) => ({
|
|
315
|
+
action: e.step?.action ?? (e.action ?? '').split(':')[0] ?? 'tap',
|
|
316
|
+
label: e.step?.value ?? e.step?.target ?? e.step?.label ?? e.step?.into ?? null,
|
|
317
|
+
to: e.to ?? null,
|
|
318
|
+
count: e.count ?? 0,
|
|
319
|
+
kind: e.kind ?? null,
|
|
320
|
+
}))
|
|
321
|
+
// A `#13` was a ref on the screen it was typed on and means nothing on the
|
|
322
|
+
// next visit; a raw coordinate is not a name either. Neither is reusable
|
|
323
|
+
// vocabulary, which is the whole point of this list.
|
|
324
|
+
.filter((e) => e.label != null && !/^#\d+$/.test(String(e.label).trim())
|
|
325
|
+
&& !/^@?-?\d+\s*,\s*-?\d+$/.test(String(e.label).trim()))
|
|
326
|
+
.sort((a, b) => b.count - a.count)
|
|
327
|
+
.slice(0, limit);
|
|
328
|
+
}
|
|
329
|
+
|
|
298
330
|
/**
|
|
299
331
|
* What to call this screen, for a human typing `goto`.
|
|
300
332
|
*
|
|
@@ -348,7 +380,7 @@ export function findScreen(udid, query) {
|
|
|
348
380
|
* 2.4 s on a screen that fetches — and because the graph is already persisted,
|
|
349
381
|
* versioned and pruned.
|
|
350
382
|
*/
|
|
351
|
-
function noteSettle(edge, settleMs, quietGapMs) {
|
|
383
|
+
function noteSettle(edge, settleMs, quietGapMs, focusMs) {
|
|
352
384
|
if (Number.isFinite(settleMs) && settleMs >= 0) {
|
|
353
385
|
edge.settles = [...(edge.settles ?? []), Math.round(settleMs)].slice(-TIMING_WINDOW);
|
|
354
386
|
}
|
|
@@ -357,6 +389,39 @@ function noteSettle(edge, settleMs, quietGapMs) {
|
|
|
357
389
|
if (Number.isFinite(quietGapMs) && quietGapMs >= 0) {
|
|
358
390
|
edge.quietGaps = [...(edge.quietGaps ?? []), Math.round(quietGapMs)].slice(-TIMING_WINDOW);
|
|
359
391
|
}
|
|
392
|
+
// How long the *field* took to take focus, which is a different duration from
|
|
393
|
+
// how long the step took: it is measured between the tap and the keyboard,
|
|
394
|
+
// inside a step whose settle is measured after the typing. One edge, two
|
|
395
|
+
// waits, so two distributions.
|
|
396
|
+
if (Number.isFinite(focusMs) && focusMs >= 0) {
|
|
397
|
+
edge.focuses = [...(edge.focuses ?? []), Math.round(focusMs)].slice(-TIMING_WINDOW);
|
|
398
|
+
}
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
/**
|
|
402
|
+
* Record the unbiased pause statistic for an edge already written.
|
|
403
|
+
*
|
|
404
|
+
* Separate from `record` because it arrives later on purpose. The biased
|
|
405
|
+
* `quietGaps` are gathered from inside the wait and are therefore bounded by
|
|
406
|
+
* when the wait chose to stop; `trueGaps` are read off the frame history once
|
|
407
|
+
* the transition is definitely over, which is one step later. So the edge has
|
|
408
|
+
* to be found again rather than passed along.
|
|
409
|
+
*
|
|
410
|
+
* Nothing reads `trueGaps` yet, and that is deliberate. Phase 11 built the
|
|
411
|
+
* learned stillness window on the biased statistic, it corrupted the graph
|
|
412
|
+
* inside an afternoon, and the lesson taken was not "use a better estimator" —
|
|
413
|
+
* it was that a number gets to *act* only after it has been watched for a while
|
|
414
|
+
* doing nothing. This is the watching.
|
|
415
|
+
*/
|
|
416
|
+
export function noteTrueGap(udid, screen, step, trueGapMs) {
|
|
417
|
+
if (!Number.isFinite(trueGapMs) || trueGapMs < 0) return null;
|
|
418
|
+
const node = screen?.hash ? nearestScreen(udid, screen)?.node : null;
|
|
419
|
+
if (!node) return null;
|
|
420
|
+
const edge = node.edges?.find((e) => e.action === actionSignature(step));
|
|
421
|
+
if (!edge) return null;
|
|
422
|
+
edge.trueGaps = [...(edge.trueGaps ?? []), Math.round(trueGapMs)].slice(-TIMING_WINDOW);
|
|
423
|
+
save(udid, node);
|
|
424
|
+
return edge;
|
|
360
425
|
}
|
|
361
426
|
|
|
362
427
|
/**
|
|
@@ -384,16 +449,76 @@ export function stillnessFor({ gapSamples, gapP95 } = {}, fallbackMs) {
|
|
|
384
449
|
};
|
|
385
450
|
}
|
|
386
451
|
|
|
452
|
+
/**
|
|
453
|
+
* How long to wait for a tapped field to take focus.
|
|
454
|
+
*
|
|
455
|
+
* The three numbers this replaces were the last genuinely fixed waits on the
|
|
456
|
+
* action path: 250 ms of stillness, a 900 ms reaction window, a 3 s timeout.
|
|
457
|
+
* What makes them different from the step budget is the shape of the failure.
|
|
458
|
+
* A step budget that is too short reports `no-visible-change` and the flow can
|
|
459
|
+
* see it. A focus wait that is too short types into a field that does not have
|
|
460
|
+
* focus yet, `typeText` succeeds because input has no feedback channel, and the
|
|
461
|
+
* step reports that it typed — the worst shape a failure can take, and the bug
|
|
462
|
+
* this helper was written to fix in the first place.
|
|
463
|
+
*
|
|
464
|
+
* So this one is asymmetric on purpose: **a learned window may only lengthen
|
|
465
|
+
* the wait**, never shorten it. p95 of what this field has actually cost, when
|
|
466
|
+
* that is longer than 3 s, is a field that was being typed into too early and
|
|
467
|
+
* now is not. Where it is shorter, the measurement is discarded rather than
|
|
468
|
+
* banked as a saving — 5% of a distribution is one silent wrong type in twenty
|
|
469
|
+
* runs, and there is no amount of median wall time worth that.
|
|
470
|
+
*
|
|
471
|
+
* The one shortening is not a learned number at all, it is positive evidence:
|
|
472
|
+
* if the keyboard was already up before the tap, this tap moves a caret. There
|
|
473
|
+
* is no keyboard animation to wait for, so the reaction window collapses to the
|
|
474
|
+
* stillness window instead of paying 900 ms to watch a screen that was never
|
|
475
|
+
* going to move. `beforeScreen` already carries `keyboard`, so the evidence is
|
|
476
|
+
* free — it is the same perception pass the step was going to run anyway.
|
|
477
|
+
*/
|
|
478
|
+
export function focusPlan(stats, { reactionMs, timeoutMs, stillnessMs, keyboardUp } = {}) {
|
|
479
|
+
if (keyboardUp) {
|
|
480
|
+
return {
|
|
481
|
+
reactionMs: stillnessMs ?? reactionMs,
|
|
482
|
+
timeoutMs,
|
|
483
|
+
cold: false,
|
|
484
|
+
from: 'the keyboard was already up, so this tap moves a caret',
|
|
485
|
+
};
|
|
486
|
+
}
|
|
487
|
+
const { focusP50, focusP95, focusSamples } = stats ?? {};
|
|
488
|
+
if (!Number.isFinite(focusP95) || !Number.isFinite(focusSamples) || focusSamples < COLD_SAMPLES) {
|
|
489
|
+
return { reactionMs, timeoutMs, cold: true, from: `fewer than ${COLD_SAMPLES} focus samples` };
|
|
490
|
+
}
|
|
491
|
+
const margin = Math.max(150, Math.round(focusP95 * 0.2));
|
|
492
|
+
const learnedTimeout = Math.min(HARD_CAP_MS, focusP95 + margin);
|
|
493
|
+
const learnedReaction = Math.min(learnedTimeout, focusP50 + margin);
|
|
494
|
+
return {
|
|
495
|
+
// max, not min. See above: only ever longer.
|
|
496
|
+
reactionMs: Math.max(reactionMs, learnedReaction),
|
|
497
|
+
timeoutMs: Math.max(timeoutMs, learnedTimeout),
|
|
498
|
+
cold: false,
|
|
499
|
+
from: `p95 ${focusP95}ms over ${focusSamples} focus samples`,
|
|
500
|
+
};
|
|
501
|
+
}
|
|
502
|
+
|
|
387
503
|
/** What this edge's observed settle durations say, or that it has none. */
|
|
388
504
|
export function timingOf(edge) {
|
|
389
505
|
const samples = edge?.settles ?? [];
|
|
390
506
|
const gaps = edge?.quietGaps ?? [];
|
|
507
|
+
const focuses = edge?.focuses ?? [];
|
|
391
508
|
return {
|
|
392
509
|
samples: samples.length,
|
|
393
510
|
p50: metrics.percentile(samples, 50),
|
|
394
511
|
p95: metrics.percentile(samples, 95),
|
|
395
512
|
gapSamples: gaps.length,
|
|
396
513
|
gapP95: metrics.percentile(gaps, 95),
|
|
514
|
+
// The same statistic measured after the transition rather than during it.
|
|
515
|
+
// Reported side by side so the size of the bias is visible rather than
|
|
516
|
+
// argued about — see `index.longestQuietGap`.
|
|
517
|
+
trueGapSamples: (edge?.trueGaps ?? []).length,
|
|
518
|
+
trueGapP95: metrics.percentile(edge?.trueGaps ?? [], 95),
|
|
519
|
+
focusSamples: focuses.length,
|
|
520
|
+
focusP50: metrics.percentile(focuses, 50),
|
|
521
|
+
focusP95: metrics.percentile(focuses, 95),
|
|
397
522
|
};
|
|
398
523
|
}
|
|
399
524
|
|
|
@@ -433,7 +558,7 @@ export function timingInto(udid, hash) {
|
|
|
433
558
|
return best ? { ...timingOf(best), action: best.action, kind: best.kind ?? null } : null;
|
|
434
559
|
}
|
|
435
560
|
|
|
436
|
-
export function record(udid, { from, action, to, kind, settleMs, quietGapMs }) {
|
|
561
|
+
export function record(udid, { from, action, to, kind, settleMs, quietGapMs, focusMs }) {
|
|
437
562
|
const fromKey = typeof from === 'string' ? { hash: from } : from;
|
|
438
563
|
const toHash = typeof to === 'string' ? to : to?.hash;
|
|
439
564
|
if (!fromKey?.hash || !toHash) return null;
|
|
@@ -500,7 +625,7 @@ export function record(udid, { from, action, to, kind, settleMs, quietGapMs }) {
|
|
|
500
625
|
save(udid, target);
|
|
501
626
|
existing.count += 1;
|
|
502
627
|
existing.lastSeen = Date.now();
|
|
503
|
-
noteSettle(existing, settleMs, quietGapMs);
|
|
628
|
+
noteSettle(existing, settleMs, quietGapMs, focusMs);
|
|
504
629
|
save(udid, node);
|
|
505
630
|
return node;
|
|
506
631
|
}
|
|
@@ -512,7 +637,13 @@ export function record(udid, { from, action, to, kind, settleMs, quietGapMs }) {
|
|
|
512
637
|
existing.kind = kind ?? existing.kind;
|
|
513
638
|
existing.count += 1;
|
|
514
639
|
existing.lastSeen = Date.now();
|
|
515
|
-
|
|
640
|
+
// Every argument, and it is worth saying why this line once passed one.
|
|
641
|
+
// `quietGapMs` was dropped here — on the *main* path, the one nearly every
|
|
642
|
+
// recorded edge takes — so the pause statistic only ever accumulated on a
|
|
643
|
+
// brand-new edge and on the variant branch. The window that reads it looked
|
|
644
|
+
// permanently cold, which is a measurement quietly not being taken rather
|
|
645
|
+
// than a wrong number, and those are the ones nothing complains about.
|
|
646
|
+
noteSettle(existing, settleMs, quietGapMs, focusMs);
|
|
516
647
|
} else {
|
|
517
648
|
node.edges.push({
|
|
518
649
|
action: signature,
|
|
@@ -525,6 +656,7 @@ export function record(udid, { from, action, to, kind, settleMs, quietGapMs }) {
|
|
|
525
656
|
lastSeen: Date.now(),
|
|
526
657
|
settles: Number.isFinite(settleMs) && settleMs >= 0 ? [Math.round(settleMs)] : [],
|
|
527
658
|
quietGaps: Number.isFinite(quietGapMs) && quietGapMs >= 0 ? [Math.round(quietGapMs)] : [],
|
|
659
|
+
focuses: Number.isFinite(focusMs) && focusMs >= 0 ? [Math.round(focusMs)] : [],
|
|
528
660
|
});
|
|
529
661
|
}
|
|
530
662
|
save(udid, node);
|
|
@@ -656,12 +788,36 @@ export const VERDICTS = ['ok', 'no-visible-change', 'unexpected-screen', 'unveri
|
|
|
656
788
|
* app and the verdict still said `unexpected-screen`, because the verdict never
|
|
657
789
|
* asked the graph.
|
|
658
790
|
*/
|
|
791
|
+
/**
|
|
792
|
+
* Are these two readings the same screen?
|
|
793
|
+
*
|
|
794
|
+
* `nearestScreen` has always had a token-similarity tolerance, precisely so
|
|
795
|
+
* that a screen whose *content* differs — a list with different rows, a form
|
|
796
|
+
* showing a different record — still resolves to the screen it is. The
|
|
797
|
+
* verification path threw that away: it passed bare hash strings, and a string
|
|
798
|
+
* carries no tokens, so only an exact hash could ever match.
|
|
799
|
+
*
|
|
800
|
+
* The cost was measured. `unexpected-screen` fired three times in one reported
|
|
801
|
+
* run and was wrong all three; two were this — the tester picked a different
|
|
802
|
+
* asset than earlier runs had, so the content differed, so the hash differed,
|
|
803
|
+
* so a correct navigation was called a wrong turn. Their conclusion: *"this
|
|
804
|
+
* will fire on every run that varies its test data — i.e. every useful run."*
|
|
805
|
+
* And because a failed step abandons the rest of its batch, each false alarm
|
|
806
|
+
* costs a round trip, which is the thing the whole design is trying to buy.
|
|
807
|
+
*
|
|
808
|
+
* So pass the reading, not just its name: `{hash, tokens}` lets the tolerance
|
|
809
|
+
* that already exists do its job. A string still works and still means "exact
|
|
810
|
+
* match only", which is right for a stored prediction that has no tokens.
|
|
811
|
+
*/
|
|
659
812
|
function sameScreen(udid, a, b) {
|
|
660
|
-
|
|
661
|
-
|
|
813
|
+
const hashOf = (v) => (typeof v === 'string' ? v : v?.hash);
|
|
814
|
+
const ha = hashOf(a);
|
|
815
|
+
const hb = hashOf(b);
|
|
816
|
+
if (!ha || !hb) return false;
|
|
817
|
+
if (ha === hb) return true;
|
|
662
818
|
if (!udid) return false;
|
|
663
|
-
const nodeA = nearestScreen(udid, a)?.node;
|
|
664
|
-
const nodeB = nearestScreen(udid, b)?.node;
|
|
819
|
+
const nodeA = nearestScreen(udid, typeof a === 'string' ? a : { hash: ha, tokens: a?.tokens })?.node;
|
|
820
|
+
const nodeB = nearestScreen(udid, typeof b === 'string' ? b : { hash: hb, tokens: b?.tokens })?.node;
|
|
665
821
|
return Boolean(nodeA && nodeB && nodeA.hash === nodeB.hash);
|
|
666
822
|
}
|
|
667
823
|
|
|
@@ -677,9 +833,35 @@ function sameScreen(udid, a, b) {
|
|
|
677
833
|
*/
|
|
678
834
|
export const CONFIDENT_OBSERVATIONS = 2;
|
|
679
835
|
|
|
680
|
-
|
|
681
|
-
|
|
682
|
-
|
|
836
|
+
/**
|
|
837
|
+
* Actions whose correct outcome is that the screen stays where it is.
|
|
838
|
+
*
|
|
839
|
+
* Typing into a field does not navigate, so screen-identity movement cannot
|
|
840
|
+
* say whether it worked — and answering with `no-visible-change` was actively
|
|
841
|
+
* harmful three ways. It printed a verdict contradicting the wait's own
|
|
842
|
+
* observation on the same line (`[a small change, in one region only] …
|
|
843
|
+
* [no-visible-change]`, both true, of different questions). It is an escalating
|
|
844
|
+
* verdict, so a clean flow that typed anything told the caller to stop and
|
|
845
|
+
* think. And it invited a re-type, which doubles a field that cannot be
|
|
846
|
+
* cleared.
|
|
847
|
+
*
|
|
848
|
+
* These steps are verified by reading the field back instead — see
|
|
849
|
+
* `fieldContents` in actions.js.
|
|
850
|
+
*/
|
|
851
|
+
export const STAYS_ON_SCREEN = new Set(['type', 'paste', 'key']);
|
|
852
|
+
|
|
853
|
+
export function verdict({ udid, prediction, before, after, kind, action }) {
|
|
854
|
+
// `before`/`after` may be a hash or a whole reading. A reading carries its
|
|
855
|
+
// tokens, which is what lets a content-varied screen still be recognised as
|
|
856
|
+
// the screen it is.
|
|
857
|
+
const hashOf = (v) => (typeof v === 'string' ? v : v?.hash);
|
|
858
|
+
const beforeHash = hashOf(before);
|
|
859
|
+
const afterHash = hashOf(after);
|
|
860
|
+
if (!beforeHash || !afterHash) return { verdict: 'unverified', detail: 'no state to compare' };
|
|
861
|
+
const moved = beforeHash !== afterHash;
|
|
862
|
+
if (STAYS_ON_SCREEN.has(action) && !moved) {
|
|
863
|
+
return { verdict: 'ok', detail: 'the screen was not expected to change, and did not' };
|
|
864
|
+
}
|
|
683
865
|
if (!prediction) {
|
|
684
866
|
if (!moved) return { verdict: 'no-visible-change', detail: 'the screen did not change, and nothing predicted it would' };
|
|
685
867
|
return { verdict: 'unverified', detail: 'this action has not been seen on this screen before' };
|