simframe 0.9.0 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/README.md +165 -5
  2. package/data/vocabulary/en.json +148 -0
  3. package/native/ocr.swift +13 -1
  4. package/native/rank.swift +87 -0
  5. package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +43 -3
  6. package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
  7. package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
  8. package/native/simframed/Sources/simframed/main.swift +13 -1
  9. package/native/supervise.swift +181 -0
  10. package/package.json +4 -1
  11. package/scripts/check-package.mjs +22 -2
  12. package/scripts/check-private.mjs +143 -0
  13. package/scripts/ci-memory.mjs +104 -20
  14. package/scripts/eval-perception.mjs +281 -0
  15. package/scripts/phase17-corpus.mjs +176 -0
  16. package/skills/simframe/SKILL.md +237 -5
  17. package/src/actions.js +1825 -38
  18. package/src/analyze.js +70 -0
  19. package/src/cli.js +214 -15
  20. package/src/control.js +1 -0
  21. package/src/fingerprint.js +19 -1
  22. package/src/graph.js +193 -11
  23. package/src/index.js +428 -16
  24. package/src/input.js +115 -8
  25. package/src/localhelper.js +155 -0
  26. package/src/matching.js +119 -3
  27. package/src/mcp.js +319 -27
  28. package/src/metrics.js +134 -8
  29. package/src/navigate.js +10 -7
  30. package/src/ocr.js +18 -1
  31. package/src/planner.js +195 -0
  32. package/src/platform/android.js +3 -2
  33. package/src/platform/ios.js +2 -1
  34. package/src/png.js +26 -0
  35. package/src/refs.js +51 -8
  36. package/src/regions.js +110 -1
  37. package/src/screenmap.js +109 -10
  38. package/src/supervisor.js +117 -0
  39. package/src/view.js +396 -7
  40. package/src/vocabulary.js +134 -0
  41. package/src/wrote.js +136 -0
package/src/graph.js CHANGED
@@ -295,6 +295,38 @@ function replayable(step) {
295
295
  return step.launch != null || step.openUrl != null ? null : step;
296
296
  }
297
297
 
298
+ /**
299
+ * What has worked from this screen before, in the caller's own vocabulary.
300
+ *
301
+ * The graph has always known this and never said it. A map reported
302
+ * `(known, 3 known exits)` — the *count* — so an agent on a screen simframe had
303
+ * driven successfully six times still had to read it to learn what was tappable.
304
+ * Measured across two peer rounds: a flow whose steps were known in advance ran
305
+ * 16 steps in **one** call, and the same agent on a screen the graph also knew
306
+ * but whose labels it did not spent 25 calls on 31 steps. The difference was not
307
+ * perception. It was whether a plan existed before execution started.
308
+ *
309
+ * Ordered by how often each has worked, because that is the order an agent
310
+ * should try them in.
311
+ */
312
+ export function exitsOf(node, { limit = 8 } = {}) {
313
+ return (node?.edges ?? [])
314
+ .map((e) => ({
315
+ action: e.step?.action ?? (e.action ?? '').split(':')[0] ?? 'tap',
316
+ label: e.step?.value ?? e.step?.target ?? e.step?.label ?? e.step?.into ?? null,
317
+ to: e.to ?? null,
318
+ count: e.count ?? 0,
319
+ kind: e.kind ?? null,
320
+ }))
321
+ // A `#13` was a ref on the screen it was typed on and means nothing on the
322
+ // next visit; a raw coordinate is not a name either. Neither is reusable
323
+ // vocabulary, which is the whole point of this list.
324
+ .filter((e) => e.label != null && !/^#\d+$/.test(String(e.label).trim())
325
+ && !/^@?-?\d+\s*,\s*-?\d+$/.test(String(e.label).trim()))
326
+ .sort((a, b) => b.count - a.count)
327
+ .slice(0, limit);
328
+ }
329
+
298
330
  /**
299
331
  * What to call this screen, for a human typing `goto`.
300
332
  *
@@ -348,7 +380,7 @@ export function findScreen(udid, query) {
348
380
  * 2.4 s on a screen that fetches — and because the graph is already persisted,
349
381
  * versioned and pruned.
350
382
  */
351
- function noteSettle(edge, settleMs, quietGapMs) {
383
+ function noteSettle(edge, settleMs, quietGapMs, focusMs) {
352
384
  if (Number.isFinite(settleMs) && settleMs >= 0) {
353
385
  edge.settles = [...(edge.settles ?? []), Math.round(settleMs)].slice(-TIMING_WINDOW);
354
386
  }
@@ -357,6 +389,39 @@ function noteSettle(edge, settleMs, quietGapMs) {
357
389
  if (Number.isFinite(quietGapMs) && quietGapMs >= 0) {
358
390
  edge.quietGaps = [...(edge.quietGaps ?? []), Math.round(quietGapMs)].slice(-TIMING_WINDOW);
359
391
  }
392
+ // How long the *field* took to take focus, which is a different duration from
393
+ // how long the step took: it is measured between the tap and the keyboard,
394
+ // inside a step whose settle is measured after the typing. One edge, two
395
+ // waits, so two distributions.
396
+ if (Number.isFinite(focusMs) && focusMs >= 0) {
397
+ edge.focuses = [...(edge.focuses ?? []), Math.round(focusMs)].slice(-TIMING_WINDOW);
398
+ }
399
+ }
400
+
401
+ /**
402
+ * Record the unbiased pause statistic for an edge already written.
403
+ *
404
+ * Separate from `record` because it arrives later on purpose. The biased
405
+ * `quietGaps` are gathered from inside the wait and are therefore bounded by
406
+ * when the wait chose to stop; `trueGaps` are read off the frame history once
407
+ * the transition is definitely over, which is one step later. So the edge has
408
+ * to be found again rather than passed along.
409
+ *
410
+ * Nothing reads `trueGaps` yet, and that is deliberate. Phase 11 built the
411
+ * learned stillness window on the biased statistic, it corrupted the graph
412
+ * inside an afternoon, and the lesson taken was not "use a better estimator" —
413
+ * it was that a number gets to *act* only after it has been watched for a while
414
+ * doing nothing. This is the watching.
415
+ */
416
+ export function noteTrueGap(udid, screen, step, trueGapMs) {
417
+ if (!Number.isFinite(trueGapMs) || trueGapMs < 0) return null;
418
+ const node = screen?.hash ? nearestScreen(udid, screen)?.node : null;
419
+ if (!node) return null;
420
+ const edge = node.edges?.find((e) => e.action === actionSignature(step));
421
+ if (!edge) return null;
422
+ edge.trueGaps = [...(edge.trueGaps ?? []), Math.round(trueGapMs)].slice(-TIMING_WINDOW);
423
+ save(udid, node);
424
+ return edge;
360
425
  }
361
426
 
362
427
  /**
@@ -384,16 +449,76 @@ export function stillnessFor({ gapSamples, gapP95 } = {}, fallbackMs) {
384
449
  };
385
450
  }
386
451
 
452
+ /**
453
+ * How long to wait for a tapped field to take focus.
454
+ *
455
+ * The three numbers this replaces were the last genuinely fixed waits on the
456
+ * action path: 250 ms of stillness, a 900 ms reaction window, a 3 s timeout.
457
+ * What makes them different from the step budget is the shape of the failure.
458
+ * A step budget that is too short reports `no-visible-change` and the flow can
459
+ * see it. A focus wait that is too short types into a field that does not have
460
+ * focus yet, `typeText` succeeds because input has no feedback channel, and the
461
+ * step reports that it typed — the worst shape a failure can take, and the bug
462
+ * this helper was written to fix in the first place.
463
+ *
464
+ * So this one is asymmetric on purpose: **a learned window may only lengthen
465
+ * the wait**, never shorten it. p95 of what this field has actually cost, when
466
+ * that is longer than 3 s, is a field that was being typed into too early and
467
+ * now is not. Where it is shorter, the measurement is discarded rather than
468
+ * banked as a saving — 5% of a distribution is one silent wrong type in twenty
469
+ * runs, and there is no amount of median wall time worth that.
470
+ *
471
+ * The one shortening is not a learned number at all, it is positive evidence:
472
+ * if the keyboard was already up before the tap, this tap moves a caret. There
473
+ * is no keyboard animation to wait for, so the reaction window collapses to the
474
+ * stillness window instead of paying 900 ms to watch a screen that was never
475
+ * going to move. `beforeScreen` already carries `keyboard`, so the evidence is
476
+ * free — it is the same perception pass the step was going to run anyway.
477
+ */
478
+ export function focusPlan(stats, { reactionMs, timeoutMs, stillnessMs, keyboardUp } = {}) {
479
+ if (keyboardUp) {
480
+ return {
481
+ reactionMs: stillnessMs ?? reactionMs,
482
+ timeoutMs,
483
+ cold: false,
484
+ from: 'the keyboard was already up, so this tap moves a caret',
485
+ };
486
+ }
487
+ const { focusP50, focusP95, focusSamples } = stats ?? {};
488
+ if (!Number.isFinite(focusP95) || !Number.isFinite(focusSamples) || focusSamples < COLD_SAMPLES) {
489
+ return { reactionMs, timeoutMs, cold: true, from: `fewer than ${COLD_SAMPLES} focus samples` };
490
+ }
491
+ const margin = Math.max(150, Math.round(focusP95 * 0.2));
492
+ const learnedTimeout = Math.min(HARD_CAP_MS, focusP95 + margin);
493
+ const learnedReaction = Math.min(learnedTimeout, focusP50 + margin);
494
+ return {
495
+ // max, not min. See above: only ever longer.
496
+ reactionMs: Math.max(reactionMs, learnedReaction),
497
+ timeoutMs: Math.max(timeoutMs, learnedTimeout),
498
+ cold: false,
499
+ from: `p95 ${focusP95}ms over ${focusSamples} focus samples`,
500
+ };
501
+ }
502
+
387
503
  /** What this edge's observed settle durations say, or that it has none. */
388
504
  export function timingOf(edge) {
389
505
  const samples = edge?.settles ?? [];
390
506
  const gaps = edge?.quietGaps ?? [];
507
+ const focuses = edge?.focuses ?? [];
391
508
  return {
392
509
  samples: samples.length,
393
510
  p50: metrics.percentile(samples, 50),
394
511
  p95: metrics.percentile(samples, 95),
395
512
  gapSamples: gaps.length,
396
513
  gapP95: metrics.percentile(gaps, 95),
514
+ // The same statistic measured after the transition rather than during it.
515
+ // Reported side by side so the size of the bias is visible rather than
516
+ // argued about — see `index.longestQuietGap`.
517
+ trueGapSamples: (edge?.trueGaps ?? []).length,
518
+ trueGapP95: metrics.percentile(edge?.trueGaps ?? [], 95),
519
+ focusSamples: focuses.length,
520
+ focusP50: metrics.percentile(focuses, 50),
521
+ focusP95: metrics.percentile(focuses, 95),
397
522
  };
398
523
  }
399
524
 
@@ -433,7 +558,7 @@ export function timingInto(udid, hash) {
433
558
  return best ? { ...timingOf(best), action: best.action, kind: best.kind ?? null } : null;
434
559
  }
435
560
 
436
- export function record(udid, { from, action, to, kind, settleMs, quietGapMs }) {
561
+ export function record(udid, { from, action, to, kind, settleMs, quietGapMs, focusMs }) {
437
562
  const fromKey = typeof from === 'string' ? { hash: from } : from;
438
563
  const toHash = typeof to === 'string' ? to : to?.hash;
439
564
  if (!fromKey?.hash || !toHash) return null;
@@ -500,7 +625,7 @@ export function record(udid, { from, action, to, kind, settleMs, quietGapMs }) {
500
625
  save(udid, target);
501
626
  existing.count += 1;
502
627
  existing.lastSeen = Date.now();
503
- noteSettle(existing, settleMs, quietGapMs);
628
+ noteSettle(existing, settleMs, quietGapMs, focusMs);
504
629
  save(udid, node);
505
630
  return node;
506
631
  }
@@ -512,7 +637,13 @@ export function record(udid, { from, action, to, kind, settleMs, quietGapMs }) {
512
637
  existing.kind = kind ?? existing.kind;
513
638
  existing.count += 1;
514
639
  existing.lastSeen = Date.now();
515
- noteSettle(existing, settleMs);
640
+ // Every argument, and it is worth saying why this line once passed one.
641
+ // `quietGapMs` was dropped here — on the *main* path, the one nearly every
642
+ // recorded edge takes — so the pause statistic only ever accumulated on a
643
+ // brand-new edge and on the variant branch. The window that reads it looked
644
+ // permanently cold, which is a measurement quietly not being taken rather
645
+ // than a wrong number, and those are the ones nothing complains about.
646
+ noteSettle(existing, settleMs, quietGapMs, focusMs);
516
647
  } else {
517
648
  node.edges.push({
518
649
  action: signature,
@@ -525,6 +656,7 @@ export function record(udid, { from, action, to, kind, settleMs, quietGapMs }) {
525
656
  lastSeen: Date.now(),
526
657
  settles: Number.isFinite(settleMs) && settleMs >= 0 ? [Math.round(settleMs)] : [],
527
658
  quietGaps: Number.isFinite(quietGapMs) && quietGapMs >= 0 ? [Math.round(quietGapMs)] : [],
659
+ focuses: Number.isFinite(focusMs) && focusMs >= 0 ? [Math.round(focusMs)] : [],
528
660
  });
529
661
  }
530
662
  save(udid, node);
@@ -656,12 +788,36 @@ export const VERDICTS = ['ok', 'no-visible-change', 'unexpected-screen', 'unveri
656
788
  * app and the verdict still said `unexpected-screen`, because the verdict never
657
789
  * asked the graph.
658
790
  */
791
+ /**
792
+ * Are these two readings the same screen?
793
+ *
794
+ * `nearestScreen` has always had a token-similarity tolerance, precisely so
795
+ * that a screen whose *content* differs — a list with different rows, a form
796
+ * showing a different record — still resolves to the screen it is. The
797
+ * verification path threw that away: it passed bare hash strings, and a string
798
+ * carries no tokens, so only an exact hash could ever match.
799
+ *
800
+ * The cost was measured. `unexpected-screen` fired three times in one reported
801
+ * run and was wrong all three; two were this — the tester picked a different
802
+ * asset than earlier runs had, so the content differed, so the hash differed,
803
+ * so a correct navigation was called a wrong turn. Their conclusion: *"this
804
+ * will fire on every run that varies its test data — i.e. every useful run."*
805
+ * And because a failed step abandons the rest of its batch, each false alarm
806
+ * costs a round trip, which is the thing the whole design is trying to buy.
807
+ *
808
+ * So pass the reading, not just its name: `{hash, tokens}` lets the tolerance
809
+ * that already exists do its job. A string still works and still means "exact
810
+ * match only", which is right for a stored prediction that has no tokens.
811
+ */
659
812
  function sameScreen(udid, a, b) {
660
- if (!a || !b) return false;
661
- if (a === b) return true;
813
+ const hashOf = (v) => (typeof v === 'string' ? v : v?.hash);
814
+ const ha = hashOf(a);
815
+ const hb = hashOf(b);
816
+ if (!ha || !hb) return false;
817
+ if (ha === hb) return true;
662
818
  if (!udid) return false;
663
- const nodeA = nearestScreen(udid, a)?.node;
664
- const nodeB = nearestScreen(udid, b)?.node;
819
+ const nodeA = nearestScreen(udid, typeof a === 'string' ? a : { hash: ha, tokens: a?.tokens })?.node;
820
+ const nodeB = nearestScreen(udid, typeof b === 'string' ? b : { hash: hb, tokens: b?.tokens })?.node;
665
821
  return Boolean(nodeA && nodeB && nodeA.hash === nodeB.hash);
666
822
  }
667
823
 
@@ -677,9 +833,35 @@ function sameScreen(udid, a, b) {
677
833
  */
678
834
  export const CONFIDENT_OBSERVATIONS = 2;
679
835
 
680
- export function verdict({ udid, prediction, before, after, kind }) {
681
- if (!before || !after) return { verdict: 'unverified', detail: 'no state to compare' };
682
- const moved = before !== after;
836
+ /**
837
+ * Actions whose correct outcome is that the screen stays where it is.
838
+ *
839
+ * Typing into a field does not navigate, so screen-identity movement cannot
840
+ * say whether it worked — and answering with `no-visible-change` was actively
841
+ * harmful three ways. It printed a verdict contradicting the wait's own
842
+ * observation on the same line (`[a small change, in one region only] …
843
+ * [no-visible-change]`, both true, of different questions). It is an escalating
844
+ * verdict, so a clean flow that typed anything told the caller to stop and
845
+ * think. And it invited a re-type, which doubles a field that cannot be
846
+ * cleared.
847
+ *
848
+ * These steps are verified by reading the field back instead — see
849
+ * `fieldContents` in actions.js.
850
+ */
851
+ export const STAYS_ON_SCREEN = new Set(['type', 'paste', 'key']);
852
+
853
+ export function verdict({ udid, prediction, before, after, kind, action }) {
854
+ // `before`/`after` may be a hash or a whole reading. A reading carries its
855
+ // tokens, which is what lets a content-varied screen still be recognised as
856
+ // the screen it is.
857
+ const hashOf = (v) => (typeof v === 'string' ? v : v?.hash);
858
+ const beforeHash = hashOf(before);
859
+ const afterHash = hashOf(after);
860
+ if (!beforeHash || !afterHash) return { verdict: 'unverified', detail: 'no state to compare' };
861
+ const moved = beforeHash !== afterHash;
862
+ if (STAYS_ON_SCREEN.has(action) && !moved) {
863
+ return { verdict: 'ok', detail: 'the screen was not expected to change, and did not' };
864
+ }
683
865
  if (!prediction) {
684
866
  if (!moved) return { verdict: 'no-visible-change', detail: 'the screen did not change, and nothing predicted it would' };
685
867
  return { verdict: 'unverified', detail: 'this action has not been seen on this screen before' };