simframe 0.15.0 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -701,6 +701,36 @@ Reproduce all of it with `npm run bench`, which prints the same table against
701
701
  your machine. Full detail, including the measurement traps, is in
702
702
  [`docs/BENCHMARKS.md`](docs/BENCHMARKS.md).
703
703
 
704
+ ### Wall clock per step — where the time really goes
705
+
706
+ Every number above is microscopic next to the one that decides how fast this
707
+ feels, and it took two field reports to see it. Per-step wall clock is
708
+
709
+ ```
710
+ (model round trip + simframe work × n) / n for n steps in one call
711
+ ```
712
+
713
+ | | per step |
714
+ | --- | --- |
715
+ | a human tester, measured | **1.95 s** |
716
+ | **a saved flow replayed — zero model calls** | **1.98 s** |
717
+ | simframe's own work inside a batch | **~1.7 s** |
718
+ | a batch of 4 steps, one model call | ~6.7 s |
719
+ | a batch of 2 steps | ~11.7 s |
720
+ | one model call per step | ~21.7 s |
721
+
722
+ **A replayed flow runs at human speed**, and simframe's own work already does.
723
+ A field report put the split at **34% simframe, 60% agent round trips** over
724
+ 462 s of wall clock — the tester's *"30+ seconds between each step"* was
725
+ accurate and was not simframe. So there is nothing left to win inside the
726
+ engine, and the only variable is `n`: `simframe hpi` reports `steps_per_call`
727
+ for exactly that reason.
728
+
729
+ Which is why **every hard-fail that drops a caller back to single-stepping is a
730
+ latency bug**. Between two field reports on the same flow, one run took 33 tool
731
+ calls and the next took **16**, for 28 executed steps and no screenshots at
732
+ all — the difference being defects fixed, not code made faster.
733
+
704
734
  ## How it works
705
735
 
706
736
  ```
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "simframe",
3
- "version": "0.15.0",
3
+ "version": "0.16.0",
4
4
  "mcpName": "io.github.lvlrSajjad/simframe",
5
5
  "description": "Always-warm iOS Simulator and Android emulator frames: agents read the screen in ~20ms instead of waiting on screenshots. MCP server + CLI.",
6
6
  "keywords": [
@@ -186,11 +186,7 @@ const report = {
186
186
 
187
187
  console.log('\nflow runs agent p50 human p50 HPI_time step_ratio');
188
188
  for (const f of report.flows) {
189
- console.log(
190
- `${f.flow.padEnd(24)} ${String(f.runs).padStart(5)} ${`${f.agent_ms.p50}ms`.padStart(9)} ` +
191
- `${(f.human_median_ms ? `${f.human_median_ms}ms` : '—').padStart(9)} ` +
192
- `${String(f.hpi_time ?? '—').padStart(8)} ${String(f.step_ratio ?? '—').padStart(10)}`,
193
- );
189
+ console.log(metrics.flowRow(f));
194
190
  }
195
191
  report.overall.hpi_time_median_of_passes = passTimes.length ? Number(metrics.median(passTimes).toFixed(3)) : null;
196
192
  const o = report.overall;
@@ -247,8 +243,11 @@ const base = committed.overall ?? {};
247
243
  const failures = metrics.gateAgainst(committed, report);
248
244
 
249
245
  console.log(`\ngate vs ${path.relative(ROOT, baselineFile)} (measured ${committed.measured_at ?? '?'})`);
250
- console.log(` HPI_accuracy ${base.hpi_accuracy ?? '—'} -> ${o.hpi_accuracy ?? '—'}`);
251
- console.log(` HPI_time ${metrics.gateTime(base) ?? '—'} -> ${metrics.gateTime(o) ?? '—'}`);
252
- console.log(` band ${metrics.TIME_REGRESSION * 100}% (median of ${passTimes.length || 1} pass(es))`);
246
+ console.log(` HPI_accuracy ${base.hpi_accuracy ?? '—'} -> ${o.hpi_accuracy ?? '—'} (any drop fails)`);
247
+ console.log(` step_ratio ${o.step_ratio ?? '—'} (fails above ${metrics.STEP_RATIO_CEILING})`);
248
+ // Printed apart from the two that gate, and labelled, because a number in a
249
+ // gate block gets read as a threshold whether or not it is one.
250
+ console.log(` ${metrics.timeTrend(committed, report)}`);
251
+ console.log(` median of ${passTimes.length || 1} pass(es), on this host`);
253
252
  for (const f of failures) console.log(`FAIL ${f}`);
254
253
  process.exit(failures.length ? 1 : 0);
@@ -635,19 +635,31 @@ check(walked.ok === true || outcomes.includes(walked.reason),
635
635
  walked.ok ? (walked.already ? 'already there' : `walked ${walked.ranSteps} step(s)`) : walked.reason);
636
636
 
637
637
  console.log('\n--- what may be saved, and what may not ---');
638
- // A flow with an unverified step in it is a recording of something that may not
639
- // have worked, and replaying it faithfully reproduces the doubt. So the
640
- // contract runs both ways and both directions are checked against whatever the
641
- // run actually produced, rather than assuming it verified.
638
+ // **A first traversal must be recordable**, which is the opposite of what this
639
+ // check used to assert. The old contract refused any flow with a non-`ok`
640
+ // verdict — and a first traversal is all-`unverified` by construction, since
641
+ // there is no prior observation to compare against. So no flow could ever be
642
+ // recorded, replay was unreachable, and the only zero-model-call path in the
643
+ // tool was sealed shut behind a gate nobody could pass.
644
+ //
645
+ // The refusal now needs evidence *against* a step rather than the absence of
646
+ // evidence for it. Checked against whatever the run actually produced rather
647
+ // than assuming a shape, as before.
642
648
  const attempt = await jsonRetry(['do', LOOP, `--save=${FLOW_NAME}`], { allowFail: true });
649
+ const contradicted = attempt.results.some((r) => /^unexpected/.test(r.verification?.verdict ?? ''));
643
650
  const clean = attempt.results.every((r) => !r.verification || r.verification.verdict === 'ok');
644
- if (clean) {
645
- check(attempt.saved?.ok === true, 'a flow whose every step verified is saved',
646
- `${attempt.saved?.steps} steps`);
647
- } else {
648
- check(attempt.saved?.ok === false && attempt.saved?.reason === 'unverified-steps',
649
- 'a flow with an unverified step is refused, not quietly saved',
651
+ if (contradicted) {
652
+ check(attempt.saved?.ok === false && attempt.saved?.reason === 'contradicted-steps',
653
+ 'a flow with a contradicted step is refused, not quietly saved',
650
654
  `${attempt.saved?.reason} (${(attempt.saved?.verdicts ?? []).join(', ')})`);
655
+ } else {
656
+ check(attempt.saved?.ok === true, 'a first traversal is recordable',
657
+ `${attempt.saved?.steps} steps`);
658
+ check(attempt.saved?.provisional === !clean,
659
+ clean
660
+ ? 'and a flow whose every step verified is confirmed outright'
661
+ : 'and it is marked provisional, because nothing had been seen before to compare against',
662
+ `provisional=${attempt.saved?.provisional}`);
651
663
  }
652
664
 
653
665
  // --force is the deliberate override, and it is what lets the replay machinery
@@ -660,6 +672,17 @@ if (check(forced.saved?.ok === true, 'and --force saves it anyway', `${forced.sa
660
672
  check(replayed.ranSteps >= 1 && Array.isArray(replayed.results),
661
673
  'and replays from disk with no model in the loop',
662
674
  `${replayed.ranSteps}/${replayed.totalSteps} steps`);
675
+ // The other half of the bootstrap, and the reason "provisional" is not a
676
+ // state nothing ever leaves: a clean replay is the confirmation a first
677
+ // traversal could not give. Only asserted when the replay actually ran to
678
+ // the end — a partial replay promotes nothing, deliberately.
679
+ if (replayed.ok) {
680
+ const after = await jsonRetry(['flow', 'list']);
681
+ const entry = after.find((f) => f.name === FLOW_NAME);
682
+ check(entry && !entry.provisional,
683
+ 'and a clean replay confirms a provisional flow',
684
+ `provisional=${entry?.provisional ?? 'gone'}`);
685
+ }
663
686
  const unknown = await cli(['flow', 'run', 'no-such-flow'], { expectFail: true });
664
687
  check(/no flow/i.test(unknown), 'an unknown flow name is refused with what is known');
665
688
  }
package/src/actions.js CHANGED
@@ -19,6 +19,29 @@ import { launchApp, openUrl, setPermission, terminateApp } from './platform/inde
19
19
 
20
20
  const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
21
21
  const MAX_PAUSE_MS = 5000;
22
+ /**
23
+ * Above this many characters a fill uses the pasteboard; at or below it, the
24
+ * keyboard.
25
+ *
26
+ * The pasteboard is the right default for long text and was the default for
27
+ * *all* text in `sweep`, which a field report showed costing far more than it
28
+ * saved. iOS 26 raises a system consent alert — *"… would like to paste from
29
+ * CoreSimulatorBridge"* — on the first paste into an app, and it fired on a
30
+ * **nine-character** value. The alert covered the form, collapsed the
31
+ * accessibility tree to OCR-only, and failed the batch; the recovery chain that
32
+ * followed was close to half of that session's wasted model round trips.
33
+ *
34
+ * `sim_type_into` already defaults to the keyboard. Only `sweep`'s fill did
35
+ * not, so the reporter got the pasteboard without asking for it.
36
+ *
37
+ * Why a threshold is safe rather than a trade against exactness — the reason
38
+ * pasteboard-by-default was chosen: `type` with `into` reads the field back,
39
+ * and retries once locally if nothing landed. So a keystroke path that a
40
+ * keyboard layout mangles is *caught*, not silently accepted. 40 characters is
41
+ * the reporter's own suggestion and covers names, emails and short
42
+ * descriptions, while a paragraph still goes by pasteboard where it belongs.
43
+ */
44
+ const KEYBOARD_UP_TO = 40;
22
45
  /**
23
46
  * A tapped field is typed into once the screen has settled, not after a fixed
24
47
  * wait.
@@ -1559,12 +1582,35 @@ async function stillOnPlan(deviceQuery, verification, nextStep, options) {
1559
1582
  // neither is evidence about where we are.
1560
1583
  if (!target || /^#\d+$/.test(String(target).trim()) || /^@?-?\d+\s*,\s*-?\d+$/.test(String(target).trim())) return false;
1561
1584
  if (!ACTION_STEPS.has(nextStep.action) && nextStep.action !== 'assert' && nextStep.action !== 'waitFor') return false;
1562
- try {
1563
- const hit = await api.locate(deviceQuery, String(target), { options });
1564
- return Boolean(hit?.target);
1565
- } catch {
1566
- return false;
1585
+ // **Settle before looking, and look twice.**
1586
+ //
1587
+ // This asked once, immediately, and the circumstance it is asked in is
1588
+ // precisely a screen that is still arriving: a content-driven screen whose
1589
+ // rows have not rendered hashes differently from the one memory expected —
1590
+ // which is what produced the `unexpected-screen` — *and* does not yet hold
1591
+ // the next step's target. So the one signal that could have rescued the batch
1592
+ // was read at the only moment it was guaranteed to be absent, and a correct
1593
+ // navigation aborted the run.
1594
+ //
1595
+ // Measured cost of getting this wrong: a field report lost a 7-step plan at
1596
+ // step 5 on a tap that had correctly advanced a wizard, and every discarded
1597
+ // step is ~20s of model latency to re-plan. A settle here costs under a
1598
+ // second and is paid only on the failure path.
1599
+ //
1600
+ // It cannot manufacture a continue: the next step's own target still has to
1601
+ // resolve, which is strong evidence about where we are.
1602
+ for (let look = 0; look < 2; look += 1) {
1603
+ await api.waitFor(deviceQuery, {
1604
+ mode: 'settle', stableMs: 300, timeoutMs: look === 0 ? 1200 : 800, options,
1605
+ }).catch(() => null);
1606
+ try {
1607
+ const hit = await api.locate(deviceQuery, String(target), { refresh: true, options });
1608
+ if (hit?.target) return true;
1609
+ } catch {
1610
+ /* not here yet; one more look, then the halt stands */
1611
+ }
1567
1612
  }
1613
+ return false;
1568
1614
  }
1569
1615
 
1570
1616
  /**
@@ -2153,6 +2199,36 @@ export const SWEEP_SECTIONS = 10;
2153
2199
 
2154
2200
  const sweepKey = (r) => `${alnum(r.label)}\u0000${Math.round((r.x ?? 0) / 8)}`;
2155
2201
 
2202
+ /**
2203
+ * Every name a caller may legitimately write for a swept row.
2204
+ *
2205
+ * `sweep` matched on `r.label` alone, and in React Native most interactive
2206
+ * controls have a `testID` and no accessibility label — so a field `sweep`
2207
+ * itself had just printed came back as **"NOT FOUND anywhere"**:
2208
+ *
2209
+ * swept 3 section(s) ... 29 distinct element(s);
2210
+ * NOT FOUND anywhere: "create-service-request-3-requested-by-input"
2211
+ * ...
2212
+ * #18 field 201,480 create-service-request-3-requested-by-input
2213
+ *
2214
+ * Four lines apart, in one response. The next call filled it by `#18` first
2215
+ * try. The reporter called it the most confidence-damaging failure of the run,
2216
+ * because "NOT FOUND anywhere" is a strong claim and it briefly convinced them
2217
+ * the form did not have the field they were looking at.
2218
+ *
2219
+ * **The third time this same assumption has been reported.** Item 152 was the
2220
+ * identifier missing from `matching.rank`; `ee305d4` was a label the matcher
2221
+ * refused; this is `sweep` carrying its own ad-hoc matcher that never learned
2222
+ * either fix. A resolver per call site is a resolver that has to be corrected
2223
+ * per call site — the names belong in one place, which is what this is.
2224
+ */
2225
+ const sweepNames = (r) => [r.label, r.identifier, ...(r.aliases ?? [])].filter(Boolean).map(alnum);
2226
+ /** Does this row answer to `needle` by any of its names? */
2227
+ const sweepHolds = (r, needle) => {
2228
+ const want = alnum(needle);
2229
+ return want ? sweepNames(r).some((n) => n.includes(want)) : false;
2230
+ };
2231
+
2156
2232
  async function sectionHere(deviceQuery, options) {
2157
2233
  try {
2158
2234
  const map = await view.screenMap(deviceQuery, { options, refresh: true });
@@ -2298,12 +2374,29 @@ async function sweep(deviceQuery, udid, step, ctx) {
2298
2374
  // scroll back to it.
2299
2375
  if (fill) {
2300
2376
  for (const [label, text] of Object.entries(fill)) {
2301
- if (!here.some((r) => alnum(r.label).includes(alnum(label)))) continue;
2377
+ if (!here.some((r) => sweepHolds(r, label))) continue;
2302
2378
  try {
2303
- await runStep(deviceQuery, udid, {
2304
- action: step.paste === false ? 'type' : 'paste', into: label, text: String(text),
2379
+ // **Keep what the step said.** `paste` and `type` both read the field
2380
+ // back and say so — "unconfirmed", "reads empty", the quiet-field
2381
+ // caveat — and this threw all of it away and printed a bare
2382
+ // `filled "X"`. A field report called that out as the worst failure
2383
+ // mode an automation tool has: a silent no-op reported as a confirmed
2384
+ // action, caught only because the tester took a screenshot on a
2385
+ // hunch. The honesty existed one function down and stopped here.
2386
+ const body = String(text);
2387
+ // Explicit wins; otherwise length decides. See KEYBOARD_UP_TO.
2388
+ const viaPaste = step.paste === true
2389
+ || (step.paste !== false && body.length > KEYBOARD_UP_TO);
2390
+ const said = await runStep(deviceQuery, udid, {
2391
+ action: viaPaste ? 'paste' : 'type', into: label, text: body,
2305
2392
  }, ctx);
2306
- filled.push(`${JSON.stringify(label)} in section ${section + 1}`);
2393
+ // Anything the step qualified travels with the claim. A step that
2394
+ // confirmed the read-back says nothing extra, so a clean fill still
2395
+ // reads cleanly.
2396
+ const caveat = /unconfirmed|NOT CONFIRMED|reads empty|did not land|nothing was read back/i.test(said ?? '')
2397
+ ? ` — ${String(said).replace(/^(pasted|typed)[^[]*/i, '').trim() || said}`
2398
+ : '';
2399
+ filled.push(`${JSON.stringify(label)} in section ${section + 1}${caveat}`);
2307
2400
  } catch (err) {
2308
2401
  filled.push(`${JSON.stringify(label)} FAILED in section ${section + 1}: ${err.message.split('\n')[0].slice(0, 90)}`);
2309
2402
  }
@@ -2311,7 +2404,7 @@ async function sweep(deviceQuery, udid, step, ctx) {
2311
2404
  }
2312
2405
  }
2313
2406
 
2314
- if (wanted && [...seen.values()].some((r) => alnum(r.label).includes(alnum(wanted)))) break;
2407
+ if (wanted && [...seen.values()].some((r) => sweepHolds(r, wanted))) break;
2315
2408
  if (fill && !Object.keys(fill).length) break;
2316
2409
  prev = here;
2317
2410
  await scrollOne(deviceQuery, udid, 'down', ctx);
@@ -2319,7 +2412,7 @@ async function sweep(deviceQuery, udid, step, ctx) {
2319
2412
 
2320
2413
  const all = [...seen.values()];
2321
2414
  ctx.sweep = all;
2322
- const hits = wanted ? all.filter((r) => alnum(r.label).includes(alnum(wanted))) : [];
2415
+ const hits = wanted ? all.filter((r) => sweepHolds(r, wanted)) : [];
2323
2416
  const listed = (wanted ? hits : all).slice(0, 30)
2324
2417
  .map((r) => `[${r.section}] ${JSON.stringify(String(r.label).slice(0, 36))} @${r.x},${r.y}`);
2325
2418
  const unfilled = fill ? Object.keys(fill) : [];
@@ -2334,6 +2427,31 @@ async function sweep(deviceQuery, udid, step, ctx) {
2334
2427
  + (listed.length ? `: ${listed.join(', ')}` : '');
2335
2428
  }
2336
2429
 
2430
+ /**
2431
+ * A `waitFor` is satisfied, but the screen is still arriving — hold or go?
2432
+ *
2433
+ * The largest single loss in the 0.15.1 field report: `waitFor "Records"` was
2434
+ * satisfied by a count header reading `78 Records`, the next `tap` fired before
2435
+ * any row had rendered, and the batch aborted with **13 steps unattempted**.
2436
+ * The reporter had even warned us in their own `supervise` text that lists
2437
+ * there render a count header before rows — and the supervisor never got asked,
2438
+ * because from `waitFor`'s point of view the wait had succeeded.
2439
+ *
2440
+ * `stillFillingIn` already recognises exactly this ("a header promises 78
2441
+ * records and only 2 rows are here yet"), and it was wired only to a note on a
2442
+ * step result — advice for the model, costing a round trip to act on. Holding
2443
+ * locally costs milliseconds.
2444
+ *
2445
+ * **It can only ever delay, never fail.** When the budget runs out the wait
2446
+ * still succeeds, carrying what it saw. A `waitFor` that turned a present
2447
+ * target into an error would be the false negative this release just fixed
2448
+ * elsewhere, and a worse trade than the race it is guarding.
2449
+ */
2450
+ function holdForContent(found, limit) {
2451
+ if (Date.now() >= limit) return null;
2452
+ return stillFillingIn(found?.entry) ?? null;
2453
+ }
2454
+
2337
2455
  async function runStep(deviceQuery, udid, step, ctx) {
2338
2456
  switch (step.action) {
2339
2457
  case 'tap': {
@@ -2778,8 +2896,12 @@ async function runStep(deviceQuery, udid, step, ctx) {
2778
2896
  for (const one of alternatives) {
2779
2897
  try {
2780
2898
  const found = await api.locate(deviceQuery, one, { refresh: step.refresh !== false, options: ctx.options });
2899
+ const arriving = holdForContent(found, limit);
2900
+ if (arriving) { lastError = arriving; continue; }
2901
+ const late = stillFillingIn(found?.entry);
2781
2902
  return `${JSON.stringify(one)} appeared at ${found.target.x},${found.target.y}`
2782
- + ` (first of ${alternatives.length} awaited)`;
2903
+ + ` (first of ${alternatives.length} awaited)`
2904
+ + (late ? ` [but ${late} — waited out the timeout]` : '');
2783
2905
  } catch (err) {
2784
2906
  lastError = err.message;
2785
2907
  }
@@ -2819,7 +2941,17 @@ async function runStep(deviceQuery, udid, step, ctx) {
2819
2941
  // is the price, and it is the same trade: a read is cheaper than the
2820
2942
  // round trip a wrong verdict causes.
2821
2943
  const found = await api.locate(deviceQuery, query, { index: step.index, refresh: step.refresh !== false, options: ctx.options });
2822
- return `"${found.target.label}" appeared at ${found.target.x},${found.target.y}`;
2944
+ const arriving = holdForContent(found, limit);
2945
+ if (arriving) {
2946
+ // Still coming. Sleep a beat and look again — the loop re-resolves
2947
+ // fresh, so this cannot lock onto a stale reading.
2948
+ await api.waitFor(deviceQuery, { mode: 'stable', stableMs: 200, timeoutMs: 700, options: ctx.options })
2949
+ .catch(() => null);
2950
+ continue;
2951
+ }
2952
+ const late = stillFillingIn(found?.entry);
2953
+ return `"${found.target.label}" appeared at ${found.target.x},${found.target.y}`
2954
+ + (late ? ` [but ${late} — waited out the timeout; the screen may still be arriving]` : '');
2823
2955
  } catch (err) {
2824
2956
  lastError = err.message;
2825
2957
  // Waiting cannot make a thing unique.
package/src/cli.js CHANGED
@@ -1234,11 +1234,7 @@ async function main() {
1234
1234
  emit(flags, report, [
1235
1235
  flags.last ? `the last ${num(flags.last)} run(s) of each flow, of ${all.filter((f) => f.flow_name).length} named runs in the log` : null,
1236
1236
  'flow runs agent p50 human p50 HPI_time step_ratio turns esc',
1237
- ...report.flows.map((f) =>
1238
- `${f.flow.padEnd(24)} ${String(f.runs).padStart(5)} ${`${f.agent_ms.p50}ms`.padStart(9)} ` +
1239
- `${(f.human_median_ms ? `${f.human_median_ms}ms` : '—').padStart(9)} ` +
1240
- `${(f.hpi_time ?? '—').toString().padStart(8)} ${(f.step_ratio ?? '—').toString().padStart(10)} ` +
1241
- `${(f.model_turns ?? '—').toString().padStart(5)} ${String(f.escalations).padStart(3)}`),
1237
+ ...report.flows.map((f) => metrics.flowRow(f, { wide: true })),
1242
1238
  '',
1243
1239
  `HPI_accuracy ${report.overall.hpi_accuracy} (${report.overall.runs} runs, ` +
1244
1240
  `${report.overall.runs - runs.filter((r) => r.completed && !r.wrong_action_taken).length} not clean)`,
@@ -1246,6 +1242,14 @@ async function main() {
1246
1242
  ? `HPI_time and HPI need a human baseline — none of ${report.overall.flows_measured} measured flow(s) has one yet.`
1247
1243
  : `HPI_time ${report.overall.hpi_time} (harmonic mean over ${report.overall.flows_with_human_baseline} flow(s)), HPI ${report.overall.hpi}`,
1248
1244
  `step_ratio ${report.overall.step_ratio ?? '—'} (target ≤1.5), model turns per flow ${report.overall.model_turns_median ?? '—'}`,
1245
+ // Printed with what it costs, because the number alone means nothing
1246
+ // to a reader and the whole point is that it is the dominant term in
1247
+ // wall clock — far larger than anything inside simframe.
1248
+ `steps per model call ${report.overall.steps_per_call ?? '—'}`
1249
+ + (report.overall.steps_per_call
1250
+ ? ` — about ${(((20000 + 1700 * report.overall.steps_per_call) / report.overall.steps_per_call) / 1000).toFixed(1)}s`
1251
+ + ' per step end to end, of which simframe is ~1.7s. Raise this, not the engine.'
1252
+ : ''),
1249
1253
  flags.out ? `wrote ${flags.out}` : null,
1250
1254
  ]);
1251
1255
  return;
package/src/matching.js CHANGED
@@ -220,7 +220,48 @@ export function rank(targets, intent, { screen } = {}) {
220
220
  // them disappears into the ceiling.
221
221
  scored.push({ target: t, score, reasons });
222
222
  }
223
- return scored.sort((a, b) => b.score - a.score);
223
+ scored.sort((a, b) => b.score - a.score);
224
+
225
+ // **A distinctive fragment of one long name, when nothing else came close.**
226
+ //
227
+ // Reported from the field: `waitFor "6322594"` gave up after 20 s on a screen
228
+ // whose own "Visible:" list printed `Work Order #6322594`. The reporter's
229
+ // guess was that `#` was being treated as significant, or that the matcher
230
+ // was anchored. It is neither — the number scores **0.287** against the 0.45
231
+ // floor, because the substring branch in `nameScore` scales by how much of
232
+ // the name the query covers, and seven digits are 41% of that label.
233
+ //
234
+ // That scaling is right and stays: it is what stops "back" beating a real
235
+ // back button from inside a long list row. But its purpose is to resolve
236
+ // *competition*, and when there is no competition it is charging a penalty
237
+ // for a risk that does not exist. So the promotion fires only when **nothing
238
+ // reached the floor** and **exactly one element** contains the query. The
239
+ // "back" case is untouched, because a screen with a back button has at least
240
+ // two names containing "back" and this never runs.
241
+ //
242
+ // The score lands just above the floor, not at 1: it is an act of
243
+ // desperation, not a confident match, and the reason says so — so an
244
+ // ambiguity check downstream still has something honest to weigh.
245
+ if (!scored.some((c) => c.score >= MINIMUM_SCORE)) {
246
+ const q = norm(bare) || norm(intent);
247
+ const holds = (t) => [t.label, t.identifier, ...(t.aliases ?? [])]
248
+ .filter(Boolean)
249
+ .some((n) => norm(n).includes(q) && norm(n) !== q);
250
+ const only = q.length >= 3 ? visible.filter(holds) : [];
251
+ if (only.length === 1) {
252
+ const t = only[0];
253
+ const existing = scored.find((c) => c.target === t);
254
+ const reason = `the only element on this screen containing "${intent}"`;
255
+ if (existing) {
256
+ existing.score = MINIMUM_SCORE + 0.01;
257
+ existing.reasons.push(reason);
258
+ } else {
259
+ scored.push({ target: t, score: MINIMUM_SCORE + 0.01, reasons: [reason] });
260
+ }
261
+ scored.sort((a, b) => b.score - a.score);
262
+ }
263
+ }
264
+ return scored;
224
265
  }
225
266
 
226
267
  /** How close two candidates may be before the answer counts as ambiguous. */
package/src/mcp.js CHANGED
@@ -13,6 +13,7 @@ import { REGION_COLS, REGION_ROWS, regionMap } from './analyze.js';
13
13
  import * as actions from './actions.js';
14
14
  import * as api from './index.js';
15
15
  import * as input from './input.js';
16
+ import * as supervisor from './supervisor.js';
16
17
  import * as metrics from './metrics.js';
17
18
  import * as navigate from './navigate.js';
18
19
  import { bootedDevices, listDevices, permissionServices, resolveDevice } from './platform/index.js';
@@ -1046,12 +1047,41 @@ async function doScript(target, args, options) {
1046
1047
  lines.push(`supervisor at step ${s_.index}: ${s_.decision} — ${s_.outcome}`
1047
1048
  + (s_.reason ? ` (it said: "${s_.reason}")` : ''));
1048
1049
  }
1050
+ // **A brief with nobody to read it says so.**
1051
+ //
1052
+ // `supervise` is standing guidance for a supervisor; *which* supervisor comes
1053
+ // from `options.supervisor` or SIMFRAME_SUPERVISOR, and with neither set the
1054
+ // brief is discarded. The code even documented that as "always safe" — safe
1055
+ // and silent, which is the trap. Two independent field reports hit it: both
1056
+ // passed a brief on every call, neither ever saw a verdict, and one had
1057
+ // written guidance describing the exact race that then aborted their batch.
1058
+ // Their words: "as written, `supervise` is a prompt I can't observe the
1059
+ // effect of."
1060
+ //
1061
+ // Said only when a brief was actually passed, so it is never noise, and it
1062
+ // names the remedy rather than the condition.
1063
+ if (args.supervise && !supervisor.requested(options)) {
1064
+ lines.push('supervisor: none enabled, so the "supervise" brief was not consulted'
1065
+ + ' — set supervisor (per call) or SIMFRAME_SUPERVISOR to enable one.');
1066
+ } else if (args.supervise && !(res.supervisions ?? []).length) {
1067
+ lines.push(`supervisor: ${supervisor.requested(options)} enabled, never consulted`
1068
+ + ' — no step failed in a way that asks for a ruling.');
1069
+ }
1049
1070
  if (args.saveAs) {
1050
1071
  const saved = navigate.saveFlow(res.device.udid, args.saveAs, res);
1072
+ // Say what to do about it. The refusal this replaced named a condition and
1073
+ // no remedy, and the reporter read it as "recording a flow is impossible"
1074
+ // — which it was.
1075
+ const how = saved.provisional
1076
+ ? ' — PROVISIONAL, because this was the first traversal and nothing had been'
1077
+ + ' seen before to compare against. Replay it once with sim_flow_run and it'
1078
+ + ' is confirmed. A replay costs no model calls.'
1079
+ : ' — confirmed; replay with sim_flow_run for zero model calls';
1051
1080
  lines.push(
1052
1081
  saved.ok
1053
- ? `saved as flow "${saved.name}" (${saved.steps} steps) — replay with sim_flow_run`
1054
- : `NOT saved as "${args.saveAs}": ${saved.reason}${saved.verdicts ? ` (${saved.verdicts.join(', ')})` : ''}`,
1082
+ ? `saved as flow "${saved.name}" (${saved.steps} steps)${how}`
1083
+ : `NOT saved as "${args.saveAs}": ${saved.reason}`
1084
+ + `${saved.verdicts ? ` (${saved.verdicts.join(', ')})` : ''}`,
1055
1085
  );
1056
1086
  }
1057
1087
  const escalated = (res.results ?? []).some((r) => metrics.ESCALATING_VERDICTS.has(r.verification?.verdict));
package/src/metrics.js CHANGED
@@ -601,6 +601,30 @@ export function harmonicMean(xs) {
601
601
  * a human baseline. A flow with no human baseline gets no HPI_time — reported
602
602
  * as null, never as 1.0, because a missing denominator is not parity.
603
603
  */
604
+ /**
605
+ * One flow's row in the HPI table, shared by `simframe hpi` and the bench
606
+ * script because they had the same row duplicated byte for byte.
607
+ *
608
+ * It lives here, next to the report it renders, for a reason the v0.15.0 tag
609
+ * paid for: making `agent_ms` null for a flow that timed nothing was correct,
610
+ * and both printers dereferenced `.p50` on it. `bench` died with exit 1 and no
611
+ * hpi.json, and `simframe hpi` would have done the same for any user whose
612
+ * flow never completed. The metric had a test; nothing rendered it. A format
613
+ * duplicated in two files is a format that gets fixed in one.
614
+ *
615
+ * `—` means "not measured", never zero.
616
+ */
617
+ export function flowRow(f, { wide = false } = {}) {
618
+ const cell = (v, unit = '') => (v === null || v === undefined ? '—' : `${v}${unit}`);
619
+ const row = `${f.flow.padEnd(24)} ${String(f.runs).padStart(5)} `
620
+ + `${cell(f.agent_ms?.p50, 'ms').padStart(9)} `
621
+ + `${cell(f.human_median_ms, 'ms').padStart(9)} `
622
+ + `${cell(f.hpi_time).padStart(8)} ${cell(f.step_ratio).padStart(10)}`;
623
+ return wide
624
+ ? `${row} ${cell(f.model_turns).padStart(5)} ${String(f.escalations).padStart(3)}`
625
+ : row;
626
+ }
627
+
604
628
  export function hpi({ flows, baselines = {} }) {
605
629
  const byName = new Map();
606
630
  for (const f of flows) {
@@ -630,7 +654,12 @@ export function hpi({ flows, baselines = {} }) {
630
654
  const agent = quartiles(finished.map((r) => r.wall_time_ms));
631
655
  const human = baselines[name]?.wall_time_ms ?? null;
632
656
  const humanMedian = human?.p50 ?? null;
633
- const stepRatios = runs.map((r) => r.step_ratio).filter((x) => Number.isFinite(x));
657
+ // Completed runs only, for the same reason the time is: a run that stops
658
+ // three steps in reports a LOW step ratio, so breakage flattered this
659
+ // number too. The runner's suite, with 3 of 14 runs completing, reported
660
+ // `step_ratio 0.375` — "the agent uses a third of the human's steps" about
661
+ // flows that mostly never arrived.
662
+ const stepRatios = finished.map((r) => r.step_ratio).filter((x) => Number.isFinite(x));
634
663
  return {
635
664
  flow: name,
636
665
  runs: runs.length,
@@ -643,6 +672,22 @@ export function hpi({ flows, baselines = {} }) {
643
672
  wrong_action: runs.filter((r) => r.wrong_action_taken).length,
644
673
  escalations: runs.reduce((acc, r) => acc + (r.escalation_count ?? 0), 0),
645
674
  model_turns: median(runs.map((r) => r.model_turns)),
675
+ // **The latency lever, and it was already in the log unsurfaced.**
676
+ //
677
+ // Wall-clock per step is dominated by the model round trip, not by
678
+ // simframe: measured in the field, ~20 s of agent latency against ~1.7 s
679
+ // of simframe work per step. So per-step time is
680
+ // `(20s + 1.7s x n) / n` for n steps in one call — 21.7 s at n=1, 11.7 s
681
+ // at n=2, 6.7 s at n=4. Nothing about making simframe faster moves that;
682
+ // only n does.
683
+ //
684
+ // Which makes this the number to watch, and every hard-fail that drops a
685
+ // caller back to single-stepping a latency regression. Over 186 recorded
686
+ // runs on the author's device the median was 2.0 and the 25th percentile
687
+ // 1.0 — the p25 tail is recovery, and it is where the time goes.
688
+ steps_per_call: median(runs
689
+ .filter((r) => Number.isFinite(r.steps_taken) && r.model_turns > 0)
690
+ .map((r) => r.steps_taken / r.model_turns)),
646
691
  };
647
692
  }).sort((a, b) => a.flow.localeCompare(b.flow));
648
693
 
@@ -660,6 +705,9 @@ export function hpi({ flows, baselines = {} }) {
660
705
  hpi_accuracy: accuracy,
661
706
  hpi_time: hpiTime == null ? null : Number(hpiTime.toFixed(3)),
662
707
  hpi: hpiTime == null || accuracy == null ? null : Number((accuracy * hpiTime).toFixed(3)),
708
+ steps_per_call: median(flows
709
+ .filter((f) => Number.isFinite(f.steps_taken) && f.model_turns > 0)
710
+ .map((f) => f.steps_taken / f.model_turns)),
663
711
  step_ratio: median(perFlow.map((f) => f.step_ratio)),
664
712
  model_turns_median: median(perFlow.map((f) => f.model_turns)),
665
713
  },
@@ -775,8 +823,34 @@ export function breakdown(records, { session = null, flow = null } = {}) {
775
823
  * than one, and accuracy stays strict at any drop at all. Numbers in
776
824
  * docs/BENCHMARKS.md under "What the gate is set to, and why".
777
825
  */
826
+ /**
827
+ * The band HPI_time is *reported* against. It no longer fails a build.
828
+ *
829
+ * It was a gate, at 25% here and written down as 10% in CLAUDE.md — and the
830
+ * measurement cannot support either. `HPI_time` is `human_p50 / agent_p50`
831
+ * where the human was recorded once, on a laptop, and the agent is measured
832
+ * wherever CI happens to run: a hosted runner put the same flows at 37 s and
833
+ * 64 s against 11.5 s and 11.7 s on that laptop, so the ratio mixes the code's
834
+ * speed with the host's. Even on one machine, identical code spans 0.406-0.558
835
+ * across device conditions — a 37% spread. A band inside that can only be
836
+ * silent or wrong, and it was silent: items 148, 152, 154 and 169 all shipped
837
+ * without it firing.
838
+ *
839
+ * So time is a trend, per host, and accuracy is the gate. Accuracy asks
840
+ * whether the agent reached the destination and whether it tapped the wrong
841
+ * thing — properties of the code, not of the machine.
842
+ */
778
843
  export const TIME_REGRESSION = 0.25;
779
844
 
845
+ /**
846
+ * The most steps-per-human-step we will accept. From the research: an agent
847
+ * taking half again as many actions as a person is wandering.
848
+ *
849
+ * An absolute target rather than a regression, deliberately — it is a ratio of
850
+ * two step counts, so unlike the time it does not change with the host.
851
+ */
852
+ export const STEP_RATIO_CEILING = 1.5;
853
+
780
854
  /**
781
855
  * The HPI_time a gate should compare: the median across passes when a report
782
856
  * has them, and the single measurement when it does not — so a baseline
@@ -792,7 +866,7 @@ export const gateTime = (o) => o?.hpi_time_median_of_passes ?? o?.hpi_time ?? nu
792
866
  * fingerprint eval both shipped unable to fail, and both looked exactly like
793
867
  * this. Returns the reasons it should fail — empty means pass.
794
868
  */
795
- export function gateAgainst(baseline, measured, { timeRegression = TIME_REGRESSION } = {}) {
869
+ export function gateAgainst(baseline, measured, { stepRatioCeiling = STEP_RATIO_CEILING } = {}) {
796
870
  const base = baseline?.overall ?? {};
797
871
  const now = measured?.overall ?? measured ?? {};
798
872
  const baseTime = gateTime(base);
@@ -801,19 +875,44 @@ export function gateAgainst(baseline, measured, { timeRegression = TIME_REGRESSI
801
875
  if (base.hpi_accuracy != null && now.hpi_accuracy != null && now.hpi_accuracy < base.hpi_accuracy) {
802
876
  failures.push(`HPI_accuracy dropped: ${now.hpi_accuracy} < ${base.hpi_accuracy} (any drop fails)`);
803
877
  }
804
- if (baseTime != null && nowTime != null && nowTime < baseTime * (1 - timeRegression)) {
878
+ // Steps, which the host cannot inflate: a ratio of two step counts.
879
+ if (now.step_ratio != null && now.step_ratio > stepRatioCeiling) {
805
880
  failures.push(
806
- `HPI_time regressed >${timeRegression * 100}%: ${nowTime} < ${(baseTime * (1 - timeRegression)).toFixed(3)}`,
881
+ `step_ratio ${now.step_ratio} is above the ${stepRatioCeiling} ceiling — the agent is taking more actions than a person`,
807
882
  );
808
883
  }
809
- // A checkout missing the human baseline the committed number was computed
810
- // against would otherwise pass by having nothing to compare.
884
+ // **HPI_time does not fail a build.** See TIME_REGRESSION. It is reported
885
+ // with its band so a trend is visible and a big move is obvious to a human,
886
+ // and `timeTrend` below is what prints it.
887
+ //
888
+ // A checkout missing the human baseline is still a failure, because it is a
889
+ // broken measurement rather than a slow one — a report with no time at all
890
+ // would otherwise pass by having nothing to say.
811
891
  if (baseTime != null && nowTime == null) {
812
892
  failures.push('HPI_time is null but the baseline has one — the human baseline it needs is missing from this checkout');
813
893
  }
814
894
  return failures;
815
895
  }
816
896
 
897
+ /**
898
+ * HPI_time as a line to read, never a verdict.
899
+ *
900
+ * Says how far it moved and whether that is outside the reporting band, and
901
+ * says plainly that it is not a gate — a CI line that looks like a threshold
902
+ * gets read as one.
903
+ */
904
+ export function timeTrend(baseline, measured, { timeRegression = TIME_REGRESSION } = {}) {
905
+ const baseTime = gateTime(baseline?.overall ?? {});
906
+ const nowTime = gateTime(measured?.overall ?? measured ?? {});
907
+ if (baseTime == null || nowTime == null) return 'HPI_time — not comparable (one side has no measurement)';
908
+ const delta = (nowTime - baseTime) / baseTime;
909
+ const pct = `${delta >= 0 ? '+' : ''}${(delta * 100).toFixed(1)}%`;
910
+ const wide = Math.abs(delta) > timeRegression;
911
+ return `HPI_time ${baseTime} -> ${nowTime} (${pct})`
912
+ + `${wide ? ` — outside the ${timeRegression * 100}% reporting band, worth a look` : ''}`
913
+ + ' — reported, not gated: this ratio mixes the code with the host';
914
+ }
915
+
817
916
  /** A flow id that sorts by time and is short enough to read in a log. */
818
917
  export function newFlowId() {
819
918
  return `${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 8)}`;
package/src/navigate.js CHANGED
@@ -139,18 +139,52 @@ export function knownScreens(udid) {
139
139
  return graph.allNodes(udid).map((n) => ({ name: graph.describe(n) ?? n.hash.slice(0, 8), hash: n.hash.slice(0, 8), edges: n.edges.length }));
140
140
  }
141
141
 
142
+ /**
143
+ * A verdict that is evidence *against* a step, as opposed to no evidence yet.
144
+ *
145
+ * This distinction is the whole of the fix below. `unexpected-*` means the step
146
+ * did something other than what memory predicted — a real objection. `unverified`
147
+ * means this edge has never been walked before, which on a first traversal is
148
+ * true of every step by definition and says nothing about whether it worked.
149
+ */
150
+ const CONTRADICTED = /^unexpected/;
151
+
142
152
  /**
143
153
  * Save a flow.
144
154
  *
145
- * Only flows that verified end to end are worth saving: a flow with an
146
- * unverified step in it is a recording of something that may not have worked,
147
- * and replaying it faithfully reproduces the doubt.
155
+ * The rule was "only flows that verified end to end", and it was right about
156
+ * replay safety and wrong about arithmetic: **a first successful traversal is
157
+ * all-`unverified` by construction**, so nothing could ever be recorded, so
158
+ * `sim_flow_run` was unreachable. A field report hit it on a clean 10-of-10
159
+ * batch — *"I never obtained a saved flow, so `sim_flow_run` went untested"* —
160
+ * and it matters far more than its severity suggests: a replayed flow costs
161
+ * **zero model calls**, which is the only path to human-level wall clock. A
162
+ * gate nobody can pass protects nothing and blocks the fastest thing here.
163
+ *
164
+ * So the refusal now needs evidence against a step, not the absence of evidence
165
+ * for it. A flow that ran every step with nothing contradicted saves as
166
+ * **provisional**, and one clean replay promotes it — a bootstrap in two runs,
167
+ * where the second run is the confirmation and is useful anyway.
168
+ *
169
+ * Still refused, because these are real objections: any `unexpected-*` verdict,
170
+ * and a flow that did not reach its own last step.
148
171
  */
149
172
  export function saveFlow(udid, name, script, { force = false } = {}) {
150
173
  const verdicts = (script.results ?? []).map((r) => r.verification?.verdict);
151
- if (!force && verdicts.some((v) => v && v !== 'ok')) {
152
- return { ok: false, reason: 'unverified-steps', verdicts };
174
+ const contradicted = verdicts.filter((v) => v && CONTRADICTED.test(v));
175
+ const ranAll = script.ranSteps == null
176
+ || script.steps == null
177
+ || script.ranSteps >= (script.steps?.length ?? 0);
178
+ if (!force && contradicted.length) {
179
+ return { ok: false, reason: 'contradicted-steps', verdicts };
153
180
  }
181
+ if (!force && !ranAll) {
182
+ return { ok: false, reason: 'incomplete-run', verdicts };
183
+ }
184
+ // Provisional whenever any step lacked a confirming verdict. Recorded on the
185
+ // flow rather than inferred later, so a replay can promote it and a listing
186
+ // can say which flows are still on their first observation.
187
+ const provisional = verdicts.some((v) => !v || v !== 'ok');
154
188
  const dir = flowDir(udid);
155
189
  fs.mkdirSync(dir, { recursive: true });
156
190
  const body = {
@@ -158,9 +192,28 @@ export function saveFlow(udid, name, script, { force = false } = {}) {
158
192
  savedAt: Date.now(),
159
193
  steps: script.steps ?? (script.results ?? []).map((r) => r.step).filter(Boolean),
160
194
  startScreen: script.startScreen ?? null,
195
+ ...(provisional ? { provisional: verdicts.filter(Boolean) } : {}),
161
196
  };
162
197
  store.writeAtomic(path.join(dir, `${encodeURIComponent(name)}.json`), JSON.stringify(body, null, 2));
163
- return { ok: true, name, steps: body.steps.length };
198
+ return { ok: true, name, steps: body.steps.length, provisional };
199
+ }
200
+
201
+ /**
202
+ * A provisional flow that replays cleanly becomes a confirmed one.
203
+ *
204
+ * The second half of the bootstrap. Without it "provisional" would be a label
205
+ * that never comes off, and the honest state of a flow that has now worked
206
+ * twice is "confirmed".
207
+ */
208
+ export function confirmFlow(udid, name) {
209
+ const flow = loadFlow(udid, name);
210
+ if (!flow?.provisional) return false;
211
+ const { provisional, ...rest } = flow;
212
+ store.writeAtomic(
213
+ path.join(flowDir(udid), `${encodeURIComponent(name)}.json`),
214
+ JSON.stringify({ ...rest, confirmedAt: Date.now() }, null, 2),
215
+ );
216
+ return true;
164
217
  }
165
218
 
166
219
  export function loadFlow(udid, name) {
@@ -174,7 +227,18 @@ export function listFlows(udid) {
174
227
  .filter((f) => f.endsWith('.json'))
175
228
  .map((f) => store.readJson(path.join(dir, f)))
176
229
  .filter(Boolean)
177
- .map((f) => ({ name: f.name, steps: f.steps?.length ?? 0, savedAt: f.savedAt }));
230
+ // `provisional` travels with the listing, for two reasons. A caller should
231
+ // be able to see which of their flows are still on a first observation
232
+ // without opening the file — and without it the integration check that
233
+ // asserts promotion reads `undefined`, which is always falsy, so the check
234
+ // would pass whether or not anything was promoted. A vacuous check is worse
235
+ // than no check.
236
+ .map((f) => ({
237
+ name: f.name,
238
+ steps: f.steps?.length ?? 0,
239
+ savedAt: f.savedAt,
240
+ ...(f.provisional ? { provisional: true } : {}),
241
+ }));
178
242
  }
179
243
 
180
244
  export async function runFlow(deviceQuery, name, { options, ...runOptions } = {}) {
@@ -192,5 +256,11 @@ export async function runFlow(deviceQuery, name, { options, ...runOptions } = {}
192
256
  minSteps: flow.minSteps ?? null,
193
257
  ...runOptions,
194
258
  });
195
- return { ok: result.ranSteps === flow.steps.length, name, ...result };
259
+ const ok = result.ranSteps === flow.steps.length;
260
+ // A clean replay is the confirmation a first traversal could not give.
261
+ // Only when nothing was contradicted — a replay that ran to the end while
262
+ // objecting to a step is not a promotion.
263
+ const objected = (result.results ?? []).some((r) => CONTRADICTED.test(r.verification?.verdict ?? ''));
264
+ const promoted = ok && !objected ? confirmFlow(device.udid, name) : false;
265
+ return { ok, name, ...result, ...(promoted ? { promoted: true } : {}) };
196
266
  }