simframe 0.10.0 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/README.md +181 -2
  2. package/data/vocabulary/en.json +148 -0
  3. package/native/ocr.swift +13 -1
  4. package/native/rank.swift +87 -0
  5. package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +72 -4
  6. package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
  7. package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
  8. package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +19 -0
  9. package/native/simframed/Sources/simframed/main.swift +33 -3
  10. package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +41 -0
  11. package/native/supervise.swift +216 -0
  12. package/package.json +4 -1
  13. package/scripts/check-package.mjs +22 -2
  14. package/scripts/check-private.mjs +9 -0
  15. package/scripts/ci-integration-local.sh +79 -0
  16. package/scripts/ci-memory.mjs +104 -20
  17. package/scripts/collect-rulings.mjs +312 -0
  18. package/scripts/eval-fingerprint.mjs +100 -23
  19. package/scripts/eval-perception.mjs +33 -0
  20. package/scripts/phase17-corpus.mjs +176 -0
  21. package/scripts/probe-network.mjs +118 -0
  22. package/scripts/soak-capture.mjs +72 -0
  23. package/skills/simframe/SKILL.md +257 -7
  24. package/src/actions.js +1791 -44
  25. package/src/cli.js +216 -14
  26. package/src/control.js +1 -0
  27. package/src/fingerprint.js +43 -1
  28. package/src/graph.js +136 -7
  29. package/src/index.js +252 -14
  30. package/src/input.js +66 -3
  31. package/src/localhelper.js +161 -0
  32. package/src/matching.js +64 -1
  33. package/src/mcp.js +333 -32
  34. package/src/metrics.js +148 -3
  35. package/src/ocr.js +18 -1
  36. package/src/planner.js +195 -0
  37. package/src/platform/android.js +25 -1
  38. package/src/platform/index.js +11 -1
  39. package/src/platform/ios.js +25 -1
  40. package/src/png.js +26 -0
  41. package/src/refs.js +51 -8
  42. package/src/regions.js +215 -1
  43. package/src/screenmap.js +89 -9
  44. package/src/supervisor.js +161 -0
  45. package/src/view.js +375 -11
  46. package/src/vocabulary.js +134 -0
  47. package/src/wrote.js +136 -0
package/src/regions.js CHANGED
@@ -24,6 +24,27 @@ export const REGIONS = [
24
24
  'content',
25
25
  ];
26
26
 
27
+ /**
28
+ * Regions the screen map will not offer as something to act on.
29
+ *
30
+ * The status bar says the time and the battery level. It is on every screen, it
31
+ * is never what anybody wants to tap, and it costs a row every time — so
32
+ * `sim_ui` hides it.
33
+ *
34
+ * It lives here rather than in `view.js`, where it started, because it is not
35
+ * only a presentation rule. A target the map refuses to *show* must also be a
36
+ * target nothing may resolve onto *behind the caller's back*, and the case that
37
+ * proved it was exactly that: a stale `#1` numbered "Reminders" in Reminders,
38
+ * re-resolved in Contacts onto the status-bar back-to-app breadcrumb
39
+ * "• Reminders", scored 0.64 and handed back a tap point at (47,40) — a place
40
+ * the map would never have put in front of anybody. One rule, one home, both
41
+ * readers.
42
+ */
43
+ const UNOFFERED_REGIONS = new Set(['status-bar']);
44
+
45
+ /** Would the screen map offer a target in this region? */
46
+ export const offerable = (region) => !UNOFFERED_REGIONS.has(region);
47
+
27
48
  /**
28
49
  * The status bar stays positional, and deliberately.
29
50
  *
@@ -36,6 +57,8 @@ const STATUS_BAR_FRACTION = 0.065;
36
57
 
37
58
  /** Keyboards occupy the bottom of the screen and are unusually tall. */
38
59
  const KEYBOARD_MIN_FRACTION = 0.28;
60
+ /** Most of a keyboard is keys. Below this it is a list that happens to be small. */
61
+ const KEYBOARD_MIN_KEYISH = 0.6;
39
62
 
40
63
  /** Chrome is short. A 90pt list cell is not a tab item however low it sits. */
41
64
  const CHROME_MAX_HEIGHT_FRACTION = 0.075;
@@ -61,6 +84,46 @@ const BOTTOM_CHROME_LIMIT = 0.82;
61
84
  * screen sets its own scale.
62
85
  */
63
86
  const MIN_BOUNDARY_GAP_PT = 10;
87
+ /**
88
+ * How wide a lone row may be and still be a screen's title rather than its
89
+ * first paragraph.
90
+ *
91
+ * 0.6 of the screen, measured: the Settings root's large title is 133 pt of
92
+ * 402 (0.33), while Reminders' empty-state heading "Welcome to Reminders" is
93
+ * 326 pt (0.81) and is content — it describes the screen instead of naming it.
94
+ * A title is a name and names are short.
95
+ */
96
+ const LARGE_TITLE_MAX_WIDTH_FRACTION = 0.6;
97
+ /**
98
+ * How far below the status bar a large title can start.
99
+ *
100
+ * iOS draws one at a **system** offset, not an app-chosen one, so this is a
101
+ * bound on a platform constant rather than a tuned threshold. Measured on the
102
+ * bench device: the Settings root's title starts 63-79 pt below the status bar
103
+ * depending on which sensor reports its box, while example.com's `<h1>` — page
104
+ * *content* that merely happens to be the first row, because Safari on iOS puts
105
+ * its chrome at the bottom — starts **122 pt** down.
106
+ *
107
+ * Without this bound the rule promoted that `<h1>` to chrome and "example
108
+ * domain" entered the screen's identity. Pulling page content into identity is
109
+ * the exact failure this module has been bitten by twice (a phantom keyboard,
110
+ * and content that merely fell into a band), so the bound is not optional.
111
+ */
112
+ const LARGE_TITLE_MAX_INSET_PT = 96;
113
+ /**
114
+ * And the least it can be, before it is just the next row.
115
+ *
116
+ * Absolute, like the maximum, and for the same reason: the inset is drawn by
117
+ * the system, so it is not a function of what the screen contains. The first
118
+ * version tested it against the screen's *median row gap* — and the testbed
119
+ * caught that on its first day, with two screens of the same app. A list of 24
120
+ * rows has a median gap of 0 and the rule fired; a list of 4 rows above a tab
121
+ * bar has a median gap of **414**, because the empty area counts as a gap, and
122
+ * the rule did not. Same title, same inset of 62.9pt, opposite answers — so one
123
+ * screen had a name in its identity and the other did not, and the graph then
124
+ * called them the same screen at 0.50 similarity.
125
+ */
126
+ const LARGE_TITLE_MIN_INSET_PT = 24;
64
127
  const BOUNDARY_GAP_FACTOR = 1.9;
65
128
 
66
129
  /** A tab bar is several things spread across the width, not one thing at the bottom. */
@@ -175,6 +238,50 @@ export function bands(elements, screen) {
175
238
  }
176
239
  }
177
240
 
241
+ // --- a large title, which has no gap under it to be found by.
242
+ //
243
+ // The loop above identifies top chrome by the whitespace *beneath* it, and an
244
+ // iOS large title is drawn tight against the content it heads: measured on
245
+ // the Settings root, 79 pt of inset above it and **5.3 pt** below, against a
246
+ // bar of `max(10, typical * 1.9)` = 66.5. No threshold reaches that, so the
247
+ // title fell into `content` and was discarded as content — leaving the screen
248
+ // with **no name at all** in its fingerprint, in either sensor mode.
249
+ //
250
+ // That is not cosmetic. Chrome labels are the only text identity keeps, and
251
+ // `fingerprint.js` names the consequence: two list screens with identical
252
+ // structure differ by their title and nothing else says so. A nameless screen
253
+ // is pure geometry, and on a hosted runner two sparse nameless readings
254
+ // matched exactly — one screen's hash for two screens.
255
+ //
256
+ // So it is found by the inset *above* it instead, which is the half iOS does
257
+ // provide. A large title sits alone, narrow, high, under a generous gap; a
258
+ // compact bar is the mirror image of that (20-33 pt above, 118-134 below) and
259
+ // is already caught by the loop. Deliberately not keyed on the `Heading`
260
+ // role: OCR has no roles, and the reading that actually collided was
261
+ // OCR-only, so a role test would work only in the case that does not fail.
262
+ //
263
+ // Measured across all 17 perception fixtures before being written here: it
264
+ // changes exactly one of them, the Settings root.
265
+ if (!navBarBottom) {
266
+ const first = rows.findIndex((r) => r.top >= statusBarBottom - 1);
267
+ const row = first >= 0 ? rows[first] : null;
268
+ if (
269
+ row
270
+ && first + 1 < rows.length
271
+ && row.items.length === 1
272
+ && (row.items[0].frame?.width ?? 0) <= screen.width * LARGE_TITLE_MAX_WIDTH_FRACTION
273
+ && row.bottom <= screen.height * TOP_CHROME_LIMIT
274
+ // A real separation from the status bar, but a system-sized one: far
275
+ // enough to be an inset, near enough to still be the app's own title.
276
+ // Both bounds absolute — see LARGE_TITLE_MIN_INSET_PT for what keying the
277
+ // lower one on the screen's own row rhythm cost.
278
+ && row.top - statusBarBottom >= LARGE_TITLE_MIN_INSET_PT
279
+ && row.top - statusBarBottom <= LARGE_TITLE_MAX_INSET_PT
280
+ ) {
281
+ navBarBottom = row.bottom;
282
+ }
283
+ }
284
+
178
285
  // --- bottom chrome. One row: a tab bar is one row by construction, and the
179
286
  // gap above it is what separates it from the list it floats over.
180
287
  let tabBarTop = Infinity;
@@ -243,6 +350,50 @@ export function navSlot(frame, screen) {
243
350
  * common case and must stay cheap. This was the first band derived from the
244
351
  * elements rather than from a fraction, and it is the model the rest now follow.
245
352
  */
353
+ /**
354
+ * Whether an element is shaped like a key rather than like content.
355
+ *
356
+ * The canonical version of this test, because two places need it and getting
357
+ * them out of step is what produced the bug below. A key is finger-sized and
358
+ * says almost nothing: a single character, a short named key, or nothing at
359
+ * all. A row of content is wider, or carries words.
360
+ */
361
+ export const KEY_MAX_WIDTH = 120;
362
+
363
+ const NAMED_KEY = /^(space|return|enter|shift|delete|backspace|done|globe|dictate|emoji|caps ?lock|number|numbers|symbols|letters|more|search|go|send|join|route|abc|123)$/i;
364
+
365
+ export function looksLikeKey(t) {
366
+ if (/^key$/i.test(String(t?.type ?? ''))) return true;
367
+ const width = t?.frame?.width;
368
+ if (Number.isFinite(width) && width > KEY_MAX_WIDTH) return false;
369
+ const label = String(t?.label ?? '').trim();
370
+ if (!label) return true;
371
+ if (label.length <= 2) return true;
372
+ return NAMED_KEY.test(label);
373
+ }
374
+
375
+ /**
376
+ * Where the software keyboard starts, or null.
377
+ *
378
+ * Size and uniformity alone were not enough, and the failure was expensive. A
379
+ * read-only summary screen stacks a dozen short text rows of near-identical
380
+ * height in the bottom half — which satisfied every test here, so a keyboard
381
+ * was detected on a screen that had none.
382
+ *
383
+ * That mattered far beyond a mislabelled band, because `fingerprint.tokens`
384
+ * discards everything below `keyboardTop`. A phantom keyboard therefore
385
+ * deleted the screen's entire content from its own identity, leaving only
386
+ * chrome — so a wizard's form step and its read-only review screen, which
387
+ * share a nav title and a step indicator, **collapsed onto one hash**. From
388
+ * there: the graph offered one screen's remembered controls on the other (three
389
+ * absent controls, one of them beside a button that submits for real), and
390
+ * `locate` resolved against the wrong screen's stored element list, which is
391
+ * why `assert` insisted a string was absent while the map printed it four lines
392
+ * below. One phantom, three findings.
393
+ *
394
+ * So the test is now what a keyboard actually is: keys. A dozen small uniform
395
+ * boxes are a keyboard only if most of them are key-shaped.
396
+ */
246
397
  export function detectKeyboardTop(elements, screen) {
247
398
  if (!screen?.height || elements.length < 12) return null;
248
399
  const threshold = screen.height * (1 - KEYBOARD_MIN_FRACTION);
@@ -253,7 +404,70 @@ export function detectKeyboardTop(elements, screen) {
253
404
  // Keys are small and uniform; a list of cells down there is not.
254
405
  const uniform = heights.filter((h) => Math.abs(h - median) <= Math.max(3, median * 0.4)).length;
255
406
  if (uniform / low.length < 0.7 || median > screen.height * 0.07) return null;
256
- return Math.min(...low.map((e) => e.frame.y));
407
+ // And they are keys. Uniformity says "a grid of something"; this says of what.
408
+ const keyish = low.filter(looksLikeKey).length;
409
+ if (keyish / low.length < KEYBOARD_MIN_KEYISH) return null;
410
+ return extendKeyboardUp(elements, Math.min(...low.map((e) => e.frame.y)), median);
411
+ }
412
+
413
+ /**
414
+ * Walk the boundary up through rows that are still keys.
415
+ *
416
+ * `KEYBOARD_MIN_FRACTION` is a **detection window**, not the keyboard's height,
417
+ * and using its edge as the boundary cut the keyboard's own top row off.
418
+ * Measured on a recorded iPhone 17 Pro screen with the software keyboard up: the
419
+ * window starts at y=629, the `q`–`p` row's frame top is **590**, so that entire
420
+ * row was excluded and the boundary landed on the `a` row at 644 — ten keys
421
+ * reported as page content, in the same map that said `keyboard up`.
422
+ *
423
+ * Widening the window instead would be the wrong fix: 0.28 of the screen is
424
+ * deliberately conservative so a list of short rows at the bottom of a page
425
+ * cannot be mistaken for a keyboard, and a real keyboard is nearer 0.38. So the
426
+ * window still *decides*, and this extends the boundary only while the rows
427
+ * above keep being key-shaped — which page content is not.
428
+ *
429
+ * The concrete cost of not having this: a sweep gesture aimed 8pt above the
430
+ * boundary still landed on the top row of keys and scrolled nothing.
431
+ */
432
+ function extendKeyboardUp(elements, top, median) {
433
+ let boundary = top;
434
+ // Four rows is a full keyboard's worth; the loop stops on its own long before
435
+ // that on anything that is not one.
436
+ for (let i = 0; i < 4; i += 1) {
437
+ const row = (elements ?? []).filter((e) => e.frame
438
+ && looksLikeKey(e)
439
+ && Math.abs(heightOf(e.frame) - median) <= Math.max(3, median * 0.4)
440
+ // Sitting directly on the current boundary, within one row's height.
441
+ && e.frame.y + heightOf(e.frame) <= boundary + 4
442
+ && e.frame.y + heightOf(e.frame) >= boundary - median * 1.6);
443
+ if (row.length < 5) break;
444
+ const next = Math.min(...row.map((e) => e.frame.y));
445
+ if (!(next < boundary)) break;
446
+ boundary = next;
447
+ }
448
+ return boundary;
449
+ }
450
+
451
+ /**
452
+ * Is this element outside the viewport?
453
+ *
454
+ * **Both axes.** Every filter in this project checked `y` and ignored `x`,
455
+ * which is fine until a horizontal row: a filter chip reported at **x=422 on a
456
+ * 402pt-wide screen** counted as visible, and `scroll_to` then said *"'Assigned
457
+ * to Me' is in view at 422,277 already"* — confidently wrong about the one thing
458
+ * it exists to answer. Off-screen chips came back at **x=-247** the same way.
459
+ *
460
+ * Reported as the most expensive finding of an agent's session, and the cost was
461
+ * not the wrong answer itself: it was that the wrong answer was *confident*, so
462
+ * the recovery was hand-tuned swipes and two overshoots.
463
+ */
464
+ export function offViewport(t, screen) {
465
+ if (!t) return false;
466
+ const w = screen?.width;
467
+ const h = screen?.height;
468
+ if (Number.isFinite(h) && (t.y < 0 || t.y > h)) return true;
469
+ if (Number.isFinite(w) && (t.x < 0 || t.x > w)) return true;
470
+ return false;
257
471
  }
258
472
 
259
473
  /** Annotate a target list with region and nav slot. Mutates and returns it. */
package/src/screenmap.js CHANGED
@@ -19,6 +19,19 @@ import * as store from './store.js';
19
19
 
20
20
  const MAP_VERSION = 9; // ax targets carry value, selected and focused
21
21
 
22
+ /**
23
+ * A stored map also holds a `structuralHash`, which the *fingerprint* rules
24
+ * produced. So a token-rule change invalidates every stored map, and relying on
25
+ * someone to remember to bump `MAP_VERSION` too is exactly how the phantom
26
+ * keyboard survived a fix: two copies of one dependency, one of them updated.
27
+ *
28
+ * Stating the dependency instead of remembering it. A map is only valid for the
29
+ * token rules that hashed it.
30
+ */
31
+ const usable = (e) => Boolean(e)
32
+ && e.version === MAP_VERSION
33
+ && e.fingerprintVersion === fingerprint.TOKEN_RULES_VERSION;
34
+
22
35
  function mapDir(udid) {
23
36
  return path.join(store.deviceDir(udid), 'screens');
24
37
  }
@@ -38,10 +51,38 @@ function mapDir(udid) {
38
51
  */
39
52
  export const DEFAULT_TOLERANCE = 20;
40
53
 
54
+ /** Comparison that ignores what OCR adds — a caret, a stray glyph, spacing. */
55
+ const alnum = (v) => String(v ?? '').toLowerCase().replace(/[^\p{L}\p{N}]+/gu, '');
56
+
57
+ /**
58
+ * May an OCR word be recorded as an alias of the element enclosing it?
59
+ *
60
+ * Only when they are plausibly the same thing. A row labelled "Kate Bell"
61
+ * containing OCR's "Kate Bell" is one element two sensors saw; a sheet's
62
+ * "Area (Optional)" enclosing a dimmed page's "Exterior Building" is two things
63
+ * at one coordinate on different z-layers, and aliasing them reads as though
64
+ * the field contains that value.
65
+ *
66
+ * An element with **no label** takes the text outright — that is how an
67
+ * icon-only control gets a name, and it cannot contradict a label it does not
68
+ * have.
69
+ *
70
+ * A function rather than three lines inline, because the inline version read
71
+ * `covering.label` before anything checked that `covering` existed, and the
72
+ * TypeError that followed was swallowed by the OCR try/catch — silently
73
+ * disabling the sensor. A pure function can be tested with the value that broke
74
+ * it, and the source-shape assertion this replaces could not.
75
+ */
76
+ export function aliasRelates(coveringLabel, text) {
77
+ const own = alnum(coveringLabel);
78
+ const seen = alnum(text);
79
+ return !own || !seen || own.includes(seen) || seen.includes(own);
80
+ }
81
+
41
82
  export function recall(udid, hash) {
42
83
  if (!hash) return null;
43
84
  const entry = store.readJson(path.join(mapDir(udid), `${hash}.json`));
44
- return entry && entry.version === MAP_VERSION ? entry : null;
85
+ return usable(entry) ? entry : null;
45
86
  }
46
87
 
47
88
  function loadAll(udid) {
@@ -53,7 +94,7 @@ function loadAll(udid) {
53
94
  }
54
95
  return files
55
96
  .map((f) => store.readJson(path.join(mapDir(udid), f)))
56
- .filter((e) => e && e.version === MAP_VERSION);
97
+ .filter(usable);
57
98
  }
58
99
 
59
100
  /**
@@ -137,6 +178,9 @@ export async function build(udid, {
137
178
  // empty `sources` rethrew — so making a layer work turned a loud failure into
138
179
  // a quiet one.
139
180
  const degraded = [];
181
+ // Pairs where an ax element and an OCR word share a coordinate and disagree
182
+ // about what is there — the signature of one layer covering another.
183
+ const occluded = [];
140
184
  // One round trip for both, because the daemon runs the tree read and the
141
185
  // recognition pass concurrently against the same instant of the screen. Asked
142
186
  // separately they would queue: the control socket serves one request at a
@@ -281,15 +325,49 @@ export async function build(udid, {
281
325
  ?? targets
282
326
  .filter((t) => eligible(t) && matching.sameElementSeenTwice(t, { ...w, frame: box, label: w.text }))
283
327
  .sort((a, b) => area(a.frame) - area(b.frame))[0];
328
+ // An alias must be the *same thing*, read twice.
329
+ //
330
+ // The geometric branch above pairs an OCR word with whichever ax
331
+ // element encloses it, and across z-layers that is simply wrong.
332
+ // Reported on a modal-heavy screen, with a Select Area sheet open over a
333
+ // dimmed page: `#15 text 167,316 Area (Optional) ~ Exterior Building`,
334
+ // which reads as though the field "Area (Optional)" contains "Exterior
335
+ // Building". They are two unrelated things at one coordinate on
336
+ // different layers, and the reporter had to fall back to a screenshot to
337
+ // count five radio options — precisely the case the text map exists to
338
+ // remove.
339
+ //
340
+ // So a labelled ax element only takes an alias that relates to its own
341
+ // label. The justification for aliasing was always "a row labelled 'Kate
342
+ // Bell' containing OCR's 'Kate Bell' is one element two sensors saw" —
343
+ // that still holds. An *unlabelled* element still takes the text
344
+ // outright, because that is how an icon-only control gets a name at all,
345
+ // and it cannot contradict a label it does not have.
346
+ // NOTE the nesting, which is the whole point of this shape: `covering`
347
+ // is undefined whenever no ax element encloses this word, which is most
348
+ // words on most screens. Reading `covering.label` before checking that
349
+ // threw a TypeError inside the OCR try/catch — so the entire OCR pass
350
+ // was swallowed and reported as `degraded: text recognition`, silently
351
+ // disabling the sensor on every screen with one uncovered word. Neither
352
+ // the unit tests nor the perception harness caught it: the harness feeds
353
+ // *already fused* element lists, so it never runs this loop. The
354
+ // integration job caught it, which is what it is for.
284
355
  if (covering) {
285
- covering.aliases = [...(covering.aliases || []), w.text];
286
- // Keep the ax role and frame — it is the hit target — and record that
287
- // both sensors saw it. Anything asking "is this the tree's element?"
288
- // must ask matching.isAxTarget, not `=== 'ax'`.
289
- if (!String(covering.source ?? '').includes('ocr')) {
290
- covering.source = `${covering.source ?? 'ax'}|ocr`;
356
+ if (aliasRelates(covering.label, w.text)) {
357
+ covering.aliases = [...(covering.aliases || []), w.text];
358
+ // Keep the ax role and frame — it is the hit target — and record
359
+ // that both sensors saw it. Anything asking "is this the tree's
360
+ // element?" must ask matching.isAxTarget, not `=== 'ax'`.
361
+ if (!String(covering.source ?? '').includes('ocr')) {
362
+ covering.source = `${covering.source ?? 'ax'}|ocr`;
363
+ }
364
+ continue;
291
365
  }
292
- continue;
366
+ // Rejected as an alias, so it falls through and becomes an element of
367
+ // its own — which is what it is. Marked, because "these two things
368
+ // overlap and disagree" is exactly the shape of an occluding layer,
369
+ // and a caller counting radio options needs to know it is there.
370
+ occluded.push({ over: covering.label, under: w.text });
293
371
  }
294
372
  targets.push({
295
373
  label: w.text,
@@ -323,6 +401,7 @@ export async function build(udid, {
323
401
  : { hash: null, tokens: [], keyboard: false };
324
402
  const entry = {
325
403
  version: MAP_VERSION,
404
+ fingerprintVersion: fingerprint.TOKEN_RULES_VERSION,
326
405
  hash,
327
406
  layoutHash,
328
407
  structuralHash: structure.hash,
@@ -336,6 +415,7 @@ export async function build(udid, {
336
415
  // map has to carry it, because this is what gets written into memory.
337
416
  if (daemonScreen?.axTruncated) degraded.push(`accessibility tree cut short: ${daemonScreen.axTruncated}`);
338
417
  if (degraded.length) entry.degraded = degraded;
418
+ if (occluded.length) entry.occluded = occluded;
339
419
  // Only a map of a settled screen is worth keeping; remembering a transition
340
420
  // fills the store with layouts that will never be seen again.
341
421
  return persist ? remember(udid, entry) : entry;
@@ -0,0 +1,161 @@
1
+ /**
2
+ * The local supervisor: three words, behind the hands, in front of Claude.
3
+ *
4
+ * The owner's design. Claude plans; the deterministic executor in `actions.js`
5
+ * runs the plan and verifies each step; and when a step fails, *this* decides
6
+ * whether the plan can proceed — before anything reaches Claude. It sits behind
7
+ * the hands and in front of the reasoner, and it is the first responder rather
8
+ * than the decision-maker.
9
+ *
10
+ * It may say **wait**, **retry** or **stop**. Nothing else. It cannot invent a
11
+ * step, skip one, substitute a target, or continue past an unexpected screen —
12
+ * not because a confidence threshold forbids it but because those are not words
13
+ * it can say. The answer space *is* the safety property. `seek` was given
14
+ * latitude over what to open and pressed "YES, THIS FIXED MY PROBLEM" in a live
15
+ * app; a component that can only choose among three words cannot do that,
16
+ * whatever it believes.
17
+ *
18
+ * **The plan briefs it**, which is the owner's second insight and the thing that
19
+ * made it work. Asked cold, it called a list that was plainly still arriving a
20
+ * dead end — because it does not know the app and Claude, by the time it writes
21
+ * the plan, does. So a batch may carry `supervise` and a step may carry
22
+ * `expect`, and both reach the supervisor as context. That costs a string and no
23
+ * round trips.
24
+ *
25
+ * One thing it is deliberately not trusted for: its own **prose**. In testing it
26
+ * returned a correct decision with a reason citing a rule that did not apply.
27
+ * The decision is used; the reason is logged and never shown as an explanation.
28
+ * Presenting a confabulated rationale as fact is the mistake `seek`'s
29
+ * documentation already made once.
30
+ *
31
+ * Off unless asked: `SIMFRAME_SUPERVISOR=apple`, or `supervisor` per call.
32
+ */
33
+ import path from 'node:path';
34
+ import { fileURLToPath } from 'node:url';
35
+ import * as store from './store.js';
36
+ import { compiler, lineServer } from './localhelper.js';
37
+
38
+ const SOURCE = path.join(path.dirname(fileURLToPath(import.meta.url)), '..', 'native', 'supervise.swift');
39
+ const BIN = path.join(store.ROOT, 'bin', 'supervise');
40
+
41
+ const helper = lineServer({
42
+ ensureBinary: compiler({ source: SOURCE, binary: BIN, what: 'local supervisor' }),
43
+ what: 'local supervisor',
44
+ });
45
+
46
+ export const DECISIONS = new Set(['wait', 'retry', 'stop']);
47
+
48
+ /**
49
+ * The vocabulary gate, as a function so it can be *tested* rather than grepped.
50
+ *
51
+ * This one line is the safety property: the supervisor cannot invent a step,
52
+ * skip one, substitute a target or continue past an unexpected screen, because
53
+ * those are not words it can say. Anything outside the three is not a decision
54
+ * and becomes `null`, which means "behave as if there is no supervisor".
55
+ *
56
+ * It was previously inline, and the test that guarded it matched the source
57
+ * text — so it broke when the branch grew an else, with nothing actually wrong.
58
+ * A property this important deserves an assertion that runs it.
59
+ */
60
+ export function decisionOf(answer) {
61
+ // A string, checked rather than coerced. `String(["wait"])` is `"wait"`, so a
62
+ // `String(...)` coercion here let `{decision: ["wait"]}` through the one gate
63
+ // that defines this component's answer space. Found the first time this
64
+ // property was *run* instead of grepped for in the source — the old test
65
+ // matched the source text of the branch and could never have caught it.
66
+ const raw = answer?.decision;
67
+ if (typeof raw !== 'string') return null;
68
+ const decision = raw.toLowerCase();
69
+ return DECISIONS.has(decision) ? decision : null;
70
+ }
71
+
72
+ /** Which backend the caller asked for, per call first and environment second. */
73
+ export function requested(options) {
74
+ const raw = String(options?.supervisor ?? process.env.SIMFRAME_SUPERVISOR ?? '').trim().toLowerCase();
75
+ if (!raw || raw === 'none' || raw === 'off' || raw === '0' || raw === 'false') return null;
76
+ return raw;
77
+ }
78
+
79
+ /**
80
+ * Can this plan proceed past the step that just failed?
81
+ *
82
+ * @returns {Promise<{decision: 'wait'|'retry'|'stop', reason: string, ms: number}|null>}
83
+ * null on every failure mode — not asked, unavailable, timed out, unparseable,
84
+ * or an answer outside the three words. A null means the executor behaves
85
+ * exactly as it does without a supervisor, which is the only safe default.
86
+ */
87
+ export async function judge({
88
+ goal, step, expected, failure, screen, stillMs, note, options, timeoutMs = 2500,
89
+ detail,
90
+ } = {}) {
91
+ if (!requested(options)) return null;
92
+ if (!step || !failure) return null;
93
+ const answer = await helper.ask({
94
+ goal: goal ? String(goal).slice(0, 200) : null,
95
+ step: String(step).slice(0, 200),
96
+ expected: expected ? String(expected).slice(0, 300) : null,
97
+ failure: String(failure).slice(0, 300),
98
+ screen: (screen ?? []).filter(Boolean).map((s) => String(s).slice(0, 40)).slice(0, 25),
99
+ stillMs: Number.isFinite(stillMs) ? Math.round(stillMs) : null,
100
+ note: note ? String(note).slice(0, 200) : null,
101
+ }, timeoutMs);
102
+ const decision = decisionOf(answer);
103
+ if (decision == null) {
104
+ // Why it did not answer, for the caller's log — through an out-parameter
105
+ // rather than the return value, because returning anything truthy here
106
+ // would change what the executor does. `null` means "behave as if there is
107
+ // no supervisor" and that safety property is the one thing in this file
108
+ // that must not become conditional.
109
+ //
110
+ // Before this, every failure reached the supervision log as "the
111
+ // supervisor did not answer": a timeout, a guardrail refusal and a model
112
+ // that was never installed were one indistinguishable line.
113
+ if (detail && typeof detail === 'object') {
114
+ detail.kind = answer?.kind
115
+ ?? (answer == null ? 'no answer' : answer.decision ? 'outside the vocabulary' : 'unparseable');
116
+ if (answer?.error) detail.error = String(answer.error).slice(0, 200);
117
+ }
118
+ return null;
119
+ }
120
+ return { decision, reason: String(answer.reason ?? '').slice(0, 120), ms: answer.ms ?? null };
121
+ }
122
+
123
+ /** For `doctor`: what the supervisor layer is, in one line. */
124
+ export async function status(options) {
125
+ const want = requested(options);
126
+ if (!want) return { supervisor: 'none', detail: 'not requested (SIMFRAME_SUPERVISOR is unset)' };
127
+ if (want !== 'apple') return { supervisor: 'none', detail: `no such supervisor backend: "${want}"` };
128
+ const live = await helper.status();
129
+ if (!live.ok) return { supervisor: 'none', detail: live.reason };
130
+ // Prove a round trip, not a presence.
131
+ //
132
+ // Reporting "available" from a fresh process was true and useless: the model
133
+ // had stopped answering inside the long-lived MCP server, `doctor` opened its
134
+ // own process, got a healthy one, and said so — for twenty calls and six
135
+ // failures during which nothing was being judged. The reporter's fix, and it
136
+ // is the right one: make it answer something.
137
+ const probe = await helper.ask({
138
+ step: 'tap "Probe"',
139
+ failure: '"Probe" is not on this screen. Visible: Probe',
140
+ screen: ['Probe'],
141
+ stillMs: 5000,
142
+ }, 6000);
143
+ if (decisionOf(probe) == null) {
144
+ return {
145
+ supervisor: 'none',
146
+ detail: 'the model loaded but did not answer a probe — it is present and not working',
147
+ };
148
+ }
149
+ // The window, read from the model rather than repeated from documentation.
150
+ // Worth printing: the worst case our own clipping allows measures 1,918
151
+ // tokens against it, and that ratio is the reason there is no per-call token
152
+ // budget check — see DEFERRED 99.
153
+ const ctx = live.hello?.contextSize;
154
+ return {
155
+ supervisor: 'apple',
156
+ detail: `Apple Foundation Models, on-device; answered a probe in ${probe.ms ?? '?'}ms;`
157
+ + `${ctx ? ` ${ctx}-token window;` : ''} may only answer wait/retry/stop`,
158
+ };
159
+ }
160
+
161
+ export const close = helper.close;