simframe 0.9.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/index.js CHANGED
@@ -9,6 +9,9 @@ import { decodePng, encodePng, scaleBitmap } from './png.js';
9
9
  import {
10
10
  REGION_COLS,
11
11
  hexToSignature,
12
+ isBlackFrame,
13
+ maxCellDelta,
14
+ CELL_CHANGE,
12
15
  regionDeltas,
13
16
  regionMap,
14
17
  signatureDiff,
@@ -398,6 +401,80 @@ export function resolveBaseline(state, since) {
398
401
  return { kind: 'unmatched', requested: key };
399
402
  }
400
403
 
404
+ /**
405
+ * Is the baseline describing a screen that had already finished moving?
406
+ *
407
+ * Pure, so the rule can be argued with in a test rather than only observed on
408
+ * a device. The full reasoning is at the call site in `waitFor`; the short
409
+ * version is that stillness cannot accumulate in the milliseconds between a
410
+ * dispatch returning and a wait beginning, so a screen that already differs
411
+ * from the baseline *and* has already been at rest for the whole stillness
412
+ * window changed for some earlier reason.
413
+ */
414
+ /** The newest frame's region signature, which the state already carries. */
415
+ function currentSig(state) {
416
+ const h = state?.history;
417
+ if (!h?.length) return null;
418
+ const newest = h.find((x) => x.seq === state.seq) ?? h[h.length - 1];
419
+ return newest?.sig ?? null;
420
+ }
421
+
422
+ /**
423
+ * The longest pause inside a transition, measured after the transition is over.
424
+ *
425
+ * This is the unbiased half of the estimator that was reverted in Phase 11, and
426
+ * the difference is entirely about *when* the measurement stops.
427
+ *
428
+ * The biased version accumulated the statistic from inside the wait: the
429
+ * longest stretch of stillness the wait itself happened to observe. Feed that
430
+ * back into how long the next wait runs and it eats itself — a wait that ends
431
+ * early never sees the pauses that come later, so the gaps read as zero, the
432
+ * window ratchets down, the next wait ends earlier still, and eventually a
433
+ * settle returns mid-transition and the graph learns a screen it never reached.
434
+ * That is not a theory; it corrupted this device's graph in one afternoon.
435
+ *
436
+ * This one reads the frame history *after* the fact, over a window whose end is
437
+ * not decided by the wait. The frames are already on disk with their timestamps
438
+ * and their hashes, so the true profile of a transition is recoverable as long
439
+ * as the history still reaches back to it.
440
+ *
441
+ * That last condition is the whole reason this returns null rather than a
442
+ * number: the ring is bounded, and during fast motion it holds a second or two.
443
+ * A partial window would produce a *shorter* gap than really occurred, which is
444
+ * the exact direction of the bias being removed. Reporting nothing is the only
445
+ * honest answer to a question the evidence cannot reach.
446
+ */
447
+ export function longestQuietGap(history, sinceMs, untilMs = Date.now()) {
448
+ if (!Array.isArray(history) || history.length < 2) return null;
449
+ const frames = history
450
+ .filter((f) => Number.isFinite(f?.at) && f.at >= sinceMs && f.at <= untilMs)
451
+ .sort((a, b) => a.at - b.at);
452
+ if (frames.length < 2) return null;
453
+ // The history has to reach back to the action itself. One frame interval of
454
+ // slack, because the frame that captures the moment of the action is not
455
+ // required to land exactly on it.
456
+ const span = frames[1].at - frames[0].at;
457
+ if (frames[0].at > sinceMs + Math.max(250, span)) return null;
458
+
459
+ let longest = 0;
460
+ let lastChangeAt = frames[0].at;
461
+ for (let i = 1; i < frames.length; i += 1) {
462
+ if (frames[i].hash !== frames[i - 1].hash) {
463
+ longest = Math.max(longest, frames[i].at - lastChangeAt);
464
+ lastChangeAt = frames[i].at;
465
+ }
466
+ }
467
+ // The quiet after the last change is not a pause *inside* the transition —
468
+ // it is the transition being over, which is what a settle already measures.
469
+ return longest;
470
+ }
471
+
472
+ export function baselineAlreadySettled({ mode, changedAtStart, stableForMs, stableMs } = {}) {
473
+ if (mode === 'stable' || !changedAtStart) return false;
474
+ if (!Number.isFinite(stableForMs) || !Number.isFinite(stableMs)) return false;
475
+ return stableForMs >= stableMs;
476
+ }
477
+
401
478
  function compareToBaseline(state, baseline) {
402
479
  if (!baseline) return null;
403
480
  if (baseline.kind === 'history') {
@@ -497,6 +574,10 @@ export async function getState(deviceQuery, { since, options, inputHealth = fals
497
574
  map: regionMap(state.regions || [], REGION_COLS),
498
575
  since: compareToBaseline(state, resolveBaseline(state, since)),
499
576
  live: liveness(device.udid, state),
577
+ // Costs 32 integer comparisons on a signature already computed, and it is
578
+ // the difference between "the screen is calm" and "the display stopped
579
+ // rendering" — which looked identical to everything above this line.
580
+ black: isBlackFrame(currentSig(state)),
500
581
  // Off by default and asked for by the state commands only. A flow step
501
582
  // calls getState twice, and the fix for a stale session runs before every
502
583
  // action anyway (input.ensureFreshSession) — this is the report, not the
@@ -568,7 +649,7 @@ export async function waitFor(
568
649
  const p = store.paths(device.udid);
569
650
  const requested = since ?? baselineHash;
570
651
  const resolved = resolveBaseline(first, requested);
571
- const baselineHashValue =
652
+ let baselineHashValue =
572
653
  resolved?.kind === 'history' ? resolved.entry.hash : (requested ?? first.hash);
573
654
  const baselineResolved = resolved?.kind === 'history' || requested == null;
574
655
 
@@ -590,10 +671,62 @@ export async function waitFor(
590
671
  */
591
672
  let quietGapMs = 0;
592
673
  let quietRun = 0;
674
+ /**
675
+ * The baseline's own signature, for changes the mean cannot see.
676
+ *
677
+ * Only when the baseline was found in the frame history — a hash we cannot
678
+ * place has no signature to compare against, and guessing one would be worse
679
+ * than not looking.
680
+ */
681
+ const baselineSig = resolved?.kind === 'history' && resolved.entry?.sig
682
+ ? hexToSignature(resolved.entry.sig)
683
+ : (requested == null ? hexToSignature(currentSig(first) ?? '') : null);
684
+ let smallChange = false;
685
+ /** Frames the display was not rendering at all. See `isBlackFrame`. */
686
+ let blackFrames = 0;
687
+ let blackSinceStart = null;
593
688
  let lastHash = first.hash;
594
689
  let sawChange = mode === 'stable' || first.hash !== baselineHashValue;
595
690
  const changedAtStart = sawChange && mode !== 'stable';
596
691
 
692
+ /**
693
+ * The baseline describes a screen that has already finished moving.
694
+ *
695
+ * `since` means "the screen as it was before the action", and the whole
696
+ * reliability of these scripts rests on it being captured *before* rather
697
+ * than after — a baseline sampled afterwards is the commonest way to wait for
698
+ * a change that already happened. What was missing is the other end of it:
699
+ * time also passes between capturing the baseline and dispatching the action,
700
+ * and in a flow step that gap holds a `locate`, a perception pass and a
701
+ * settle wait — hundreds of milliseconds, not microseconds.
702
+ *
703
+ * So a transition can begin *and finish* in that gap, and then `sawChange` is
704
+ * true at wait start because of the previous action's animation. Measured:
705
+ * `tap Accessibility` returned `settled 124ms` against a 500 ms stillness
706
+ * window, the screen had never left the Settings root, and the graph recorded
707
+ * `root -> root` as a verified edge — count 11, changedOutcomes 5, flipping
708
+ * between the real destination and itself all day.
709
+ *
710
+ * The test is unambiguous rather than clever: the screen differs from the
711
+ * baseline *and has already been at rest for the full stillness window*.
712
+ * Stillness cannot have accumulated in the milliseconds between a dispatch
713
+ * returning and this call starting, so whatever changed, changed and settled
714
+ * before we looked, and it is not this action's doing. Re-baseline to what is
715
+ * actually on screen and wait for a further change — which is what the caller
716
+ * asked for and what a stale hash prevented.
717
+ *
718
+ * A screen that differs and is *still moving* is left alone: that is
719
+ * genuinely ambiguous, and after an action the usual reading is the right
720
+ * one.
721
+ */
722
+ const staleBaseline = baselineAlreadySettled({
723
+ mode, changedAtStart, stableForMs: first.stableForMs, stableMs,
724
+ });
725
+ if (staleBaseline) {
726
+ baselineHashValue = first.hash;
727
+ sawChange = false;
728
+ }
729
+
597
730
  const done = (satisfied, extra = {}) => ({
598
731
  device,
599
732
  state: last,
@@ -602,6 +735,15 @@ export async function waitFor(
602
735
  quietGapMs,
603
736
  sawChange,
604
737
  changedBeforeWait: changedAtStart,
738
+ // Something moved, but only in one region — a control changing state
739
+ // rather than a screen changing.
740
+ smallChange,
741
+ staleBaseline,
742
+ blackFrames,
743
+ // Said as an observation, never as a diagnosis: a screen can be black
744
+ // because the app drew black. What makes it the capture wedge is that it
745
+ // stays black while input is being delivered, and the caller knows that.
746
+ blackMs: blackSinceStart ? Date.now() - blackSinceStart : 0,
605
747
  baselineHash: baselineHashValue,
606
748
  baselineResolved,
607
749
  waitedMs: Date.now() - startedAt,
@@ -617,8 +759,44 @@ export async function waitFor(
617
759
  const live = liveness(device.udid, state);
618
760
  if (!live.ok) return done(false, { stalled: true });
619
761
 
762
+ // A black frame is not evidence, in either direction.
763
+ //
764
+ // The capture wedge leaves every frame black while the whole capture path
765
+ // reports success, so before this a settle read the black screen as a
766
+ // change (the hash differs from anything) and then as a calm one (nothing
767
+ // moves), and returned `ok` for an action nobody could see the result of.
768
+ // It self-recovers most times, so the useful behaviour is to keep waiting
769
+ // rather than to conclude.
770
+ const black = isBlackFrame(currentSig(state));
771
+ if (black) {
772
+ blackFrames += 1;
773
+ blackSinceStart = blackSinceStart ?? Date.now();
774
+ await sleep(60);
775
+ continue;
776
+ }
777
+ blackSinceStart = null;
778
+
620
779
  if (!sawChange && state.hash !== baselineHashValue) sawChange = true;
621
780
 
781
+ // A change too small for the whole-screen mean to see.
782
+ //
783
+ // A switch flipping moves one cell of thirty-two by 0.043 and the mean by
784
+ // 0.0013 — a third of the threshold — so every switch, radio dot,
785
+ // checkbox and segment highlight was an action that "changed nothing",
786
+ // and `no-visible-change` is a verdict that escalates. See
787
+ // analyze.CELL_CHANGE for the measured gap this sits in.
788
+ //
789
+ // Deliberately feeding `sawChange` and *not* stillness: `stableForMs`
790
+ // stays on the mean, because a blinking text caret is a small localised
791
+ // change and a screen with a cursor would otherwise never settle.
792
+ if (!sawChange && baselineSig) {
793
+ const sig = currentSig(state);
794
+ if (sig && maxCellDelta(hexToSignature(sig), baselineSig) > CELL_CHANGE) {
795
+ sawChange = true;
796
+ smallChange = true;
797
+ }
798
+ }
799
+
622
800
  // A pause that turned out not to be the end of the transition. Only
623
801
  // pauses followed by more movement count: the quiet at the end of a
624
802
  // settle is the answer, not a gap.
@@ -985,7 +1163,11 @@ export async function locate(
985
1163
  `"${query}" matches ${outcome.alternatives.length} things on this screen — say which, or pass index: ${list}`,
986
1164
  ),
987
1165
  'ambiguous_intent',
988
- { candidates: outcome.alternatives },
1166
+ // Present, several times over — as opposed to absent, which also tags
1167
+ // ambiguous_intent when the screen was one we thought we knew. A
1168
+ // waiting caller needs the difference: more time cannot make a thing
1169
+ // unique, and it can make an absent thing arrive.
1170
+ { candidates: outcome.alternatives, ambiguous: true },
989
1171
  );
990
1172
  }
991
1173
  if (outcome.status === 'ok') {
@@ -1039,6 +1221,38 @@ export async function locate(
1039
1221
  */
1040
1222
  export const STRUCTURAL_SETTLE_MS = 300;
1041
1223
 
1224
+ /**
1225
+ * How much of the structural window is still owed, given when the last sample's
1226
+ * frame was captured.
1227
+ *
1228
+ * This was `sleep(300)` between the two readings, and unlike every other fixed
1229
+ * wait in the engine it cannot be replaced by waiting for a signal — because
1230
+ * there is no signal. The race it guards is a screen whose *pixels* have gone
1231
+ * still while its structure has not: a list whose spinner has gone and whose
1232
+ * rows have not landed is perfectly quiet and structurally wrong, so the settle
1233
+ * detector, which watches pixels, has nothing to report. Only elapsed time
1234
+ * separates the two readings.
1235
+ *
1236
+ * What can be fixed is that the wait was *additional*. The guarantee wanted is
1237
+ * 300 ms between the frames the two samples read; the code slept 300 ms after a
1238
+ * sample that had already spent an unbounded settle wait and a full perception
1239
+ * pass getting there. On a screen that took a second to go quiet the separation
1240
+ * was already there and the sleep bought nothing but a second of it. So credit
1241
+ * what has passed and wait only for the remainder — the same guarantee, and
1242
+ * usually none of the sleep.
1243
+ *
1244
+ * The window itself is per-screen learnable, and worth noting that its
1245
+ * estimator has the *opposite* feedback sign to the one that corrupted the
1246
+ * graph: a window too short produces disagreeing samples, which lengthens it.
1247
+ * Self-correcting rather than self-reinforcing. It still waits on the
1248
+ * perception eval harness, because "the samples agreed" is only evidence the
1249
+ * window was long enough if the readings themselves are trustworthy.
1250
+ */
1251
+ export function structuralSettleOwed(capturedAt, now = Date.now(), windowMs = STRUCTURAL_SETTLE_MS) {
1252
+ if (!Number.isFinite(capturedAt)) return windowMs;
1253
+ return Math.max(0, Math.min(windowMs, windowMs - (now - capturedAt)));
1254
+ }
1255
+
1042
1256
  /**
1043
1257
  * How long identity will wait for pixels to go quiet.
1044
1258
  *
@@ -1122,7 +1336,8 @@ export async function screenIdentity(deviceQuery, { options, confirmNovel = true
1122
1336
  // Nothing recognises this, or the pixels have not gone quiet. Either way, make
1123
1337
  // it prove it is the same screen twice running before it becomes a node.
1124
1338
  for (let i = 1; i < STRUCTURAL_SETTLE_SAMPLES; i += 1) {
1125
- await sleep(STRUCTURAL_SETTLE_MS);
1339
+ const owed = structuralSettleOwed(identity.state?.capturedAt);
1340
+ if (owed > 0) await sleep(owed);
1126
1341
  const again = await read({ fresh: true });
1127
1342
  // Two readings agree if they are the same screen — the same test identity
1128
1343
  // itself uses. Demanding an identical hash is a stricter question than the
package/src/input.js CHANGED
@@ -248,6 +248,11 @@ export function elementToNode(e) {
248
248
  type: e.role ?? null,
249
249
  identifier: e.identifier ?? null,
250
250
  enabled: e.state?.enabled ?? null,
251
+ // The daemon batches AXSelected and AXFocused alongside AXEnabled and has
252
+ // since 0.6.0. This converter took one of the three, and it is the one on
253
+ // the path that actually runs — `normalizeNode` below is the idb fallback.
254
+ selected: e.state?.selected ?? null,
255
+ focused: e.state?.focused ?? null,
251
256
  frame: e.frame ?? null,
252
257
  raw: e,
253
258
  };
@@ -273,6 +278,12 @@ function normalizeNode(node) {
273
278
  type: node.type ?? node.AXType ?? null,
274
279
  identifier: node.AXUniqueId ?? node.identifier ?? null,
275
280
  enabled: node.AXEnabled ?? node.enabled ?? null,
281
+ // The daemon has asked the tree for AXSelected and AXFocused since 0.6.0 —
282
+ // they are two of the eight attributes in its batched round trip — and this
283
+ // function dropped both. `view.renderRow` has printed `selected` for as
284
+ // long as it has existed, against a field nobody set.
285
+ selected: node.AXSelected ?? node.selected ?? null,
286
+ focused: node.AXFocused ?? node.focused ?? null,
276
287
  frame: frame
277
288
  ? {
278
289
  x: frame.x ?? frame.X ?? 0,
@@ -506,19 +517,52 @@ export async function sessionHealth(udid) {
506
517
  }
507
518
 
508
519
  /**
509
- * Rebuild the session if the device outlived it. Once per process, per device.
520
+ * Should this staleness be acted on, given what has already been rebuilt?
521
+ *
522
+ * Pure, and separate from the check because the *key* is the whole bug. This
523
+ * gate used to be a set of udids — "once per process, per device" — and the
524
+ * reasoning was to avoid statting on every action. What it actually bought was
525
+ * that the feature could not fire in the one process that matters. A CLI
526
+ * command is a new process every time, so per-process is per-call there and
527
+ * the gate never showed; the MCP server is a single process that lives for a
528
+ * whole session, so it checked once, at the first action, and then never
529
+ * again — and a device that reboots *mid-session* is precisely the case this
530
+ * exists to catch. Reported from a real session: capture kept working, input
531
+ * died, every tap returned `ok`, and about ten calls went into two wrong
532
+ * conclusions about the app.
533
+ *
534
+ * The right key is the boot the rebuild was for. One attempt per device boot:
535
+ * enough that a failed rebuild does not retry on every tap forever, and not so
536
+ * much that the next boot is invisible.
537
+ */
538
+ export function shouldRebuildSession({ stale, bootedAt }, rebuiltFor) {
539
+ if (!stale) return false;
540
+ // A boot we cannot date cannot be memoised against, and re-attempting on
541
+ // every action would be worse than not detecting it. sessionStaleness only
542
+ // reports stale with a finite bootedAt, so this is a belt, not a case.
543
+ if (!Number.isFinite(bootedAt)) return false;
544
+ return rebuiltFor !== bootedAt;
545
+ }
546
+
547
+ /**
548
+ * Rebuild the session if the device outlived it. Once per device boot.
510
549
  *
511
550
  * Rebuild, and retry nothing: this runs *before* the action, so the action is
512
551
  * delivered on a session known to be current. Retrying afterwards is how an
513
552
  * action fires twice, which is the hazard the verify barrier exists to
514
553
  * prevent — and it is why the existing recovery covers hardware buttons only.
554
+ *
555
+ * The check now runs on every dispatch rather than once. It costs two small
556
+ * `readJson`s and, at most every three seconds, one stat — `bootedAtCached`
557
+ * already caps the part that was expensive, which is what made the
558
+ * once-per-process gate unnecessary as well as wrong.
515
559
  */
516
- const freshened = new Set();
560
+ const rebuiltForBoot = new Map();
517
561
  export async function ensureFreshSession(udid) {
518
- if (!udid || freshened.has(udid)) return null;
519
- freshened.add(udid);
562
+ if (!udid) return null;
520
563
  const health = await sessionHealth(udid);
521
- if (!health.stale) return null;
564
+ if (!shouldRebuildSession(health, rebuiltForBoot.get(udid))) return null;
565
+ rebuiltForBoot.set(udid, health.bootedAt);
522
566
  const rebuilt = await resetSession(udid);
523
567
  return { ...health, rebuilt };
524
568
  }
package/src/matching.js CHANGED
@@ -59,7 +59,25 @@ export function nameScore(name, query) {
59
59
  const q = norm(query);
60
60
  if (!n || !q) return 0;
61
61
  if (n === q) return 1;
62
- if (n.startsWith(q) || q.startsWith(n)) return 0.86;
62
+ // A prefix match is only as good as the share it covers, and this branch had
63
+ // to be taught that twice.
64
+ //
65
+ // `n.startsWith(q)` — the name begins with the query, "Acce" for
66
+ // "Accessibility" — is the ordinary case and keeps most of its score: a
67
+ // prefix of a name is how people abbreviate. `q.startsWith(n)` is the
68
+ // opposite direction, where the *name* is a fragment of the query, and it
69
+ // returned the same flat 0.86 no matter how little of the query it was. So
70
+ // the section-index letter "S" scored 0.86 against the query "Search" and
71
+ // beat the search field's own label "Q Search" at 0.585 — and simframe
72
+ // tapped a scrubber and typed into it.
73
+ //
74
+ // The sibling branch below already carries this lesson in a comment about
75
+ // "back" matching a list row. Only one of the two had learned it. Both scale
76
+ // now, and the floor differs by direction on purpose: a query that is a
77
+ // prefix of a name is usually deliberate, while a name that is a fragment of
78
+ // the query is usually a coincidence, and one character is always one.
79
+ if (n.startsWith(q)) return 0.86 * Math.max(PREFIX_FLOOR, Math.min(1, q.length / n.length + 0.35));
80
+ if (q.startsWith(n)) return 0.8 * Math.max(0.15, n.length / q.length);
63
81
  // A substring match is only as good as the share of the name it covers.
64
82
  // Without this, "back" scores 0.78 against a two-hundred-character list row
65
83
  // that happens to contain "Back of House", and beats the actual back button.
@@ -76,7 +94,31 @@ export function nameScore(name, query) {
76
94
  if (distance > cap) return 0;
77
95
  const longest = Math.max(n.length, q.length);
78
96
  const similarity = 1 - distance / longest;
79
- return similarity >= 0.7 ? similarity * 0.72 : 0;
97
+ if (similarity >= 0.7) return similarity * 0.72;
98
+ // Last tier: OCR read a confusable character.
99
+ //
100
+ // Language correction is deliberately off, which is right for labels and
101
+ // wrong for exactly this. Measured on a real app, `(All)` reads back as
102
+ // `(AII)` and would fail an assert against the string it is; capital-I,
103
+ // lowercase-l, the digit one and a pipe are one shape in most UI fonts, as
104
+ // are capital-O and zero.
105
+ //
106
+ // Deliberately the *last* tier and discounted, not part of `norm`. Folding
107
+ // in `norm` would make it change what an exact match means — "Log in" and
108
+ // "1og in" would become the same string everywhere — and identity is not
109
+ // something to be fuzzy about. Here it only ever rescues a comparison that
110
+ // had already scored zero.
111
+ const foldedScore = confusableFold(n) === confusableFold(q) ? 0.62 : 0;
112
+ return foldedScore;
113
+ }
114
+
115
+ /** One shape per glyph family, for comparison only. Never for identity. */
116
+ export function confusableFold(s) {
117
+ return String(s ?? '')
118
+ .replace(/[il1|!]/gi, '1')
119
+ .replace(/[o0]/gi, '0')
120
+ .replace(/[s5]/gi, '5')
121
+ .replace(/[b8]/gi, '8');
80
122
  }
81
123
 
82
124
  function synonymGroup(query) {
@@ -159,6 +201,17 @@ export function rank(targets, intent, { screen } = {}) {
159
201
  export const AMBIGUITY_MARGIN = 0.08;
160
202
  /** Below this, no candidate is worth acting on. */
161
203
  export const MINIMUM_SCORE = 0.45;
204
+ /**
205
+ * How much of a name's score survives scaling a prefix by its coverage.
206
+ *
207
+ * Tied to `MINIMUM_SCORE` rather than chosen: at 0.5 the shortest useful
208
+ * abbreviation — "Ac" for "Accessibility" — scored 0.433 and fell *below* the
209
+ * threshold to resolve at all, which would have turned a ranking fix into a
210
+ * feature removal. 0.6 puts the worst case at 0.516, comfortably resolvable and
211
+ * still well under a fuller match. The unit test asserts the relationship so
212
+ * the cliff cannot come back by someone tuning one of the two numbers.
213
+ */
214
+ export const PREFIX_FLOOR = 0.6;
162
215
 
163
216
  /**
164
217
  * How close two tap points have to be to mean the same control.
package/src/metrics.js CHANGED
@@ -40,8 +40,16 @@ export const FACULTY = {
40
40
  no_plan: 'exploration (Phase 14)',
41
41
  };
42
42
 
43
- /** Faculties that exist. Empty until Phase 11 lands the first one. */
44
- export const BUILT_FACULTIES = new Set();
43
+ /**
44
+ * Faculties that exist.
45
+ *
46
+ * Phase 11 landed the first one, so `verification_failed` no longer maps to
47
+ * something unbuilt — which changes what those records *mean*. Before, they
48
+ * were a queue waiting on a phase. Now they are evidence that the phase which
49
+ * shipped is not sufficient, and that is a more useful thing for the log to be
50
+ * able to say than a count of things nobody has written yet.
51
+ */
52
+ export const BUILT_FACULTIES = new Set(['sense of time (Phase 11)']);
45
53
 
46
54
  function metricPaths(udid) {
47
55
  const dir = store.deviceDir(udid);
@@ -111,9 +119,15 @@ export const readFlows = (udid, opts) => readJsonl(metricPaths(udid).flows, opts
111
119
  * matching error strings at the boundary — a regexed message is a reason that
112
120
  * silently becomes "unknown" the day somebody rewords it.
113
121
  */
114
- export function tag(err, reason, { candidates = [], tried = [] } = {}) {
122
+ export function tag(err, reason, { candidates = [], tried = [], ambiguous = false } = {}) {
115
123
  if (!REASONS.includes(reason)) throw new Error(`not an escalation reason: ${reason}`);
116
- err.escalation = { reason, candidates, tried };
124
+ // `ambiguous` is narrower than the reason, and that is the point. Two very
125
+ // different failures both tag `ambiguous_intent`: the target is on screen
126
+ // several times over, and the target is not on screen at all on a screen we
127
+ // thought we knew. Only the first is resolvable by *choosing*, and only the
128
+ // first tells a waiting caller that waiting is pointless — the thing it is
129
+ // waiting for has already arrived.
130
+ err.escalation = { reason, candidates, tried, ambiguous };
117
131
  return err;
118
132
  }
119
133
 
@@ -213,8 +227,50 @@ export function fingerprintNow(udid, screenmap) {
213
227
  * measurement's clothes, and every number in this project is supposed to say
214
228
  * where it came from.
215
229
  */
230
+ /**
231
+ * Which run of which program wrote a record.
232
+ *
233
+ * The log is per-device and, until now, anonymous — so two agents driving one
234
+ * booted simulator wrote one interleaved file with no way to separate them.
235
+ * Measured on the bench device in a single evening: 57 records to 81, a third
236
+ * of the new ones naming screens from an app the suite has never launched.
237
+ *
238
+ * That is not a corrupted file, it is a corrupted instrument. CLAUDE.md makes
239
+ * the reason breakdown of this log the thing that chooses which faculty gets
240
+ * built next, and a breakdown that silently pools two sessions errs toward
241
+ * whichever of them made more mistakes — which is not the same question as
242
+ * which faculty is missing.
243
+ *
244
+ * A pid alone would not do: pids are reused, and the useful grouping is "one
245
+ * agent's run", which for the MCP server is the life of the process and for
246
+ * the CLI is a single command. So: the start time, the pid, and a random tail,
247
+ * computed once per process. `client` says what kind of process it was, since
248
+ * "the MCP server did this" and "somebody ran a CLI command" deserve different
249
+ * readings of the same reason.
250
+ *
251
+ * Deliberately not a device id, a username, or anything about the machine. This
252
+ * file is committed to a public repo in summary form, and the question it has
253
+ * to answer is "was this all one agent", which needs no identity to answer.
254
+ */
255
+ const SESSION_ID = `${process.pid.toString(36)}-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 8)}`;
256
+
257
+ /** How this process is being used, for reading a breakdown afterwards. */
258
+ function clientKind() {
259
+ const argv = process.argv.join(' ');
260
+ if (/\bmcp\b/.test(argv)) return 'mcp';
261
+ if (/bench-hpi|scripts\//.test(argv)) return 'script';
262
+ if (/cli\.js|\bsimframe\b/.test(argv)) return 'cli';
263
+ return 'library';
264
+ }
265
+ const CLIENT = clientKind();
266
+
267
+ /** The session this process's records belong to. Exported for `escalations`. */
268
+ export const sessionId = () => SESSION_ID;
269
+ export const clientName = () => CLIENT;
270
+
216
271
  export function recordEscalation(udid, {
217
272
  flowId = null,
273
+ flowName = null,
218
274
  stepIndex = null,
219
275
  fingerprint = null,
220
276
  reason,
@@ -229,7 +285,13 @@ export function recordEscalation(udid, {
229
285
  if (!OUTCOMES.includes(outcome)) throw new Error(`not an escalation outcome: ${outcome}`);
230
286
  const record = {
231
287
  timestamp: new Date().toISOString(),
288
+ // Added after the log turned out to pool two agents' work invisibly. Both
289
+ // are cheap and neither is derivable afterwards, which is the test for
290
+ // whether a field belongs in a log at all.
291
+ session_id: SESSION_ID,
292
+ client: CLIENT,
232
293
  flow_id: flowId,
294
+ flow_name: flowName,
233
295
  step_index: stepIndex,
234
296
  screen_fingerprint: fingerprint,
235
297
  reason,
@@ -408,14 +470,31 @@ export function hpi({ flows, baselines = {} }) {
408
470
  * rate is 1.0 by construction and says nothing. The per-reason breakdown is
409
471
  * the part that decides the next phase, and it is informative today.
410
472
  */
411
- export function breakdown(records) {
473
+ export function breakdown(records, { session = null, flow = null } = {}) {
412
474
  const byReason = {};
413
475
  for (const r of REASONS) byReason[r] = 0;
414
476
  const byScreen = new Map();
415
477
  const byOutcome = {};
478
+ const bySession = new Map();
479
+ const byFlow = new Map();
416
480
  let avoidable = 0;
417
- for (const r of records) {
481
+ let unattributed = 0;
482
+ // Filtering happens here rather than at the call site so `total` and every
483
+ // rate below it describe the same set of records.
484
+ const kept = records.filter((r) => (session ? r?.session_id === session : true))
485
+ .filter((r) => (flow ? r?.flow_name === flow : true));
486
+ for (const r of kept) {
418
487
  if (!REASONS.includes(r?.reason)) continue;
488
+ // Records written before sessions were recorded cannot be attributed, and
489
+ // saying how many there are is the difference between a breakdown that
490
+ // pools two agents and one that says it might be.
491
+ if (r.session_id) {
492
+ const key = `${r.session_id}|${r.client ?? '?'}`;
493
+ bySession.set(key, (bySession.get(key) ?? 0) + 1);
494
+ } else {
495
+ unattributed += 1;
496
+ }
497
+ if (r.flow_name) byFlow.set(r.flow_name, (byFlow.get(r.flow_name) ?? 0) + 1);
419
498
  byReason[r.reason] += 1;
420
499
  byOutcome[r.outcome] = (byOutcome[r.outcome] ?? 0) + 1;
421
500
  // Already avoided locally, so not avoidable by anything unbuilt.
@@ -423,9 +502,25 @@ export function breakdown(records) {
423
502
  const key = r.screen_fingerprint ?? '(no fingerprint)';
424
503
  byScreen.set(key, (byScreen.get(key) ?? 0) + 1);
425
504
  }
426
- const total = records.filter((r) => REASONS.includes(r?.reason)).length;
505
+ const total = kept.filter((r) => REASONS.includes(r?.reason)).length;
506
+ const sessions = [...bySession.entries()]
507
+ .map(([key, count]) => {
508
+ const [id, client] = key.split('|');
509
+ return { session_id: id, client, count };
510
+ })
511
+ .sort((a, b) => b.count - a.count);
427
512
  return {
428
513
  total,
514
+ // The log is per-device and shared: two agents on one booted simulator
515
+ // write one interleaved file. More than one session here means the counts
516
+ // below are a pool, and CLAUDE.md uses those counts to choose a phase.
517
+ sessions,
518
+ session_count: sessions.length,
519
+ unattributed,
520
+ // Any unattributed record at all makes this a pool: the whole point is
521
+ // that they cannot be told apart, and 92 of them is not "one session".
522
+ pooled: sessions.length > 1 || unattributed > 0,
523
+ by_flow: Object.fromEntries([...byFlow.entries()].sort((a, b) => b[1] - a[1])),
429
524
  by_reason: byReason,
430
525
  by_outcome: byOutcome,
431
526
  faculty: Object.fromEntries(
@@ -437,7 +532,9 @@ export function breakdown(records) {
437
532
  .sort((a, b) => b[1] - a[1])
438
533
  .slice(0, 10)
439
534
  .map(([fingerprint, count]) => ({ fingerprint, count })),
440
- model_turns_spent: records.reduce((acc, r) => acc + (r.model_turns_spent ?? 0), 0),
535
+ // `kept`, not `records` — a filtered breakdown that reports the whole
536
+ // log's model turns is the same class of mistake as pooling two sessions.
537
+ model_turns_spent: kept.reduce((acc, r) => acc + (r.model_turns_spent ?? 0), 0),
441
538
  };
442
539
  }
443
540