simframe 0.9.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -3
- package/package.json +1 -1
- package/scripts/check-private.mjs +143 -0
- package/scripts/eval-perception.mjs +248 -0
- package/src/actions.js +150 -11
- package/src/analyze.js +70 -0
- package/src/cli.js +83 -6
- package/src/graph.js +104 -4
- package/src/index.js +218 -3
- package/src/input.js +49 -5
- package/src/matching.js +55 -2
- package/src/metrics.js +105 -8
- package/src/navigate.js +10 -7
- package/src/platform/android.js +1 -1
- package/src/screenmap.js +20 -1
- package/src/view.js +61 -0
package/src/index.js
CHANGED
|
@@ -9,6 +9,9 @@ import { decodePng, encodePng, scaleBitmap } from './png.js';
|
|
|
9
9
|
import {
|
|
10
10
|
REGION_COLS,
|
|
11
11
|
hexToSignature,
|
|
12
|
+
isBlackFrame,
|
|
13
|
+
maxCellDelta,
|
|
14
|
+
CELL_CHANGE,
|
|
12
15
|
regionDeltas,
|
|
13
16
|
regionMap,
|
|
14
17
|
signatureDiff,
|
|
@@ -398,6 +401,80 @@ export function resolveBaseline(state, since) {
|
|
|
398
401
|
return { kind: 'unmatched', requested: key };
|
|
399
402
|
}
|
|
400
403
|
|
|
404
|
+
/**
|
|
405
|
+
* Is the baseline describing a screen that had already finished moving?
|
|
406
|
+
*
|
|
407
|
+
* Pure, so the rule can be argued with in a test rather than only observed on
|
|
408
|
+
* a device. The full reasoning is at the call site in `waitFor`; the short
|
|
409
|
+
* version is that stillness cannot accumulate in the milliseconds between a
|
|
410
|
+
* dispatch returning and a wait beginning, so a screen that already differs
|
|
411
|
+
* from the baseline *and* has already been at rest for the whole stillness
|
|
412
|
+
* window changed for some earlier reason.
|
|
413
|
+
*/
|
|
414
|
+
/** The newest frame's region signature, which the state already carries. */
|
|
415
|
+
function currentSig(state) {
|
|
416
|
+
const h = state?.history;
|
|
417
|
+
if (!h?.length) return null;
|
|
418
|
+
const newest = h.find((x) => x.seq === state.seq) ?? h[h.length - 1];
|
|
419
|
+
return newest?.sig ?? null;
|
|
420
|
+
}
|
|
421
|
+
|
|
422
|
+
/**
|
|
423
|
+
* The longest pause inside a transition, measured after the transition is over.
|
|
424
|
+
*
|
|
425
|
+
* This is the unbiased half of the estimator that was reverted in Phase 11, and
|
|
426
|
+
* the difference is entirely about *when* the measurement stops.
|
|
427
|
+
*
|
|
428
|
+
* The biased version accumulated the statistic from inside the wait: the
|
|
429
|
+
* longest stretch of stillness the wait itself happened to observe. Feed that
|
|
430
|
+
* back into how long the next wait runs and it eats itself — a wait that ends
|
|
431
|
+
* early never sees the pauses that come later, so the gaps read as zero, the
|
|
432
|
+
* window ratchets down, the next wait ends earlier still, and eventually a
|
|
433
|
+
* settle returns mid-transition and the graph learns a screen it never reached.
|
|
434
|
+
* That is not a theory; it corrupted this device's graph in one afternoon.
|
|
435
|
+
*
|
|
436
|
+
* This one reads the frame history *after* the fact, over a window whose end is
|
|
437
|
+
* not decided by the wait. The frames are already on disk with their timestamps
|
|
438
|
+
* and their hashes, so the true profile of a transition is recoverable as long
|
|
439
|
+
* as the history still reaches back to it.
|
|
440
|
+
*
|
|
441
|
+
* That last condition is the whole reason this returns null rather than a
|
|
442
|
+
* number: the ring is bounded, and during fast motion it holds a second or two.
|
|
443
|
+
* A partial window would produce a *shorter* gap than really occurred, which is
|
|
444
|
+
* the exact direction of the bias being removed. Reporting nothing is the only
|
|
445
|
+
* honest answer to a question the evidence cannot reach.
|
|
446
|
+
*/
|
|
447
|
+
export function longestQuietGap(history, sinceMs, untilMs = Date.now()) {
|
|
448
|
+
if (!Array.isArray(history) || history.length < 2) return null;
|
|
449
|
+
const frames = history
|
|
450
|
+
.filter((f) => Number.isFinite(f?.at) && f.at >= sinceMs && f.at <= untilMs)
|
|
451
|
+
.sort((a, b) => a.at - b.at);
|
|
452
|
+
if (frames.length < 2) return null;
|
|
453
|
+
// The history has to reach back to the action itself. One frame interval of
|
|
454
|
+
// slack, because the frame that captures the moment of the action is not
|
|
455
|
+
// required to land exactly on it.
|
|
456
|
+
const span = frames[1].at - frames[0].at;
|
|
457
|
+
if (frames[0].at > sinceMs + Math.max(250, span)) return null;
|
|
458
|
+
|
|
459
|
+
let longest = 0;
|
|
460
|
+
let lastChangeAt = frames[0].at;
|
|
461
|
+
for (let i = 1; i < frames.length; i += 1) {
|
|
462
|
+
if (frames[i].hash !== frames[i - 1].hash) {
|
|
463
|
+
longest = Math.max(longest, frames[i].at - lastChangeAt);
|
|
464
|
+
lastChangeAt = frames[i].at;
|
|
465
|
+
}
|
|
466
|
+
}
|
|
467
|
+
// The quiet after the last change is not a pause *inside* the transition —
|
|
468
|
+
// it is the transition being over, which is what a settle already measures.
|
|
469
|
+
return longest;
|
|
470
|
+
}
|
|
471
|
+
|
|
472
|
+
export function baselineAlreadySettled({ mode, changedAtStart, stableForMs, stableMs } = {}) {
|
|
473
|
+
if (mode === 'stable' || !changedAtStart) return false;
|
|
474
|
+
if (!Number.isFinite(stableForMs) || !Number.isFinite(stableMs)) return false;
|
|
475
|
+
return stableForMs >= stableMs;
|
|
476
|
+
}
|
|
477
|
+
|
|
401
478
|
function compareToBaseline(state, baseline) {
|
|
402
479
|
if (!baseline) return null;
|
|
403
480
|
if (baseline.kind === 'history') {
|
|
@@ -497,6 +574,10 @@ export async function getState(deviceQuery, { since, options, inputHealth = fals
|
|
|
497
574
|
map: regionMap(state.regions || [], REGION_COLS),
|
|
498
575
|
since: compareToBaseline(state, resolveBaseline(state, since)),
|
|
499
576
|
live: liveness(device.udid, state),
|
|
577
|
+
// Costs 32 integer comparisons on a signature already computed, and it is
|
|
578
|
+
// the difference between "the screen is calm" and "the display stopped
|
|
579
|
+
// rendering" — which looked identical to everything above this line.
|
|
580
|
+
black: isBlackFrame(currentSig(state)),
|
|
500
581
|
// Off by default and asked for by the state commands only. A flow step
|
|
501
582
|
// calls getState twice, and the fix for a stale session runs before every
|
|
502
583
|
// action anyway (input.ensureFreshSession) — this is the report, not the
|
|
@@ -568,7 +649,7 @@ export async function waitFor(
|
|
|
568
649
|
const p = store.paths(device.udid);
|
|
569
650
|
const requested = since ?? baselineHash;
|
|
570
651
|
const resolved = resolveBaseline(first, requested);
|
|
571
|
-
|
|
652
|
+
let baselineHashValue =
|
|
572
653
|
resolved?.kind === 'history' ? resolved.entry.hash : (requested ?? first.hash);
|
|
573
654
|
const baselineResolved = resolved?.kind === 'history' || requested == null;
|
|
574
655
|
|
|
@@ -590,10 +671,62 @@ export async function waitFor(
|
|
|
590
671
|
*/
|
|
591
672
|
let quietGapMs = 0;
|
|
592
673
|
let quietRun = 0;
|
|
674
|
+
/**
|
|
675
|
+
* The baseline's own signature, for changes the mean cannot see.
|
|
676
|
+
*
|
|
677
|
+
* Only when the baseline was found in the frame history — a hash we cannot
|
|
678
|
+
* place has no signature to compare against, and guessing one would be worse
|
|
679
|
+
* than not looking.
|
|
680
|
+
*/
|
|
681
|
+
const baselineSig = resolved?.kind === 'history' && resolved.entry?.sig
|
|
682
|
+
? hexToSignature(resolved.entry.sig)
|
|
683
|
+
: (requested == null ? hexToSignature(currentSig(first) ?? '') : null);
|
|
684
|
+
let smallChange = false;
|
|
685
|
+
/** Frames the display was not rendering at all. See `isBlackFrame`. */
|
|
686
|
+
let blackFrames = 0;
|
|
687
|
+
let blackSinceStart = null;
|
|
593
688
|
let lastHash = first.hash;
|
|
594
689
|
let sawChange = mode === 'stable' || first.hash !== baselineHashValue;
|
|
595
690
|
const changedAtStart = sawChange && mode !== 'stable';
|
|
596
691
|
|
|
692
|
+
/**
|
|
693
|
+
* The baseline describes a screen that has already finished moving.
|
|
694
|
+
*
|
|
695
|
+
* `since` means "the screen as it was before the action", and the whole
|
|
696
|
+
* reliability of these scripts rests on it being captured *before* rather
|
|
697
|
+
* than after — a baseline sampled afterwards is the commonest way to wait for
|
|
698
|
+
* a change that already happened. What was missing is the other end of it:
|
|
699
|
+
* time also passes between capturing the baseline and dispatching the action,
|
|
700
|
+
* and in a flow step that gap holds a `locate`, a perception pass and a
|
|
701
|
+
* settle wait — hundreds of milliseconds, not microseconds.
|
|
702
|
+
*
|
|
703
|
+
* So a transition can begin *and finish* in that gap, and then `sawChange` is
|
|
704
|
+
* true at wait start because of the previous action's animation. Measured:
|
|
705
|
+
* `tap Accessibility` returned `settled 124ms` against a 500 ms stillness
|
|
706
|
+
* window, the screen had never left the Settings root, and the graph recorded
|
|
707
|
+
* `root -> root` as a verified edge — count 11, changedOutcomes 5, flipping
|
|
708
|
+
* between the real destination and itself all day.
|
|
709
|
+
*
|
|
710
|
+
* The test is unambiguous rather than clever: the screen differs from the
|
|
711
|
+
* baseline *and has already been at rest for the full stillness window*.
|
|
712
|
+
* Stillness cannot have accumulated in the milliseconds between a dispatch
|
|
713
|
+
* returning and this call starting, so whatever changed, changed and settled
|
|
714
|
+
* before we looked, and it is not this action's doing. Re-baseline to what is
|
|
715
|
+
* actually on screen and wait for a further change — which is what the caller
|
|
716
|
+
* asked for and what a stale hash prevented.
|
|
717
|
+
*
|
|
718
|
+
* A screen that differs and is *still moving* is left alone: that is
|
|
719
|
+
* genuinely ambiguous, and after an action the usual reading is the right
|
|
720
|
+
* one.
|
|
721
|
+
*/
|
|
722
|
+
const staleBaseline = baselineAlreadySettled({
|
|
723
|
+
mode, changedAtStart, stableForMs: first.stableForMs, stableMs,
|
|
724
|
+
});
|
|
725
|
+
if (staleBaseline) {
|
|
726
|
+
baselineHashValue = first.hash;
|
|
727
|
+
sawChange = false;
|
|
728
|
+
}
|
|
729
|
+
|
|
597
730
|
const done = (satisfied, extra = {}) => ({
|
|
598
731
|
device,
|
|
599
732
|
state: last,
|
|
@@ -602,6 +735,15 @@ export async function waitFor(
|
|
|
602
735
|
quietGapMs,
|
|
603
736
|
sawChange,
|
|
604
737
|
changedBeforeWait: changedAtStart,
|
|
738
|
+
// Something moved, but only in one region — a control changing state
|
|
739
|
+
// rather than a screen changing.
|
|
740
|
+
smallChange,
|
|
741
|
+
staleBaseline,
|
|
742
|
+
blackFrames,
|
|
743
|
+
// Said as an observation, never as a diagnosis: a screen can be black
|
|
744
|
+
// because the app drew black. What makes it the capture wedge is that it
|
|
745
|
+
// stays black while input is being delivered, and the caller knows that.
|
|
746
|
+
blackMs: blackSinceStart ? Date.now() - blackSinceStart : 0,
|
|
605
747
|
baselineHash: baselineHashValue,
|
|
606
748
|
baselineResolved,
|
|
607
749
|
waitedMs: Date.now() - startedAt,
|
|
@@ -617,8 +759,44 @@ export async function waitFor(
|
|
|
617
759
|
const live = liveness(device.udid, state);
|
|
618
760
|
if (!live.ok) return done(false, { stalled: true });
|
|
619
761
|
|
|
762
|
+
// A black frame is not evidence, in either direction.
|
|
763
|
+
//
|
|
764
|
+
// The capture wedge leaves every frame black while the whole capture path
|
|
765
|
+
// reports success, so before this a settle read the black screen as a
|
|
766
|
+
// change (the hash differs from anything) and then as a calm one (nothing
|
|
767
|
+
// moves), and returned `ok` for an action nobody could see the result of.
|
|
768
|
+
// It self-recovers most times, so the useful behaviour is to keep waiting
|
|
769
|
+
// rather than to conclude.
|
|
770
|
+
const black = isBlackFrame(currentSig(state));
|
|
771
|
+
if (black) {
|
|
772
|
+
blackFrames += 1;
|
|
773
|
+
blackSinceStart = blackSinceStart ?? Date.now();
|
|
774
|
+
await sleep(60);
|
|
775
|
+
continue;
|
|
776
|
+
}
|
|
777
|
+
blackSinceStart = null;
|
|
778
|
+
|
|
620
779
|
if (!sawChange && state.hash !== baselineHashValue) sawChange = true;
|
|
621
780
|
|
|
781
|
+
// A change too small for the whole-screen mean to see.
|
|
782
|
+
//
|
|
783
|
+
// A switch flipping moves one cell of thirty-two by 0.043 and the mean by
|
|
784
|
+
// 0.0013 — a third of the threshold — so every switch, radio dot,
|
|
785
|
+
// checkbox and segment highlight was an action that "changed nothing",
|
|
786
|
+
// and `no-visible-change` is a verdict that escalates. See
|
|
787
|
+
// analyze.CELL_CHANGE for the measured gap this sits in.
|
|
788
|
+
//
|
|
789
|
+
// Deliberately feeding `sawChange` and *not* stillness: `stableForMs`
|
|
790
|
+
// stays on the mean, because a blinking text caret is a small localised
|
|
791
|
+
// change and a screen with a cursor would otherwise never settle.
|
|
792
|
+
if (!sawChange && baselineSig) {
|
|
793
|
+
const sig = currentSig(state);
|
|
794
|
+
if (sig && maxCellDelta(hexToSignature(sig), baselineSig) > CELL_CHANGE) {
|
|
795
|
+
sawChange = true;
|
|
796
|
+
smallChange = true;
|
|
797
|
+
}
|
|
798
|
+
}
|
|
799
|
+
|
|
622
800
|
// A pause that turned out not to be the end of the transition. Only
|
|
623
801
|
// pauses followed by more movement count: the quiet at the end of a
|
|
624
802
|
// settle is the answer, not a gap.
|
|
@@ -985,7 +1163,11 @@ export async function locate(
|
|
|
985
1163
|
`"${query}" matches ${outcome.alternatives.length} things on this screen — say which, or pass index: ${list}`,
|
|
986
1164
|
),
|
|
987
1165
|
'ambiguous_intent',
|
|
988
|
-
|
|
1166
|
+
// Present, several times over — as opposed to absent, which also tags
|
|
1167
|
+
// ambiguous_intent when the screen was one we thought we knew. A
|
|
1168
|
+
// waiting caller needs the difference: more time cannot make a thing
|
|
1169
|
+
// unique, and it can make an absent thing arrive.
|
|
1170
|
+
{ candidates: outcome.alternatives, ambiguous: true },
|
|
989
1171
|
);
|
|
990
1172
|
}
|
|
991
1173
|
if (outcome.status === 'ok') {
|
|
@@ -1039,6 +1221,38 @@ export async function locate(
|
|
|
1039
1221
|
*/
|
|
1040
1222
|
export const STRUCTURAL_SETTLE_MS = 300;
|
|
1041
1223
|
|
|
1224
|
+
/**
|
|
1225
|
+
* How much of the structural window is still owed, given when the last sample's
|
|
1226
|
+
* frame was captured.
|
|
1227
|
+
*
|
|
1228
|
+
* This was `sleep(300)` between the two readings, and unlike every other fixed
|
|
1229
|
+
* wait in the engine it cannot be replaced by waiting for a signal — because
|
|
1230
|
+
* there is no signal. The race it guards is a screen whose *pixels* have gone
|
|
1231
|
+
* still while its structure has not: a list whose spinner has gone and whose
|
|
1232
|
+
* rows have not landed is perfectly quiet and structurally wrong, so the settle
|
|
1233
|
+
* detector, which watches pixels, has nothing to report. Only elapsed time
|
|
1234
|
+
* separates the two readings.
|
|
1235
|
+
*
|
|
1236
|
+
* What can be fixed is that the wait was *additional*. The guarantee wanted is
|
|
1237
|
+
* 300 ms between the frames the two samples read; the code slept 300 ms after a
|
|
1238
|
+
* sample that had already spent an unbounded settle wait and a full perception
|
|
1239
|
+
* pass getting there. On a screen that took a second to go quiet the separation
|
|
1240
|
+
* was already there and the sleep bought nothing but a second of it. So credit
|
|
1241
|
+
* what has passed and wait only for the remainder — the same guarantee, and
|
|
1242
|
+
* usually none of the sleep.
|
|
1243
|
+
*
|
|
1244
|
+
* The window itself is per-screen learnable, and worth noting that its
|
|
1245
|
+
* estimator has the *opposite* feedback sign to the one that corrupted the
|
|
1246
|
+
* graph: a window too short produces disagreeing samples, which lengthens it.
|
|
1247
|
+
* Self-correcting rather than self-reinforcing. It still waits on the
|
|
1248
|
+
* perception eval harness, because "the samples agreed" is only evidence the
|
|
1249
|
+
* window was long enough if the readings themselves are trustworthy.
|
|
1250
|
+
*/
|
|
1251
|
+
export function structuralSettleOwed(capturedAt, now = Date.now(), windowMs = STRUCTURAL_SETTLE_MS) {
|
|
1252
|
+
if (!Number.isFinite(capturedAt)) return windowMs;
|
|
1253
|
+
return Math.max(0, Math.min(windowMs, windowMs - (now - capturedAt)));
|
|
1254
|
+
}
|
|
1255
|
+
|
|
1042
1256
|
/**
|
|
1043
1257
|
* How long identity will wait for pixels to go quiet.
|
|
1044
1258
|
*
|
|
@@ -1122,7 +1336,8 @@ export async function screenIdentity(deviceQuery, { options, confirmNovel = true
|
|
|
1122
1336
|
// Nothing recognises this, or the pixels have not gone quiet. Either way, make
|
|
1123
1337
|
// it prove it is the same screen twice running before it becomes a node.
|
|
1124
1338
|
for (let i = 1; i < STRUCTURAL_SETTLE_SAMPLES; i += 1) {
|
|
1125
|
-
|
|
1339
|
+
const owed = structuralSettleOwed(identity.state?.capturedAt);
|
|
1340
|
+
if (owed > 0) await sleep(owed);
|
|
1126
1341
|
const again = await read({ fresh: true });
|
|
1127
1342
|
// Two readings agree if they are the same screen — the same test identity
|
|
1128
1343
|
// itself uses. Demanding an identical hash is a stricter question than the
|
package/src/input.js
CHANGED
|
@@ -248,6 +248,11 @@ export function elementToNode(e) {
|
|
|
248
248
|
type: e.role ?? null,
|
|
249
249
|
identifier: e.identifier ?? null,
|
|
250
250
|
enabled: e.state?.enabled ?? null,
|
|
251
|
+
// The daemon batches AXSelected and AXFocused alongside AXEnabled and has
|
|
252
|
+
// since 0.6.0. This converter took one of the three, and it is the one on
|
|
253
|
+
// the path that actually runs — `normalizeNode` below is the idb fallback.
|
|
254
|
+
selected: e.state?.selected ?? null,
|
|
255
|
+
focused: e.state?.focused ?? null,
|
|
251
256
|
frame: e.frame ?? null,
|
|
252
257
|
raw: e,
|
|
253
258
|
};
|
|
@@ -273,6 +278,12 @@ function normalizeNode(node) {
|
|
|
273
278
|
type: node.type ?? node.AXType ?? null,
|
|
274
279
|
identifier: node.AXUniqueId ?? node.identifier ?? null,
|
|
275
280
|
enabled: node.AXEnabled ?? node.enabled ?? null,
|
|
281
|
+
// The daemon has asked the tree for AXSelected and AXFocused since 0.6.0 —
|
|
282
|
+
// they are two of the eight attributes in its batched round trip — and this
|
|
283
|
+
// function dropped both. `view.renderRow` has printed `selected` for as
|
|
284
|
+
// long as it has existed, against a field nobody set.
|
|
285
|
+
selected: node.AXSelected ?? node.selected ?? null,
|
|
286
|
+
focused: node.AXFocused ?? node.focused ?? null,
|
|
276
287
|
frame: frame
|
|
277
288
|
? {
|
|
278
289
|
x: frame.x ?? frame.X ?? 0,
|
|
@@ -506,19 +517,52 @@ export async function sessionHealth(udid) {
|
|
|
506
517
|
}
|
|
507
518
|
|
|
508
519
|
/**
|
|
509
|
-
*
|
|
520
|
+
* Should this staleness be acted on, given what has already been rebuilt?
|
|
521
|
+
*
|
|
522
|
+
* Pure, and separate from the check because the *key* is the whole bug. This
|
|
523
|
+
* gate used to be a set of udids — "once per process, per device" — and the
|
|
524
|
+
* reasoning was to avoid statting on every action. What it actually bought was
|
|
525
|
+
* that the feature could not fire in the one process that matters. A CLI
|
|
526
|
+
* command is a new process every time, so per-process is per-call there and
|
|
527
|
+
* the gate never showed; the MCP server is a single process that lives for a
|
|
528
|
+
* whole session, so it checked once, at the first action, and then never
|
|
529
|
+
* again — and a device that reboots *mid-session* is precisely the case this
|
|
530
|
+
* exists to catch. Reported from a real session: capture kept working, input
|
|
531
|
+
* died, every tap returned `ok`, and about ten calls went into two wrong
|
|
532
|
+
* conclusions about the app.
|
|
533
|
+
*
|
|
534
|
+
* The right key is the boot the rebuild was for. One attempt per device boot:
|
|
535
|
+
* enough that a failed rebuild does not retry on every tap forever, and not so
|
|
536
|
+
* much that the next boot is invisible.
|
|
537
|
+
*/
|
|
538
|
+
export function shouldRebuildSession({ stale, bootedAt }, rebuiltFor) {
|
|
539
|
+
if (!stale) return false;
|
|
540
|
+
// A boot we cannot date cannot be memoised against, and re-attempting on
|
|
541
|
+
// every action would be worse than not detecting it. sessionStaleness only
|
|
542
|
+
// reports stale with a finite bootedAt, so this is a belt, not a case.
|
|
543
|
+
if (!Number.isFinite(bootedAt)) return false;
|
|
544
|
+
return rebuiltFor !== bootedAt;
|
|
545
|
+
}
|
|
546
|
+
|
|
547
|
+
/**
|
|
548
|
+
* Rebuild the session if the device outlived it. Once per device boot.
|
|
510
549
|
*
|
|
511
550
|
* Rebuild, and retry nothing: this runs *before* the action, so the action is
|
|
512
551
|
* delivered on a session known to be current. Retrying afterwards is how an
|
|
513
552
|
* action fires twice, which is the hazard the verify barrier exists to
|
|
514
553
|
* prevent — and it is why the existing recovery covers hardware buttons only.
|
|
554
|
+
*
|
|
555
|
+
* The check now runs on every dispatch rather than once. It costs two small
|
|
556
|
+
* `readJson`s and, at most every three seconds, one stat — `bootedAtCached`
|
|
557
|
+
* already caps the part that was expensive, which is what made the
|
|
558
|
+
* once-per-process gate unnecessary as well as wrong.
|
|
515
559
|
*/
|
|
516
|
-
const
|
|
560
|
+
const rebuiltForBoot = new Map();
|
|
517
561
|
export async function ensureFreshSession(udid) {
|
|
518
|
-
if (!udid
|
|
519
|
-
freshened.add(udid);
|
|
562
|
+
if (!udid) return null;
|
|
520
563
|
const health = await sessionHealth(udid);
|
|
521
|
-
if (!health.
|
|
564
|
+
if (!shouldRebuildSession(health, rebuiltForBoot.get(udid))) return null;
|
|
565
|
+
rebuiltForBoot.set(udid, health.bootedAt);
|
|
522
566
|
const rebuilt = await resetSession(udid);
|
|
523
567
|
return { ...health, rebuilt };
|
|
524
568
|
}
|
package/src/matching.js
CHANGED
|
@@ -59,7 +59,25 @@ export function nameScore(name, query) {
|
|
|
59
59
|
const q = norm(query);
|
|
60
60
|
if (!n || !q) return 0;
|
|
61
61
|
if (n === q) return 1;
|
|
62
|
-
|
|
62
|
+
// A prefix match is only as good as the share it covers, and this branch had
|
|
63
|
+
// to be taught that twice.
|
|
64
|
+
//
|
|
65
|
+
// `n.startsWith(q)` — the name begins with the query, "Acce" for
|
|
66
|
+
// "Accessibility" — is the ordinary case and keeps most of its score: a
|
|
67
|
+
// prefix of a name is how people abbreviate. `q.startsWith(n)` is the
|
|
68
|
+
// opposite direction, where the *name* is a fragment of the query, and it
|
|
69
|
+
// returned the same flat 0.86 no matter how little of the query it was. So
|
|
70
|
+
// the section-index letter "S" scored 0.86 against the query "Search" and
|
|
71
|
+
// beat the search field's own label "Q Search" at 0.585 — and simframe
|
|
72
|
+
// tapped a scrubber and typed into it.
|
|
73
|
+
//
|
|
74
|
+
// The sibling branch below already carries this lesson in a comment about
|
|
75
|
+
// "back" matching a list row. Only one of the two had learned it. Both scale
|
|
76
|
+
// now, and the floor differs by direction on purpose: a query that is a
|
|
77
|
+
// prefix of a name is usually deliberate, while a name that is a fragment of
|
|
78
|
+
// the query is usually a coincidence, and one character is always one.
|
|
79
|
+
if (n.startsWith(q)) return 0.86 * Math.max(PREFIX_FLOOR, Math.min(1, q.length / n.length + 0.35));
|
|
80
|
+
if (q.startsWith(n)) return 0.8 * Math.max(0.15, n.length / q.length);
|
|
63
81
|
// A substring match is only as good as the share of the name it covers.
|
|
64
82
|
// Without this, "back" scores 0.78 against a two-hundred-character list row
|
|
65
83
|
// that happens to contain "Back of House", and beats the actual back button.
|
|
@@ -76,7 +94,31 @@ export function nameScore(name, query) {
|
|
|
76
94
|
if (distance > cap) return 0;
|
|
77
95
|
const longest = Math.max(n.length, q.length);
|
|
78
96
|
const similarity = 1 - distance / longest;
|
|
79
|
-
|
|
97
|
+
if (similarity >= 0.7) return similarity * 0.72;
|
|
98
|
+
// Last tier: OCR read a confusable character.
|
|
99
|
+
//
|
|
100
|
+
// Language correction is deliberately off, which is right for labels and
|
|
101
|
+
// wrong for exactly this. Measured on a real app, `(All)` reads back as
|
|
102
|
+
// `(AII)` and would fail an assert against the string it is; capital-I,
|
|
103
|
+
// lowercase-l, the digit one and a pipe are one shape in most UI fonts, as
|
|
104
|
+
// are capital-O and zero.
|
|
105
|
+
//
|
|
106
|
+
// Deliberately the *last* tier and discounted, not part of `norm`. Folding
|
|
107
|
+
// in `norm` would make it change what an exact match means — "Log in" and
|
|
108
|
+
// "1og in" would become the same string everywhere — and identity is not
|
|
109
|
+
// something to be fuzzy about. Here it only ever rescues a comparison that
|
|
110
|
+
// had already scored zero.
|
|
111
|
+
const foldedScore = confusableFold(n) === confusableFold(q) ? 0.62 : 0;
|
|
112
|
+
return foldedScore;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/** One shape per glyph family, for comparison only. Never for identity. */
|
|
116
|
+
export function confusableFold(s) {
|
|
117
|
+
return String(s ?? '')
|
|
118
|
+
.replace(/[il1|!]/gi, '1')
|
|
119
|
+
.replace(/[o0]/gi, '0')
|
|
120
|
+
.replace(/[s5]/gi, '5')
|
|
121
|
+
.replace(/[b8]/gi, '8');
|
|
80
122
|
}
|
|
81
123
|
|
|
82
124
|
function synonymGroup(query) {
|
|
@@ -159,6 +201,17 @@ export function rank(targets, intent, { screen } = {}) {
|
|
|
159
201
|
export const AMBIGUITY_MARGIN = 0.08;
|
|
160
202
|
/** Below this, no candidate is worth acting on. */
|
|
161
203
|
export const MINIMUM_SCORE = 0.45;
|
|
204
|
+
/**
|
|
205
|
+
* How much of a name's score survives scaling a prefix by its coverage.
|
|
206
|
+
*
|
|
207
|
+
* Tied to `MINIMUM_SCORE` rather than chosen: at 0.5 the shortest useful
|
|
208
|
+
* abbreviation — "Ac" for "Accessibility" — scored 0.433 and fell *below* the
|
|
209
|
+
* threshold to resolve at all, which would have turned a ranking fix into a
|
|
210
|
+
* feature removal. 0.6 puts the worst case at 0.516, comfortably resolvable and
|
|
211
|
+
* still well under a fuller match. The unit test asserts the relationship so
|
|
212
|
+
* the cliff cannot come back by someone tuning one of the two numbers.
|
|
213
|
+
*/
|
|
214
|
+
export const PREFIX_FLOOR = 0.6;
|
|
162
215
|
|
|
163
216
|
/**
|
|
164
217
|
* How close two tap points have to be to mean the same control.
|
package/src/metrics.js
CHANGED
|
@@ -40,8 +40,16 @@ export const FACULTY = {
|
|
|
40
40
|
no_plan: 'exploration (Phase 14)',
|
|
41
41
|
};
|
|
42
42
|
|
|
43
|
-
/**
|
|
44
|
-
|
|
43
|
+
/**
|
|
44
|
+
* Faculties that exist.
|
|
45
|
+
*
|
|
46
|
+
* Phase 11 landed the first one, so `verification_failed` no longer maps to
|
|
47
|
+
* something unbuilt — which changes what those records *mean*. Before, they
|
|
48
|
+
* were a queue waiting on a phase. Now they are evidence that the phase which
|
|
49
|
+
* shipped is not sufficient, and that is a more useful thing for the log to be
|
|
50
|
+
* able to say than a count of things nobody has written yet.
|
|
51
|
+
*/
|
|
52
|
+
export const BUILT_FACULTIES = new Set(['sense of time (Phase 11)']);
|
|
45
53
|
|
|
46
54
|
function metricPaths(udid) {
|
|
47
55
|
const dir = store.deviceDir(udid);
|
|
@@ -111,9 +119,15 @@ export const readFlows = (udid, opts) => readJsonl(metricPaths(udid).flows, opts
|
|
|
111
119
|
* matching error strings at the boundary — a regexed message is a reason that
|
|
112
120
|
* silently becomes "unknown" the day somebody rewords it.
|
|
113
121
|
*/
|
|
114
|
-
export function tag(err, reason, { candidates = [], tried = [] } = {}) {
|
|
122
|
+
export function tag(err, reason, { candidates = [], tried = [], ambiguous = false } = {}) {
|
|
115
123
|
if (!REASONS.includes(reason)) throw new Error(`not an escalation reason: ${reason}`);
|
|
116
|
-
|
|
124
|
+
// `ambiguous` is narrower than the reason, and that is the point. Two very
|
|
125
|
+
// different failures both tag `ambiguous_intent`: the target is on screen
|
|
126
|
+
// several times over, and the target is not on screen at all on a screen we
|
|
127
|
+
// thought we knew. Only the first is resolvable by *choosing*, and only the
|
|
128
|
+
// first tells a waiting caller that waiting is pointless — the thing it is
|
|
129
|
+
// waiting for has already arrived.
|
|
130
|
+
err.escalation = { reason, candidates, tried, ambiguous };
|
|
117
131
|
return err;
|
|
118
132
|
}
|
|
119
133
|
|
|
@@ -213,8 +227,50 @@ export function fingerprintNow(udid, screenmap) {
|
|
|
213
227
|
* measurement's clothes, and every number in this project is supposed to say
|
|
214
228
|
* where it came from.
|
|
215
229
|
*/
|
|
230
|
+
/**
|
|
231
|
+
* Which run of which program wrote a record.
|
|
232
|
+
*
|
|
233
|
+
* The log is per-device and, until now, anonymous — so two agents driving one
|
|
234
|
+
* booted simulator wrote one interleaved file with no way to separate them.
|
|
235
|
+
* Measured on the bench device in a single evening: 57 records to 81, a third
|
|
236
|
+
* of the new ones naming screens from an app the suite has never launched.
|
|
237
|
+
*
|
|
238
|
+
* That is not a corrupted file, it is a corrupted instrument. CLAUDE.md makes
|
|
239
|
+
* the reason breakdown of this log the thing that chooses which faculty gets
|
|
240
|
+
* built next, and a breakdown that silently pools two sessions errs toward
|
|
241
|
+
* whichever of them made more mistakes — which is not the same question as
|
|
242
|
+
* which faculty is missing.
|
|
243
|
+
*
|
|
244
|
+
* A pid alone would not do: pids are reused, and the useful grouping is "one
|
|
245
|
+
* agent's run", which for the MCP server is the life of the process and for
|
|
246
|
+
* the CLI is a single command. So: the start time, the pid, and a random tail,
|
|
247
|
+
* computed once per process. `client` says what kind of process it was, since
|
|
248
|
+
* "the MCP server did this" and "somebody ran a CLI command" deserve different
|
|
249
|
+
* readings of the same reason.
|
|
250
|
+
*
|
|
251
|
+
* Deliberately not a device id, a username, or anything about the machine. This
|
|
252
|
+
* file is committed to a public repo in summary form, and the question it has
|
|
253
|
+
* to answer is "was this all one agent", which needs no identity to answer.
|
|
254
|
+
*/
|
|
255
|
+
const SESSION_ID = `${process.pid.toString(36)}-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 8)}`;
|
|
256
|
+
|
|
257
|
+
/** How this process is being used, for reading a breakdown afterwards. */
|
|
258
|
+
function clientKind() {
|
|
259
|
+
const argv = process.argv.join(' ');
|
|
260
|
+
if (/\bmcp\b/.test(argv)) return 'mcp';
|
|
261
|
+
if (/bench-hpi|scripts\//.test(argv)) return 'script';
|
|
262
|
+
if (/cli\.js|\bsimframe\b/.test(argv)) return 'cli';
|
|
263
|
+
return 'library';
|
|
264
|
+
}
|
|
265
|
+
const CLIENT = clientKind();
|
|
266
|
+
|
|
267
|
+
/** The session this process's records belong to. Exported for `escalations`. */
|
|
268
|
+
export const sessionId = () => SESSION_ID;
|
|
269
|
+
export const clientName = () => CLIENT;
|
|
270
|
+
|
|
216
271
|
export function recordEscalation(udid, {
|
|
217
272
|
flowId = null,
|
|
273
|
+
flowName = null,
|
|
218
274
|
stepIndex = null,
|
|
219
275
|
fingerprint = null,
|
|
220
276
|
reason,
|
|
@@ -229,7 +285,13 @@ export function recordEscalation(udid, {
|
|
|
229
285
|
if (!OUTCOMES.includes(outcome)) throw new Error(`not an escalation outcome: ${outcome}`);
|
|
230
286
|
const record = {
|
|
231
287
|
timestamp: new Date().toISOString(),
|
|
288
|
+
// Added after the log turned out to pool two agents' work invisibly. Both
|
|
289
|
+
// are cheap and neither is derivable afterwards, which is the test for
|
|
290
|
+
// whether a field belongs in a log at all.
|
|
291
|
+
session_id: SESSION_ID,
|
|
292
|
+
client: CLIENT,
|
|
232
293
|
flow_id: flowId,
|
|
294
|
+
flow_name: flowName,
|
|
233
295
|
step_index: stepIndex,
|
|
234
296
|
screen_fingerprint: fingerprint,
|
|
235
297
|
reason,
|
|
@@ -408,14 +470,31 @@ export function hpi({ flows, baselines = {} }) {
|
|
|
408
470
|
* rate is 1.0 by construction and says nothing. The per-reason breakdown is
|
|
409
471
|
* the part that decides the next phase, and it is informative today.
|
|
410
472
|
*/
|
|
411
|
-
export function breakdown(records) {
|
|
473
|
+
export function breakdown(records, { session = null, flow = null } = {}) {
|
|
412
474
|
const byReason = {};
|
|
413
475
|
for (const r of REASONS) byReason[r] = 0;
|
|
414
476
|
const byScreen = new Map();
|
|
415
477
|
const byOutcome = {};
|
|
478
|
+
const bySession = new Map();
|
|
479
|
+
const byFlow = new Map();
|
|
416
480
|
let avoidable = 0;
|
|
417
|
-
|
|
481
|
+
let unattributed = 0;
|
|
482
|
+
// Filtering happens here rather than at the call site so `total` and every
|
|
483
|
+
// rate below it describe the same set of records.
|
|
484
|
+
const kept = records.filter((r) => (session ? r?.session_id === session : true))
|
|
485
|
+
.filter((r) => (flow ? r?.flow_name === flow : true));
|
|
486
|
+
for (const r of kept) {
|
|
418
487
|
if (!REASONS.includes(r?.reason)) continue;
|
|
488
|
+
// Records written before sessions were recorded cannot be attributed, and
|
|
489
|
+
// saying how many there are is the difference between a breakdown that
|
|
490
|
+
// pools two agents and one that says it might be.
|
|
491
|
+
if (r.session_id) {
|
|
492
|
+
const key = `${r.session_id}|${r.client ?? '?'}`;
|
|
493
|
+
bySession.set(key, (bySession.get(key) ?? 0) + 1);
|
|
494
|
+
} else {
|
|
495
|
+
unattributed += 1;
|
|
496
|
+
}
|
|
497
|
+
if (r.flow_name) byFlow.set(r.flow_name, (byFlow.get(r.flow_name) ?? 0) + 1);
|
|
419
498
|
byReason[r.reason] += 1;
|
|
420
499
|
byOutcome[r.outcome] = (byOutcome[r.outcome] ?? 0) + 1;
|
|
421
500
|
// Already avoided locally, so not avoidable by anything unbuilt.
|
|
@@ -423,9 +502,25 @@ export function breakdown(records) {
|
|
|
423
502
|
const key = r.screen_fingerprint ?? '(no fingerprint)';
|
|
424
503
|
byScreen.set(key, (byScreen.get(key) ?? 0) + 1);
|
|
425
504
|
}
|
|
426
|
-
const total =
|
|
505
|
+
const total = kept.filter((r) => REASONS.includes(r?.reason)).length;
|
|
506
|
+
const sessions = [...bySession.entries()]
|
|
507
|
+
.map(([key, count]) => {
|
|
508
|
+
const [id, client] = key.split('|');
|
|
509
|
+
return { session_id: id, client, count };
|
|
510
|
+
})
|
|
511
|
+
.sort((a, b) => b.count - a.count);
|
|
427
512
|
return {
|
|
428
513
|
total,
|
|
514
|
+
// The log is per-device and shared: two agents on one booted simulator
|
|
515
|
+
// write one interleaved file. More than one session here means the counts
|
|
516
|
+
// below are a pool, and CLAUDE.md uses those counts to choose a phase.
|
|
517
|
+
sessions,
|
|
518
|
+
session_count: sessions.length,
|
|
519
|
+
unattributed,
|
|
520
|
+
// Any unattributed record at all makes this a pool: the whole point is
|
|
521
|
+
// that they cannot be told apart, and 92 of them is not "one session".
|
|
522
|
+
pooled: sessions.length > 1 || unattributed > 0,
|
|
523
|
+
by_flow: Object.fromEntries([...byFlow.entries()].sort((a, b) => b[1] - a[1])),
|
|
429
524
|
by_reason: byReason,
|
|
430
525
|
by_outcome: byOutcome,
|
|
431
526
|
faculty: Object.fromEntries(
|
|
@@ -437,7 +532,9 @@ export function breakdown(records) {
|
|
|
437
532
|
.sort((a, b) => b[1] - a[1])
|
|
438
533
|
.slice(0, 10)
|
|
439
534
|
.map(([fingerprint, count]) => ({ fingerprint, count })),
|
|
440
|
-
|
|
535
|
+
// `kept`, not `records` — a filtered breakdown that reports the whole
|
|
536
|
+
// log's model turns is the same class of mistake as pooling two sessions.
|
|
537
|
+
model_turns_spent: kept.reduce((acc, r) => acc + (r.model_turns_spent ?? 0), 0),
|
|
441
538
|
};
|
|
442
539
|
}
|
|
443
540
|
|