simframe 0.16.0 → 0.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +90 -11
- package/package.json +1 -1
- package/scripts/bench-hpi.mjs +8 -0
- package/scripts/ci-device-guard.mjs +27 -0
- package/scripts/ci-memory.mjs +24 -2
- package/scripts/device-state.mjs +5 -62
- package/src/actions.js +118 -3
- package/src/cli.js +84 -4
- package/src/device-state.js +64 -0
- package/src/graph.js +1 -1
- package/src/matching.js +3 -2
- package/src/mcp.js +24 -2
- package/src/metrics.js +144 -3
- package/src/navigate.js +59 -3
- package/src/wedge.js +218 -0
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
// Is a failure the device's fault or the code's?
|
|
2
|
+
//
|
|
3
|
+
// Moved out of `scripts/` on 2026-09-17 so `src/metrics.js` can use it. The
|
|
4
|
+
// escalation log needed it: 78 of the `verification_failed` records on the
|
|
5
|
+
// bench device are `xcrun simctl openurl` failing, an app that would not
|
|
6
|
+
// launch, or capture stopping — the device, filed under a code faculty and
|
|
7
|
+
// reported as evidence for "sense of time (Phase 11)". A steering wheel that
|
|
8
|
+
// counts item 173's own occurrences as a perception problem points somewhere
|
|
9
|
+
// nobody chose.
|
|
10
|
+
//
|
|
11
|
+
// Deliberately dependency-free. `wedge.js` imports `index.js`, and `metrics.js`
|
|
12
|
+
// is imported *by* `index.js`, so a shared table living in either would close a
|
|
13
|
+
// cycle. This module imports nothing and is imported by both.
|
|
14
|
+
|
|
15
|
+
/** Conditions that are the simulator, not the code. Each seen in a real run. */
|
|
16
|
+
export const DEVICE_STATE = [
|
|
17
|
+
[/NSPOSIXErrorDomain.*code=?\s*60|Operation timed out/i, 'simctl stopped answering (NSPOSIXErrorDomain 60)'],
|
|
18
|
+
[/did not produce a frame|produced no frame in \d+s/i, 'the daemon is up and the display renders nothing'],
|
|
19
|
+
[/Timeout waiting for screen surfaces|display surface is not answering|display surface could not be read/i, 'the display surface is wedged'],
|
|
20
|
+
[/no frames buffered|capture is wedged/i, 'capture stopped'],
|
|
21
|
+
[/the second app never launched|could not be dispatched/i, 'an app would not launch'],
|
|
22
|
+
// A launched app that never comes to the front, seen as the tour waiting for
|
|
23
|
+
// one of its landmarks on a screen that is showing a clock and nothing else.
|
|
24
|
+
//
|
|
25
|
+
// Measured on a runner: `ok launch — launched com.apple.Preferences
|
|
26
|
+
// (relaunched)` followed by `waited 8000ms for General: "General" is not on
|
|
27
|
+
// this screen. Visible: 10:50, .?o (the screen has not moved for 6181ms)`.
|
|
28
|
+
// Two labels, one of them a clock, on a still screen — the device is not
|
|
29
|
+
// presenting the app, and the guard called that a check failing on its
|
|
30
|
+
// merits and declined to revive.
|
|
31
|
+
//
|
|
32
|
+
// Deliberately narrow. It requires the wait to have failed AND the screen to
|
|
33
|
+
// have been still AND almost nothing readable: a tour that genuinely asks for
|
|
34
|
+
// the wrong label has a screen full of other labels, and must keep failing
|
|
35
|
+
// rather than being retried into a pass.
|
|
36
|
+
[
|
|
37
|
+
/never arrived[\s\S]*?Visible:[^\n]{0,24}\(the screen has not moved for \d+ms/i,
|
|
38
|
+
'a launched app never came to the front (the screen shows a clock and nothing else)',
|
|
39
|
+
],
|
|
40
|
+
// The same condition, now said outright by the step that suffered it instead
|
|
41
|
+
// of inferred from the shape of the screen afterwards. Item 169 gave `launch`
|
|
42
|
+
// a pid to compare, so a launch that starts a process and never fronts it
|
|
43
|
+
// reports itself; this signature fires on the cause rather than on a
|
|
44
|
+
// consequence that had to be recognised by "two labels, one a clock".
|
|
45
|
+
//
|
|
46
|
+
// It cannot be triggered by a tour asking for the wrong label — only a failed
|
|
47
|
+
// launch emits this sentence — so it needs none of the narrowing above.
|
|
48
|
+
[
|
|
49
|
+
/never came to the front within \d+ms/i,
|
|
50
|
+
'a launched app never came to the front (the launch said so itself, by pid)',
|
|
51
|
+
],
|
|
52
|
+
// Seen on the v0.14.3 bench run: `could not launch com.apple.Preferences:
|
|
53
|
+
// The system shell (SpringBoard:36454) probably crashed.` The guest's window
|
|
54
|
+
// server going down is the device, not the check, and nothing here matched it.
|
|
55
|
+
[
|
|
56
|
+
/system shell \(SpringBoard[^)]*\) probably crashed/i,
|
|
57
|
+
"the guest's SpringBoard crashed, so nothing can be fronted",
|
|
58
|
+
],
|
|
59
|
+
];
|
|
60
|
+
|
|
61
|
+
/** The condition this output shows, or null when the check failed on its merits. */
|
|
62
|
+
export function deviceCause(text) {
|
|
63
|
+
return DEVICE_STATE.find(([re]) => re.test(String(text ?? '')))?.[1] ?? null;
|
|
64
|
+
}
|
package/src/graph.js
CHANGED
|
@@ -869,7 +869,7 @@ export const VERDICTS = ['ok', 'no-visible-change', 'unexpected-screen', 'unveri
|
|
|
869
869
|
* that already exists do its job. A string still works and still means "exact
|
|
870
870
|
* match only", which is right for a stored prediction that has no tokens.
|
|
871
871
|
*/
|
|
872
|
-
function sameScreen(udid, a, b) {
|
|
872
|
+
export function sameScreen(udid, a, b) {
|
|
873
873
|
const hashOf = (v) => (typeof v === 'string' ? v : v?.hash);
|
|
874
874
|
const ha = hashOf(a);
|
|
875
875
|
const hb = hashOf(b);
|
package/src/matching.js
CHANGED
|
@@ -224,8 +224,9 @@ export function rank(targets, intent, { screen } = {}) {
|
|
|
224
224
|
|
|
225
225
|
// **A distinctive fragment of one long name, when nothing else came close.**
|
|
226
226
|
//
|
|
227
|
-
// Reported from the field: `waitFor
|
|
228
|
-
// whose own "Visible:" list printed
|
|
227
|
+
// Reported from the field: a `waitFor` on a seven-digit record number gave up
|
|
228
|
+
// after 20 s on a screen whose own "Visible:" list printed that number inside
|
|
229
|
+
// a heading — `Record #<digits>`, seventeen characters. The reporter's
|
|
229
230
|
// guess was that `#` was being treated as significant, or that the matcher
|
|
230
231
|
// was anchored. It is neither — the number scores **0.287** against the 0.45
|
|
231
232
|
// floor, because the substring branch in `nameScore` scales by how much of
|
package/src/mcp.js
CHANGED
|
@@ -134,7 +134,7 @@ const TOOLS = [
|
|
|
134
134
|
steps: {
|
|
135
135
|
type: 'array',
|
|
136
136
|
description:
|
|
137
|
-
'Ordered steps. Every selector below accepts "Save" | "#3" | "@120,400", in that order of preference. Act: {"tap":"Save"} (add "index" if a label is ambiguous), {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"clear":"Notes"} to empty a field and "clear":true on a type/paste to replace rather than append (drop "into" to type into whatever already has focus, which is how you follow a browser next-field chevron — nothing can be read back then, and the step says so), {"scroll":"down"}, {"scrollTo":"Delete account"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"key":"return"} (the keyboard return/enter key, which is how a mobile search field submits — also escape, tab, space, backspace, and the arrows), {"launch":{"value":"com.example.app","relaunch":true,"args":["-uiTest","1"]}}, {"openUrl":"myapp://x"}, {"permission":{"value":"photos","grant":"grant","bundleId":"com.example.app"}}. Check: {"assert":{"value":"Saved","is":"visible"}} (also gone | enabled | disabled | value with "equals"), {"waitFor":{"value":"Saved","timeoutMs":5000}} (add "failIfStillFor":15000 to stop early once the screen has plainly stopped changing — a 180s wait once burned three minutes on an app that had logged itself out; without it a timeout still reports how long the screen had been still), {"settle":{"stableMs":600}}, {"pause":300}. Recover without a round trip: add "or" to any step for fallback selectors tried locally — {"tap":"Save","or":["Done","Confirm"]} — and {"seek":"change username","budget":6} explores for something not on this screen: it OPENS containers (a real action — state changes), checks, and returns to where it started, refusing to open anything that commits, abandons or answers. It does not tap the target; it leaves you on the screen where the target resolves so you tap it next. Do not point it into a flow whose progress you cannot afford to lose. A long screen is only knowable a viewport at a time, so {"sweep":"all","fill":{"Last Name":"Asadi","Email":"a@b.c"}} goes to the top, then reads and fills section by section to the bottom — filling each field while it is on screen, which beats finding one and scrolling back. Add "from":"here" to sweep down from where you are. It reports which section each element was in, what it filled, and what it never found at any scroll position. Prefer it to scrollTo on forms and long lists. Brief the supervisor from the plan: top-level "supervise" is standing guidance for the whole batch ("lists here render a count header before rows; REVIEW stays disabled until a provider is chosen") and per-step "expect" adds to it. When it stops a run the result names the steps it did not attempt — re-issue them with a corrected "supervise" note if the judgement was wrong.',
|
|
137
|
+
'Ordered steps. Every selector below accepts "Save" | "#3" | "@120,400", in that order of preference. Act: {"tap":"Save"} (add "index" if a label is ambiguous), {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"clear":"Notes"} to empty a field and "clear":true on a type/paste to replace rather than append (drop "into" to type into whatever already has focus, which is how you follow a browser next-field chevron — nothing can be read back then, and the step says so), {"scroll":"down"}, {"scrollTo":"Delete account"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"key":"return"} (the keyboard return/enter key, which is how a mobile search field submits — also escape, tab, space, backspace, and the arrows), {"launch":{"value":"com.example.app","relaunch":true,"args":["-uiTest","1"]}}, {"openUrl":"myapp://x"}, {"permission":{"value":"photos","grant":"grant","bundleId":"com.example.app"}}. Check: {"assert":{"value":"Saved","is":"visible"}} (also gone | enabled | disabled | value with "equals"), {"waitFor":{"value":"Saved","timeoutMs":5000}} (add "failIfStillFor":15000 to stop early once the screen has plainly stopped changing — a 180s wait once burned three minutes on an app that had logged itself out; without it a timeout still reports how long the screen had been still), {"settle":{"stableMs":600}}, {"pause":300}. Recover without a round trip: add "or" to any step for fallback selectors tried locally — {"tap":"Save","or":["Done","Confirm"]} — and "optional":true to any step that may legitimately have nothing to act on, which is how you cross a first-launch nag, a permission sheet or a "What\'s New" in one batch — {"tap":"Not Now","optional":true} is skipped when nothing matches and runs normally when it does, so the rest of the plan survives an interstitial that did not appear — and {"seek":"change username","budget":6} explores for something not on this screen: it OPENS containers (a real action — state changes), checks, and returns to where it started, refusing to open anything that commits, abandons or answers. It does not tap the target; it leaves you on the screen where the target resolves so you tap it next. Do not point it into a flow whose progress you cannot afford to lose. A long screen is only knowable a viewport at a time, so {"sweep":"all","fill":{"Last Name":"Asadi","Email":"a@b.c"}} goes to the top, then reads and fills section by section to the bottom — filling each field while it is on screen, which beats finding one and scrolling back. Add "from":"here" to sweep down from where you are. It reports which section each element was in, what it filled, and what it never found at any scroll position. Prefer it to scrollTo on forms and long lists. Brief the supervisor from the plan: top-level "supervise" is standing guidance for the whole batch ("lists here render a count header before rows; REVIEW stays disabled until a provider is chosen") and per-step "expect" adds to it. When it stops a run the result names the steps it did not attempt — re-issue them with a corrected "supervise" note if the judgement was wrong.',
|
|
138
138
|
items: { type: 'object' },
|
|
139
139
|
},
|
|
140
140
|
autoSettle: {
|
|
@@ -1077,11 +1077,25 @@ async function doScript(target, args, options) {
|
|
|
1077
1077
|
+ ' seen before to compare against. Replay it once with sim_flow_run and it'
|
|
1078
1078
|
+ ' is confirmed. A replay costs no model calls.'
|
|
1079
1079
|
: ' — confirmed; replay with sim_flow_run for zero model calls';
|
|
1080
|
+
// One sentence per reason, because "incomplete-run" and "failed-steps" ask
|
|
1081
|
+
// for opposite things from the caller and the bare reason names neither.
|
|
1082
|
+
const why = {
|
|
1083
|
+
'contradicted-steps': 'a step landed somewhere memory says it should not have,'
|
|
1084
|
+
+ ' so this route is not safe to replay. Check the flagged step before re-running it.',
|
|
1085
|
+
'failed-steps': 'a step ran and failed, so replaying this would fail the same way'
|
|
1086
|
+
+ ' every time. Fix the step and re-run the batch, then save.',
|
|
1087
|
+
'incomplete-run': 'the run stopped before its last step, so the rest of the route'
|
|
1088
|
+
+ ' was never observed. Re-run it to the end, then save.',
|
|
1089
|
+
}[saved.reason];
|
|
1090
|
+
// `verdicts` has a hole wherever a step never produced one — a failed step
|
|
1091
|
+
// has no verification — and joining those printed `(unverified, unverified, )`.
|
|
1092
|
+
const seen = (saved.verdicts ?? []).filter(Boolean);
|
|
1080
1093
|
lines.push(
|
|
1081
1094
|
saved.ok
|
|
1082
1095
|
? `saved as flow "${saved.name}" (${saved.steps} steps)${how}`
|
|
1083
1096
|
: `NOT saved as "${args.saveAs}": ${saved.reason}`
|
|
1084
|
-
+ `${
|
|
1097
|
+
+ `${why ? ` — ${why}` : ''}`
|
|
1098
|
+
+ `${seen.length ? ` (verdicts: ${seen.join(', ')})` : ''}`,
|
|
1085
1099
|
);
|
|
1086
1100
|
}
|
|
1087
1101
|
const escalated = (res.results ?? []).some((r) => metrics.ESCALATING_VERDICTS.has(r.verification?.verdict));
|
|
@@ -1167,6 +1181,14 @@ async function flowRun(target, args, options) {
|
|
|
1167
1181
|
};
|
|
1168
1182
|
}
|
|
1169
1183
|
const lines = [`flow "${args.name}"`, ...stepLines(res)];
|
|
1184
|
+
// Said out loud or not said at all. A flow now records the screen it was
|
|
1185
|
+
// recorded on, and a mismatch that only exists in the return value is a fact
|
|
1186
|
+
// nobody reads — which is the same shape as the check that could not fail.
|
|
1187
|
+
if (res.startedElsewhere) {
|
|
1188
|
+
lines.push(`NOTE: this flow was recorded starting on screen ${res.startedElsewhere.recorded},`
|
|
1189
|
+
+ ` and this replay started on ${res.startedElsewhere.here}. Not refused — content-driven`
|
|
1190
|
+
+ ' screens legitimately change identity — but if the steps below fail to resolve, this is why.');
|
|
1191
|
+
}
|
|
1170
1192
|
lines.push('', await mapFrom(target, options, res.endScreen, { verdictLine: verdictLineFor(res.results) }));
|
|
1171
1193
|
return { content: [text(lines.join('\n'))], isError: !res.ok };
|
|
1172
1194
|
}
|
package/src/metrics.js
CHANGED
|
@@ -12,6 +12,7 @@
|
|
|
12
12
|
import fs from 'node:fs';
|
|
13
13
|
import path from 'node:path';
|
|
14
14
|
import * as store from './store.js';
|
|
15
|
+
import { deviceCause } from './device-state.js';
|
|
15
16
|
|
|
16
17
|
/** The five reasons, from docs/research/03-human-parity.md §8. Nothing else is a reason. */
|
|
17
18
|
export const REASONS = [
|
|
@@ -66,6 +67,52 @@ export const FACULTY = {
|
|
|
66
67
|
*/
|
|
67
68
|
export const BUILT_FACULTIES = new Set(['sense of time (Phase 11)']);
|
|
68
69
|
|
|
70
|
+
/**
|
|
71
|
+
* The faculty a `verification_failed` record points at, by **verdict**.
|
|
72
|
+
*
|
|
73
|
+
* `FACULTY` is keyed on the reason, and for this class the reason is too coarse
|
|
74
|
+
* to steer by. Measured on the bench device's 1022 records: of the
|
|
75
|
+
* `verification_failed` ones, 162 are `no-visible-change`, 26 are
|
|
76
|
+
* `unexpected-screen`, and about 172 are a wait that timed out — three
|
|
77
|
+
* different faculties, all of which the report named as "sense of time
|
|
78
|
+
* (Phase 11)" because that is what the reason maps to.
|
|
79
|
+
*
|
|
80
|
+
* `unexpected-screen` is the clearest case: it is item 174, screen identity
|
|
81
|
+
* fragmenting on content-driven screens, and reporting it as a timing problem
|
|
82
|
+
* is how a tester was once told their unlabeled-control problem was a timing
|
|
83
|
+
* problem.
|
|
84
|
+
*/
|
|
85
|
+
/**
|
|
86
|
+
* The verdict a legacy record carries in the first token of its `detail`.
|
|
87
|
+
*
|
|
88
|
+
* Written as `${verdict}: ${detail}` since the field existed, so the prefix is
|
|
89
|
+
* reliable — but only for records whose detail came from a verdict at all,
|
|
90
|
+
* which is why this returns null rather than guessing on anything else.
|
|
91
|
+
*/
|
|
92
|
+
export function verdictFromDetail(detail) {
|
|
93
|
+
const m = /^([a-z][a-z-]{3,30}):\s/.exec(String(detail ?? ''));
|
|
94
|
+
// Only a verdict that exists. The first draft returned any lowercase prefix
|
|
95
|
+
// and duly reported a verdict called **"capture"** with a count of 8, from
|
|
96
|
+
// details reading `capture: ...`. A parser that invents a category gets it
|
|
97
|
+
// counted, named in a report, and eventually used to choose a phase — which
|
|
98
|
+
// is the whole failure this grouping was added to fix, reproduced inside the
|
|
99
|
+
// fix.
|
|
100
|
+
return m && KNOWN_VERDICTS.has(m[1]) ? m[1] : null;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
export const VERDICT_FACULTY = {
|
|
104
|
+
'unexpected-screen': 'screen identity (item 174)',
|
|
105
|
+
'still-filling-in': 'sense of time (Phase 11)',
|
|
106
|
+
// **`no-visible-change` is deliberately absent, and it is the largest verdict
|
|
107
|
+
// in the log (162 of 1022).** The escalation site has a measured argument for
|
|
108
|
+
// leaving it unclassified and it is right: in the field it came
|
|
109
|
+
// overwhelmingly from tapping an inert text label whose real hit target was
|
|
110
|
+
// an invisible chevron — icon semantics, Phase 15 — but it also covers a
|
|
111
|
+
// switch moving 0.1% of the screen, which is neither faculty. Two causes, one
|
|
112
|
+
// verdict, and no way to tell them apart from here. Putting it in this map
|
|
113
|
+
// would name a faculty for 162 records on a coin flip.
|
|
114
|
+
};
|
|
115
|
+
|
|
69
116
|
function metricPaths(udid) {
|
|
70
117
|
const dir = store.deviceDir(udid);
|
|
71
118
|
return {
|
|
@@ -323,6 +370,17 @@ export function reasonForStepError(step, err) {
|
|
|
323
370
|
// admits "other" collects a pile of "other". What changes is that the record
|
|
324
371
|
// carries whether the reason was *read off the failure* or *assumed*, and
|
|
325
372
|
// the report declines to recommend a faculty for the assumed ones.
|
|
373
|
+
// The device, not a faculty.
|
|
374
|
+
//
|
|
375
|
+
// 78 of this class on the bench device are `xcrun simctl openurl` failing, an
|
|
376
|
+
// app that would not launch, or capture stopping. Those are item 173, and
|
|
377
|
+
// filing them under a code faculty is how the log came to offer "sense of
|
|
378
|
+
// time (Phase 11)" as the remedy for a simulator that had stopped answering.
|
|
379
|
+
// `classified` stays false because no *faculty* was read; `device` says what
|
|
380
|
+
// was, so the report can take these out of the faculty count instead of
|
|
381
|
+
// counting them towards a phase.
|
|
382
|
+
const device = deviceCause(err?.message);
|
|
383
|
+
if (device) return { reason: 'verification_failed', candidates: [], tried: [], classified: false, device };
|
|
326
384
|
return { reason: 'verification_failed', candidates: [], tried: [], classified: false };
|
|
327
385
|
}
|
|
328
386
|
|
|
@@ -336,6 +394,17 @@ export function reasonForStepError(step, err) {
|
|
|
336
394
|
*/
|
|
337
395
|
export const ESCALATING_VERDICTS = new Set(['unexpected-screen', 'no-visible-change']);
|
|
338
396
|
|
|
397
|
+
/**
|
|
398
|
+
* Every verdict the engine can report, for reading legacy details back.
|
|
399
|
+
*
|
|
400
|
+
* Wider than `ESCALATING_VERDICTS` — which is the two that hand back to a model
|
|
401
|
+
* — because a record's detail may carry any of them, and narrower than "any
|
|
402
|
+
* lowercase word", which is what a prefix parser accepts if nobody bounds it.
|
|
403
|
+
*/
|
|
404
|
+
export const KNOWN_VERDICTS = new Set([
|
|
405
|
+
'unexpected-screen', 'no-visible-change', 'still-filling-in', 'unverified', 'ok',
|
|
406
|
+
]);
|
|
407
|
+
|
|
339
408
|
/** How `goto`/`flow run` refusals map. They refuse rather than guess, and the refusal is the hand-back. */
|
|
340
409
|
export const PLAN_REASONS = {
|
|
341
410
|
'unknown-screen': 'unknown_screen',
|
|
@@ -463,6 +532,14 @@ export function recordEscalation(udid, {
|
|
|
463
532
|
// Default `false`, so a caller that does not think about it cannot
|
|
464
533
|
// accidentally claim precision it does not have.
|
|
465
534
|
classified = false,
|
|
535
|
+
// Which device-state condition this failure shows, when it shows one. Kept
|
|
536
|
+
// separate from `reason` because the five-reason vocabulary is fixed and a
|
|
537
|
+
// sixth reason collects a pile of "other" — see REASONS.
|
|
538
|
+
device = null,
|
|
539
|
+
// Which local verdict fired, for the records that have one. Derivable from
|
|
540
|
+
// `detail` today by parsing a prefix, which is exactly the fragility the log
|
|
541
|
+
// should not depend on.
|
|
542
|
+
verdict = null,
|
|
466
543
|
} = {}) {
|
|
467
544
|
if (!REASONS.includes(reason)) throw new Error(`not an escalation reason: ${reason}`);
|
|
468
545
|
if (!OUTCOMES.includes(outcome)) throw new Error(`not an escalation outcome: ${outcome}`);
|
|
@@ -479,6 +556,10 @@ export function recordEscalation(udid, {
|
|
|
479
556
|
// go/no-go, and on its own it answers "what kind of decision is costing us".
|
|
480
557
|
intent: intent ? String(intent).slice(0, 120) : null,
|
|
481
558
|
classified: Boolean(classified),
|
|
559
|
+
// Null unless read. Both of these separate "we know" from "we assumed" for
|
|
560
|
+
// a class that is otherwise one coarse bucket.
|
|
561
|
+
device_cause: device ? String(device).slice(0, 120) : null,
|
|
562
|
+
verdict: verdict ? String(verdict).slice(0, 60) : null,
|
|
482
563
|
step_index: stepIndex,
|
|
483
564
|
screen_fingerprint: fingerprint,
|
|
484
565
|
reason,
|
|
@@ -538,7 +619,9 @@ export function flowRecordFrom({
|
|
|
538
619
|
images_sent: imagesSent,
|
|
539
620
|
input_tokens: null,
|
|
540
621
|
output_tokens: null,
|
|
541
|
-
escalations: escalations.map((e) => ({
|
|
622
|
+
escalations: escalations.map((e) => ({
|
|
623
|
+
reason: e.reason, step_index: e.step_index, outcome: e.outcome, device_cause: e.device_cause ?? null,
|
|
624
|
+
})),
|
|
542
625
|
escalation_count: escalations.length,
|
|
543
626
|
mis_taps: misTaps,
|
|
544
627
|
verdict_histogram: histogram,
|
|
@@ -546,6 +629,22 @@ export function flowRecordFrom({
|
|
|
546
629
|
exploration_events: [],
|
|
547
630
|
completed: Boolean(completed),
|
|
548
631
|
wrong_action_taken: verdicts.includes('unexpected-screen'),
|
|
632
|
+
// Did this run fail because the *simulator* failed?
|
|
633
|
+
//
|
|
634
|
+
// Measured, on a full local suite: of 17 runs, 5 failed because the guest's
|
|
635
|
+
// SpringBoard crashed or `simctl` stopped answering for 90 s. `hpi`
|
|
636
|
+
// counted every one against `HPI_accuracy`, so the gate CI reads was partly
|
|
637
|
+
// measuring SpringBoard's stability.
|
|
638
|
+
//
|
|
639
|
+
// This is item 172 in the other column. There, `HPI_time` took every run's
|
|
640
|
+
// wall clock regardless of completion, so breaking a flow registered as the
|
|
641
|
+
// agent getting quicker; the fix was to time only completed runs. Accuracy
|
|
642
|
+
// had the mirror-image fault and kept it.
|
|
643
|
+
//
|
|
644
|
+
// Derived from the escalations this run already wrote — `device_cause` is
|
|
645
|
+
// set at the step that suffered it — so nothing new has to be plumbed and a
|
|
646
|
+
// run cannot claim a device fault that its own log does not show.
|
|
647
|
+
device_cause: escalations.map((e) => e.device_cause).find(Boolean) ?? null,
|
|
549
648
|
};
|
|
550
649
|
}
|
|
551
650
|
|
|
@@ -691,8 +790,14 @@ export function hpi({ flows, baselines = {} }) {
|
|
|
691
790
|
};
|
|
692
791
|
}).sort((a, b) => a.flow.localeCompare(b.flow));
|
|
693
792
|
|
|
694
|
-
|
|
695
|
-
|
|
793
|
+
// A run the simulator broke is not a run the code got wrong. It is an
|
|
794
|
+
// **unmeasured** run, and it leaves the denominator rather than lowering the
|
|
795
|
+
// score — the same discipline as timing only completed runs (172), and the
|
|
796
|
+
// same discipline as the suite refusing to publish a partial HPI at all.
|
|
797
|
+
const lostToDevice = flows.filter((f) => f.device_cause && !f.completed);
|
|
798
|
+
const measurable = flows.filter((f) => !(f.device_cause && !f.completed));
|
|
799
|
+
const total = measurable.length;
|
|
800
|
+
const clean = measurable.filter((f) => f.completed && !f.wrong_action_taken).length;
|
|
696
801
|
const accuracy = total ? Number((clean / total).toFixed(3)) : null;
|
|
697
802
|
const times = perFlow.map((f) => f.hpi_time).filter((x) => Number.isFinite(x));
|
|
698
803
|
const hpiTime = harmonicMean(times);
|
|
@@ -700,6 +805,11 @@ export function hpi({ flows, baselines = {} }) {
|
|
|
700
805
|
flows: perFlow,
|
|
701
806
|
overall: {
|
|
702
807
|
runs: total,
|
|
808
|
+
// Said out loud, because a denominator that quietly shrinks is worse than
|
|
809
|
+
// one that is wrong: an accuracy of 1.0 over two measurable runs is not
|
|
810
|
+
// the same claim as 1.0 over eighteen.
|
|
811
|
+
runs_lost_to_device: lostToDevice.length,
|
|
812
|
+
device_causes: [...new Set(lostToDevice.map((f) => f.device_cause))],
|
|
703
813
|
flows_measured: perFlow.length,
|
|
704
814
|
flows_with_human_baseline: times.length,
|
|
705
815
|
hpi_accuracy: accuracy,
|
|
@@ -728,12 +838,20 @@ export function breakdown(records, { session = null, flow = null } = {}) {
|
|
|
728
838
|
const classifiedByReason = {};
|
|
729
839
|
const assumedByReason = {};
|
|
730
840
|
for (const r of REASONS) { byReason[r] = 0; classifiedByReason[r] = 0; assumedByReason[r] = 0; }
|
|
841
|
+
// Two more groupings, because the reason alone could not steer. `byVerdict`
|
|
842
|
+
// splits the largest class into the three different things it holds, and
|
|
843
|
+
// `byDevice` takes out the records that are the simulator rather than the
|
|
844
|
+
// code — those were being counted towards a perception phase.
|
|
845
|
+
const byVerdict = new Map();
|
|
846
|
+
const byDevice = new Map();
|
|
731
847
|
const byScreen = new Map();
|
|
732
848
|
const byOutcome = {};
|
|
733
849
|
const bySession = new Map();
|
|
734
850
|
const byFlow = new Map();
|
|
735
851
|
let avoidable = 0;
|
|
736
852
|
let unattributed = 0;
|
|
853
|
+
let derivedVerdicts = 0;
|
|
854
|
+
let derivedDevice = 0;
|
|
737
855
|
// Filtering happens here rather than at the call site so `total` and every
|
|
738
856
|
// rate below it describe the same set of records.
|
|
739
857
|
const kept = records.filter((r) => (session ? r?.session_id === session : true))
|
|
@@ -757,6 +875,18 @@ export function breakdown(records, { session = null, flow = null } = {}) {
|
|
|
757
875
|
// real one gets ignored, which this file already knows in another place.
|
|
758
876
|
if (r.classified === true) classifiedByReason[r.reason] += 1;
|
|
759
877
|
else if (r.classified === false) assumedByReason[r.reason] += 1;
|
|
878
|
+
// Derived for the records that predate the fields, rather than waiting for
|
|
879
|
+
// a fresh corpus. `detail` already carries both facts — the verdict as its
|
|
880
|
+
// prefix, the device signature inside the message — so a read-time
|
|
881
|
+
// derivation turns 1022 existing records into signal without rewriting a
|
|
882
|
+
// single line of the log. Marked in the output as derived, because a
|
|
883
|
+
// recorded fact and a parsed one are not the same evidence.
|
|
884
|
+
const verdict = r.verdict ?? verdictFromDetail(r.detail);
|
|
885
|
+
const device = r.device_cause ?? deviceCause(r.detail);
|
|
886
|
+
if (verdict) byVerdict.set(verdict, (byVerdict.get(verdict) ?? 0) + 1);
|
|
887
|
+
if (device) byDevice.set(device, (byDevice.get(device) ?? 0) + 1);
|
|
888
|
+
if (!r.verdict && verdict) derivedVerdicts += 1;
|
|
889
|
+
if (!r.device_cause && device) derivedDevice += 1;
|
|
760
890
|
byOutcome[r.outcome] = (byOutcome[r.outcome] ?? 0) + 1;
|
|
761
891
|
// Already avoided locally, so not avoidable by anything unbuilt.
|
|
762
892
|
if (r.outcome !== 'resolved_locally') avoidable += 1;
|
|
@@ -770,8 +900,19 @@ export function breakdown(records, { session = null, flow = null } = {}) {
|
|
|
770
900
|
return { session_id: id, client, count };
|
|
771
901
|
})
|
|
772
902
|
.sort((a, b) => b.count - a.count);
|
|
903
|
+
const sorted = (m) => [...m.entries()].sort((a, b) => b[1] - a[1]).map(([name, count]) => ({ name, count }));
|
|
773
904
|
return {
|
|
774
905
|
total,
|
|
906
|
+
// What the reason could not say. `verdicts` is the largest class split into
|
|
907
|
+
// the faculties it actually implies; `device` is the part that is item 173
|
|
908
|
+
// wearing a code reason.
|
|
909
|
+
verdicts: sorted(byVerdict),
|
|
910
|
+
device: sorted(byDevice),
|
|
911
|
+
device_total: [...byDevice.values()].reduce((a, b) => a + b, 0),
|
|
912
|
+
// How much of the two groupings above was parsed out of `detail` rather
|
|
913
|
+
// than recorded at the time. A reader deciding a phase order should know
|
|
914
|
+
// which half they are looking at.
|
|
915
|
+
derived: { verdicts: derivedVerdicts, device: derivedDevice },
|
|
775
916
|
// The log is per-device and shared: two agents on one booted simulator
|
|
776
917
|
// write one interleaved file. More than one session here means the counts
|
|
777
918
|
// below are a pool, and CLAUDE.md uses those counts to choose a phase.
|
package/src/navigate.js
CHANGED
|
@@ -47,6 +47,16 @@ function refuse(udid, result, { detail = null, flowName = null } = {}) {
|
|
|
47
47
|
if (reason) {
|
|
48
48
|
metrics.recordEscalation(udid, {
|
|
49
49
|
reason,
|
|
50
|
+
// Read, not assumed. `recordEscalation` defaults `classified` to false
|
|
51
|
+
// so a careless caller cannot claim precision it does not have, which
|
|
52
|
+
// is right — and this caller is not careless: `result.reason` is a
|
|
53
|
+
// named refusal (`no-route`, `unreplayable-edge`, `unknown-flow`,
|
|
54
|
+
// `arrived-elsewhere`…) and PLAN_REASONS maps it deterministically.
|
|
55
|
+
// Saying nothing filed 82 records on the bench device as "reason
|
|
56
|
+
// assumed" when the reason was known exactly, which makes the log
|
|
57
|
+
// understate its own knowledge and the report decline to name a
|
|
58
|
+
// faculty it was entitled to name.
|
|
59
|
+
classified: true,
|
|
50
60
|
// A refusal by `goto` is about a destination and one by `flow run` is
|
|
51
61
|
// about a named flow. Either is what a breakdown wants to group by.
|
|
52
62
|
flowName,
|
|
@@ -170,14 +180,31 @@ const CONTRADICTED = /^unexpected/;
|
|
|
170
180
|
* and a flow that did not reach its own last step.
|
|
171
181
|
*/
|
|
172
182
|
export function saveFlow(udid, name, script, { force = false } = {}) {
|
|
173
|
-
const
|
|
183
|
+
const results = script.results ?? [];
|
|
184
|
+
const verdicts = results.map((r) => r.verification?.verdict);
|
|
174
185
|
const contradicted = verdicts.filter((v) => v && CONTRADICTED.test(v));
|
|
175
186
|
const ranAll = script.ranSteps == null
|
|
176
187
|
|| script.steps == null
|
|
177
188
|
|| script.ranSteps >= (script.steps?.length ?? 0);
|
|
189
|
+
// A step that ran and failed is not a step that ran.
|
|
190
|
+
//
|
|
191
|
+
// `ranSteps` counts steps *attempted*, and a failing step stops the batch —
|
|
192
|
+
// so a run whose failure is on the **last** step has `ranSteps === steps.length`
|
|
193
|
+
// and passed the gate above. A peer watched `FLOW FAILED — 4 ok, 1 failed (of
|
|
194
|
+
// 5)` save itself as a flow, which is a flow guaranteed to fail forever. The
|
|
195
|
+
// predicate wanted "every step succeeded" and was written as "every step was
|
|
196
|
+
// reached"; on every run except one they are the same sentence.
|
|
197
|
+
//
|
|
198
|
+
// `!r.ok` rather than `r.ok === false` on purpose: a result that does not say
|
|
199
|
+
// it succeeded has not said it succeeded, and a gate that only catches an
|
|
200
|
+
// explicit `false` is one absent field away from being unable to fail.
|
|
201
|
+
const failedSteps = results.filter((r) => !r.ok);
|
|
178
202
|
if (!force && contradicted.length) {
|
|
179
203
|
return { ok: false, reason: 'contradicted-steps', verdicts };
|
|
180
204
|
}
|
|
205
|
+
if (!force && failedSteps.length) {
|
|
206
|
+
return { ok: false, reason: 'failed-steps', verdicts, failed: failedSteps.map((r) => r.index ?? null) };
|
|
207
|
+
}
|
|
181
208
|
if (!force && !ranAll) {
|
|
182
209
|
return { ok: false, reason: 'incomplete-run', verdicts };
|
|
183
210
|
}
|
|
@@ -249,6 +276,28 @@ export async function runFlow(deviceQuery, name, { options, ...runOptions } = {}
|
|
|
249
276
|
// a human doing the same thing. `minSteps` comes from the flow definition or
|
|
250
277
|
// stays null — the step count of a recorded route is not a claim about the
|
|
251
278
|
// shortest one.
|
|
279
|
+
// Is this the screen the flow was recorded on?
|
|
280
|
+
//
|
|
281
|
+
// Reported, not refused, and the distinction is deliberate. A replay from the
|
|
282
|
+
// wrong screen is mostly self-limiting — the first selector does not resolve
|
|
283
|
+
// and the run stops one step in, and the destructive vocabulary is barred by
|
|
284
|
+
// the verify barrier either way. Refusing on a mismatch would put the exact
|
|
285
|
+
// failure mode of item 174 — content-driven screens fragmenting into several
|
|
286
|
+
// identities — in front of the one path that costs zero model calls. So this
|
|
287
|
+
// says what it saw and lets the run proceed, which also measures how often
|
|
288
|
+
// the mismatch is spurious. If it turns out to be rare, it can become a gate;
|
|
289
|
+
// deciding that by reasoning is how 174 got built in the first place.
|
|
290
|
+
let startedElsewhere = null;
|
|
291
|
+
if (flow.startScreen?.hash) {
|
|
292
|
+
try {
|
|
293
|
+
const here = await api.screenIdentity(device.udid, {});
|
|
294
|
+
if (here.hash && !graph.sameScreen(device.udid, here, flow.startScreen)) {
|
|
295
|
+
startedElsewhere = { recorded: flow.startScreen.hash.slice(0, 8), here: here.hash.slice(0, 8) };
|
|
296
|
+
}
|
|
297
|
+
} catch {
|
|
298
|
+
/* not knowing where we are is not a reason to refuse to try */
|
|
299
|
+
}
|
|
300
|
+
}
|
|
252
301
|
const result = await runScript(device.udid, {
|
|
253
302
|
steps: flow.steps,
|
|
254
303
|
stopOnUnexpected: true,
|
|
@@ -256,11 +305,18 @@ export async function runFlow(deviceQuery, name, { options, ...runOptions } = {}
|
|
|
256
305
|
minSteps: flow.minSteps ?? null,
|
|
257
306
|
...runOptions,
|
|
258
307
|
});
|
|
259
|
-
|
|
308
|
+
// Same arithmetic as `saveFlow`, and the same defect: reaching the last step
|
|
309
|
+
// is not passing it. `result.ok` is the run's own verdict and was ignored
|
|
310
|
+
// here, so a replay that failed on its final assert promoted the flow it had
|
|
311
|
+
// just disproved. Both of a peer's saved flows were marked confirmed by
|
|
312
|
+
// replays that failed; the word "clean" in "confirmed by one clean replay"
|
|
313
|
+
// did not exist in the code path.
|
|
314
|
+
const everyStepPassed = (result.results ?? []).every((r) => r.ok);
|
|
315
|
+
const ok = result.ok !== false && everyStepPassed && result.ranSteps === flow.steps.length;
|
|
260
316
|
// A clean replay is the confirmation a first traversal could not give.
|
|
261
317
|
// Only when nothing was contradicted — a replay that ran to the end while
|
|
262
318
|
// objecting to a step is not a promotion.
|
|
263
319
|
const objected = (result.results ?? []).some((r) => CONTRADICTED.test(r.verification?.verdict ?? ''));
|
|
264
320
|
const promoted = ok && !objected ? confirmFlow(device.udid, name) : false;
|
|
265
|
-
return { ok, name, ...result, ...(promoted ? { promoted: true } : {}) };
|
|
321
|
+
return { ok, name, ...result, ...(promoted ? { promoted: true } : {}), ...(startedElsewhere ? { startedElsewhere } : {}) };
|
|
266
322
|
}
|