simframe 0.9.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +165 -5
- package/data/vocabulary/en.json +148 -0
- package/native/ocr.swift +13 -1
- package/native/rank.swift +87 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +43 -3
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
- package/native/simframed/Sources/simframed/main.swift +13 -1
- package/native/supervise.swift +181 -0
- package/package.json +4 -1
- package/scripts/check-package.mjs +22 -2
- package/scripts/check-private.mjs +143 -0
- package/scripts/ci-memory.mjs +104 -20
- package/scripts/eval-perception.mjs +281 -0
- package/scripts/phase17-corpus.mjs +176 -0
- package/skills/simframe/SKILL.md +237 -5
- package/src/actions.js +1825 -38
- package/src/analyze.js +70 -0
- package/src/cli.js +214 -15
- package/src/control.js +1 -0
- package/src/fingerprint.js +19 -1
- package/src/graph.js +193 -11
- package/src/index.js +428 -16
- package/src/input.js +115 -8
- package/src/localhelper.js +155 -0
- package/src/matching.js +119 -3
- package/src/mcp.js +319 -27
- package/src/metrics.js +134 -8
- package/src/navigate.js +10 -7
- package/src/ocr.js +18 -1
- package/src/planner.js +195 -0
- package/src/platform/android.js +3 -2
- package/src/platform/ios.js +2 -1
- package/src/png.js +26 -0
- package/src/refs.js +51 -8
- package/src/regions.js +110 -1
- package/src/screenmap.js +109 -10
- package/src/supervisor.js +117 -0
- package/src/view.js +396 -7
- package/src/vocabulary.js +134 -0
- package/src/wrote.js +136 -0
|
@@ -0,0 +1,281 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// The perception eval harness. Deferred since Phase 5; four things wait on it.
|
|
3
|
+
//
|
|
4
|
+
// What it is, and the design decision behind it. The obvious harness replays
|
|
5
|
+
// stored frames through the perception path and diffs the element lists, which
|
|
6
|
+
// is what Phase 13 step 5 describes. That harness cannot exist: the element
|
|
7
|
+
// list is the accessibility tree fused with OCR, the tree is not in the frame,
|
|
8
|
+
// and OCR runs in the daemon against a live framebuffer. A frame on disk is
|
|
9
|
+
// half the input.
|
|
10
|
+
//
|
|
11
|
+
// So the split is different and, for what actually needs gating, better. The
|
|
12
|
+
// machine records the *input* — the fused element list for a screen, exactly as
|
|
13
|
+
// perception produced it. A person authors the *expected output*. Everything
|
|
14
|
+
// downstream of the element list is a pure function, so the check runs offline,
|
|
15
|
+
// deterministically, with no simulator and no daemon:
|
|
16
|
+
//
|
|
17
|
+
// resolution matching.resolve(targets, query) — which element a query picks
|
|
18
|
+
// identity fingerprint.tokens(targets) — which screen this is
|
|
19
|
+
// change analyze.signatureDiff(a, b) — whether a frame moved
|
|
20
|
+
//
|
|
21
|
+
// Those three are precisely the thresholds every blocked item wants to change.
|
|
22
|
+
// What it does *not* cover is whether perception found the elements at all,
|
|
23
|
+
// which is inherently live and stays with eval-fingerprint.mjs and the
|
|
24
|
+
// integration job. Said plainly here rather than implied, because a harness
|
|
25
|
+
// that is trusted for more than it measures is worse than no harness.
|
|
26
|
+
//
|
|
27
|
+
// Two rules from this repo:
|
|
28
|
+
// - The gate is an exit code. Nothing is verified through a pipe: this
|
|
29
|
+
// project has already shipped a check that could not fail because
|
|
30
|
+
// `node script.mjs | tail` reports tail's status.
|
|
31
|
+
// - A fixture with no authored expectations is reported as unauthored, never
|
|
32
|
+
// counted as a pass. An empty test suite is green.
|
|
33
|
+
import fs from 'node:fs';
|
|
34
|
+
import path from 'node:path';
|
|
35
|
+
import { fileURLToPath } from 'node:url';
|
|
36
|
+
import * as fingerprint from '../src/fingerprint.js';
|
|
37
|
+
import * as matching from '../src/matching.js';
|
|
38
|
+
import * as analyze from '../src/analyze.js';
|
|
39
|
+
import * as view from '../src/view.js';
|
|
40
|
+
|
|
41
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
42
|
+
const DIR = path.join(ROOT, 'test', 'perception', 'screens');
|
|
43
|
+
const arg = (n, d = null) => {
|
|
44
|
+
const hit = process.argv.find((a) => a.startsWith(`--${n}=`));
|
|
45
|
+
return hit ? hit.slice(n.length + 3) : d;
|
|
46
|
+
};
|
|
47
|
+
const has = (n) => process.argv.includes(`--${n}`);
|
|
48
|
+
|
|
49
|
+
export function fixtures(dir = DIR, only = null) {
|
|
50
|
+
if (!fs.existsSync(dir)) return [];
|
|
51
|
+
return fs.readdirSync(dir)
|
|
52
|
+
.filter((f) => f.endsWith('.json'))
|
|
53
|
+
.filter((f) => (only ? f.includes(only) : true))
|
|
54
|
+
.sort()
|
|
55
|
+
.map((f) => ({ file: f, ...JSON.parse(fs.readFileSync(path.join(dir, f), 'utf8')) }));
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Check one screen's authored expectations against the pure layers.
|
|
60
|
+
*
|
|
61
|
+
* Returns findings rather than printing, so the same function backs the CLI and
|
|
62
|
+
* the unit tests — and so a finding can be counted without being formatted.
|
|
63
|
+
*/
|
|
64
|
+
export function checkScreen(fx) {
|
|
65
|
+
const findings = [];
|
|
66
|
+
const screen = fx.points ?? null;
|
|
67
|
+
const targets = fx.targets ?? [];
|
|
68
|
+
const expect = fx.expect ?? {};
|
|
69
|
+
const authored = (expect.resolutions?.length ?? 0)
|
|
70
|
+
+ (expect.ambiguous?.length ?? 0)
|
|
71
|
+
+ (expect.none?.length ?? 0)
|
|
72
|
+
+ (expect.discoverable?.length ?? 0)
|
|
73
|
+
+ (fx.frame_pairs?.length ?? 0)
|
|
74
|
+
+ (expect.identity ? 1 : 0);
|
|
75
|
+
|
|
76
|
+
for (const r of expect.resolutions ?? []) {
|
|
77
|
+
const out = matching.resolve(targets, r.query, { screen });
|
|
78
|
+
if (out.status !== 'ok') {
|
|
79
|
+
findings.push({
|
|
80
|
+
kind: 'resolution',
|
|
81
|
+
query: r.query,
|
|
82
|
+
want: r.label,
|
|
83
|
+
got: out.status === 'ambiguous'
|
|
84
|
+
? `ambiguous between ${out.alternatives.map((a) => a.label).join(', ')}`
|
|
85
|
+
: 'nothing resolved',
|
|
86
|
+
});
|
|
87
|
+
continue;
|
|
88
|
+
}
|
|
89
|
+
const got = out.target.label ?? '(icon-only)';
|
|
90
|
+
// Compared on the label because that is what the author can see and mean.
|
|
91
|
+
// Coordinates would make a fixture break every time a row moved a point.
|
|
92
|
+
if (got !== r.label) {
|
|
93
|
+
findings.push({ kind: 'resolution', query: r.query, want: r.label, got: `"${got}" at ${out.target.x},${out.target.y}` });
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
// The other half of the contract, and the half with the wrong-tap risk in it:
|
|
98
|
+
// "when two things answer equally well it says so rather than guessing".
|
|
99
|
+
for (const q of expect.ambiguous ?? []) {
|
|
100
|
+
const out = matching.resolve(targets, typeof q === 'string' ? q : q.query, { screen });
|
|
101
|
+
if (out.status !== 'ambiguous') {
|
|
102
|
+
findings.push({
|
|
103
|
+
kind: 'should-ask',
|
|
104
|
+
query: typeof q === 'string' ? q : q.query,
|
|
105
|
+
want: 'ambiguous',
|
|
106
|
+
got: out.status === 'ok' ? `picked "${out.target.label}" at score ${out.score}` : 'nothing resolved',
|
|
107
|
+
});
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
for (const q of expect.none ?? []) {
|
|
112
|
+
const out = matching.resolve(targets, q, { screen });
|
|
113
|
+
if (out.status !== 'none') {
|
|
114
|
+
findings.push({
|
|
115
|
+
kind: 'should-find-nothing',
|
|
116
|
+
query: q,
|
|
117
|
+
want: 'none',
|
|
118
|
+
got: out.status === 'ok' ? `picked "${out.target.label}"` : 'ambiguous',
|
|
119
|
+
});
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
// Discovery: does the default map still SAY what is on this screen?
|
|
124
|
+
//
|
|
125
|
+
// This check exists because of a regression it would have caught. A rule that
|
|
126
|
+
// dropped long non-interactive rows left every React Native list card out of
|
|
127
|
+
// the map — the cards expose their children as one concatenated accessibility
|
|
128
|
+
// label, so they look exactly like a Settings caption. Nothing became
|
|
129
|
+
// untappable, because `locate` reads targets rather than rows, so every
|
|
130
|
+
// resolution expectation above still passed. What was lost was the map's
|
|
131
|
+
// account of what is there, and an agent that cannot see a row falls back to
|
|
132
|
+
// a ~1600-token screenshot.
|
|
133
|
+
//
|
|
134
|
+
// Resolution and discovery are different claims, and the harness could only
|
|
135
|
+
// make the first one.
|
|
136
|
+
if (expect.discoverable?.length && screen) {
|
|
137
|
+
const { rows } = view.rowsFor({ targets }, { screen });
|
|
138
|
+
const shown = rows.map((r) => String(r.label ?? ''));
|
|
139
|
+
for (const want of expect.discoverable) {
|
|
140
|
+
// Compared as a prefix, because a long label is legitimately truncated in
|
|
141
|
+
// a row — truncated is discoverable, absent is not.
|
|
142
|
+
const head = want.slice(0, 24);
|
|
143
|
+
if (!shown.some((got) => got.startsWith(head))) {
|
|
144
|
+
findings.push({
|
|
145
|
+
kind: 'discovery',
|
|
146
|
+
query: `${want.slice(0, 40)}${want.length > 40 ? '…' : ''}`,
|
|
147
|
+
want: 'listed in the default map',
|
|
148
|
+
got: `${rows.length} row(s), none starting with it`,
|
|
149
|
+
});
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
// Identity drift. A token-rule change that silently alters a screen's
|
|
155
|
+
// structural hash discards every stored map and graph for it, and the failure
|
|
156
|
+
// is invisible — an old hash is a well-formed hash that matches nothing.
|
|
157
|
+
if (expect.identity && screen) {
|
|
158
|
+
const now = fingerprint.fingerprint(targets, screen);
|
|
159
|
+
if (now.hash !== expect.identity.hash) {
|
|
160
|
+
const similarity = fingerprint.similarity(now.tokens, expect.identity.tokens ?? []);
|
|
161
|
+
findings.push({
|
|
162
|
+
kind: 'identity',
|
|
163
|
+
query: '(structural hash)',
|
|
164
|
+
want: `${expect.identity.hash?.slice(0, 12)} (${expect.identity.tokens?.length ?? 0} tokens)`,
|
|
165
|
+
got: `${now.hash?.slice(0, 12)} (${now.tokens.length} tokens), similarity ${similarity.toFixed(2)}`,
|
|
166
|
+
});
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
// Frame pairs, for the change detector. `changed: true` means an action that
|
|
171
|
+
// a person would call visible — a switch flipping, a radio dot moving.
|
|
172
|
+
for (const pair of fx.frame_pairs ?? []) {
|
|
173
|
+
const a = analyze.hexToSignature(pair.after);
|
|
174
|
+
const b = analyze.hexToSignature(pair.before);
|
|
175
|
+
// Two different questions about the same pair of frames: has the *screen*
|
|
176
|
+
// changed (the mean, which drives stillness) and has a *control* changed
|
|
177
|
+
// (the largest single region, which drives the verdict). A switch flip
|
|
178
|
+
// answers no to the first and yes to the second, which is the whole reason
|
|
179
|
+
// both exist.
|
|
180
|
+
const diff = pair.per_cell ? analyze.maxCellDelta(a, b) : analyze.signatureDiff(a, b);
|
|
181
|
+
const seen = diff > (pair.threshold ?? 0.004);
|
|
182
|
+
if (seen !== Boolean(pair.changed)) {
|
|
183
|
+
findings.push({
|
|
184
|
+
kind: 'change',
|
|
185
|
+
query: pair.note ?? '(frame pair)',
|
|
186
|
+
want: pair.changed ? 'a visible change' : 'no change',
|
|
187
|
+
got: `${pair.per_cell ? 'max cell' : 'mean'} ${diff.toFixed(5)} against a threshold of ${pair.threshold ?? 0.004}`,
|
|
188
|
+
});
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
return { authored, findings };
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
async function record() {
|
|
196
|
+
const api = await import('../src/index.js');
|
|
197
|
+
const device = arg('device');
|
|
198
|
+
const name = arg('name');
|
|
199
|
+
if (!name) throw new Error('--record needs --name=<app>/<screen>');
|
|
200
|
+
const id = await api.screenIdentity(device, { fresh: true, confirmNovel: false });
|
|
201
|
+
const entry = id.entry ?? {};
|
|
202
|
+
const out = {
|
|
203
|
+
app: name.split('/')[0],
|
|
204
|
+
screen: name.split('/').slice(1).join('/') || name,
|
|
205
|
+
note: arg('note') ?? null,
|
|
206
|
+
recorded_at: new Date().toISOString(),
|
|
207
|
+
points: id.points,
|
|
208
|
+
// The recorded input. Trimmed to what the pure layers read, so a fixture
|
|
209
|
+
// is reviewable by a person rather than a wall of machine state.
|
|
210
|
+
targets: (entry.targets ?? []).map((t) => ({
|
|
211
|
+
label: t.label ?? null,
|
|
212
|
+
value: t.value ?? null,
|
|
213
|
+
type: t.type ?? null,
|
|
214
|
+
x: t.x, y: t.y,
|
|
215
|
+
frame: t.frame ?? null,
|
|
216
|
+
region: t.region ?? null,
|
|
217
|
+
source: t.source ?? null,
|
|
218
|
+
aliases: t.aliases ?? undefined,
|
|
219
|
+
navSlot: t.navSlot ?? undefined,
|
|
220
|
+
enabled: t.enabled ?? undefined,
|
|
221
|
+
selected: t.selected ?? undefined,
|
|
222
|
+
})),
|
|
223
|
+
expect: {
|
|
224
|
+
identity: { hash: entry.structuralHash, tokens: entry.structuralTokens ?? [] },
|
|
225
|
+
// Authored by hand. The machine records what perception saw; a person
|
|
226
|
+
// says what it should mean. Deriving these from current behaviour would
|
|
227
|
+
// bake today's bugs in as the specification.
|
|
228
|
+
resolutions: [],
|
|
229
|
+
ambiguous: [],
|
|
230
|
+
none: [],
|
|
231
|
+
},
|
|
232
|
+
};
|
|
233
|
+
fs.mkdirSync(DIR, { recursive: true });
|
|
234
|
+
const file = path.join(DIR, `${name.replace(/\//g, '__')}.json`);
|
|
235
|
+
fs.writeFileSync(file, `${JSON.stringify(out, null, 2)}\n`);
|
|
236
|
+
console.log(`recorded ${out.targets.length} element(s) from ${name} -> ${path.relative(ROOT, file)}`);
|
|
237
|
+
console.log(' now author expect.resolutions / expect.ambiguous / expect.none by hand');
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
function check() {
|
|
241
|
+
const all = fixtures(DIR, arg('only'));
|
|
242
|
+
if (!all.length) {
|
|
243
|
+
console.error('check-perception: no fixtures. Record some with --record --name=<app>/<screen>');
|
|
244
|
+
process.exitCode = 1;
|
|
245
|
+
return;
|
|
246
|
+
}
|
|
247
|
+
let failed = 0;
|
|
248
|
+
let unauthored = 0;
|
|
249
|
+
let checks = 0;
|
|
250
|
+
const apps = new Set();
|
|
251
|
+
for (const fx of all) {
|
|
252
|
+
apps.add(fx.app);
|
|
253
|
+
const { authored, findings } = checkScreen(fx);
|
|
254
|
+
checks += authored;
|
|
255
|
+
if (!authored) {
|
|
256
|
+
unauthored += 1;
|
|
257
|
+
console.log(` ?? ${fx.app}/${fx.screen} recorded, no expectations authored`);
|
|
258
|
+
continue;
|
|
259
|
+
}
|
|
260
|
+
if (!findings.length) {
|
|
261
|
+
console.log(` ok ${fx.app}/${fx.screen} ${authored} expectation(s)`);
|
|
262
|
+
continue;
|
|
263
|
+
}
|
|
264
|
+
failed += findings.length;
|
|
265
|
+
console.log(`FAIL ${fx.app}/${fx.screen}`);
|
|
266
|
+
for (const f of findings) {
|
|
267
|
+
console.log(` ${f.kind}: ${f.query}\n want ${f.want}\n got ${f.got}`);
|
|
268
|
+
}
|
|
269
|
+
}
|
|
270
|
+
console.log(`\n${all.length} screen(s) across ${apps.size} app(s), ${checks} expectation(s), ${failed} failure(s)`
|
|
271
|
+
+ (unauthored ? `, ${unauthored} unauthored` : ''));
|
|
272
|
+
// An unauthored fixture is not a pass. A suite that counts recordings as
|
|
273
|
+
// successes is a suite that goes green by adding files.
|
|
274
|
+
if (failed || unauthored) process.exitCode = 1;
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
if (import.meta.url === `file://${process.argv[1]}`) {
|
|
278
|
+
if (has('record')) await record();
|
|
279
|
+
else if (has('list')) for (const f of fixtures()) console.log(`${f.app}/${f.screen} ${f.targets?.length ?? 0} elements`);
|
|
280
|
+
else check();
|
|
281
|
+
}
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* Phase 17, go/no-go step 1: build the corpus and ask whether there is a job.
|
|
4
|
+
*
|
|
5
|
+
* The plan in docs/PHASES-HUMAN-PARITY.md said to export 200 `ambiguous_intent`
|
|
6
|
+
* and `no_plan` escalations. That corpus cannot be assembled: `no_plan` has
|
|
7
|
+
* never been logged once, `ambiguous_intent` is the reason the ranking bug
|
|
8
|
+
* produced (so the pre-fix records encode a bug), and the session id is minted
|
|
9
|
+
* per process — so a CLI-driven agent gets one "session" per command and the
|
|
10
|
+
* "filter to one session id" step has nothing to filter.
|
|
11
|
+
*
|
|
12
|
+
* There is a better corpus and it was there all along. Every verified graph edge
|
|
13
|
+
* is a decision that *worked*: the goal is in `step`, the screen it was taken on
|
|
14
|
+
* is the node, and the element list for that screen is in the screen map. That
|
|
15
|
+
* is (goal, element list, action) for every successful step, not just the rare
|
|
16
|
+
* failures — and the ground truth is stronger, because the tap was verified.
|
|
17
|
+
*
|
|
18
|
+
* What this measures is the question the whole phase turns on: of the decisions
|
|
19
|
+
* an agent actually makes, how many does the local matcher already resolve? A
|
|
20
|
+
* local planner can only earn its keep on the remainder. If the remainder is
|
|
21
|
+
* small, Phase 17 is a no-go regardless of how good the model is.
|
|
22
|
+
*
|
|
23
|
+
* Output is aggregate by design. This reads a real device's memory of real
|
|
24
|
+
* third-party apps, so it prints counts and never a label, an app name or a
|
|
25
|
+
* screen's contents. See the standing rule in scripts/check-private.mjs.
|
|
26
|
+
*
|
|
27
|
+
* Usage: node scripts/phase17-corpus.mjs [--device <udid>] [--json]
|
|
28
|
+
*/
|
|
29
|
+
import { readdirSync, readFileSync, existsSync } from 'node:fs';
|
|
30
|
+
import { join } from 'node:path';
|
|
31
|
+
import { homedir } from 'node:os';
|
|
32
|
+
import * as matching from '../src/matching.js';
|
|
33
|
+
import { goalOf } from '../src/actions.js';
|
|
34
|
+
|
|
35
|
+
const ROOT = process.env.SIMFRAME_HOME || join(homedir(), '.simframe');
|
|
36
|
+
|
|
37
|
+
const readJson = (p) => {
|
|
38
|
+
try { return JSON.parse(readFileSync(p, 'utf8')); } catch { return null; }
|
|
39
|
+
};
|
|
40
|
+
const listDir = (p) => (existsSync(p) ? readdirSync(p).filter((f) => f.endsWith('.json')) : []);
|
|
41
|
+
|
|
42
|
+
/** Screens are keyed by frame hash; graph nodes carry a layoutHash. Index both. */
|
|
43
|
+
function screenIndex(dev) {
|
|
44
|
+
const idx = new Map();
|
|
45
|
+
for (const f of listDir(join(dev, 'screens'))) {
|
|
46
|
+
const s = readJson(join(dev, 'screens', f));
|
|
47
|
+
if (!s?.targets?.length) continue;
|
|
48
|
+
for (const k of ['layoutHash', 'structuralHash', 'hash']) {
|
|
49
|
+
if (s[k] && !idx.has(s[k])) idx.set(s[k], s);
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
return idx;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Only some steps involve choosing an element. `launch` names an app, `button`
|
|
57
|
+
* names hardware, `openUrl` names a URL, `scroll` names a direction — none of
|
|
58
|
+
* them is a decision a planner could help with, and counting them was inflating
|
|
59
|
+
* the "not found" bucket to 45% on the first run of this script.
|
|
60
|
+
*/
|
|
61
|
+
const CHOOSES_AN_ELEMENT = new Set(['tap', 'type', 'scroll_to', 'scrollTo', 'assert', 'swipe_from', 'longPress']);
|
|
62
|
+
|
|
63
|
+
/** A "#3" is answered by the ref table and a "@x,y" by arithmetic. Neither is a decision. */
|
|
64
|
+
const isSelectorNotIntent = (goal) => /^#\d+$/.test(goal) || /^@?-?\d+\s*,\s*-?\d+$/.test(goal);
|
|
65
|
+
|
|
66
|
+
export function classify(dev) {
|
|
67
|
+
const idx = screenIndex(dev);
|
|
68
|
+
const out = {
|
|
69
|
+
screens_stored: idx.size,
|
|
70
|
+
edges: 0,
|
|
71
|
+
unjoinable: 0,
|
|
72
|
+
no_goal: 0,
|
|
73
|
+
not_an_element_step: 0,
|
|
74
|
+
selector_not_intent: 0,
|
|
75
|
+
cases: 0,
|
|
76
|
+
resolved: 0,
|
|
77
|
+
ambiguous: 0,
|
|
78
|
+
none: 0,
|
|
79
|
+
};
|
|
80
|
+
for (const f of listDir(join(dev, 'graph'))) {
|
|
81
|
+
const g = readJson(join(dev, 'graph', f));
|
|
82
|
+
const screen = idx.get(g?.layoutHash) ?? idx.get(g?.hash);
|
|
83
|
+
for (const e of g?.edges ?? []) {
|
|
84
|
+
out.edges += 1;
|
|
85
|
+
if (!screen) { out.unjoinable += 1; continue; }
|
|
86
|
+
const goal = goalOf(e.step);
|
|
87
|
+
if (!goal) { out.no_goal += 1; continue; }
|
|
88
|
+
if (!CHOOSES_AN_ELEMENT.has(e.step?.action)) { out.not_an_element_step += 1; continue; }
|
|
89
|
+
if (isSelectorNotIntent(String(goal).trim())) { out.selector_not_intent += 1; continue; }
|
|
90
|
+
out.cases += 1;
|
|
91
|
+
const r = matching.resolve(screen.targets, String(goal));
|
|
92
|
+
if (r.status === 'ok') out.resolved += 1;
|
|
93
|
+
else if (r.status === 'ambiguous') out.ambiguous += 1;
|
|
94
|
+
else out.none += 1;
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
return out;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* The second instrument, with the opposite bias.
|
|
102
|
+
*
|
|
103
|
+
* A verified edge is a decision that *succeeded*, so the graph cannot see the
|
|
104
|
+
* ones the matcher fumbled — those became escalations, Claude fixed them, and
|
|
105
|
+
* the edge was written with the corrected goal. Measuring the matcher on its
|
|
106
|
+
* own successes is partly circular. So also count resolution failures from the
|
|
107
|
+
* log against total edge traversals: same question, denominator built from
|
|
108
|
+
* failures instead of successes.
|
|
109
|
+
*/
|
|
110
|
+
export function resolutionFailureRate(dev) {
|
|
111
|
+
let traversals = 0;
|
|
112
|
+
for (const f of listDir(join(dev, 'graph'))) {
|
|
113
|
+
for (const e of readJson(join(dev, 'graph', f))?.edges ?? []) traversals += e.count ?? 0;
|
|
114
|
+
}
|
|
115
|
+
let failures = 0;
|
|
116
|
+
let escalations = 0;
|
|
117
|
+
const log = join(dev, 'escalations.jsonl');
|
|
118
|
+
if (existsSync(log)) {
|
|
119
|
+
for (const line of readFileSync(log, 'utf8').split('\n')) {
|
|
120
|
+
if (!line.trim()) continue;
|
|
121
|
+
let r;
|
|
122
|
+
try { r = JSON.parse(line); } catch { continue; }
|
|
123
|
+
escalations += 1;
|
|
124
|
+
if (r.reason === 'ambiguous_intent' || r.reason === 'unknown_screen') failures += 1;
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
return { traversals, escalations, failures };
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
const flags = process.argv.slice(2);
|
|
131
|
+
const only = flags.includes('--device') ? flags[flags.indexOf('--device') + 1] : null;
|
|
132
|
+
const devices = (existsSync(ROOT) ? readdirSync(ROOT) : [])
|
|
133
|
+
.filter((d) => !d.startsWith('TEST-') && d !== 'bin')
|
|
134
|
+
.filter((d) => existsSync(join(ROOT, d, 'graph')))
|
|
135
|
+
.filter((d) => !only || d === only);
|
|
136
|
+
|
|
137
|
+
const totals = { cases: 0, resolved: 0, ambiguous: 0, none: 0, edges: 0, unjoinable: 0, not_an_element_step: 0, selector_not_intent: 0 };
|
|
138
|
+
const rows = [];
|
|
139
|
+
for (const d of devices) {
|
|
140
|
+
const r = classify(join(ROOT, d));
|
|
141
|
+
if (!r.edges) continue;
|
|
142
|
+
rows.push({ device: `${d.slice(0, 8)}…`, ...r, ...resolutionFailureRate(join(ROOT, d)) });
|
|
143
|
+
for (const k of Object.keys(totals)) totals[k] += r[k];
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
if (flags.includes('--json')) {
|
|
147
|
+
console.log(JSON.stringify({ devices: rows, totals }, null, 2));
|
|
148
|
+
} else {
|
|
149
|
+
console.log('Phase 17 corpus — verified graph edges as (goal, element list, action)\n');
|
|
150
|
+
for (const r of rows) {
|
|
151
|
+
console.log(` ${r.device} edges ${r.edges} joinable ${r.edges - r.unjoinable} decisions ${r.cases}` +
|
|
152
|
+
` → matcher ok ${r.resolved}, ambiguous ${r.ambiguous}, not found ${r.none}`);
|
|
153
|
+
}
|
|
154
|
+
const { cases, resolved, ambiguous, none } = totals;
|
|
155
|
+
const pct = (n) => (cases ? `${Math.round((n / cases) * 1000) / 10}%` : '—');
|
|
156
|
+
console.log(`\n ${totals.edges} edges: ${totals.unjoinable} unjoinable, ` +
|
|
157
|
+
`${totals.not_an_element_step} chose no element (launch/button/openUrl/scroll), ` +
|
|
158
|
+
`${totals.selector_not_intent} used a #ref or a coordinate.`);
|
|
159
|
+
console.log(` TOTAL element decisions: ${cases}`);
|
|
160
|
+
console.log(` already resolved locally ${resolved} ${pct(resolved)} <- a planner adds nothing here`);
|
|
161
|
+
console.log(` ambiguous ${ambiguous} ${pct(ambiguous)} <- a planner could pick`);
|
|
162
|
+
console.log(` not found ${none} ${pct(none)} <- a planner cannot invent an element`);
|
|
163
|
+
const addressable = ambiguous;
|
|
164
|
+
console.log(`\n Addressable by a local planner: ${addressable} of ${cases} (${pct(addressable)}).`);
|
|
165
|
+
|
|
166
|
+
console.log('\nSecond instrument — resolution failures from the log, per traversal.');
|
|
167
|
+
console.log('A rate over 100% means the graph was discarded while the log kept appending');
|
|
168
|
+
console.log('(a MAP_VERSION or GRAPH_VERSION bump), so it is not a rate. Read the device');
|
|
169
|
+
console.log('that was driven by an agent doing real work, not the bench device.\n');
|
|
170
|
+
for (const r of rows) {
|
|
171
|
+
const rate = r.traversals ? `${Math.round((r.failures / r.traversals) * 1000) / 10}%` : '—';
|
|
172
|
+
console.log(` ${r.device} traversals ${String(r.traversals).padStart(4)}` +
|
|
173
|
+
` escalations ${String(r.escalations).padStart(4)}` +
|
|
174
|
+
` resolution failures ${String(r.failures).padStart(3)} ${rate.padStart(6)}`);
|
|
175
|
+
}
|
|
176
|
+
}
|