simframe 0.8.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +31 -3
- package/flows/hpi-suite.json +68 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +49 -1
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +7 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +11 -0
- package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +48 -0
- package/native/simframed/Sources/simframed/main.swift +128 -73
- package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +40 -0
- package/package.json +2 -1
- package/scripts/bench-hpi.mjs +254 -0
- package/scripts/check-package.mjs +7 -0
- package/scripts/check-private.mjs +143 -0
- package/scripts/eval-perception.mjs +248 -0
- package/src/actions.js +333 -14
- package/src/analyze.js +70 -0
- package/src/baseline.js +333 -0
- package/src/cli.js +410 -4
- package/src/daemon.js +9 -0
- package/src/fingerprint.js +7 -1
- package/src/graph.js +262 -1
- package/src/index.js +335 -22
- package/src/input.js +155 -1
- package/src/intent.js +11 -2
- package/src/matching.js +136 -4
- package/src/mcp.js +14 -1
- package/src/metrics.js +596 -0
- package/src/navigate.js +47 -7
- package/src/platform/android.js +16 -1
- package/src/platform/index.js +3 -0
- package/src/platform/ios.js +40 -0
- package/src/screenmap.js +55 -14
- package/src/view.js +65 -3
|
@@ -0,0 +1,254 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// The agent half of the Human Parity Index.
|
|
3
|
+
//
|
|
4
|
+
// Runs every flow in the suite N times, exactly as an agent would run it, and
|
|
5
|
+
// reduces the flow records to an HPI report. The human half is recorded once
|
|
6
|
+
// by a person (`simframe baseline record`) and committed; this side is
|
|
7
|
+
// re-measured on every run, which is what makes HPI a trend rather than a
|
|
8
|
+
// claim.
|
|
9
|
+
//
|
|
10
|
+
// Two rules this file follows, both learned in this repo:
|
|
11
|
+
//
|
|
12
|
+
// - Nothing is verified through a pipe. `node script.mjs | tail -3` exits
|
|
13
|
+
// with tail's status, and this project has already shipped a check that
|
|
14
|
+
// could not fail because of it. Every gate here is an exit code.
|
|
15
|
+
// - A missing human baseline is reported, never defaulted. HPI_time with no
|
|
16
|
+
// denominator is null; it is not 1.0, and it is not "parity".
|
|
17
|
+
import fs from 'node:fs';
|
|
18
|
+
import path from 'node:path';
|
|
19
|
+
import { fileURLToPath } from 'node:url';
|
|
20
|
+
import { runScript } from '../src/actions.js';
|
|
21
|
+
import * as api from '../src/index.js';
|
|
22
|
+
import * as baseline from '../src/baseline.js';
|
|
23
|
+
import * as metrics from '../src/metrics.js';
|
|
24
|
+
|
|
25
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
26
|
+
const arg = (name, fallback = null) => {
|
|
27
|
+
const hit = process.argv.find((a) => a.startsWith(`--${name}=`));
|
|
28
|
+
return hit ? hit.slice(name.length + 3) : fallback;
|
|
29
|
+
};
|
|
30
|
+
const has = (name) => process.argv.includes(`--${name}`);
|
|
31
|
+
|
|
32
|
+
const device = arg('device');
|
|
33
|
+
const runs = Math.max(1, Number(arg('runs', '3')));
|
|
34
|
+
const only = arg('flow');
|
|
35
|
+
const out = arg('out');
|
|
36
|
+
const baselineFile = arg('baseline', path.join(ROOT, 'docs', 'research', 'hpi-baseline.json'));
|
|
37
|
+
const gate = has('gate');
|
|
38
|
+
/**
|
|
39
|
+
* How many times to measure the whole suite.
|
|
40
|
+
*
|
|
41
|
+
* Three when gating, because one measurement is mostly noise: identical code
|
|
42
|
+
* measured three times the same afternoon gave HPI_time 0.475, 0.413 and
|
|
43
|
+
* 0.371. The gate reads the median of the passes, which is what lets the time
|
|
44
|
+
* band stay as tight as 25% instead of being widened to cover a single run's
|
|
45
|
+
* spread.
|
|
46
|
+
*/
|
|
47
|
+
const passes = Math.max(1, Number(arg('passes', gate ? '3' : '1')));
|
|
48
|
+
/**
|
|
49
|
+
* A pause between runs, and it is not superstition.
|
|
50
|
+
*
|
|
51
|
+
* Measured on one 8-run loop: 15.2 s, 15.2 s, then 6.9 s failing, then 25.7 s,
|
|
52
|
+
* 25.7 s, then the display wedged — and it wedged again within a couple of
|
|
53
|
+
* minutes of the same load after a device restart. Relaunching an app as fast
|
|
54
|
+
* as a script can is not what this suite is trying to measure, and a simulator
|
|
55
|
+
* asked to do it degrades and then stops rendering. The suite paces itself so
|
|
56
|
+
* the numbers describe simframe rather than the simulator's tolerance for
|
|
57
|
+
* being hammered.
|
|
58
|
+
*/
|
|
59
|
+
const cooldownMs = Math.max(0, Number(arg('cooldown', '1500')));
|
|
60
|
+
|
|
61
|
+
const suite = baseline.loadSuite(arg('suite', baseline.SUITE_FILE)).filter((f) => !only || f.name === only);
|
|
62
|
+
if (!suite.length) {
|
|
63
|
+
console.error(`no flows to run${only ? ` matching "${only}"` : ''}`);
|
|
64
|
+
process.exit(2);
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
const { device: dev } = await api.ensureDaemon(device);
|
|
68
|
+
console.log(`device: ${dev.name} (${dev.udid})`);
|
|
69
|
+
console.log(`suite: ${suite.map((f) => f.name).join(', ')} × ${runs} run(s) × ${passes} pass(es)\n`);
|
|
70
|
+
|
|
71
|
+
const ids = new Set();
|
|
72
|
+
let hardFailures = 0;
|
|
73
|
+
/**
|
|
74
|
+
* A wedged capture loop is not a slow flow, and every run after it is doomed.
|
|
75
|
+
*
|
|
76
|
+
* Measured four times in one session: the display surface stops being readable
|
|
77
|
+
* mid-suite, the daemon re-resolves the display port and still gets nothing,
|
|
78
|
+
* and only restarting the device cures it. Grinding through the remaining runs
|
|
79
|
+
* produced three identical "could not run at all" lines and an HPI computed
|
|
80
|
+
* from whatever happened to finish first — a number with a hole in it, which
|
|
81
|
+
* is worse than no number.
|
|
82
|
+
*/
|
|
83
|
+
const WEDGED = /display surface could not be read|did not produce a frame/;
|
|
84
|
+
/**
|
|
85
|
+
* The host could not do the thing, as distinct from simframe doing it slowly.
|
|
86
|
+
*
|
|
87
|
+
* A hosted runner takes 47-55 s to fail `simctl launch com.apple.Preferences`
|
|
88
|
+
* and then fails it again, three runs in a row — the same class of fault this
|
|
89
|
+
* repo already records for `simctl openurl`, which returns "Operation timed
|
|
90
|
+
* out" on a loaded runner. Six of those is nine minutes of CI spent measuring
|
|
91
|
+
* the runner's patience, and the resulting HPI describes nothing.
|
|
92
|
+
*/
|
|
93
|
+
const ENVIRONMENT = /could not launch|Command failed: xcrun simctl (launch|terminate)|Operation timed out|timed out/i;
|
|
94
|
+
/** Two environmental failures of the same flow is the environment, not a flake. */
|
|
95
|
+
const ENV_GIVE_UP = 2;
|
|
96
|
+
let abort = null;
|
|
97
|
+
const envFailures = new Map();
|
|
98
|
+
|
|
99
|
+
const passSets = [];
|
|
100
|
+
outer: for (let pass = 1; pass <= passes; pass += 1) {
|
|
101
|
+
const thisPass = new Set();
|
|
102
|
+
passSets.push(thisPass);
|
|
103
|
+
if (passes > 1) console.log(`pass ${pass}/${passes}`);
|
|
104
|
+
for (const flow of suite) {
|
|
105
|
+
for (let run = 1; run <= runs; run += 1) {
|
|
106
|
+
// The same start state the human baseline was recorded from: app not
|
|
107
|
+
// running, on the home screen. Not part of the timed flow, and
|
|
108
|
+
// deliberately not expressed as flow steps — see baseline.resetFor.
|
|
109
|
+
// The reset and its settle live inside the same guard as the run.
|
|
110
|
+
//
|
|
111
|
+
// They did not, and a device whose display had failed threw out of
|
|
112
|
+
// `waitFor` — outside the try — killing the process with a stack trace
|
|
113
|
+
// instead of the one classification this script exists to make. An
|
|
114
|
+
// environmental failure has to be classified wherever it happens, not
|
|
115
|
+
// only where it was convenient to catch.
|
|
116
|
+
let res;
|
|
117
|
+
try {
|
|
118
|
+
const reset = await baseline.resetFor(dev.udid, flow);
|
|
119
|
+
if (reset.failures.length) console.log(` (reset: ${reset.failures.join('; ')})`);
|
|
120
|
+
await api.waitFor(dev.udid, { mode: 'stable', stableMs: 400, timeoutMs: 4000 });
|
|
121
|
+
res = await runScript(dev.udid, {
|
|
122
|
+
steps: flow.steps,
|
|
123
|
+
flowName: flow.name,
|
|
124
|
+
minSteps: flow.minSteps ?? null,
|
|
125
|
+
});
|
|
126
|
+
} catch (err) {
|
|
127
|
+
// A flow that could not start at all is not a slow flow. It has no
|
|
128
|
+
// record, so it cannot be averaged into anything; it is reported and
|
|
129
|
+
// counted.
|
|
130
|
+
hardFailures += 1;
|
|
131
|
+
console.log(`FAIL ${flow.name} run ${run}: ${err.message}`);
|
|
132
|
+
if (WEDGED.test(err.message)) {
|
|
133
|
+
abort = { kind: 'capture is wedged', message: err.message };
|
|
134
|
+
break outer;
|
|
135
|
+
}
|
|
136
|
+
if (ENVIRONMENT.test(err.message)) {
|
|
137
|
+
const n = (envFailures.get(flow.name) ?? 0) + 1;
|
|
138
|
+
envFailures.set(flow.name, n);
|
|
139
|
+
if (n >= ENV_GIVE_UP) {
|
|
140
|
+
abort = { kind: `the host cannot run "${flow.name}"`, message: err.message.split('\n')[0] };
|
|
141
|
+
break outer;
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
continue;
|
|
145
|
+
}
|
|
146
|
+
ids.add(res.flowId);
|
|
147
|
+
thisPass.add(res.flowId);
|
|
148
|
+
const verdicts = res.results.map((r) => r.verification?.verdict).filter(Boolean);
|
|
149
|
+
console.log(
|
|
150
|
+
`${res.ok ? 'ok ' : 'FAIL'} ${flow.name.padEnd(24)} run ${run}/${runs} ` +
|
|
151
|
+
`${String(res.totalMs).padStart(6)}ms ${res.ranSteps}/${res.totalSteps} steps ` +
|
|
152
|
+
`${verdicts.filter((v) => v !== 'ok').length ? `verdicts: ${verdicts.join(',')}` : 'all ok'}`,
|
|
153
|
+
);
|
|
154
|
+
if (!res.ok) console.log(` ${res.results.filter((r) => !r.ok).map((r) => r.error).join('; ')}`);
|
|
155
|
+
if (cooldownMs) await new Promise((r) => setTimeout(r, cooldownMs));
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
// This process's runs only. The log is append-only and holds every earlier
|
|
161
|
+
// measurement, which is what makes it a trend — but a CI number computed over
|
|
162
|
+
// somebody's local runs from last week is not this commit's number.
|
|
163
|
+
const flows = metrics.readFlows(dev.udid).filter((f) => ids.has(f.flow_id));
|
|
164
|
+
const humans = baseline.readBaselines();
|
|
165
|
+
// Each pass measured on its own, so the gate can take a median over them
|
|
166
|
+
// rather than trusting one. Accuracy is pooled over every run instead: one
|
|
167
|
+
// wrong action in thirty is a wrong action, and a median would hide it.
|
|
168
|
+
const allFlows = metrics.readFlows(dev.udid);
|
|
169
|
+
const passReports = passSets
|
|
170
|
+
.map((set) => metrics.hpi({ flows: allFlows.filter((f) => set.has(f.flow_id)), baselines: humans }))
|
|
171
|
+
.filter((r) => r.overall.runs > 0);
|
|
172
|
+
const passTimes = passReports.map((r) => r.overall.hpi_time).filter((t) => Number.isFinite(t));
|
|
173
|
+
|
|
174
|
+
const report = {
|
|
175
|
+
...metrics.hpi({ flows, baselines: humans }),
|
|
176
|
+
measured_at: new Date().toISOString(),
|
|
177
|
+
device: { udid: dev.udid, name: dev.name, runtime: dev.runtime },
|
|
178
|
+
runs_per_flow: runs,
|
|
179
|
+
passes,
|
|
180
|
+
pass_hpi_time: passTimes,
|
|
181
|
+
hard_failures: hardFailures,
|
|
182
|
+
human_baselines: Object.fromEntries(
|
|
183
|
+
Object.entries(humans).map(([k, v]) => [k, { runs: v.runs, p50: v.wall_time_ms?.p50 ?? null }]),
|
|
184
|
+
),
|
|
185
|
+
};
|
|
186
|
+
|
|
187
|
+
console.log('\nflow runs agent p50 human p50 HPI_time step_ratio');
|
|
188
|
+
for (const f of report.flows) {
|
|
189
|
+
console.log(
|
|
190
|
+
`${f.flow.padEnd(24)} ${String(f.runs).padStart(5)} ${`${f.agent_ms.p50}ms`.padStart(9)} ` +
|
|
191
|
+
`${(f.human_median_ms ? `${f.human_median_ms}ms` : '—').padStart(9)} ` +
|
|
192
|
+
`${String(f.hpi_time ?? '—').padStart(8)} ${String(f.step_ratio ?? '—').padStart(10)}`,
|
|
193
|
+
);
|
|
194
|
+
}
|
|
195
|
+
report.overall.hpi_time_median_of_passes = passTimes.length ? Number(metrics.median(passTimes).toFixed(3)) : null;
|
|
196
|
+
const o = report.overall;
|
|
197
|
+
console.log(`\nHPI_accuracy ${o.hpi_accuracy} HPI_time ${o.hpi_time ?? '—'} HPI ${o.hpi ?? '—'} step_ratio ${o.step_ratio ?? '—'}`);
|
|
198
|
+
if (passTimes.length > 1) {
|
|
199
|
+
console.log(`HPI_time per pass: ${passTimes.join(', ')} — median ${o.hpi_time_median_of_passes} (what the gate reads)`);
|
|
200
|
+
}
|
|
201
|
+
if (o.hpi_time == null) {
|
|
202
|
+
console.log(`no human baseline for any measured flow — HPI_time and HPI are null, not 1.0.`);
|
|
203
|
+
console.log(`record one: simframe baseline record <flow> --device=${dev.udid} --runs=5`);
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
if (out) {
|
|
207
|
+
fs.writeFileSync(out, `${JSON.stringify(report, null, 2)}\n`);
|
|
208
|
+
console.log(`wrote ${out}`);
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
const escalations = metrics.breakdown(metrics.readEscalations(dev.udid));
|
|
212
|
+
console.log(`\nescalations in this device's log: ${escalations.total} (${Object.entries(escalations.by_reason).filter(([, n]) => n).map(([r, n]) => `${r} ${n}`).join(', ') || 'none'})`);
|
|
213
|
+
|
|
214
|
+
if (abort) {
|
|
215
|
+
console.error(`\n${abort.kind}: ${abort.message}`);
|
|
216
|
+
console.error('No comparable HPI was measured — what is above is a partial suite and');
|
|
217
|
+
console.error('must not be adopted as a baseline or read as a regression.');
|
|
218
|
+
if (/wedged/.test(abort.kind)) {
|
|
219
|
+
console.error('Only restarting the device is known to cure a wedge:');
|
|
220
|
+
console.error(` xcrun simctl shutdown ${dev.udid} && xcrun simctl boot ${dev.udid}`);
|
|
221
|
+
}
|
|
222
|
+
// 2, not 1. "I could not measure" and "it got worse" are different answers,
|
|
223
|
+
// and a job that reports them with the same exit code teaches people to
|
|
224
|
+
// ignore both. The workflow treats 2 as a loud warning and 1 as a failure.
|
|
225
|
+
process.exit(2);
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
if (hardFailures) {
|
|
229
|
+
console.error(`\n${hardFailures} flow run(s) could not run at all.`);
|
|
230
|
+
process.exit(1);
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
if (!gate) process.exit(0);
|
|
234
|
+
|
|
235
|
+
// ------------------------------------------------------------------ the gate
|
|
236
|
+
const committed = fs.existsSync(baselineFile) ? JSON.parse(fs.readFileSync(baselineFile, 'utf8')) : null;
|
|
237
|
+
if (!committed) {
|
|
238
|
+
// Loudly inactive rather than quietly passing. A gate with nothing to
|
|
239
|
+
// compare against cannot detect a regression, and saying "ok" here is how a
|
|
240
|
+
// CI job comes to mean nothing.
|
|
241
|
+
console.log(`\nGATE INACTIVE — no committed baseline at ${path.relative(ROOT, baselineFile)}.`);
|
|
242
|
+
console.log('Commit this run as the baseline to arm it.');
|
|
243
|
+
process.exit(0);
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
const base = committed.overall ?? {};
|
|
247
|
+
const failures = metrics.gateAgainst(committed, report);
|
|
248
|
+
|
|
249
|
+
console.log(`\ngate vs ${path.relative(ROOT, baselineFile)} (measured ${committed.measured_at ?? '?'})`);
|
|
250
|
+
console.log(` HPI_accuracy ${base.hpi_accuracy ?? '—'} -> ${o.hpi_accuracy ?? '—'}`);
|
|
251
|
+
console.log(` HPI_time ${metrics.gateTime(base) ?? '—'} -> ${metrics.gateTime(o) ?? '—'}`);
|
|
252
|
+
console.log(` band ${metrics.TIME_REGRESSION * 100}% (median of ${passTimes.length || 1} pass(es))`);
|
|
253
|
+
for (const f of failures) console.log(`FAIL ${f}`);
|
|
254
|
+
process.exit(failures.length ? 1 : 0);
|
|
@@ -75,6 +75,13 @@ function requiredFiles() {
|
|
|
75
75
|
throw new Error('no skill found under skills/ — has the layout moved? This check would silently pass.');
|
|
76
76
|
}
|
|
77
77
|
|
|
78
|
+
// The flow suite. `simframe baseline` and `simframe hpi` read it at
|
|
79
|
+
// runtime, so an absent one is a command that only fails once installed.
|
|
80
|
+
for (const f of walk(path.join(ROOT, 'flows'), (p) => p.endsWith('.json'))) required.add(rel(f));
|
|
81
|
+
if (!walk(path.join(ROOT, 'flows'), (p) => p.endsWith('.json')).length) {
|
|
82
|
+
throw new Error('no flow suite found under flows/ — has the layout moved? This check would silently pass.');
|
|
83
|
+
}
|
|
84
|
+
|
|
78
85
|
// Anything package.json points at to run.
|
|
79
86
|
const pkg = JSON.parse(fs.readFileSync(path.join(ROOT, 'package.json'), 'utf8'));
|
|
80
87
|
for (const target of Object.values(pkg.bin ?? {})) required.add(target.replace(/^\.\//, ''));
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// No third-party app identifiers in this repository. Ever, from anyone.
|
|
3
|
+
//
|
|
4
|
+
// simframe is a general-purpose tool: you install it and Claude Code drives
|
|
5
|
+
// *your* app on the simulator. It has no relationship with any particular app,
|
|
6
|
+
// so no particular app's bundle id belongs in it — not in the source, not in
|
|
7
|
+
// the docs, and not in a secret either. A denylist of specific strings would
|
|
8
|
+
// assume there is one app to protect, which is the wrong shape for this.
|
|
9
|
+
//
|
|
10
|
+
// So the rule is a pattern, not a list, and it needs no configuration at all.
|
|
11
|
+
// Anything shaped like a reverse-DNS bundle id is flagged unless it is one of:
|
|
12
|
+
//
|
|
13
|
+
// * a platform's own com.apple.*, com.android.*, com.google.*
|
|
14
|
+
// * a documentation placeholder com.example.*, com.acme.*, com.mycompany.*
|
|
15
|
+
// * this project's own identifiers
|
|
16
|
+
//
|
|
17
|
+
// That works on a fresh clone, on a fork, and in a pull request from a stranger,
|
|
18
|
+
// which a secret does not. An optional `.private-strings` file (gitignored) or
|
|
19
|
+
// $SIMFRAME_PRIVATE_STRINGS still adds extra patterns for anyone who wants them,
|
|
20
|
+
// but nothing depends on either existing.
|
|
21
|
+
//
|
|
22
|
+
// How this got written: a 282-line field-notes file about a real third-party app
|
|
23
|
+
// was committed here by `git add -A` and pushed, an hour after the first version
|
|
24
|
+
// of this script was written to prevent exactly that. It could not fire, because
|
|
25
|
+
// it was waiting for a denylist nobody had supplied. A guard with a
|
|
26
|
+
// precondition is a guard that is off.
|
|
27
|
+
//
|
|
28
|
+
// Nothing here ever prints a match. It prints the file and the line number, so
|
|
29
|
+
// the output of a failed run is safe to paste into an issue, a CI log, or a
|
|
30
|
+
// conversation with an agent — which is where the last one would have gone.
|
|
31
|
+
import fs from 'node:fs';
|
|
32
|
+
import path from 'node:path';
|
|
33
|
+
import { execFileSync } from 'node:child_process';
|
|
34
|
+
import { fileURLToPath } from 'node:url';
|
|
35
|
+
|
|
36
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
37
|
+
const LIST_FILE = path.join(ROOT, '.private-strings');
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Bundle-id-shaped strings, and the ones that are fine.
|
|
41
|
+
*
|
|
42
|
+
* The first segment is restricted to real reverse-DNS prefixes, which is what
|
|
43
|
+
* keeps ordinary property chains out: `res.state.seq`, `registry.paths.dir` and
|
|
44
|
+
* `import.meta.url` all look exactly like bundle ids until you require the head
|
|
45
|
+
* to be a TLD.
|
|
46
|
+
*/
|
|
47
|
+
const BUNDLE = /\b(?:com|io|org|net|dev|co|app|me|xyz|uk|de|fr|jp|nl|se|ca|au)\.[A-Za-z][A-Za-z0-9_-]{1,30}(?:\.[A-Za-z][A-Za-z0-9_-]{0,30}){1,3}\b/g;
|
|
48
|
+
|
|
49
|
+
/** Platform-owned, placeholder, or ours. Anything else is somebody's app. */
|
|
50
|
+
export const ALLOWED = [
|
|
51
|
+
/^com\.apple\./i,
|
|
52
|
+
/^com\.android\./i,
|
|
53
|
+
/^com\.google\./i,
|
|
54
|
+
/^org\.swift\./i,
|
|
55
|
+
/^org\.json\./i,
|
|
56
|
+
/^com\.facebook\./i, // idb, a reference implementation named in the docs
|
|
57
|
+
/^com\.example\./i,
|
|
58
|
+
/^com\.acme\./i,
|
|
59
|
+
/^com\.mycompany\./i,
|
|
60
|
+
/^com\.yourcompany\./i,
|
|
61
|
+
/^io\.github\./i,
|
|
62
|
+
];
|
|
63
|
+
|
|
64
|
+
export function isAllowedIdentifier(id) {
|
|
65
|
+
return ALLOWED.some((re) => re.test(id));
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/** Extra patterns, for anyone who wants them. Nothing depends on this existing. */
|
|
69
|
+
export function patternsFrom({ env, file } = {}) {
|
|
70
|
+
const raw = [
|
|
71
|
+
...String(env ?? '').split(/[\n,]/),
|
|
72
|
+
...String(file ?? '').split(/\n/),
|
|
73
|
+
];
|
|
74
|
+
return [...new Set(
|
|
75
|
+
raw
|
|
76
|
+
.map((s) => s.trim())
|
|
77
|
+
.filter((s) => s && !s.startsWith('#'))
|
|
78
|
+
// One character would match everything, which is a check that only ever
|
|
79
|
+
// fails and therefore only ever gets disabled.
|
|
80
|
+
.filter((s) => s.length >= 3)
|
|
81
|
+
.map((s) => s.toLowerCase()),
|
|
82
|
+
)];
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/** Which lines of `text` are a problem. Line numbers only, never matches. */
|
|
86
|
+
export function offendingLines(text, patterns = []) {
|
|
87
|
+
const hits = [];
|
|
88
|
+
const lines = String(text).split('\n');
|
|
89
|
+
for (let i = 0; i < lines.length; i += 1) {
|
|
90
|
+
const lower = lines[i].toLowerCase();
|
|
91
|
+
const why = [];
|
|
92
|
+
const extra = patterns.filter((p) => lower.includes(p)).length;
|
|
93
|
+
if (extra) why.push(`${extra} denied pattern(s)`);
|
|
94
|
+
const ids = (lines[i].match(BUNDLE) ?? []).filter((id) => !isAllowedIdentifier(id));
|
|
95
|
+
// Reported as a count and a shape, never as the identifier: knowing which
|
|
96
|
+
// app leaked is worth less to a bug report than not restating it.
|
|
97
|
+
if (ids.length) why.push(`${ids.length} third-party bundle id(s)`);
|
|
98
|
+
if (why.length) hits.push({ line: i + 1, why: why.join(', ') });
|
|
99
|
+
}
|
|
100
|
+
return hits;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
function trackedFiles() {
|
|
104
|
+
return execFileSync('git', ['ls-files', '-z'], { cwd: ROOT, encoding: 'utf8' })
|
|
105
|
+
.split('\0')
|
|
106
|
+
.filter(Boolean);
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
function main() {
|
|
110
|
+
const patterns = patternsFrom({
|
|
111
|
+
env: process.env.SIMFRAME_PRIVATE_STRINGS,
|
|
112
|
+
file: fs.existsSync(LIST_FILE) ? fs.readFileSync(LIST_FILE, 'utf8') : '',
|
|
113
|
+
});
|
|
114
|
+
const files = trackedFiles();
|
|
115
|
+
let failed = 0;
|
|
116
|
+
for (const rel of files) {
|
|
117
|
+
const full = path.join(ROOT, rel);
|
|
118
|
+
let text;
|
|
119
|
+
try {
|
|
120
|
+
const stat = fs.statSync(full);
|
|
121
|
+
if (!stat.isFile() || stat.size > 4_000_000) continue;
|
|
122
|
+
text = fs.readFileSync(full, 'utf8');
|
|
123
|
+
} catch {
|
|
124
|
+
continue;
|
|
125
|
+
}
|
|
126
|
+
// A NUL byte means this is not text, and a substring hit in it is noise.
|
|
127
|
+
if (text.includes('\0')) continue;
|
|
128
|
+
for (const hit of offendingLines(text, patterns)) {
|
|
129
|
+
console.error(`check-private: ${rel}:${hit.line} — ${hit.why}`);
|
|
130
|
+
failed += 1;
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
console.log(`check-private: ${files.length} tracked file(s); no third-party bundle ids`
|
|
134
|
+
+ (patterns.length ? `, plus ${patterns.length} local pattern(s)` : '')
|
|
135
|
+
+ ` — ${failed} line(s) flagged`);
|
|
136
|
+
if (failed) {
|
|
137
|
+
console.error('check-private: nothing above prints the match itself. '
|
|
138
|
+
+ 'A third-party app identifier does not belong in a general-purpose tool.');
|
|
139
|
+
process.exitCode = 1;
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
if (import.meta.url === `file://${process.argv[1]}`) main();
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// The perception eval harness. Deferred since Phase 5; four things wait on it.
|
|
3
|
+
//
|
|
4
|
+
// What it is, and the design decision behind it. The obvious harness replays
|
|
5
|
+
// stored frames through the perception path and diffs the element lists, which
|
|
6
|
+
// is what Phase 13 step 5 describes. That harness cannot exist: the element
|
|
7
|
+
// list is the accessibility tree fused with OCR, the tree is not in the frame,
|
|
8
|
+
// and OCR runs in the daemon against a live framebuffer. A frame on disk is
|
|
9
|
+
// half the input.
|
|
10
|
+
//
|
|
11
|
+
// So the split is different and, for what actually needs gating, better. The
|
|
12
|
+
// machine records the *input* — the fused element list for a screen, exactly as
|
|
13
|
+
// perception produced it. A person authors the *expected output*. Everything
|
|
14
|
+
// downstream of the element list is a pure function, so the check runs offline,
|
|
15
|
+
// deterministically, with no simulator and no daemon:
|
|
16
|
+
//
|
|
17
|
+
// resolution matching.resolve(targets, query) — which element a query picks
|
|
18
|
+
// identity fingerprint.tokens(targets) — which screen this is
|
|
19
|
+
// change analyze.signatureDiff(a, b) — whether a frame moved
|
|
20
|
+
//
|
|
21
|
+
// Those three are precisely the thresholds every blocked item wants to change.
|
|
22
|
+
// What it does *not* cover is whether perception found the elements at all,
|
|
23
|
+
// which is inherently live and stays with eval-fingerprint.mjs and the
|
|
24
|
+
// integration job. Said plainly here rather than implied, because a harness
|
|
25
|
+
// that is trusted for more than it measures is worse than no harness.
|
|
26
|
+
//
|
|
27
|
+
// Two rules from this repo:
|
|
28
|
+
// - The gate is an exit code. Nothing is verified through a pipe: this
|
|
29
|
+
// project has already shipped a check that could not fail because
|
|
30
|
+
// `node script.mjs | tail` reports tail's status.
|
|
31
|
+
// - A fixture with no authored expectations is reported as unauthored, never
|
|
32
|
+
// counted as a pass. An empty test suite is green.
|
|
33
|
+
import fs from 'node:fs';
|
|
34
|
+
import path from 'node:path';
|
|
35
|
+
import { fileURLToPath } from 'node:url';
|
|
36
|
+
import * as fingerprint from '../src/fingerprint.js';
|
|
37
|
+
import * as matching from '../src/matching.js';
|
|
38
|
+
import * as analyze from '../src/analyze.js';
|
|
39
|
+
|
|
40
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
41
|
+
const DIR = path.join(ROOT, 'test', 'perception', 'screens');
|
|
42
|
+
const arg = (n, d = null) => {
|
|
43
|
+
const hit = process.argv.find((a) => a.startsWith(`--${n}=`));
|
|
44
|
+
return hit ? hit.slice(n.length + 3) : d;
|
|
45
|
+
};
|
|
46
|
+
const has = (n) => process.argv.includes(`--${n}`);
|
|
47
|
+
|
|
48
|
+
export function fixtures(dir = DIR, only = null) {
|
|
49
|
+
if (!fs.existsSync(dir)) return [];
|
|
50
|
+
return fs.readdirSync(dir)
|
|
51
|
+
.filter((f) => f.endsWith('.json'))
|
|
52
|
+
.filter((f) => (only ? f.includes(only) : true))
|
|
53
|
+
.sort()
|
|
54
|
+
.map((f) => ({ file: f, ...JSON.parse(fs.readFileSync(path.join(dir, f), 'utf8')) }));
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Check one screen's authored expectations against the pure layers.
|
|
59
|
+
*
|
|
60
|
+
* Returns findings rather than printing, so the same function backs the CLI and
|
|
61
|
+
* the unit tests — and so a finding can be counted without being formatted.
|
|
62
|
+
*/
|
|
63
|
+
export function checkScreen(fx) {
|
|
64
|
+
const findings = [];
|
|
65
|
+
const screen = fx.points ?? null;
|
|
66
|
+
const targets = fx.targets ?? [];
|
|
67
|
+
const expect = fx.expect ?? {};
|
|
68
|
+
const authored = (expect.resolutions?.length ?? 0)
|
|
69
|
+
+ (expect.ambiguous?.length ?? 0)
|
|
70
|
+
+ (expect.none?.length ?? 0)
|
|
71
|
+
+ (fx.frame_pairs?.length ?? 0)
|
|
72
|
+
+ (expect.identity ? 1 : 0);
|
|
73
|
+
|
|
74
|
+
for (const r of expect.resolutions ?? []) {
|
|
75
|
+
const out = matching.resolve(targets, r.query, { screen });
|
|
76
|
+
if (out.status !== 'ok') {
|
|
77
|
+
findings.push({
|
|
78
|
+
kind: 'resolution',
|
|
79
|
+
query: r.query,
|
|
80
|
+
want: r.label,
|
|
81
|
+
got: out.status === 'ambiguous'
|
|
82
|
+
? `ambiguous between ${out.alternatives.map((a) => a.label).join(', ')}`
|
|
83
|
+
: 'nothing resolved',
|
|
84
|
+
});
|
|
85
|
+
continue;
|
|
86
|
+
}
|
|
87
|
+
const got = out.target.label ?? '(icon-only)';
|
|
88
|
+
// Compared on the label because that is what the author can see and mean.
|
|
89
|
+
// Coordinates would make a fixture break every time a row moved a point.
|
|
90
|
+
if (got !== r.label) {
|
|
91
|
+
findings.push({ kind: 'resolution', query: r.query, want: r.label, got: `"${got}" at ${out.target.x},${out.target.y}` });
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
// The other half of the contract, and the half with the wrong-tap risk in it:
|
|
96
|
+
// "when two things answer equally well it says so rather than guessing".
|
|
97
|
+
for (const q of expect.ambiguous ?? []) {
|
|
98
|
+
const out = matching.resolve(targets, typeof q === 'string' ? q : q.query, { screen });
|
|
99
|
+
if (out.status !== 'ambiguous') {
|
|
100
|
+
findings.push({
|
|
101
|
+
kind: 'should-ask',
|
|
102
|
+
query: typeof q === 'string' ? q : q.query,
|
|
103
|
+
want: 'ambiguous',
|
|
104
|
+
got: out.status === 'ok' ? `picked "${out.target.label}" at score ${out.score}` : 'nothing resolved',
|
|
105
|
+
});
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
for (const q of expect.none ?? []) {
|
|
110
|
+
const out = matching.resolve(targets, q, { screen });
|
|
111
|
+
if (out.status !== 'none') {
|
|
112
|
+
findings.push({
|
|
113
|
+
kind: 'should-find-nothing',
|
|
114
|
+
query: q,
|
|
115
|
+
want: 'none',
|
|
116
|
+
got: out.status === 'ok' ? `picked "${out.target.label}"` : 'ambiguous',
|
|
117
|
+
});
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
// Identity drift. A token-rule change that silently alters a screen's
|
|
122
|
+
// structural hash discards every stored map and graph for it, and the failure
|
|
123
|
+
// is invisible — an old hash is a well-formed hash that matches nothing.
|
|
124
|
+
if (expect.identity && screen) {
|
|
125
|
+
const now = fingerprint.fingerprint(targets, screen);
|
|
126
|
+
if (now.hash !== expect.identity.hash) {
|
|
127
|
+
const similarity = fingerprint.similarity(now.tokens, expect.identity.tokens ?? []);
|
|
128
|
+
findings.push({
|
|
129
|
+
kind: 'identity',
|
|
130
|
+
query: '(structural hash)',
|
|
131
|
+
want: `${expect.identity.hash?.slice(0, 12)} (${expect.identity.tokens?.length ?? 0} tokens)`,
|
|
132
|
+
got: `${now.hash?.slice(0, 12)} (${now.tokens.length} tokens), similarity ${similarity.toFixed(2)}`,
|
|
133
|
+
});
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
// Frame pairs, for the change detector. `changed: true` means an action that
|
|
138
|
+
// a person would call visible — a switch flipping, a radio dot moving.
|
|
139
|
+
for (const pair of fx.frame_pairs ?? []) {
|
|
140
|
+
const a = analyze.hexToSignature(pair.after);
|
|
141
|
+
const b = analyze.hexToSignature(pair.before);
|
|
142
|
+
// Two different questions about the same pair of frames: has the *screen*
|
|
143
|
+
// changed (the mean, which drives stillness) and has a *control* changed
|
|
144
|
+
// (the largest single region, which drives the verdict). A switch flip
|
|
145
|
+
// answers no to the first and yes to the second, which is the whole reason
|
|
146
|
+
// both exist.
|
|
147
|
+
const diff = pair.per_cell ? analyze.maxCellDelta(a, b) : analyze.signatureDiff(a, b);
|
|
148
|
+
const seen = diff > (pair.threshold ?? 0.004);
|
|
149
|
+
if (seen !== Boolean(pair.changed)) {
|
|
150
|
+
findings.push({
|
|
151
|
+
kind: 'change',
|
|
152
|
+
query: pair.note ?? '(frame pair)',
|
|
153
|
+
want: pair.changed ? 'a visible change' : 'no change',
|
|
154
|
+
got: `${pair.per_cell ? 'max cell' : 'mean'} ${diff.toFixed(5)} against a threshold of ${pair.threshold ?? 0.004}`,
|
|
155
|
+
});
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
return { authored, findings };
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
async function record() {
|
|
163
|
+
const api = await import('../src/index.js');
|
|
164
|
+
const device = arg('device');
|
|
165
|
+
const name = arg('name');
|
|
166
|
+
if (!name) throw new Error('--record needs --name=<app>/<screen>');
|
|
167
|
+
const id = await api.screenIdentity(device, { fresh: true, confirmNovel: false });
|
|
168
|
+
const entry = id.entry ?? {};
|
|
169
|
+
const out = {
|
|
170
|
+
app: name.split('/')[0],
|
|
171
|
+
screen: name.split('/').slice(1).join('/') || name,
|
|
172
|
+
note: arg('note') ?? null,
|
|
173
|
+
recorded_at: new Date().toISOString(),
|
|
174
|
+
points: id.points,
|
|
175
|
+
// The recorded input. Trimmed to what the pure layers read, so a fixture
|
|
176
|
+
// is reviewable by a person rather than a wall of machine state.
|
|
177
|
+
targets: (entry.targets ?? []).map((t) => ({
|
|
178
|
+
label: t.label ?? null,
|
|
179
|
+
value: t.value ?? null,
|
|
180
|
+
type: t.type ?? null,
|
|
181
|
+
x: t.x, y: t.y,
|
|
182
|
+
frame: t.frame ?? null,
|
|
183
|
+
region: t.region ?? null,
|
|
184
|
+
source: t.source ?? null,
|
|
185
|
+
aliases: t.aliases ?? undefined,
|
|
186
|
+
navSlot: t.navSlot ?? undefined,
|
|
187
|
+
enabled: t.enabled ?? undefined,
|
|
188
|
+
selected: t.selected ?? undefined,
|
|
189
|
+
})),
|
|
190
|
+
expect: {
|
|
191
|
+
identity: { hash: entry.structuralHash, tokens: entry.structuralTokens ?? [] },
|
|
192
|
+
// Authored by hand. The machine records what perception saw; a person
|
|
193
|
+
// says what it should mean. Deriving these from current behaviour would
|
|
194
|
+
// bake today's bugs in as the specification.
|
|
195
|
+
resolutions: [],
|
|
196
|
+
ambiguous: [],
|
|
197
|
+
none: [],
|
|
198
|
+
},
|
|
199
|
+
};
|
|
200
|
+
fs.mkdirSync(DIR, { recursive: true });
|
|
201
|
+
const file = path.join(DIR, `${name.replace(/\//g, '__')}.json`);
|
|
202
|
+
fs.writeFileSync(file, `${JSON.stringify(out, null, 2)}\n`);
|
|
203
|
+
console.log(`recorded ${out.targets.length} element(s) from ${name} -> ${path.relative(ROOT, file)}`);
|
|
204
|
+
console.log(' now author expect.resolutions / expect.ambiguous / expect.none by hand');
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
function check() {
|
|
208
|
+
const all = fixtures(DIR, arg('only'));
|
|
209
|
+
if (!all.length) {
|
|
210
|
+
console.error('check-perception: no fixtures. Record some with --record --name=<app>/<screen>');
|
|
211
|
+
process.exitCode = 1;
|
|
212
|
+
return;
|
|
213
|
+
}
|
|
214
|
+
let failed = 0;
|
|
215
|
+
let unauthored = 0;
|
|
216
|
+
let checks = 0;
|
|
217
|
+
const apps = new Set();
|
|
218
|
+
for (const fx of all) {
|
|
219
|
+
apps.add(fx.app);
|
|
220
|
+
const { authored, findings } = checkScreen(fx);
|
|
221
|
+
checks += authored;
|
|
222
|
+
if (!authored) {
|
|
223
|
+
unauthored += 1;
|
|
224
|
+
console.log(` ?? ${fx.app}/${fx.screen} recorded, no expectations authored`);
|
|
225
|
+
continue;
|
|
226
|
+
}
|
|
227
|
+
if (!findings.length) {
|
|
228
|
+
console.log(` ok ${fx.app}/${fx.screen} ${authored} expectation(s)`);
|
|
229
|
+
continue;
|
|
230
|
+
}
|
|
231
|
+
failed += findings.length;
|
|
232
|
+
console.log(`FAIL ${fx.app}/${fx.screen}`);
|
|
233
|
+
for (const f of findings) {
|
|
234
|
+
console.log(` ${f.kind}: ${f.query}\n want ${f.want}\n got ${f.got}`);
|
|
235
|
+
}
|
|
236
|
+
}
|
|
237
|
+
console.log(`\n${all.length} screen(s) across ${apps.size} app(s), ${checks} expectation(s), ${failed} failure(s)`
|
|
238
|
+
+ (unauthored ? `, ${unauthored} unauthored` : ''));
|
|
239
|
+
// An unauthored fixture is not a pass. A suite that counts recordings as
|
|
240
|
+
// successes is a suite that goes green by adding files.
|
|
241
|
+
if (failed || unauthored) process.exitCode = 1;
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
if (import.meta.url === `file://${process.argv[1]}`) {
|
|
245
|
+
if (has('record')) await record();
|
|
246
|
+
else if (has('list')) for (const f of fixtures()) console.log(`${f.app}/${f.screen} ${f.targets?.length ?? 0} elements`);
|
|
247
|
+
else check();
|
|
248
|
+
}
|