simframe 0.8.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -0
- package/flows/hpi-suite.json +68 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +49 -1
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +7 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +11 -0
- package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +48 -0
- package/native/simframed/Sources/simframed/main.swift +128 -73
- package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +40 -0
- package/package.json +2 -1
- package/scripts/bench-hpi.mjs +254 -0
- package/scripts/check-package.mjs +7 -0
- package/src/actions.js +184 -4
- package/src/baseline.js +333 -0
- package/src/cli.js +331 -2
- package/src/daemon.js +9 -0
- package/src/fingerprint.js +7 -1
- package/src/graph.js +162 -1
- package/src/index.js +118 -20
- package/src/input.js +111 -1
- package/src/intent.js +11 -2
- package/src/matching.js +81 -2
- package/src/mcp.js +14 -1
- package/src/metrics.js +499 -0
- package/src/navigate.js +44 -7
- package/src/platform/android.js +15 -0
- package/src/platform/index.js +3 -0
- package/src/platform/ios.js +40 -0
- package/src/screenmap.js +36 -14
- package/src/view.js +4 -3
package/src/metrics.js
ADDED
|
@@ -0,0 +1,499 @@
|
|
|
1
|
+
// Phase 10: simframe measuring itself.
|
|
2
|
+
//
|
|
3
|
+
// Two logs, both JSONL, both per device. `escalations.jsonl` records every
|
|
4
|
+
// point where simframe handed a decision back to the agent; `flows.jsonl`
|
|
5
|
+
// records one line per flow run. Nothing here changes what simframe does — it
|
|
6
|
+
// only writes down what happened, which is the only way the phases after this
|
|
7
|
+
// one can be prioritised or proven.
|
|
8
|
+
//
|
|
9
|
+
// The escalation log is the steering wheel: the reason breakdown decides which
|
|
10
|
+
// faculty is built next. So a reason is mandatory, "unknown" is not one of
|
|
11
|
+
// them, and every existing hand-back path maps to exactly one of the five.
|
|
12
|
+
import fs from 'node:fs';
|
|
13
|
+
import path from 'node:path';
|
|
14
|
+
import * as store from './store.js';
|
|
15
|
+
|
|
16
|
+
/** The five reasons, from docs/research/03-human-parity.md §8. Nothing else is a reason. */
|
|
17
|
+
export const REASONS = [
|
|
18
|
+
'unknown_screen',
|
|
19
|
+
'ambiguous_intent',
|
|
20
|
+
'verification_failed',
|
|
21
|
+
'novel_dialog',
|
|
22
|
+
'no_plan',
|
|
23
|
+
];
|
|
24
|
+
|
|
25
|
+
export const OUTCOMES = ['resolved_locally', 'escalated_to_model', 'failed'];
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Which faculty would have removed this escalation.
|
|
29
|
+
*
|
|
30
|
+
* This is the mapping that turns the reason breakdown into a phase order, so
|
|
31
|
+
* it lives next to the reasons rather than in prose. `built` is empty today
|
|
32
|
+
* and each phase moves its own faculty into it — see avoidableRate() for why
|
|
33
|
+
* that matters and what the rate honestly means before then.
|
|
34
|
+
*/
|
|
35
|
+
export const FACULTY = {
|
|
36
|
+
unknown_screen: 'exploration (Phase 14)',
|
|
37
|
+
ambiguous_intent: 'icon semantics (Phase 15)',
|
|
38
|
+
verification_failed: 'sense of time (Phase 11)',
|
|
39
|
+
novel_dialog: 'reflexes (Phase 12)',
|
|
40
|
+
no_plan: 'exploration (Phase 14)',
|
|
41
|
+
};
|
|
42
|
+
|
|
43
|
+
/** Faculties that exist. Empty until Phase 11 lands the first one. */
|
|
44
|
+
export const BUILT_FACULTIES = new Set();
|
|
45
|
+
|
|
46
|
+
function metricPaths(udid) {
|
|
47
|
+
const dir = store.deviceDir(udid);
|
|
48
|
+
return {
|
|
49
|
+
dir,
|
|
50
|
+
escalations: path.join(dir, 'escalations.jsonl'),
|
|
51
|
+
flows: path.join(dir, 'flows.jsonl'),
|
|
52
|
+
baselines: path.join(dir, 'baselines'),
|
|
53
|
+
};
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
export { metricPaths as paths };
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Append one record. Bookkeeping must never be able to fail a flow, so this
|
|
60
|
+
* swallows — but it swallows loudly enough to be found: the reason lands in
|
|
61
|
+
* `lastWriteError`, which `simframe escalations` prints.
|
|
62
|
+
*/
|
|
63
|
+
let lastWriteError = null;
|
|
64
|
+
export const writeError = () => lastWriteError;
|
|
65
|
+
|
|
66
|
+
export function appendJsonl(file, record) {
|
|
67
|
+
try {
|
|
68
|
+
fs.mkdirSync(path.dirname(file), { recursive: true });
|
|
69
|
+
fs.appendFileSync(file, `${JSON.stringify(record)}\n`);
|
|
70
|
+
return true;
|
|
71
|
+
} catch (err) {
|
|
72
|
+
lastWriteError = `${file}: ${err.message}`;
|
|
73
|
+
return false;
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Read a JSONL log, tolerating a torn last line.
|
|
79
|
+
*
|
|
80
|
+
* Two processes appending is not forbidden here, and a half-written line is a
|
|
81
|
+
* thing to skip rather than a reason to report no history at all.
|
|
82
|
+
*/
|
|
83
|
+
export function readJsonl(file, { limit } = {}) {
|
|
84
|
+
let raw;
|
|
85
|
+
try {
|
|
86
|
+
raw = fs.readFileSync(file, 'utf8');
|
|
87
|
+
} catch {
|
|
88
|
+
return [];
|
|
89
|
+
}
|
|
90
|
+
const out = [];
|
|
91
|
+
for (const line of raw.split('\n')) {
|
|
92
|
+
if (!line.trim()) continue;
|
|
93
|
+
try {
|
|
94
|
+
out.push(JSON.parse(line));
|
|
95
|
+
} catch {
|
|
96
|
+
/* a torn append, or something else's file. Skip the line, keep the log. */
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
return limit ? out.slice(-limit) : out;
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
export const readEscalations = (udid, opts) => readJsonl(metricPaths(udid).escalations, opts);
|
|
103
|
+
export const readFlows = (udid, opts) => readJsonl(metricPaths(udid).flows, opts);
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Mark an error as an escalation with a reason, at the site that knows why.
|
|
107
|
+
*
|
|
108
|
+
* Only ever adds a property: the message and the error class stay exactly what
|
|
109
|
+
* they were, so nothing above can behave differently for having been told.
|
|
110
|
+
* That is the whole reason classification happens by tagging rather than by
|
|
111
|
+
* matching error strings at the boundary — a regexed message is a reason that
|
|
112
|
+
* silently becomes "unknown" the day somebody rewords it.
|
|
113
|
+
*/
|
|
114
|
+
export function tag(err, reason, { candidates = [], tried = [] } = {}) {
|
|
115
|
+
if (!REASONS.includes(reason)) throw new Error(`not an escalation reason: ${reason}`);
|
|
116
|
+
err.escalation = { reason, candidates, tried };
|
|
117
|
+
return err;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/** The tag an error carries, or null. */
|
|
121
|
+
export function escalationOf(err) {
|
|
122
|
+
const e = err?.escalation;
|
|
123
|
+
if (!e || !REASONS.includes(e.reason)) return null;
|
|
124
|
+
return e;
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* Steps whose failure is a verification failure rather than a perception one.
|
|
129
|
+
*
|
|
130
|
+
* An assert that did not hold, or a wait that timed out, is simframe saying
|
|
131
|
+
* "I could not confirm this" — which is `verification_failed`, and is a
|
|
132
|
+
* different question from not knowing what is on screen.
|
|
133
|
+
*/
|
|
134
|
+
const VERIFYING_STEPS = new Set([
|
|
135
|
+
'assert', 'assertText', 'assertGone', 'waitText', 'waitFor', 'settle',
|
|
136
|
+
]);
|
|
137
|
+
|
|
138
|
+
/**
|
|
139
|
+
* The reason a thrown step is an escalation.
|
|
140
|
+
*
|
|
141
|
+
* A tagged error wins, because the site that threw it knew more than this
|
|
142
|
+
* does. Everything else maps by step, and the fallback is
|
|
143
|
+
* `verification_failed` — a step that threw is a step whose effect could not
|
|
144
|
+
* be confirmed. There is no "unknown" branch on purpose.
|
|
145
|
+
*/
|
|
146
|
+
export function reasonForStepError(step, err) {
|
|
147
|
+
const tagged = escalationOf(err);
|
|
148
|
+
if (tagged) return tagged;
|
|
149
|
+
const action = step?.action;
|
|
150
|
+
if (action === 'confirm' || action === 'chooseAny') return { reason: 'novel_dialog', candidates: [], tried: [] };
|
|
151
|
+
if (VERIFYING_STEPS.has(action)) return { reason: 'verification_failed', candidates: [], tried: [] };
|
|
152
|
+
return { reason: 'verification_failed', candidates: [], tried: [] };
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/**
|
|
156
|
+
* Which verdicts hand a decision back.
|
|
157
|
+
*
|
|
158
|
+
* `unverified` does not: it means "this action has not been seen here before",
|
|
159
|
+
* which is a fact about the graph and not a question for anybody. The other
|
|
160
|
+
* two are — an unexpected screen halts the flow, and a step that moved nothing
|
|
161
|
+
* leaves the agent to judge whether the flow really did what it says.
|
|
162
|
+
*/
|
|
163
|
+
export const ESCALATING_VERDICTS = new Set(['unexpected-screen', 'no-visible-change']);
|
|
164
|
+
|
|
165
|
+
/** How `goto`/`flow run` refusals map. They refuse rather than guess, and the refusal is the hand-back. */
|
|
166
|
+
export const PLAN_REASONS = {
|
|
167
|
+
'unknown-screen': 'unknown_screen',
|
|
168
|
+
'no-identity': 'unknown_screen',
|
|
169
|
+
ambiguous: 'ambiguous_intent',
|
|
170
|
+
'no-route': 'no_plan',
|
|
171
|
+
'unreplayable-edge': 'no_plan',
|
|
172
|
+
'unknown-flow': 'no_plan',
|
|
173
|
+
};
|
|
174
|
+
|
|
175
|
+
/** A short, bounded description of a candidate element, for the log. */
|
|
176
|
+
export function candidateOf(t) {
|
|
177
|
+
if (!t) return null;
|
|
178
|
+
if (typeof t === 'string') return { label: t.slice(0, 40) };
|
|
179
|
+
return {
|
|
180
|
+
label: typeof t.label === 'string' ? t.label.slice(0, 40) : null,
|
|
181
|
+
x: t.x ?? null,
|
|
182
|
+
y: t.y ?? null,
|
|
183
|
+
region: t.region ?? null,
|
|
184
|
+
score: t.score ?? null,
|
|
185
|
+
};
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* What screen this is, for free.
|
|
190
|
+
*
|
|
191
|
+
* Read off what is already on disk — the newest frame's layout hash, and the
|
|
192
|
+
* structural fingerprint screen memory filed under it. A perception pass here
|
|
193
|
+
* would make logging cost something, and an instrumentation phase that slows
|
|
194
|
+
* the thing it measures has broken its own numbers.
|
|
195
|
+
*/
|
|
196
|
+
export function fingerprintNow(udid, screenmap) {
|
|
197
|
+
try {
|
|
198
|
+
const state = store.readJson(store.paths(udid).state);
|
|
199
|
+
if (!state?.layoutHash) return null;
|
|
200
|
+
const near = screenmap?.recallNearest?.(udid, state.layoutHash);
|
|
201
|
+
return near?.entry?.structuralHash ?? null;
|
|
202
|
+
} catch {
|
|
203
|
+
return null;
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
/**
|
|
208
|
+
* Write one escalation.
|
|
209
|
+
*
|
|
210
|
+
* `tokens_spent` is null and stays null: the schema asks for it, and simframe
|
|
211
|
+
* is on the other side of the model from the thing that counts tokens. A
|
|
212
|
+
* plausible number derived from output length would be a guess wearing a
|
|
213
|
+
* measurement's clothes, and every number in this project is supposed to say
|
|
214
|
+
* where it came from.
|
|
215
|
+
*/
|
|
216
|
+
export function recordEscalation(udid, {
|
|
217
|
+
flowId = null,
|
|
218
|
+
stepIndex = null,
|
|
219
|
+
fingerprint = null,
|
|
220
|
+
reason,
|
|
221
|
+
candidates = [],
|
|
222
|
+
tried = [],
|
|
223
|
+
outcome = 'escalated_to_model',
|
|
224
|
+
modelTurns = 1,
|
|
225
|
+
wallMs = null,
|
|
226
|
+
detail = null,
|
|
227
|
+
} = {}) {
|
|
228
|
+
if (!REASONS.includes(reason)) throw new Error(`not an escalation reason: ${reason}`);
|
|
229
|
+
if (!OUTCOMES.includes(outcome)) throw new Error(`not an escalation outcome: ${outcome}`);
|
|
230
|
+
const record = {
|
|
231
|
+
timestamp: new Date().toISOString(),
|
|
232
|
+
flow_id: flowId,
|
|
233
|
+
step_index: stepIndex,
|
|
234
|
+
screen_fingerprint: fingerprint,
|
|
235
|
+
reason,
|
|
236
|
+
candidate_elements: candidates.map(candidateOf).filter(Boolean).slice(0, 8),
|
|
237
|
+
reflex_or_exploration_tried: tried,
|
|
238
|
+
outcome,
|
|
239
|
+
model_turns_spent: modelTurns,
|
|
240
|
+
tokens_spent: null,
|
|
241
|
+
wall_time_ms: wallMs,
|
|
242
|
+
detail: detail ? String(detail).slice(0, 200) : null,
|
|
243
|
+
};
|
|
244
|
+
appendJsonl(metricPaths(udid).escalations, record);
|
|
245
|
+
return record;
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
/** Write one flow record. Shape is docs/research/03-human-parity.md §1. */
|
|
249
|
+
export function recordFlow(udid, record) {
|
|
250
|
+
appendJsonl(metricPaths(udid).flows, record);
|
|
251
|
+
return record;
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
/**
|
|
255
|
+
* Assemble a flow record from what a run already knows.
|
|
256
|
+
*
|
|
257
|
+
* `model_turns` counts the call itself as one, then adds one per escalation
|
|
258
|
+
* the agent has to answer. That is the number the whole series is trying to
|
|
259
|
+
* drive down, so it is defined here rather than left to whoever reads the log.
|
|
260
|
+
*/
|
|
261
|
+
export function flowRecordFrom({
|
|
262
|
+
flowId,
|
|
263
|
+
flowName = null,
|
|
264
|
+
udid,
|
|
265
|
+
startedAt,
|
|
266
|
+
wallMs,
|
|
267
|
+
stepsTaken,
|
|
268
|
+
totalSteps,
|
|
269
|
+
minSteps = null,
|
|
270
|
+
imagesSent = 0,
|
|
271
|
+
escalations = [],
|
|
272
|
+
verdicts = [],
|
|
273
|
+
completed,
|
|
274
|
+
}) {
|
|
275
|
+
const histogram = {};
|
|
276
|
+
for (const v of verdicts) if (v) histogram[v] = (histogram[v] ?? 0) + 1;
|
|
277
|
+
const misTaps = verdicts.filter((v) => ESCALATING_VERDICTS.has(v)).length;
|
|
278
|
+
return {
|
|
279
|
+
flow_id: flowId,
|
|
280
|
+
flow_name: flowName,
|
|
281
|
+
device: udid,
|
|
282
|
+
started_at: new Date(startedAt).toISOString(),
|
|
283
|
+
wall_time_ms: wallMs,
|
|
284
|
+
steps_taken: stepsTaken,
|
|
285
|
+
total_steps: totalSteps,
|
|
286
|
+
min_steps: minSteps,
|
|
287
|
+
step_ratio: minSteps ? Number((stepsTaken / minSteps).toFixed(3)) : null,
|
|
288
|
+
model_turns: 1 + escalations.filter((e) => e.outcome !== 'resolved_locally').length,
|
|
289
|
+
images_sent: imagesSent,
|
|
290
|
+
input_tokens: null,
|
|
291
|
+
output_tokens: null,
|
|
292
|
+
escalations: escalations.map((e) => ({ reason: e.reason, step_index: e.step_index, outcome: e.outcome })),
|
|
293
|
+
escalation_count: escalations.length,
|
|
294
|
+
mis_taps: misTaps,
|
|
295
|
+
verdict_histogram: histogram,
|
|
296
|
+
reflex_firings: [],
|
|
297
|
+
exploration_events: [],
|
|
298
|
+
completed: Boolean(completed),
|
|
299
|
+
wrong_action_taken: verdicts.includes('unexpected-screen'),
|
|
300
|
+
};
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
// ---------------------------------------------------------------- statistics
|
|
304
|
+
|
|
305
|
+
export function median(xs) {
|
|
306
|
+
const s = xs.filter((x) => Number.isFinite(x)).sort((a, b) => a - b);
|
|
307
|
+
if (!s.length) return null;
|
|
308
|
+
const mid = s.length >> 1;
|
|
309
|
+
return s.length % 2 ? s[mid] : (s[mid - 1] + s[mid]) / 2;
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
/**
|
|
313
|
+
* Quartiles by the "exclusive median" convention: each half excludes the
|
|
314
|
+
* median of an odd-length sample. Named because IQR is meaningless without it.
|
|
315
|
+
*/
|
|
316
|
+
export function quartiles(xs) {
|
|
317
|
+
const s = xs.filter((x) => Number.isFinite(x)).sort((a, b) => a - b);
|
|
318
|
+
if (!s.length) return null;
|
|
319
|
+
const mid = s.length >> 1;
|
|
320
|
+
const lower = s.slice(0, mid);
|
|
321
|
+
const upper = s.length % 2 ? s.slice(mid + 1) : s.slice(mid);
|
|
322
|
+
const p25 = median(lower) ?? s[0];
|
|
323
|
+
const p75 = median(upper) ?? s[s.length - 1];
|
|
324
|
+
return { min: s[0], p25, p50: median(s), p75, max: s[s.length - 1], iqr: p75 - p25, n: s.length };
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
/**
|
|
328
|
+
* The p-th percentile, by nearest-rank on the sorted sample.
|
|
329
|
+
*
|
|
330
|
+
* Nearest-rank rather than interpolation: with the 5-50 samples a graph edge
|
|
331
|
+
* carries, an interpolated p95 invents a value between two observations, and
|
|
332
|
+
* every number here is supposed to be one that actually happened.
|
|
333
|
+
*/
|
|
334
|
+
export function percentile(xs, p) {
|
|
335
|
+
const s = xs.filter((x) => Number.isFinite(x)).sort((a, b) => a - b);
|
|
336
|
+
if (!s.length) return null;
|
|
337
|
+
const rank = Math.ceil((p / 100) * s.length);
|
|
338
|
+
return s[Math.min(s.length - 1, Math.max(0, rank - 1))];
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
/** The mean that punishes one slow flow, which is why §1 asks for it. */
|
|
342
|
+
export function harmonicMean(xs) {
|
|
343
|
+
const s = xs.filter((x) => Number.isFinite(x) && x > 0);
|
|
344
|
+
if (!s.length) return null;
|
|
345
|
+
return s.length / s.reduce((acc, x) => acc + 1 / x, 0);
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
/**
|
|
349
|
+
* HPI, exactly as §1 defines it.
|
|
350
|
+
*
|
|
351
|
+
* `flows` are agent runs (flows.jsonl records); `baselines` maps flow name to
|
|
352
|
+
* a human baseline. A flow with no human baseline gets no HPI_time — reported
|
|
353
|
+
* as null, never as 1.0, because a missing denominator is not parity.
|
|
354
|
+
*/
|
|
355
|
+
export function hpi({ flows, baselines = {} }) {
|
|
356
|
+
const byName = new Map();
|
|
357
|
+
for (const f of flows) {
|
|
358
|
+
if (!f?.flow_name) continue;
|
|
359
|
+
if (!byName.has(f.flow_name)) byName.set(f.flow_name, []);
|
|
360
|
+
byName.get(f.flow_name).push(f);
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
const perFlow = [...byName.entries()].map(([name, runs]) => {
|
|
364
|
+
const agent = quartiles(runs.map((r) => r.wall_time_ms));
|
|
365
|
+
const human = baselines[name]?.wall_time_ms ?? null;
|
|
366
|
+
const humanMedian = human?.p50 ?? null;
|
|
367
|
+
const stepRatios = runs.map((r) => r.step_ratio).filter((x) => Number.isFinite(x));
|
|
368
|
+
return {
|
|
369
|
+
flow: name,
|
|
370
|
+
runs: runs.length,
|
|
371
|
+
agent_ms: agent,
|
|
372
|
+
human_median_ms: humanMedian,
|
|
373
|
+
hpi_time: humanMedian && agent?.p50 ? Number((humanMedian / agent.p50).toFixed(3)) : null,
|
|
374
|
+
step_ratio: median(stepRatios),
|
|
375
|
+
completed: runs.filter((r) => r.completed).length,
|
|
376
|
+
wrong_action: runs.filter((r) => r.wrong_action_taken).length,
|
|
377
|
+
escalations: runs.reduce((acc, r) => acc + (r.escalation_count ?? 0), 0),
|
|
378
|
+
model_turns: median(runs.map((r) => r.model_turns)),
|
|
379
|
+
};
|
|
380
|
+
}).sort((a, b) => a.flow.localeCompare(b.flow));
|
|
381
|
+
|
|
382
|
+
const total = flows.length;
|
|
383
|
+
const clean = flows.filter((f) => f.completed && !f.wrong_action_taken).length;
|
|
384
|
+
const accuracy = total ? Number((clean / total).toFixed(3)) : null;
|
|
385
|
+
const times = perFlow.map((f) => f.hpi_time).filter((x) => Number.isFinite(x));
|
|
386
|
+
const hpiTime = harmonicMean(times);
|
|
387
|
+
return {
|
|
388
|
+
flows: perFlow,
|
|
389
|
+
overall: {
|
|
390
|
+
runs: total,
|
|
391
|
+
flows_measured: perFlow.length,
|
|
392
|
+
flows_with_human_baseline: times.length,
|
|
393
|
+
hpi_accuracy: accuracy,
|
|
394
|
+
hpi_time: hpiTime == null ? null : Number(hpiTime.toFixed(3)),
|
|
395
|
+
hpi: hpiTime == null || accuracy == null ? null : Number((accuracy * hpiTime).toFixed(3)),
|
|
396
|
+
step_ratio: median(perFlow.map((f) => f.step_ratio)),
|
|
397
|
+
model_turns_median: median(perFlow.map((f) => f.model_turns)),
|
|
398
|
+
},
|
|
399
|
+
};
|
|
400
|
+
}
|
|
401
|
+
|
|
402
|
+
/**
|
|
403
|
+
* The escalation dashboard.
|
|
404
|
+
*
|
|
405
|
+
* `avoidable` is §8's definition — an escalation whose reason maps to a
|
|
406
|
+
* faculty that is not built yet, or is built and still let it through. One
|
|
407
|
+
* consequence is worth stating rather than hiding: with no faculty built, the
|
|
408
|
+
* rate is 1.0 by construction and says nothing. The per-reason breakdown is
|
|
409
|
+
* the part that decides the next phase, and it is informative today.
|
|
410
|
+
*/
|
|
411
|
+
export function breakdown(records) {
|
|
412
|
+
const byReason = {};
|
|
413
|
+
for (const r of REASONS) byReason[r] = 0;
|
|
414
|
+
const byScreen = new Map();
|
|
415
|
+
const byOutcome = {};
|
|
416
|
+
let avoidable = 0;
|
|
417
|
+
for (const r of records) {
|
|
418
|
+
if (!REASONS.includes(r?.reason)) continue;
|
|
419
|
+
byReason[r.reason] += 1;
|
|
420
|
+
byOutcome[r.outcome] = (byOutcome[r.outcome] ?? 0) + 1;
|
|
421
|
+
// Already avoided locally, so not avoidable by anything unbuilt.
|
|
422
|
+
if (r.outcome !== 'resolved_locally') avoidable += 1;
|
|
423
|
+
const key = r.screen_fingerprint ?? '(no fingerprint)';
|
|
424
|
+
byScreen.set(key, (byScreen.get(key) ?? 0) + 1);
|
|
425
|
+
}
|
|
426
|
+
const total = records.filter((r) => REASONS.includes(r?.reason)).length;
|
|
427
|
+
return {
|
|
428
|
+
total,
|
|
429
|
+
by_reason: byReason,
|
|
430
|
+
by_outcome: byOutcome,
|
|
431
|
+
faculty: Object.fromEntries(
|
|
432
|
+
REASONS.filter((r) => byReason[r]).map((r) => [r, `${FACULTY[r]}${BUILT_FACULTIES.has(FACULTY[r]) ? ' [built]' : ''}`]),
|
|
433
|
+
),
|
|
434
|
+
avoidable,
|
|
435
|
+
avoidable_escalation_rate: total ? Number((avoidable / total).toFixed(3)) : null,
|
|
436
|
+
top_screens: [...byScreen.entries()]
|
|
437
|
+
.sort((a, b) => b[1] - a[1])
|
|
438
|
+
.slice(0, 10)
|
|
439
|
+
.map(([fingerprint, count]) => ({ fingerprint, count })),
|
|
440
|
+
model_turns_spent: records.reduce((acc, r) => acc + (r.model_turns_spent ?? 0), 0),
|
|
441
|
+
};
|
|
442
|
+
}
|
|
443
|
+
|
|
444
|
+
/**
|
|
445
|
+
* How much slower than the committed baseline fails the build.
|
|
446
|
+
*
|
|
447
|
+
* 25%, not the 10% research §1 proposed, and the number is traceable to a
|
|
448
|
+
* measurement rather than to taste. Three HPI measurements of *identical code*
|
|
449
|
+
* on the same device the same afternoon gave HPI_time 0.475, 0.413 and 0.371,
|
|
450
|
+
* and per flow the medians moved up to 18% between runs — a 10% gate would
|
|
451
|
+
* have failed on noise roughly half the time, and a gate that cries wolf gets
|
|
452
|
+
* ignored, which costs more than no gate. Two things narrow the band instead
|
|
453
|
+
* of loosening it further: the gate reads the median of three passes rather
|
|
454
|
+
* than one, and accuracy stays strict at any drop at all. Numbers in
|
|
455
|
+
* docs/BENCHMARKS.md under "What the gate is set to, and why".
|
|
456
|
+
*/
|
|
457
|
+
export const TIME_REGRESSION = 0.25;
|
|
458
|
+
|
|
459
|
+
/**
|
|
460
|
+
* The HPI_time a gate should compare: the median across passes when a report
|
|
461
|
+
* has them, and the single measurement when it does not — so a baseline
|
|
462
|
+
* committed before passes existed still gates.
|
|
463
|
+
*/
|
|
464
|
+
export const gateTime = (o) => o?.hpi_time_median_of_passes ?? o?.hpi_time ?? null;
|
|
465
|
+
|
|
466
|
+
/**
|
|
467
|
+
* Compare a measurement against the committed baseline.
|
|
468
|
+
*
|
|
469
|
+
* A pure function rather than a few lines inside the CI script, because an
|
|
470
|
+
* untested gate is this project's recurring failure: the packaging check and a
|
|
471
|
+
* fingerprint eval both shipped unable to fail, and both looked exactly like
|
|
472
|
+
* this. Returns the reasons it should fail — empty means pass.
|
|
473
|
+
*/
|
|
474
|
+
export function gateAgainst(baseline, measured, { timeRegression = TIME_REGRESSION } = {}) {
|
|
475
|
+
const base = baseline?.overall ?? {};
|
|
476
|
+
const now = measured?.overall ?? measured ?? {};
|
|
477
|
+
const baseTime = gateTime(base);
|
|
478
|
+
const nowTime = gateTime(now);
|
|
479
|
+
const failures = [];
|
|
480
|
+
if (base.hpi_accuracy != null && now.hpi_accuracy != null && now.hpi_accuracy < base.hpi_accuracy) {
|
|
481
|
+
failures.push(`HPI_accuracy dropped: ${now.hpi_accuracy} < ${base.hpi_accuracy} (any drop fails)`);
|
|
482
|
+
}
|
|
483
|
+
if (baseTime != null && nowTime != null && nowTime < baseTime * (1 - timeRegression)) {
|
|
484
|
+
failures.push(
|
|
485
|
+
`HPI_time regressed >${timeRegression * 100}%: ${nowTime} < ${(baseTime * (1 - timeRegression)).toFixed(3)}`,
|
|
486
|
+
);
|
|
487
|
+
}
|
|
488
|
+
// A checkout missing the human baseline the committed number was computed
|
|
489
|
+
// against would otherwise pass by having nothing to compare.
|
|
490
|
+
if (baseTime != null && nowTime == null) {
|
|
491
|
+
failures.push('HPI_time is null but the baseline has one — the human baseline it needs is missing from this checkout');
|
|
492
|
+
}
|
|
493
|
+
return failures;
|
|
494
|
+
}
|
|
495
|
+
|
|
496
|
+
/** A flow id that sorts by time and is short enough to read in a log. */
|
|
497
|
+
export function newFlowId() {
|
|
498
|
+
return `${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 8)}`;
|
|
499
|
+
}
|
package/src/navigate.js
CHANGED
|
@@ -6,6 +6,8 @@
|
|
|
6
6
|
// somebody already walked, saved so it can be walked again.
|
|
7
7
|
import { runScript } from './actions.js';
|
|
8
8
|
import * as graph from './graph.js';
|
|
9
|
+
import * as metrics from './metrics.js';
|
|
10
|
+
import * as screenmap from './screenmap.js';
|
|
9
11
|
import * as api from './index.js';
|
|
10
12
|
import * as store from './store.js';
|
|
11
13
|
import fs from 'node:fs';
|
|
@@ -31,6 +33,31 @@ export function stepFor(edge) {
|
|
|
31
33
|
return null;
|
|
32
34
|
}
|
|
33
35
|
|
|
36
|
+
/**
|
|
37
|
+
* A refusal to act, written down.
|
|
38
|
+
*
|
|
39
|
+
* `goto` and `flow run` refuse rather than guess, and a refusal is exactly
|
|
40
|
+
* "the tool handed the decision back" — the thing the escalation log exists to
|
|
41
|
+
* count. The reason mapping lives in metrics.PLAN_REASONS so the five reasons
|
|
42
|
+
* have one owner.
|
|
43
|
+
*/
|
|
44
|
+
function refuse(udid, result, { detail = null } = {}) {
|
|
45
|
+
try {
|
|
46
|
+
const reason = metrics.PLAN_REASONS[result.reason];
|
|
47
|
+
if (reason) {
|
|
48
|
+
metrics.recordEscalation(udid, {
|
|
49
|
+
reason,
|
|
50
|
+
fingerprint: metrics.fingerprintNow(udid, screenmap),
|
|
51
|
+
outcome: 'escalated_to_model',
|
|
52
|
+
detail: detail ?? result.reason,
|
|
53
|
+
});
|
|
54
|
+
}
|
|
55
|
+
} catch {
|
|
56
|
+
/* a log that cannot be written must not change what is returned */
|
|
57
|
+
}
|
|
58
|
+
return result;
|
|
59
|
+
}
|
|
60
|
+
|
|
34
61
|
/**
|
|
35
62
|
* Walk to a known screen.
|
|
36
63
|
*
|
|
@@ -43,8 +70,8 @@ export async function goto(deviceQuery, target, { options, ...runOptions } = {})
|
|
|
43
70
|
const udid = device.udid;
|
|
44
71
|
|
|
45
72
|
const found = graph.findScreen(udid, target);
|
|
46
|
-
if (!found) return { ok: false, reason: 'unknown-screen', known: knownScreens(udid) };
|
|
47
|
-
if (found.ambiguous) return { ok: false, reason: 'ambiguous', candidates: found.ambiguous };
|
|
73
|
+
if (!found) return refuse(udid, { ok: false, reason: 'unknown-screen', known: knownScreens(udid) }, { detail: `no screen matches "${target}"` });
|
|
74
|
+
if (found.ambiguous) return refuse(udid, { ok: false, reason: 'ambiguous', candidates: found.ambiguous }, { detail: `"${target}" fits ${found.ambiguous.length} screens` });
|
|
48
75
|
|
|
49
76
|
const here = await api.screenIdentity(udid, {});
|
|
50
77
|
if (here.hash === found.node.hash) {
|
|
@@ -57,12 +84,12 @@ export async function goto(deviceQuery, target, { options, ...runOptions } = {})
|
|
|
57
84
|
// `Cannot read properties of null (reading 'slice')` instead of answering.
|
|
58
85
|
// Not hypothetical on Android, where README's own table puts the launcher at
|
|
59
86
|
// one token.
|
|
60
|
-
if (!here.hash) return { ok: false, reason: 'no-identity', to: found.name };
|
|
87
|
+
if (!here.hash) return refuse(udid, { ok: false, reason: 'no-identity', to: found.name });
|
|
61
88
|
const path_ = graph.route(udid, { hash: here.hash, tokens: here.tokens }, found.node.hash);
|
|
62
|
-
if (!path_) return { ok: false, reason: 'no-route', from: here.hash.slice(0, 8), to: found.name };
|
|
89
|
+
if (!path_) return refuse(udid, { ok: false, reason: 'no-route', from: here.hash.slice(0, 8), to: found.name });
|
|
63
90
|
|
|
64
91
|
const steps = path_.map(stepFor);
|
|
65
|
-
if (steps.some((s) => !s)) return { ok: false, reason: 'unreplayable-edge', to: found.name };
|
|
92
|
+
if (steps.some((s) => !s)) return refuse(udid, { ok: false, reason: 'unreplayable-edge', to: found.name });
|
|
66
93
|
|
|
67
94
|
const result = await runScript(udid, { steps, stopOnUnexpected: true, ...runOptions });
|
|
68
95
|
const arrived = await api.screenIdentity(udid, {});
|
|
@@ -121,7 +148,17 @@ export function listFlows(udid) {
|
|
|
121
148
|
export async function runFlow(deviceQuery, name, { options, ...runOptions } = {}) {
|
|
122
149
|
const { device } = await api.ensureDaemon(deviceQuery, options);
|
|
123
150
|
const flow = loadFlow(device.udid, name);
|
|
124
|
-
if (!flow) return { ok: false, reason: 'unknown-flow', known: listFlows(device.udid).map((f) => f.name) };
|
|
125
|
-
|
|
151
|
+
if (!flow) return refuse(device.udid, { ok: false, reason: 'unknown-flow', known: listFlows(device.udid).map((f) => f.name) }, { detail: `no saved flow "${name}"` });
|
|
152
|
+
// A replayed flow knows its own name, so its record can be compared against
|
|
153
|
+
// a human doing the same thing. `minSteps` comes from the flow definition or
|
|
154
|
+
// stays null — the step count of a recorded route is not a claim about the
|
|
155
|
+
// shortest one.
|
|
156
|
+
const result = await runScript(device.udid, {
|
|
157
|
+
steps: flow.steps,
|
|
158
|
+
stopOnUnexpected: true,
|
|
159
|
+
flowName: name,
|
|
160
|
+
minSteps: flow.minSteps ?? null,
|
|
161
|
+
...runOptions,
|
|
162
|
+
});
|
|
126
163
|
return { ok: result.ranSteps === flow.steps.length, name, ...result };
|
|
127
164
|
}
|
package/src/platform/android.js
CHANGED
|
@@ -974,6 +974,16 @@ function toolchain() {
|
|
|
974
974
|
* driver's name. `uiautomator dump` costs 2,012 ms a read, which is why it is
|
|
975
975
|
* not the answer; see docs/DEFERRED.md for the shape of the one that would be.
|
|
976
976
|
*/
|
|
977
|
+
async function bootedAt(serial) {
|
|
978
|
+
try {
|
|
979
|
+
const out = await adb(serial, ['shell', 'cat', '/proc/uptime']);
|
|
980
|
+
const seconds = Number(String(out.stdout ?? out).trim().split(/\s+/)[0]);
|
|
981
|
+
return Number.isFinite(seconds) ? Date.now() - seconds * 1000 : null;
|
|
982
|
+
} catch {
|
|
983
|
+
return null;
|
|
984
|
+
}
|
|
985
|
+
}
|
|
986
|
+
|
|
977
987
|
function capabilities() {
|
|
978
988
|
return {
|
|
979
989
|
captureEngines: ['screenshot'],
|
|
@@ -1004,6 +1014,11 @@ export const platform = {
|
|
|
1004
1014
|
setPasteboard,
|
|
1005
1015
|
getPasteboard,
|
|
1006
1016
|
permissionServices: () => PERMISSION_SERVICES,
|
|
1017
|
+
// Uptime, because there is no CoreSimulator directory to stat. `/proc/uptime`
|
|
1018
|
+
// is seconds since boot, so boot is now minus that — and it is a real
|
|
1019
|
+
// answer rather than the other platform's vocabulary, which is the rule a
|
|
1020
|
+
// backend that cannot answer has to follow.
|
|
1021
|
+
bootedAt,
|
|
1007
1022
|
capabilities,
|
|
1008
1023
|
toolchain,
|
|
1009
1024
|
};
|
package/src/platform/index.js
CHANGED
|
@@ -63,6 +63,7 @@ export const PLATFORM_SURFACE = Object.freeze([
|
|
|
63
63
|
'geometry', 'inputDriver',
|
|
64
64
|
'screenshot', 'launchApp', 'terminateApp', 'openUrl',
|
|
65
65
|
'setPermission', 'setPasteboard', 'permissionServices', 'capabilities', 'toolchain',
|
|
66
|
+
'bootedAt',
|
|
66
67
|
]);
|
|
67
68
|
|
|
68
69
|
/** @type {Record<string, Platform>} */
|
|
@@ -237,6 +238,8 @@ export function permissionServices(udid) {
|
|
|
237
238
|
* knows. Tap points computed from that are wrong, and nothing says so.
|
|
238
239
|
*/
|
|
239
240
|
export const geometryFor = (udid) => platformFor(udid).geometry(udid);
|
|
241
|
+
/** When the device last booted, epoch ms, or null. Both backends answer; neither guesses. */
|
|
242
|
+
export const bootedAtFor = (udid) => platformFor(udid).bootedAt(udid);
|
|
240
243
|
|
|
241
244
|
/**
|
|
242
245
|
* The backend's own input path, or null when input comes from above the
|
package/src/platform/ios.js
CHANGED
|
@@ -5,6 +5,9 @@
|
|
|
5
5
|
// and reachable only through the `platform` object at the bottom — the
|
|
6
6
|
// JavaScript counterpart of the `SimulatorPlatform` protocol in Swift.
|
|
7
7
|
import { execFile, execFileSync } from 'node:child_process';
|
|
8
|
+
import fs from 'node:fs';
|
|
9
|
+
import os from 'node:os';
|
|
10
|
+
import path from 'node:path';
|
|
8
11
|
import { promisify } from 'node:util';
|
|
9
12
|
|
|
10
13
|
const run = promisify(execFile);
|
|
@@ -272,6 +275,42 @@ function geometry() {
|
|
|
272
275
|
* *is* the platform's — the emulator console — which is why this is a question
|
|
273
276
|
* a backend gets asked at all.
|
|
274
277
|
*/
|
|
278
|
+
/**
|
|
279
|
+
* When this device last booted, in epoch ms, or null if it cannot be told.
|
|
280
|
+
*
|
|
281
|
+
* Why it matters: the HID session lives in the daemon, and a device restart
|
|
282
|
+
* kills it while leaving the daemon perfectly healthy. Every tap after that is
|
|
283
|
+
* dispatched successfully and moves nothing — measured, five runs in a row,
|
|
284
|
+
* on the correct coordinates for the correct element. Only hardware buttons
|
|
285
|
+
* recover on their own, deliberately, because retrying a tap can act twice.
|
|
286
|
+
*
|
|
287
|
+
* The signal is a stat, not a `simctl` call: CoreSimulator writes
|
|
288
|
+
* `data/var/run/syslog.pid` when the device's syslogd starts, and touches
|
|
289
|
+
* `device.plist` on every state change. Both read 21:21:26 on a device booted
|
|
290
|
+
* at 21:21:26. A stat costs microseconds, which matters because this is
|
|
291
|
+
* checked before input.
|
|
292
|
+
*
|
|
293
|
+
* A false positive costs one session rebuild and no action, so the ordering
|
|
294
|
+
* prefers the most boot-specific marker and falls back rather than guessing.
|
|
295
|
+
*/
|
|
296
|
+
function bootedAt(udid) {
|
|
297
|
+
const dir = path.join(
|
|
298
|
+
os.homedir(), 'Library', 'Developer', 'CoreSimulator', 'Devices', udid,
|
|
299
|
+
);
|
|
300
|
+
for (const marker of [
|
|
301
|
+
path.join(dir, 'data', 'var', 'run', 'syslog.pid'),
|
|
302
|
+
path.join(dir, 'data', 'var', 'run'),
|
|
303
|
+
path.join(dir, 'device.plist'),
|
|
304
|
+
]) {
|
|
305
|
+
try {
|
|
306
|
+
return fs.statSync(marker).mtimeMs;
|
|
307
|
+
} catch {
|
|
308
|
+
/* try the next marker */
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
return null;
|
|
312
|
+
}
|
|
313
|
+
|
|
275
314
|
function inputDriver() {
|
|
276
315
|
return null;
|
|
277
316
|
}
|
|
@@ -301,6 +340,7 @@ export const platform = {
|
|
|
301
340
|
isBootedSync,
|
|
302
341
|
ownsUdid,
|
|
303
342
|
geometry,
|
|
343
|
+
bootedAt,
|
|
304
344
|
inputDriver,
|
|
305
345
|
screenshot,
|
|
306
346
|
launchApp,
|