simframe 0.8.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/metrics.js ADDED
@@ -0,0 +1,596 @@
1
+ // Phase 10: simframe measuring itself.
2
+ //
3
+ // Two logs, both JSONL, both per device. `escalations.jsonl` records every
4
+ // point where simframe handed a decision back to the agent; `flows.jsonl`
5
+ // records one line per flow run. Nothing here changes what simframe does — it
6
+ // only writes down what happened, which is the only way the phases after this
7
+ // one can be prioritised or proven.
8
+ //
9
+ // The escalation log is the steering wheel: the reason breakdown decides which
10
+ // faculty is built next. So a reason is mandatory, "unknown" is not one of
11
+ // them, and every existing hand-back path maps to exactly one of the five.
12
+ import fs from 'node:fs';
13
+ import path from 'node:path';
14
+ import * as store from './store.js';
15
+
16
+ /** The five reasons, from docs/research/03-human-parity.md §8. Nothing else is a reason. */
17
+ export const REASONS = [
18
+ 'unknown_screen',
19
+ 'ambiguous_intent',
20
+ 'verification_failed',
21
+ 'novel_dialog',
22
+ 'no_plan',
23
+ ];
24
+
25
+ export const OUTCOMES = ['resolved_locally', 'escalated_to_model', 'failed'];
26
+
27
+ /**
28
+ * Which faculty would have removed this escalation.
29
+ *
30
+ * This is the mapping that turns the reason breakdown into a phase order, so
31
+ * it lives next to the reasons rather than in prose. `built` is empty today
32
+ * and each phase moves its own faculty into it — see avoidableRate() for why
33
+ * that matters and what the rate honestly means before then.
34
+ */
35
+ export const FACULTY = {
36
+ unknown_screen: 'exploration (Phase 14)',
37
+ ambiguous_intent: 'icon semantics (Phase 15)',
38
+ verification_failed: 'sense of time (Phase 11)',
39
+ novel_dialog: 'reflexes (Phase 12)',
40
+ no_plan: 'exploration (Phase 14)',
41
+ };
42
+
43
+ /**
44
+ * Faculties that exist.
45
+ *
46
+ * Phase 11 landed the first one, so `verification_failed` no longer maps to
47
+ * something unbuilt — which changes what those records *mean*. Before, they
48
+ * were a queue waiting on a phase. Now they are evidence that the phase which
49
+ * shipped is not sufficient, and that is a more useful thing for the log to be
50
+ * able to say than a count of things nobody has written yet.
51
+ */
52
+ export const BUILT_FACULTIES = new Set(['sense of time (Phase 11)']);
53
+
54
+ function metricPaths(udid) {
55
+ const dir = store.deviceDir(udid);
56
+ return {
57
+ dir,
58
+ escalations: path.join(dir, 'escalations.jsonl'),
59
+ flows: path.join(dir, 'flows.jsonl'),
60
+ baselines: path.join(dir, 'baselines'),
61
+ };
62
+ }
63
+
64
+ export { metricPaths as paths };
65
+
66
+ /**
67
+ * Append one record. Bookkeeping must never be able to fail a flow, so this
68
+ * swallows — but it swallows loudly enough to be found: the reason lands in
69
+ * `lastWriteError`, which `simframe escalations` prints.
70
+ */
71
+ let lastWriteError = null;
72
+ export const writeError = () => lastWriteError;
73
+
74
+ export function appendJsonl(file, record) {
75
+ try {
76
+ fs.mkdirSync(path.dirname(file), { recursive: true });
77
+ fs.appendFileSync(file, `${JSON.stringify(record)}\n`);
78
+ return true;
79
+ } catch (err) {
80
+ lastWriteError = `${file}: ${err.message}`;
81
+ return false;
82
+ }
83
+ }
84
+
85
+ /**
86
+ * Read a JSONL log, tolerating a torn last line.
87
+ *
88
+ * Two processes appending is not forbidden here, and a half-written line is a
89
+ * thing to skip rather than a reason to report no history at all.
90
+ */
91
+ export function readJsonl(file, { limit } = {}) {
92
+ let raw;
93
+ try {
94
+ raw = fs.readFileSync(file, 'utf8');
95
+ } catch {
96
+ return [];
97
+ }
98
+ const out = [];
99
+ for (const line of raw.split('\n')) {
100
+ if (!line.trim()) continue;
101
+ try {
102
+ out.push(JSON.parse(line));
103
+ } catch {
104
+ /* a torn append, or something else's file. Skip the line, keep the log. */
105
+ }
106
+ }
107
+ return limit ? out.slice(-limit) : out;
108
+ }
109
+
110
+ export const readEscalations = (udid, opts) => readJsonl(metricPaths(udid).escalations, opts);
111
+ export const readFlows = (udid, opts) => readJsonl(metricPaths(udid).flows, opts);
112
+
113
+ /**
114
+ * Mark an error as an escalation with a reason, at the site that knows why.
115
+ *
116
+ * Only ever adds a property: the message and the error class stay exactly what
117
+ * they were, so nothing above can behave differently for having been told.
118
+ * That is the whole reason classification happens by tagging rather than by
119
+ * matching error strings at the boundary — a regexed message is a reason that
120
+ * silently becomes "unknown" the day somebody rewords it.
121
+ */
122
+ export function tag(err, reason, { candidates = [], tried = [], ambiguous = false } = {}) {
123
+ if (!REASONS.includes(reason)) throw new Error(`not an escalation reason: ${reason}`);
124
+ // `ambiguous` is narrower than the reason, and that is the point. Two very
125
+ // different failures both tag `ambiguous_intent`: the target is on screen
126
+ // several times over, and the target is not on screen at all on a screen we
127
+ // thought we knew. Only the first is resolvable by *choosing*, and only the
128
+ // first tells a waiting caller that waiting is pointless — the thing it is
129
+ // waiting for has already arrived.
130
+ err.escalation = { reason, candidates, tried, ambiguous };
131
+ return err;
132
+ }
133
+
134
+ /** The tag an error carries, or null. */
135
+ export function escalationOf(err) {
136
+ const e = err?.escalation;
137
+ if (!e || !REASONS.includes(e.reason)) return null;
138
+ return e;
139
+ }
140
+
141
+ /**
142
+ * Steps whose failure is a verification failure rather than a perception one.
143
+ *
144
+ * An assert that did not hold, or a wait that timed out, is simframe saying
145
+ * "I could not confirm this" — which is `verification_failed`, and is a
146
+ * different question from not knowing what is on screen.
147
+ */
148
+ const VERIFYING_STEPS = new Set([
149
+ 'assert', 'assertText', 'assertGone', 'waitText', 'waitFor', 'settle',
150
+ ]);
151
+
152
+ /**
153
+ * The reason a thrown step is an escalation.
154
+ *
155
+ * A tagged error wins, because the site that threw it knew more than this
156
+ * does. Everything else maps by step, and the fallback is
157
+ * `verification_failed` — a step that threw is a step whose effect could not
158
+ * be confirmed. There is no "unknown" branch on purpose.
159
+ */
160
+ export function reasonForStepError(step, err) {
161
+ const tagged = escalationOf(err);
162
+ if (tagged) return tagged;
163
+ const action = step?.action;
164
+ if (action === 'confirm' || action === 'chooseAny') return { reason: 'novel_dialog', candidates: [], tried: [] };
165
+ if (VERIFYING_STEPS.has(action)) return { reason: 'verification_failed', candidates: [], tried: [] };
166
+ return { reason: 'verification_failed', candidates: [], tried: [] };
167
+ }
168
+
169
+ /**
170
+ * Which verdicts hand a decision back.
171
+ *
172
+ * `unverified` does not: it means "this action has not been seen here before",
173
+ * which is a fact about the graph and not a question for anybody. The other
174
+ * two are — an unexpected screen halts the flow, and a step that moved nothing
175
+ * leaves the agent to judge whether the flow really did what it says.
176
+ */
177
+ export const ESCALATING_VERDICTS = new Set(['unexpected-screen', 'no-visible-change']);
178
+
179
+ /** How `goto`/`flow run` refusals map. They refuse rather than guess, and the refusal is the hand-back. */
180
+ export const PLAN_REASONS = {
181
+ 'unknown-screen': 'unknown_screen',
182
+ 'no-identity': 'unknown_screen',
183
+ ambiguous: 'ambiguous_intent',
184
+ 'no-route': 'no_plan',
185
+ 'unreplayable-edge': 'no_plan',
186
+ 'unknown-flow': 'no_plan',
187
+ };
188
+
189
+ /** A short, bounded description of a candidate element, for the log. */
190
+ export function candidateOf(t) {
191
+ if (!t) return null;
192
+ if (typeof t === 'string') return { label: t.slice(0, 40) };
193
+ return {
194
+ label: typeof t.label === 'string' ? t.label.slice(0, 40) : null,
195
+ x: t.x ?? null,
196
+ y: t.y ?? null,
197
+ region: t.region ?? null,
198
+ score: t.score ?? null,
199
+ };
200
+ }
201
+
202
+ /**
203
+ * What screen this is, for free.
204
+ *
205
+ * Read off what is already on disk — the newest frame's layout hash, and the
206
+ * structural fingerprint screen memory filed under it. A perception pass here
207
+ * would make logging cost something, and an instrumentation phase that slows
208
+ * the thing it measures has broken its own numbers.
209
+ */
210
+ export function fingerprintNow(udid, screenmap) {
211
+ try {
212
+ const state = store.readJson(store.paths(udid).state);
213
+ if (!state?.layoutHash) return null;
214
+ const near = screenmap?.recallNearest?.(udid, state.layoutHash);
215
+ return near?.entry?.structuralHash ?? null;
216
+ } catch {
217
+ return null;
218
+ }
219
+ }
220
+
221
+ /**
222
+ * Write one escalation.
223
+ *
224
+ * `tokens_spent` is null and stays null: the schema asks for it, and simframe
225
+ * is on the other side of the model from the thing that counts tokens. A
226
+ * plausible number derived from output length would be a guess wearing a
227
+ * measurement's clothes, and every number in this project is supposed to say
228
+ * where it came from.
229
+ */
230
+ /**
231
+ * Which run of which program wrote a record.
232
+ *
233
+ * The log is per-device and, until now, anonymous — so two agents driving one
234
+ * booted simulator wrote one interleaved file with no way to separate them.
235
+ * Measured on the bench device in a single evening: 57 records to 81, a third
236
+ * of the new ones naming screens from an app the suite has never launched.
237
+ *
238
+ * That is not a corrupted file, it is a corrupted instrument. CLAUDE.md makes
239
+ * the reason breakdown of this log the thing that chooses which faculty gets
240
+ * built next, and a breakdown that silently pools two sessions errs toward
241
+ * whichever of them made more mistakes — which is not the same question as
242
+ * which faculty is missing.
243
+ *
244
+ * A pid alone would not do: pids are reused, and the useful grouping is "one
245
+ * agent's run", which for the MCP server is the life of the process and for
246
+ * the CLI is a single command. So: the start time, the pid, and a random tail,
247
+ * computed once per process. `client` says what kind of process it was, since
248
+ * "the MCP server did this" and "somebody ran a CLI command" deserve different
249
+ * readings of the same reason.
250
+ *
251
+ * Deliberately not a device id, a username, or anything about the machine. This
252
+ * file is committed to a public repo in summary form, and the question it has
253
+ * to answer is "was this all one agent", which needs no identity to answer.
254
+ */
255
+ const SESSION_ID = `${process.pid.toString(36)}-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 8)}`;
256
+
257
+ /** How this process is being used, for reading a breakdown afterwards. */
258
+ function clientKind() {
259
+ const argv = process.argv.join(' ');
260
+ if (/\bmcp\b/.test(argv)) return 'mcp';
261
+ if (/bench-hpi|scripts\//.test(argv)) return 'script';
262
+ if (/cli\.js|\bsimframe\b/.test(argv)) return 'cli';
263
+ return 'library';
264
+ }
265
+ const CLIENT = clientKind();
266
+
267
+ /** The session this process's records belong to. Exported for `escalations`. */
268
+ export const sessionId = () => SESSION_ID;
269
+ export const clientName = () => CLIENT;
270
+
271
+ export function recordEscalation(udid, {
272
+ flowId = null,
273
+ flowName = null,
274
+ stepIndex = null,
275
+ fingerprint = null,
276
+ reason,
277
+ candidates = [],
278
+ tried = [],
279
+ outcome = 'escalated_to_model',
280
+ modelTurns = 1,
281
+ wallMs = null,
282
+ detail = null,
283
+ } = {}) {
284
+ if (!REASONS.includes(reason)) throw new Error(`not an escalation reason: ${reason}`);
285
+ if (!OUTCOMES.includes(outcome)) throw new Error(`not an escalation outcome: ${outcome}`);
286
+ const record = {
287
+ timestamp: new Date().toISOString(),
288
+ // Added after the log turned out to pool two agents' work invisibly. Both
289
+ // are cheap and neither is derivable afterwards, which is the test for
290
+ // whether a field belongs in a log at all.
291
+ session_id: SESSION_ID,
292
+ client: CLIENT,
293
+ flow_id: flowId,
294
+ flow_name: flowName,
295
+ step_index: stepIndex,
296
+ screen_fingerprint: fingerprint,
297
+ reason,
298
+ candidate_elements: candidates.map(candidateOf).filter(Boolean).slice(0, 8),
299
+ reflex_or_exploration_tried: tried,
300
+ outcome,
301
+ model_turns_spent: modelTurns,
302
+ tokens_spent: null,
303
+ wall_time_ms: wallMs,
304
+ detail: detail ? String(detail).slice(0, 200) : null,
305
+ };
306
+ appendJsonl(metricPaths(udid).escalations, record);
307
+ return record;
308
+ }
309
+
310
+ /** Write one flow record. Shape is docs/research/03-human-parity.md §1. */
311
+ export function recordFlow(udid, record) {
312
+ appendJsonl(metricPaths(udid).flows, record);
313
+ return record;
314
+ }
315
+
316
+ /**
317
+ * Assemble a flow record from what a run already knows.
318
+ *
319
+ * `model_turns` counts the call itself as one, then adds one per escalation
320
+ * the agent has to answer. That is the number the whole series is trying to
321
+ * drive down, so it is defined here rather than left to whoever reads the log.
322
+ */
323
+ export function flowRecordFrom({
324
+ flowId,
325
+ flowName = null,
326
+ udid,
327
+ startedAt,
328
+ wallMs,
329
+ stepsTaken,
330
+ totalSteps,
331
+ minSteps = null,
332
+ imagesSent = 0,
333
+ escalations = [],
334
+ verdicts = [],
335
+ completed,
336
+ }) {
337
+ const histogram = {};
338
+ for (const v of verdicts) if (v) histogram[v] = (histogram[v] ?? 0) + 1;
339
+ const misTaps = verdicts.filter((v) => ESCALATING_VERDICTS.has(v)).length;
340
+ return {
341
+ flow_id: flowId,
342
+ flow_name: flowName,
343
+ device: udid,
344
+ started_at: new Date(startedAt).toISOString(),
345
+ wall_time_ms: wallMs,
346
+ steps_taken: stepsTaken,
347
+ total_steps: totalSteps,
348
+ min_steps: minSteps,
349
+ step_ratio: minSteps ? Number((stepsTaken / minSteps).toFixed(3)) : null,
350
+ model_turns: 1 + escalations.filter((e) => e.outcome !== 'resolved_locally').length,
351
+ images_sent: imagesSent,
352
+ input_tokens: null,
353
+ output_tokens: null,
354
+ escalations: escalations.map((e) => ({ reason: e.reason, step_index: e.step_index, outcome: e.outcome })),
355
+ escalation_count: escalations.length,
356
+ mis_taps: misTaps,
357
+ verdict_histogram: histogram,
358
+ reflex_firings: [],
359
+ exploration_events: [],
360
+ completed: Boolean(completed),
361
+ wrong_action_taken: verdicts.includes('unexpected-screen'),
362
+ };
363
+ }
364
+
365
+ // ---------------------------------------------------------------- statistics
366
+
367
+ export function median(xs) {
368
+ const s = xs.filter((x) => Number.isFinite(x)).sort((a, b) => a - b);
369
+ if (!s.length) return null;
370
+ const mid = s.length >> 1;
371
+ return s.length % 2 ? s[mid] : (s[mid - 1] + s[mid]) / 2;
372
+ }
373
+
374
+ /**
375
+ * Quartiles by the "exclusive median" convention: each half excludes the
376
+ * median of an odd-length sample. Named because IQR is meaningless without it.
377
+ */
378
+ export function quartiles(xs) {
379
+ const s = xs.filter((x) => Number.isFinite(x)).sort((a, b) => a - b);
380
+ if (!s.length) return null;
381
+ const mid = s.length >> 1;
382
+ const lower = s.slice(0, mid);
383
+ const upper = s.length % 2 ? s.slice(mid + 1) : s.slice(mid);
384
+ const p25 = median(lower) ?? s[0];
385
+ const p75 = median(upper) ?? s[s.length - 1];
386
+ return { min: s[0], p25, p50: median(s), p75, max: s[s.length - 1], iqr: p75 - p25, n: s.length };
387
+ }
388
+
389
+ /**
390
+ * The p-th percentile, by nearest-rank on the sorted sample.
391
+ *
392
+ * Nearest-rank rather than interpolation: with the 5-50 samples a graph edge
393
+ * carries, an interpolated p95 invents a value between two observations, and
394
+ * every number here is supposed to be one that actually happened.
395
+ */
396
+ export function percentile(xs, p) {
397
+ const s = xs.filter((x) => Number.isFinite(x)).sort((a, b) => a - b);
398
+ if (!s.length) return null;
399
+ const rank = Math.ceil((p / 100) * s.length);
400
+ return s[Math.min(s.length - 1, Math.max(0, rank - 1))];
401
+ }
402
+
403
+ /** The mean that punishes one slow flow, which is why §1 asks for it. */
404
+ export function harmonicMean(xs) {
405
+ const s = xs.filter((x) => Number.isFinite(x) && x > 0);
406
+ if (!s.length) return null;
407
+ return s.length / s.reduce((acc, x) => acc + 1 / x, 0);
408
+ }
409
+
410
+ /**
411
+ * HPI, exactly as §1 defines it.
412
+ *
413
+ * `flows` are agent runs (flows.jsonl records); `baselines` maps flow name to
414
+ * a human baseline. A flow with no human baseline gets no HPI_time — reported
415
+ * as null, never as 1.0, because a missing denominator is not parity.
416
+ */
417
+ export function hpi({ flows, baselines = {} }) {
418
+ const byName = new Map();
419
+ for (const f of flows) {
420
+ if (!f?.flow_name) continue;
421
+ if (!byName.has(f.flow_name)) byName.set(f.flow_name, []);
422
+ byName.get(f.flow_name).push(f);
423
+ }
424
+
425
+ const perFlow = [...byName.entries()].map(([name, runs]) => {
426
+ const agent = quartiles(runs.map((r) => r.wall_time_ms));
427
+ const human = baselines[name]?.wall_time_ms ?? null;
428
+ const humanMedian = human?.p50 ?? null;
429
+ const stepRatios = runs.map((r) => r.step_ratio).filter((x) => Number.isFinite(x));
430
+ return {
431
+ flow: name,
432
+ runs: runs.length,
433
+ agent_ms: agent,
434
+ human_median_ms: humanMedian,
435
+ hpi_time: humanMedian && agent?.p50 ? Number((humanMedian / agent.p50).toFixed(3)) : null,
436
+ step_ratio: median(stepRatios),
437
+ completed: runs.filter((r) => r.completed).length,
438
+ wrong_action: runs.filter((r) => r.wrong_action_taken).length,
439
+ escalations: runs.reduce((acc, r) => acc + (r.escalation_count ?? 0), 0),
440
+ model_turns: median(runs.map((r) => r.model_turns)),
441
+ };
442
+ }).sort((a, b) => a.flow.localeCompare(b.flow));
443
+
444
+ const total = flows.length;
445
+ const clean = flows.filter((f) => f.completed && !f.wrong_action_taken).length;
446
+ const accuracy = total ? Number((clean / total).toFixed(3)) : null;
447
+ const times = perFlow.map((f) => f.hpi_time).filter((x) => Number.isFinite(x));
448
+ const hpiTime = harmonicMean(times);
449
+ return {
450
+ flows: perFlow,
451
+ overall: {
452
+ runs: total,
453
+ flows_measured: perFlow.length,
454
+ flows_with_human_baseline: times.length,
455
+ hpi_accuracy: accuracy,
456
+ hpi_time: hpiTime == null ? null : Number(hpiTime.toFixed(3)),
457
+ hpi: hpiTime == null || accuracy == null ? null : Number((accuracy * hpiTime).toFixed(3)),
458
+ step_ratio: median(perFlow.map((f) => f.step_ratio)),
459
+ model_turns_median: median(perFlow.map((f) => f.model_turns)),
460
+ },
461
+ };
462
+ }
463
+
464
+ /**
465
+ * The escalation dashboard.
466
+ *
467
+ * `avoidable` is §8's definition — an escalation whose reason maps to a
468
+ * faculty that is not built yet, or is built and still let it through. One
469
+ * consequence is worth stating rather than hiding: with no faculty built, the
470
+ * rate is 1.0 by construction and says nothing. The per-reason breakdown is
471
+ * the part that decides the next phase, and it is informative today.
472
+ */
473
+ export function breakdown(records, { session = null, flow = null } = {}) {
474
+ const byReason = {};
475
+ for (const r of REASONS) byReason[r] = 0;
476
+ const byScreen = new Map();
477
+ const byOutcome = {};
478
+ const bySession = new Map();
479
+ const byFlow = new Map();
480
+ let avoidable = 0;
481
+ let unattributed = 0;
482
+ // Filtering happens here rather than at the call site so `total` and every
483
+ // rate below it describe the same set of records.
484
+ const kept = records.filter((r) => (session ? r?.session_id === session : true))
485
+ .filter((r) => (flow ? r?.flow_name === flow : true));
486
+ for (const r of kept) {
487
+ if (!REASONS.includes(r?.reason)) continue;
488
+ // Records written before sessions were recorded cannot be attributed, and
489
+ // saying how many there are is the difference between a breakdown that
490
+ // pools two agents and one that says it might be.
491
+ if (r.session_id) {
492
+ const key = `${r.session_id}|${r.client ?? '?'}`;
493
+ bySession.set(key, (bySession.get(key) ?? 0) + 1);
494
+ } else {
495
+ unattributed += 1;
496
+ }
497
+ if (r.flow_name) byFlow.set(r.flow_name, (byFlow.get(r.flow_name) ?? 0) + 1);
498
+ byReason[r.reason] += 1;
499
+ byOutcome[r.outcome] = (byOutcome[r.outcome] ?? 0) + 1;
500
+ // Already avoided locally, so not avoidable by anything unbuilt.
501
+ if (r.outcome !== 'resolved_locally') avoidable += 1;
502
+ const key = r.screen_fingerprint ?? '(no fingerprint)';
503
+ byScreen.set(key, (byScreen.get(key) ?? 0) + 1);
504
+ }
505
+ const total = kept.filter((r) => REASONS.includes(r?.reason)).length;
506
+ const sessions = [...bySession.entries()]
507
+ .map(([key, count]) => {
508
+ const [id, client] = key.split('|');
509
+ return { session_id: id, client, count };
510
+ })
511
+ .sort((a, b) => b.count - a.count);
512
+ return {
513
+ total,
514
+ // The log is per-device and shared: two agents on one booted simulator
515
+ // write one interleaved file. More than one session here means the counts
516
+ // below are a pool, and CLAUDE.md uses those counts to choose a phase.
517
+ sessions,
518
+ session_count: sessions.length,
519
+ unattributed,
520
+ // Any unattributed record at all makes this a pool: the whole point is
521
+ // that they cannot be told apart, and 92 of them is not "one session".
522
+ pooled: sessions.length > 1 || unattributed > 0,
523
+ by_flow: Object.fromEntries([...byFlow.entries()].sort((a, b) => b[1] - a[1])),
524
+ by_reason: byReason,
525
+ by_outcome: byOutcome,
526
+ faculty: Object.fromEntries(
527
+ REASONS.filter((r) => byReason[r]).map((r) => [r, `${FACULTY[r]}${BUILT_FACULTIES.has(FACULTY[r]) ? ' [built]' : ''}`]),
528
+ ),
529
+ avoidable,
530
+ avoidable_escalation_rate: total ? Number((avoidable / total).toFixed(3)) : null,
531
+ top_screens: [...byScreen.entries()]
532
+ .sort((a, b) => b[1] - a[1])
533
+ .slice(0, 10)
534
+ .map(([fingerprint, count]) => ({ fingerprint, count })),
535
+ // `kept`, not `records` — a filtered breakdown that reports the whole
536
+ // log's model turns is the same class of mistake as pooling two sessions.
537
+ model_turns_spent: kept.reduce((acc, r) => acc + (r.model_turns_spent ?? 0), 0),
538
+ };
539
+ }
540
+
541
+ /**
542
+ * How much slower than the committed baseline fails the build.
543
+ *
544
+ * 25%, not the 10% research §1 proposed, and the number is traceable to a
545
+ * measurement rather than to taste. Three HPI measurements of *identical code*
546
+ * on the same device the same afternoon gave HPI_time 0.475, 0.413 and 0.371,
547
+ * and per flow the medians moved up to 18% between runs — a 10% gate would
548
+ * have failed on noise roughly half the time, and a gate that cries wolf gets
549
+ * ignored, which costs more than no gate. Two things narrow the band instead
550
+ * of loosening it further: the gate reads the median of three passes rather
551
+ * than one, and accuracy stays strict at any drop at all. Numbers in
552
+ * docs/BENCHMARKS.md under "What the gate is set to, and why".
553
+ */
554
+ export const TIME_REGRESSION = 0.25;
555
+
556
+ /**
557
+ * The HPI_time a gate should compare: the median across passes when a report
558
+ * has them, and the single measurement when it does not — so a baseline
559
+ * committed before passes existed still gates.
560
+ */
561
+ export const gateTime = (o) => o?.hpi_time_median_of_passes ?? o?.hpi_time ?? null;
562
+
563
+ /**
564
+ * Compare a measurement against the committed baseline.
565
+ *
566
+ * A pure function rather than a few lines inside the CI script, because an
567
+ * untested gate is this project's recurring failure: the packaging check and a
568
+ * fingerprint eval both shipped unable to fail, and both looked exactly like
569
+ * this. Returns the reasons it should fail — empty means pass.
570
+ */
571
+ export function gateAgainst(baseline, measured, { timeRegression = TIME_REGRESSION } = {}) {
572
+ const base = baseline?.overall ?? {};
573
+ const now = measured?.overall ?? measured ?? {};
574
+ const baseTime = gateTime(base);
575
+ const nowTime = gateTime(now);
576
+ const failures = [];
577
+ if (base.hpi_accuracy != null && now.hpi_accuracy != null && now.hpi_accuracy < base.hpi_accuracy) {
578
+ failures.push(`HPI_accuracy dropped: ${now.hpi_accuracy} < ${base.hpi_accuracy} (any drop fails)`);
579
+ }
580
+ if (baseTime != null && nowTime != null && nowTime < baseTime * (1 - timeRegression)) {
581
+ failures.push(
582
+ `HPI_time regressed >${timeRegression * 100}%: ${nowTime} < ${(baseTime * (1 - timeRegression)).toFixed(3)}`,
583
+ );
584
+ }
585
+ // A checkout missing the human baseline the committed number was computed
586
+ // against would otherwise pass by having nothing to compare.
587
+ if (baseTime != null && nowTime == null) {
588
+ failures.push('HPI_time is null but the baseline has one — the human baseline it needs is missing from this checkout');
589
+ }
590
+ return failures;
591
+ }
592
+
593
+ /** A flow id that sorts by time and is short enough to read in a log. */
594
+ export function newFlowId() {
595
+ return `${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 8)}`;
596
+ }
package/src/navigate.js CHANGED
@@ -6,6 +6,8 @@
6
6
  // somebody already walked, saved so it can be walked again.
7
7
  import { runScript } from './actions.js';
8
8
  import * as graph from './graph.js';
9
+ import * as metrics from './metrics.js';
10
+ import * as screenmap from './screenmap.js';
9
11
  import * as api from './index.js';
10
12
  import * as store from './store.js';
11
13
  import fs from 'node:fs';
@@ -31,6 +33,34 @@ export function stepFor(edge) {
31
33
  return null;
32
34
  }
33
35
 
36
+ /**
37
+ * A refusal to act, written down.
38
+ *
39
+ * `goto` and `flow run` refuse rather than guess, and a refusal is exactly
40
+ * "the tool handed the decision back" — the thing the escalation log exists to
41
+ * count. The reason mapping lives in metrics.PLAN_REASONS so the five reasons
42
+ * have one owner.
43
+ */
44
+ function refuse(udid, result, { detail = null, flowName = null } = {}) {
45
+ try {
46
+ const reason = metrics.PLAN_REASONS[result.reason];
47
+ if (reason) {
48
+ metrics.recordEscalation(udid, {
49
+ reason,
50
+ // A refusal by `goto` is about a destination and one by `flow run` is
51
+ // about a named flow. Either is what a breakdown wants to group by.
52
+ flowName,
53
+ fingerprint: metrics.fingerprintNow(udid, screenmap),
54
+ outcome: 'escalated_to_model',
55
+ detail: detail ?? result.reason,
56
+ });
57
+ }
58
+ } catch {
59
+ /* a log that cannot be written must not change what is returned */
60
+ }
61
+ return result;
62
+ }
63
+
34
64
  /**
35
65
  * Walk to a known screen.
36
66
  *
@@ -43,8 +73,8 @@ export async function goto(deviceQuery, target, { options, ...runOptions } = {})
43
73
  const udid = device.udid;
44
74
 
45
75
  const found = graph.findScreen(udid, target);
46
- if (!found) return { ok: false, reason: 'unknown-screen', known: knownScreens(udid) };
47
- if (found.ambiguous) return { ok: false, reason: 'ambiguous', candidates: found.ambiguous };
76
+ if (!found) return refuse(udid, { ok: false, reason: 'unknown-screen', known: knownScreens(udid) }, { detail: `no screen matches "${target}"`, flowName: `goto:${target}` });
77
+ if (found.ambiguous) return refuse(udid, { ok: false, reason: 'ambiguous', candidates: found.ambiguous }, { detail: `"${target}" fits ${found.ambiguous.length} screens`, flowName: `goto:${target}` });
48
78
 
49
79
  const here = await api.screenIdentity(udid, {});
50
80
  if (here.hash === found.node.hash) {
@@ -57,12 +87,12 @@ export async function goto(deviceQuery, target, { options, ...runOptions } = {})
57
87
  // `Cannot read properties of null (reading 'slice')` instead of answering.
58
88
  // Not hypothetical on Android, where README's own table puts the launcher at
59
89
  // one token.
60
- if (!here.hash) return { ok: false, reason: 'no-identity', to: found.name };
90
+ if (!here.hash) return refuse(udid, { ok: false, reason: 'no-identity', to: found.name }, { flowName: `goto:${target}` });
61
91
  const path_ = graph.route(udid, { hash: here.hash, tokens: here.tokens }, found.node.hash);
62
- if (!path_) return { ok: false, reason: 'no-route', from: here.hash.slice(0, 8), to: found.name };
92
+ if (!path_) return refuse(udid, { ok: false, reason: 'no-route', from: here.hash.slice(0, 8), to: found.name }, { flowName: `goto:${target}` });
63
93
 
64
94
  const steps = path_.map(stepFor);
65
- if (steps.some((s) => !s)) return { ok: false, reason: 'unreplayable-edge', to: found.name };
95
+ if (steps.some((s) => !s)) return refuse(udid, { ok: false, reason: 'unreplayable-edge', to: found.name }, { flowName: `goto:${target}` });
66
96
 
67
97
  const result = await runScript(udid, { steps, stopOnUnexpected: true, ...runOptions });
68
98
  const arrived = await api.screenIdentity(udid, {});
@@ -121,7 +151,17 @@ export function listFlows(udid) {
121
151
  export async function runFlow(deviceQuery, name, { options, ...runOptions } = {}) {
122
152
  const { device } = await api.ensureDaemon(deviceQuery, options);
123
153
  const flow = loadFlow(device.udid, name);
124
- if (!flow) return { ok: false, reason: 'unknown-flow', known: listFlows(device.udid).map((f) => f.name) };
125
- const result = await runScript(device.udid, { steps: flow.steps, stopOnUnexpected: true, ...runOptions });
154
+ if (!flow) return refuse(device.udid, { ok: false, reason: 'unknown-flow', known: listFlows(device.udid).map((f) => f.name) }, { detail: `no saved flow "${name}"`, flowName: name });
155
+ // A replayed flow knows its own name, so its record can be compared against
156
+ // a human doing the same thing. `minSteps` comes from the flow definition or
157
+ // stays null — the step count of a recorded route is not a claim about the
158
+ // shortest one.
159
+ const result = await runScript(device.udid, {
160
+ steps: flow.steps,
161
+ stopOnUnexpected: true,
162
+ flowName: name,
163
+ minSteps: flow.minSteps ?? null,
164
+ ...runOptions,
165
+ });
126
166
  return { ok: result.ranSteps === flow.steps.length, name, ...result };
127
167
  }