simframe 0.8.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +31 -3
- package/flows/hpi-suite.json +68 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +49 -1
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +7 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +11 -0
- package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +48 -0
- package/native/simframed/Sources/simframed/main.swift +128 -73
- package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +40 -0
- package/package.json +2 -1
- package/scripts/bench-hpi.mjs +254 -0
- package/scripts/check-package.mjs +7 -0
- package/scripts/check-private.mjs +143 -0
- package/scripts/eval-perception.mjs +248 -0
- package/src/actions.js +333 -14
- package/src/analyze.js +70 -0
- package/src/baseline.js +333 -0
- package/src/cli.js +410 -4
- package/src/daemon.js +9 -0
- package/src/fingerprint.js +7 -1
- package/src/graph.js +262 -1
- package/src/index.js +335 -22
- package/src/input.js +155 -1
- package/src/intent.js +11 -2
- package/src/matching.js +136 -4
- package/src/mcp.js +14 -1
- package/src/metrics.js +596 -0
- package/src/navigate.js +47 -7
- package/src/platform/android.js +16 -1
- package/src/platform/index.js +3 -0
- package/src/platform/ios.js +40 -0
- package/src/screenmap.js +55 -14
- package/src/view.js +65 -3
package/src/metrics.js
ADDED
|
@@ -0,0 +1,596 @@
|
|
|
1
|
+
// Phase 10: simframe measuring itself.
|
|
2
|
+
//
|
|
3
|
+
// Two logs, both JSONL, both per device. `escalations.jsonl` records every
|
|
4
|
+
// point where simframe handed a decision back to the agent; `flows.jsonl`
|
|
5
|
+
// records one line per flow run. Nothing here changes what simframe does — it
|
|
6
|
+
// only writes down what happened, which is the only way the phases after this
|
|
7
|
+
// one can be prioritised or proven.
|
|
8
|
+
//
|
|
9
|
+
// The escalation log is the steering wheel: the reason breakdown decides which
|
|
10
|
+
// faculty is built next. So a reason is mandatory, "unknown" is not one of
|
|
11
|
+
// them, and every existing hand-back path maps to exactly one of the five.
|
|
12
|
+
import fs from 'node:fs';
|
|
13
|
+
import path from 'node:path';
|
|
14
|
+
import * as store from './store.js';
|
|
15
|
+
|
|
16
|
+
/** The five reasons, from docs/research/03-human-parity.md §8. Nothing else is a reason. */
|
|
17
|
+
export const REASONS = [
|
|
18
|
+
'unknown_screen',
|
|
19
|
+
'ambiguous_intent',
|
|
20
|
+
'verification_failed',
|
|
21
|
+
'novel_dialog',
|
|
22
|
+
'no_plan',
|
|
23
|
+
];
|
|
24
|
+
|
|
25
|
+
export const OUTCOMES = ['resolved_locally', 'escalated_to_model', 'failed'];
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Which faculty would have removed this escalation.
|
|
29
|
+
*
|
|
30
|
+
* This is the mapping that turns the reason breakdown into a phase order, so
|
|
31
|
+
* it lives next to the reasons rather than in prose. `built` is empty today
|
|
32
|
+
* and each phase moves its own faculty into it — see avoidableRate() for why
|
|
33
|
+
* that matters and what the rate honestly means before then.
|
|
34
|
+
*/
|
|
35
|
+
export const FACULTY = {
|
|
36
|
+
unknown_screen: 'exploration (Phase 14)',
|
|
37
|
+
ambiguous_intent: 'icon semantics (Phase 15)',
|
|
38
|
+
verification_failed: 'sense of time (Phase 11)',
|
|
39
|
+
novel_dialog: 'reflexes (Phase 12)',
|
|
40
|
+
no_plan: 'exploration (Phase 14)',
|
|
41
|
+
};
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Faculties that exist.
|
|
45
|
+
*
|
|
46
|
+
* Phase 11 landed the first one, so `verification_failed` no longer maps to
|
|
47
|
+
* something unbuilt — which changes what those records *mean*. Before, they
|
|
48
|
+
* were a queue waiting on a phase. Now they are evidence that the phase which
|
|
49
|
+
* shipped is not sufficient, and that is a more useful thing for the log to be
|
|
50
|
+
* able to say than a count of things nobody has written yet.
|
|
51
|
+
*/
|
|
52
|
+
export const BUILT_FACULTIES = new Set(['sense of time (Phase 11)']);
|
|
53
|
+
|
|
54
|
+
function metricPaths(udid) {
|
|
55
|
+
const dir = store.deviceDir(udid);
|
|
56
|
+
return {
|
|
57
|
+
dir,
|
|
58
|
+
escalations: path.join(dir, 'escalations.jsonl'),
|
|
59
|
+
flows: path.join(dir, 'flows.jsonl'),
|
|
60
|
+
baselines: path.join(dir, 'baselines'),
|
|
61
|
+
};
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
export { metricPaths as paths };
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* Append one record. Bookkeeping must never be able to fail a flow, so this
|
|
68
|
+
* swallows — but it swallows loudly enough to be found: the reason lands in
|
|
69
|
+
* `lastWriteError`, which `simframe escalations` prints.
|
|
70
|
+
*/
|
|
71
|
+
let lastWriteError = null;
|
|
72
|
+
export const writeError = () => lastWriteError;
|
|
73
|
+
|
|
74
|
+
export function appendJsonl(file, record) {
|
|
75
|
+
try {
|
|
76
|
+
fs.mkdirSync(path.dirname(file), { recursive: true });
|
|
77
|
+
fs.appendFileSync(file, `${JSON.stringify(record)}\n`);
|
|
78
|
+
return true;
|
|
79
|
+
} catch (err) {
|
|
80
|
+
lastWriteError = `${file}: ${err.message}`;
|
|
81
|
+
return false;
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* Read a JSONL log, tolerating a torn last line.
|
|
87
|
+
*
|
|
88
|
+
* Two processes appending is not forbidden here, and a half-written line is a
|
|
89
|
+
* thing to skip rather than a reason to report no history at all.
|
|
90
|
+
*/
|
|
91
|
+
export function readJsonl(file, { limit } = {}) {
|
|
92
|
+
let raw;
|
|
93
|
+
try {
|
|
94
|
+
raw = fs.readFileSync(file, 'utf8');
|
|
95
|
+
} catch {
|
|
96
|
+
return [];
|
|
97
|
+
}
|
|
98
|
+
const out = [];
|
|
99
|
+
for (const line of raw.split('\n')) {
|
|
100
|
+
if (!line.trim()) continue;
|
|
101
|
+
try {
|
|
102
|
+
out.push(JSON.parse(line));
|
|
103
|
+
} catch {
|
|
104
|
+
/* a torn append, or something else's file. Skip the line, keep the log. */
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
return limit ? out.slice(-limit) : out;
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
export const readEscalations = (udid, opts) => readJsonl(metricPaths(udid).escalations, opts);
|
|
111
|
+
export const readFlows = (udid, opts) => readJsonl(metricPaths(udid).flows, opts);
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* Mark an error as an escalation with a reason, at the site that knows why.
|
|
115
|
+
*
|
|
116
|
+
* Only ever adds a property: the message and the error class stay exactly what
|
|
117
|
+
* they were, so nothing above can behave differently for having been told.
|
|
118
|
+
* That is the whole reason classification happens by tagging rather than by
|
|
119
|
+
* matching error strings at the boundary — a regexed message is a reason that
|
|
120
|
+
* silently becomes "unknown" the day somebody rewords it.
|
|
121
|
+
*/
|
|
122
|
+
export function tag(err, reason, { candidates = [], tried = [], ambiguous = false } = {}) {
|
|
123
|
+
if (!REASONS.includes(reason)) throw new Error(`not an escalation reason: ${reason}`);
|
|
124
|
+
// `ambiguous` is narrower than the reason, and that is the point. Two very
|
|
125
|
+
// different failures both tag `ambiguous_intent`: the target is on screen
|
|
126
|
+
// several times over, and the target is not on screen at all on a screen we
|
|
127
|
+
// thought we knew. Only the first is resolvable by *choosing*, and only the
|
|
128
|
+
// first tells a waiting caller that waiting is pointless — the thing it is
|
|
129
|
+
// waiting for has already arrived.
|
|
130
|
+
err.escalation = { reason, candidates, tried, ambiguous };
|
|
131
|
+
return err;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/** The tag an error carries, or null. */
|
|
135
|
+
export function escalationOf(err) {
|
|
136
|
+
const e = err?.escalation;
|
|
137
|
+
if (!e || !REASONS.includes(e.reason)) return null;
|
|
138
|
+
return e;
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* Steps whose failure is a verification failure rather than a perception one.
|
|
143
|
+
*
|
|
144
|
+
* An assert that did not hold, or a wait that timed out, is simframe saying
|
|
145
|
+
* "I could not confirm this" — which is `verification_failed`, and is a
|
|
146
|
+
* different question from not knowing what is on screen.
|
|
147
|
+
*/
|
|
148
|
+
const VERIFYING_STEPS = new Set([
|
|
149
|
+
'assert', 'assertText', 'assertGone', 'waitText', 'waitFor', 'settle',
|
|
150
|
+
]);
|
|
151
|
+
|
|
152
|
+
/**
|
|
153
|
+
* The reason a thrown step is an escalation.
|
|
154
|
+
*
|
|
155
|
+
* A tagged error wins, because the site that threw it knew more than this
|
|
156
|
+
* does. Everything else maps by step, and the fallback is
|
|
157
|
+
* `verification_failed` — a step that threw is a step whose effect could not
|
|
158
|
+
* be confirmed. There is no "unknown" branch on purpose.
|
|
159
|
+
*/
|
|
160
|
+
export function reasonForStepError(step, err) {
|
|
161
|
+
const tagged = escalationOf(err);
|
|
162
|
+
if (tagged) return tagged;
|
|
163
|
+
const action = step?.action;
|
|
164
|
+
if (action === 'confirm' || action === 'chooseAny') return { reason: 'novel_dialog', candidates: [], tried: [] };
|
|
165
|
+
if (VERIFYING_STEPS.has(action)) return { reason: 'verification_failed', candidates: [], tried: [] };
|
|
166
|
+
return { reason: 'verification_failed', candidates: [], tried: [] };
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
/**
|
|
170
|
+
* Which verdicts hand a decision back.
|
|
171
|
+
*
|
|
172
|
+
* `unverified` does not: it means "this action has not been seen here before",
|
|
173
|
+
* which is a fact about the graph and not a question for anybody. The other
|
|
174
|
+
* two are — an unexpected screen halts the flow, and a step that moved nothing
|
|
175
|
+
* leaves the agent to judge whether the flow really did what it says.
|
|
176
|
+
*/
|
|
177
|
+
export const ESCALATING_VERDICTS = new Set(['unexpected-screen', 'no-visible-change']);
|
|
178
|
+
|
|
179
|
+
/** How `goto`/`flow run` refusals map. They refuse rather than guess, and the refusal is the hand-back. */
|
|
180
|
+
export const PLAN_REASONS = {
|
|
181
|
+
'unknown-screen': 'unknown_screen',
|
|
182
|
+
'no-identity': 'unknown_screen',
|
|
183
|
+
ambiguous: 'ambiguous_intent',
|
|
184
|
+
'no-route': 'no_plan',
|
|
185
|
+
'unreplayable-edge': 'no_plan',
|
|
186
|
+
'unknown-flow': 'no_plan',
|
|
187
|
+
};
|
|
188
|
+
|
|
189
|
+
/** A short, bounded description of a candidate element, for the log. */
|
|
190
|
+
export function candidateOf(t) {
|
|
191
|
+
if (!t) return null;
|
|
192
|
+
if (typeof t === 'string') return { label: t.slice(0, 40) };
|
|
193
|
+
return {
|
|
194
|
+
label: typeof t.label === 'string' ? t.label.slice(0, 40) : null,
|
|
195
|
+
x: t.x ?? null,
|
|
196
|
+
y: t.y ?? null,
|
|
197
|
+
region: t.region ?? null,
|
|
198
|
+
score: t.score ?? null,
|
|
199
|
+
};
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
/**
|
|
203
|
+
* What screen this is, for free.
|
|
204
|
+
*
|
|
205
|
+
* Read off what is already on disk — the newest frame's layout hash, and the
|
|
206
|
+
* structural fingerprint screen memory filed under it. A perception pass here
|
|
207
|
+
* would make logging cost something, and an instrumentation phase that slows
|
|
208
|
+
* the thing it measures has broken its own numbers.
|
|
209
|
+
*/
|
|
210
|
+
export function fingerprintNow(udid, screenmap) {
|
|
211
|
+
try {
|
|
212
|
+
const state = store.readJson(store.paths(udid).state);
|
|
213
|
+
if (!state?.layoutHash) return null;
|
|
214
|
+
const near = screenmap?.recallNearest?.(udid, state.layoutHash);
|
|
215
|
+
return near?.entry?.structuralHash ?? null;
|
|
216
|
+
} catch {
|
|
217
|
+
return null;
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
/**
|
|
222
|
+
* Write one escalation.
|
|
223
|
+
*
|
|
224
|
+
* `tokens_spent` is null and stays null: the schema asks for it, and simframe
|
|
225
|
+
* is on the other side of the model from the thing that counts tokens. A
|
|
226
|
+
* plausible number derived from output length would be a guess wearing a
|
|
227
|
+
* measurement's clothes, and every number in this project is supposed to say
|
|
228
|
+
* where it came from.
|
|
229
|
+
*/
|
|
230
|
+
/**
|
|
231
|
+
* Which run of which program wrote a record.
|
|
232
|
+
*
|
|
233
|
+
* The log is per-device and, until now, anonymous — so two agents driving one
|
|
234
|
+
* booted simulator wrote one interleaved file with no way to separate them.
|
|
235
|
+
* Measured on the bench device in a single evening: 57 records to 81, a third
|
|
236
|
+
* of the new ones naming screens from an app the suite has never launched.
|
|
237
|
+
*
|
|
238
|
+
* That is not a corrupted file, it is a corrupted instrument. CLAUDE.md makes
|
|
239
|
+
* the reason breakdown of this log the thing that chooses which faculty gets
|
|
240
|
+
* built next, and a breakdown that silently pools two sessions errs toward
|
|
241
|
+
* whichever of them made more mistakes — which is not the same question as
|
|
242
|
+
* which faculty is missing.
|
|
243
|
+
*
|
|
244
|
+
* A pid alone would not do: pids are reused, and the useful grouping is "one
|
|
245
|
+
* agent's run", which for the MCP server is the life of the process and for
|
|
246
|
+
* the CLI is a single command. So: the start time, the pid, and a random tail,
|
|
247
|
+
* computed once per process. `client` says what kind of process it was, since
|
|
248
|
+
* "the MCP server did this" and "somebody ran a CLI command" deserve different
|
|
249
|
+
* readings of the same reason.
|
|
250
|
+
*
|
|
251
|
+
* Deliberately not a device id, a username, or anything about the machine. This
|
|
252
|
+
* file is committed to a public repo in summary form, and the question it has
|
|
253
|
+
* to answer is "was this all one agent", which needs no identity to answer.
|
|
254
|
+
*/
|
|
255
|
+
const SESSION_ID = `${process.pid.toString(36)}-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 8)}`;
|
|
256
|
+
|
|
257
|
+
/** How this process is being used, for reading a breakdown afterwards. */
|
|
258
|
+
function clientKind() {
|
|
259
|
+
const argv = process.argv.join(' ');
|
|
260
|
+
if (/\bmcp\b/.test(argv)) return 'mcp';
|
|
261
|
+
if (/bench-hpi|scripts\//.test(argv)) return 'script';
|
|
262
|
+
if (/cli\.js|\bsimframe\b/.test(argv)) return 'cli';
|
|
263
|
+
return 'library';
|
|
264
|
+
}
|
|
265
|
+
const CLIENT = clientKind();
|
|
266
|
+
|
|
267
|
+
/** The session this process's records belong to. Exported for `escalations`. */
|
|
268
|
+
export const sessionId = () => SESSION_ID;
|
|
269
|
+
export const clientName = () => CLIENT;
|
|
270
|
+
|
|
271
|
+
export function recordEscalation(udid, {
|
|
272
|
+
flowId = null,
|
|
273
|
+
flowName = null,
|
|
274
|
+
stepIndex = null,
|
|
275
|
+
fingerprint = null,
|
|
276
|
+
reason,
|
|
277
|
+
candidates = [],
|
|
278
|
+
tried = [],
|
|
279
|
+
outcome = 'escalated_to_model',
|
|
280
|
+
modelTurns = 1,
|
|
281
|
+
wallMs = null,
|
|
282
|
+
detail = null,
|
|
283
|
+
} = {}) {
|
|
284
|
+
if (!REASONS.includes(reason)) throw new Error(`not an escalation reason: ${reason}`);
|
|
285
|
+
if (!OUTCOMES.includes(outcome)) throw new Error(`not an escalation outcome: ${outcome}`);
|
|
286
|
+
const record = {
|
|
287
|
+
timestamp: new Date().toISOString(),
|
|
288
|
+
// Added after the log turned out to pool two agents' work invisibly. Both
|
|
289
|
+
// are cheap and neither is derivable afterwards, which is the test for
|
|
290
|
+
// whether a field belongs in a log at all.
|
|
291
|
+
session_id: SESSION_ID,
|
|
292
|
+
client: CLIENT,
|
|
293
|
+
flow_id: flowId,
|
|
294
|
+
flow_name: flowName,
|
|
295
|
+
step_index: stepIndex,
|
|
296
|
+
screen_fingerprint: fingerprint,
|
|
297
|
+
reason,
|
|
298
|
+
candidate_elements: candidates.map(candidateOf).filter(Boolean).slice(0, 8),
|
|
299
|
+
reflex_or_exploration_tried: tried,
|
|
300
|
+
outcome,
|
|
301
|
+
model_turns_spent: modelTurns,
|
|
302
|
+
tokens_spent: null,
|
|
303
|
+
wall_time_ms: wallMs,
|
|
304
|
+
detail: detail ? String(detail).slice(0, 200) : null,
|
|
305
|
+
};
|
|
306
|
+
appendJsonl(metricPaths(udid).escalations, record);
|
|
307
|
+
return record;
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
/** Write one flow record. Shape is docs/research/03-human-parity.md §1. */
|
|
311
|
+
export function recordFlow(udid, record) {
|
|
312
|
+
appendJsonl(metricPaths(udid).flows, record);
|
|
313
|
+
return record;
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
/**
|
|
317
|
+
* Assemble a flow record from what a run already knows.
|
|
318
|
+
*
|
|
319
|
+
* `model_turns` counts the call itself as one, then adds one per escalation
|
|
320
|
+
* the agent has to answer. That is the number the whole series is trying to
|
|
321
|
+
* drive down, so it is defined here rather than left to whoever reads the log.
|
|
322
|
+
*/
|
|
323
|
+
export function flowRecordFrom({
|
|
324
|
+
flowId,
|
|
325
|
+
flowName = null,
|
|
326
|
+
udid,
|
|
327
|
+
startedAt,
|
|
328
|
+
wallMs,
|
|
329
|
+
stepsTaken,
|
|
330
|
+
totalSteps,
|
|
331
|
+
minSteps = null,
|
|
332
|
+
imagesSent = 0,
|
|
333
|
+
escalations = [],
|
|
334
|
+
verdicts = [],
|
|
335
|
+
completed,
|
|
336
|
+
}) {
|
|
337
|
+
const histogram = {};
|
|
338
|
+
for (const v of verdicts) if (v) histogram[v] = (histogram[v] ?? 0) + 1;
|
|
339
|
+
const misTaps = verdicts.filter((v) => ESCALATING_VERDICTS.has(v)).length;
|
|
340
|
+
return {
|
|
341
|
+
flow_id: flowId,
|
|
342
|
+
flow_name: flowName,
|
|
343
|
+
device: udid,
|
|
344
|
+
started_at: new Date(startedAt).toISOString(),
|
|
345
|
+
wall_time_ms: wallMs,
|
|
346
|
+
steps_taken: stepsTaken,
|
|
347
|
+
total_steps: totalSteps,
|
|
348
|
+
min_steps: minSteps,
|
|
349
|
+
step_ratio: minSteps ? Number((stepsTaken / minSteps).toFixed(3)) : null,
|
|
350
|
+
model_turns: 1 + escalations.filter((e) => e.outcome !== 'resolved_locally').length,
|
|
351
|
+
images_sent: imagesSent,
|
|
352
|
+
input_tokens: null,
|
|
353
|
+
output_tokens: null,
|
|
354
|
+
escalations: escalations.map((e) => ({ reason: e.reason, step_index: e.step_index, outcome: e.outcome })),
|
|
355
|
+
escalation_count: escalations.length,
|
|
356
|
+
mis_taps: misTaps,
|
|
357
|
+
verdict_histogram: histogram,
|
|
358
|
+
reflex_firings: [],
|
|
359
|
+
exploration_events: [],
|
|
360
|
+
completed: Boolean(completed),
|
|
361
|
+
wrong_action_taken: verdicts.includes('unexpected-screen'),
|
|
362
|
+
};
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
// ---------------------------------------------------------------- statistics
|
|
366
|
+
|
|
367
|
+
export function median(xs) {
|
|
368
|
+
const s = xs.filter((x) => Number.isFinite(x)).sort((a, b) => a - b);
|
|
369
|
+
if (!s.length) return null;
|
|
370
|
+
const mid = s.length >> 1;
|
|
371
|
+
return s.length % 2 ? s[mid] : (s[mid - 1] + s[mid]) / 2;
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
/**
|
|
375
|
+
* Quartiles by the "exclusive median" convention: each half excludes the
|
|
376
|
+
* median of an odd-length sample. Named because IQR is meaningless without it.
|
|
377
|
+
*/
|
|
378
|
+
export function quartiles(xs) {
|
|
379
|
+
const s = xs.filter((x) => Number.isFinite(x)).sort((a, b) => a - b);
|
|
380
|
+
if (!s.length) return null;
|
|
381
|
+
const mid = s.length >> 1;
|
|
382
|
+
const lower = s.slice(0, mid);
|
|
383
|
+
const upper = s.length % 2 ? s.slice(mid + 1) : s.slice(mid);
|
|
384
|
+
const p25 = median(lower) ?? s[0];
|
|
385
|
+
const p75 = median(upper) ?? s[s.length - 1];
|
|
386
|
+
return { min: s[0], p25, p50: median(s), p75, max: s[s.length - 1], iqr: p75 - p25, n: s.length };
|
|
387
|
+
}
|
|
388
|
+
|
|
389
|
+
/**
|
|
390
|
+
* The p-th percentile, by nearest-rank on the sorted sample.
|
|
391
|
+
*
|
|
392
|
+
* Nearest-rank rather than interpolation: with the 5-50 samples a graph edge
|
|
393
|
+
* carries, an interpolated p95 invents a value between two observations, and
|
|
394
|
+
* every number here is supposed to be one that actually happened.
|
|
395
|
+
*/
|
|
396
|
+
export function percentile(xs, p) {
|
|
397
|
+
const s = xs.filter((x) => Number.isFinite(x)).sort((a, b) => a - b);
|
|
398
|
+
if (!s.length) return null;
|
|
399
|
+
const rank = Math.ceil((p / 100) * s.length);
|
|
400
|
+
return s[Math.min(s.length - 1, Math.max(0, rank - 1))];
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
/** The mean that punishes one slow flow, which is why §1 asks for it. */
|
|
404
|
+
export function harmonicMean(xs) {
|
|
405
|
+
const s = xs.filter((x) => Number.isFinite(x) && x > 0);
|
|
406
|
+
if (!s.length) return null;
|
|
407
|
+
return s.length / s.reduce((acc, x) => acc + 1 / x, 0);
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
/**
|
|
411
|
+
* HPI, exactly as §1 defines it.
|
|
412
|
+
*
|
|
413
|
+
* `flows` are agent runs (flows.jsonl records); `baselines` maps flow name to
|
|
414
|
+
* a human baseline. A flow with no human baseline gets no HPI_time — reported
|
|
415
|
+
* as null, never as 1.0, because a missing denominator is not parity.
|
|
416
|
+
*/
|
|
417
|
+
export function hpi({ flows, baselines = {} }) {
|
|
418
|
+
const byName = new Map();
|
|
419
|
+
for (const f of flows) {
|
|
420
|
+
if (!f?.flow_name) continue;
|
|
421
|
+
if (!byName.has(f.flow_name)) byName.set(f.flow_name, []);
|
|
422
|
+
byName.get(f.flow_name).push(f);
|
|
423
|
+
}
|
|
424
|
+
|
|
425
|
+
const perFlow = [...byName.entries()].map(([name, runs]) => {
|
|
426
|
+
const agent = quartiles(runs.map((r) => r.wall_time_ms));
|
|
427
|
+
const human = baselines[name]?.wall_time_ms ?? null;
|
|
428
|
+
const humanMedian = human?.p50 ?? null;
|
|
429
|
+
const stepRatios = runs.map((r) => r.step_ratio).filter((x) => Number.isFinite(x));
|
|
430
|
+
return {
|
|
431
|
+
flow: name,
|
|
432
|
+
runs: runs.length,
|
|
433
|
+
agent_ms: agent,
|
|
434
|
+
human_median_ms: humanMedian,
|
|
435
|
+
hpi_time: humanMedian && agent?.p50 ? Number((humanMedian / agent.p50).toFixed(3)) : null,
|
|
436
|
+
step_ratio: median(stepRatios),
|
|
437
|
+
completed: runs.filter((r) => r.completed).length,
|
|
438
|
+
wrong_action: runs.filter((r) => r.wrong_action_taken).length,
|
|
439
|
+
escalations: runs.reduce((acc, r) => acc + (r.escalation_count ?? 0), 0),
|
|
440
|
+
model_turns: median(runs.map((r) => r.model_turns)),
|
|
441
|
+
};
|
|
442
|
+
}).sort((a, b) => a.flow.localeCompare(b.flow));
|
|
443
|
+
|
|
444
|
+
const total = flows.length;
|
|
445
|
+
const clean = flows.filter((f) => f.completed && !f.wrong_action_taken).length;
|
|
446
|
+
const accuracy = total ? Number((clean / total).toFixed(3)) : null;
|
|
447
|
+
const times = perFlow.map((f) => f.hpi_time).filter((x) => Number.isFinite(x));
|
|
448
|
+
const hpiTime = harmonicMean(times);
|
|
449
|
+
return {
|
|
450
|
+
flows: perFlow,
|
|
451
|
+
overall: {
|
|
452
|
+
runs: total,
|
|
453
|
+
flows_measured: perFlow.length,
|
|
454
|
+
flows_with_human_baseline: times.length,
|
|
455
|
+
hpi_accuracy: accuracy,
|
|
456
|
+
hpi_time: hpiTime == null ? null : Number(hpiTime.toFixed(3)),
|
|
457
|
+
hpi: hpiTime == null || accuracy == null ? null : Number((accuracy * hpiTime).toFixed(3)),
|
|
458
|
+
step_ratio: median(perFlow.map((f) => f.step_ratio)),
|
|
459
|
+
model_turns_median: median(perFlow.map((f) => f.model_turns)),
|
|
460
|
+
},
|
|
461
|
+
};
|
|
462
|
+
}
|
|
463
|
+
|
|
464
|
+
/**
|
|
465
|
+
* The escalation dashboard.
|
|
466
|
+
*
|
|
467
|
+
* `avoidable` is §8's definition — an escalation whose reason maps to a
|
|
468
|
+
* faculty that is not built yet, or is built and still let it through. One
|
|
469
|
+
* consequence is worth stating rather than hiding: with no faculty built, the
|
|
470
|
+
* rate is 1.0 by construction and says nothing. The per-reason breakdown is
|
|
471
|
+
* the part that decides the next phase, and it is informative today.
|
|
472
|
+
*/
|
|
473
|
+
export function breakdown(records, { session = null, flow = null } = {}) {
|
|
474
|
+
const byReason = {};
|
|
475
|
+
for (const r of REASONS) byReason[r] = 0;
|
|
476
|
+
const byScreen = new Map();
|
|
477
|
+
const byOutcome = {};
|
|
478
|
+
const bySession = new Map();
|
|
479
|
+
const byFlow = new Map();
|
|
480
|
+
let avoidable = 0;
|
|
481
|
+
let unattributed = 0;
|
|
482
|
+
// Filtering happens here rather than at the call site so `total` and every
|
|
483
|
+
// rate below it describe the same set of records.
|
|
484
|
+
const kept = records.filter((r) => (session ? r?.session_id === session : true))
|
|
485
|
+
.filter((r) => (flow ? r?.flow_name === flow : true));
|
|
486
|
+
for (const r of kept) {
|
|
487
|
+
if (!REASONS.includes(r?.reason)) continue;
|
|
488
|
+
// Records written before sessions were recorded cannot be attributed, and
|
|
489
|
+
// saying how many there are is the difference between a breakdown that
|
|
490
|
+
// pools two agents and one that says it might be.
|
|
491
|
+
if (r.session_id) {
|
|
492
|
+
const key = `${r.session_id}|${r.client ?? '?'}`;
|
|
493
|
+
bySession.set(key, (bySession.get(key) ?? 0) + 1);
|
|
494
|
+
} else {
|
|
495
|
+
unattributed += 1;
|
|
496
|
+
}
|
|
497
|
+
if (r.flow_name) byFlow.set(r.flow_name, (byFlow.get(r.flow_name) ?? 0) + 1);
|
|
498
|
+
byReason[r.reason] += 1;
|
|
499
|
+
byOutcome[r.outcome] = (byOutcome[r.outcome] ?? 0) + 1;
|
|
500
|
+
// Already avoided locally, so not avoidable by anything unbuilt.
|
|
501
|
+
if (r.outcome !== 'resolved_locally') avoidable += 1;
|
|
502
|
+
const key = r.screen_fingerprint ?? '(no fingerprint)';
|
|
503
|
+
byScreen.set(key, (byScreen.get(key) ?? 0) + 1);
|
|
504
|
+
}
|
|
505
|
+
const total = kept.filter((r) => REASONS.includes(r?.reason)).length;
|
|
506
|
+
const sessions = [...bySession.entries()]
|
|
507
|
+
.map(([key, count]) => {
|
|
508
|
+
const [id, client] = key.split('|');
|
|
509
|
+
return { session_id: id, client, count };
|
|
510
|
+
})
|
|
511
|
+
.sort((a, b) => b.count - a.count);
|
|
512
|
+
return {
|
|
513
|
+
total,
|
|
514
|
+
// The log is per-device and shared: two agents on one booted simulator
|
|
515
|
+
// write one interleaved file. More than one session here means the counts
|
|
516
|
+
// below are a pool, and CLAUDE.md uses those counts to choose a phase.
|
|
517
|
+
sessions,
|
|
518
|
+
session_count: sessions.length,
|
|
519
|
+
unattributed,
|
|
520
|
+
// Any unattributed record at all makes this a pool: the whole point is
|
|
521
|
+
// that they cannot be told apart, and 92 of them is not "one session".
|
|
522
|
+
pooled: sessions.length > 1 || unattributed > 0,
|
|
523
|
+
by_flow: Object.fromEntries([...byFlow.entries()].sort((a, b) => b[1] - a[1])),
|
|
524
|
+
by_reason: byReason,
|
|
525
|
+
by_outcome: byOutcome,
|
|
526
|
+
faculty: Object.fromEntries(
|
|
527
|
+
REASONS.filter((r) => byReason[r]).map((r) => [r, `${FACULTY[r]}${BUILT_FACULTIES.has(FACULTY[r]) ? ' [built]' : ''}`]),
|
|
528
|
+
),
|
|
529
|
+
avoidable,
|
|
530
|
+
avoidable_escalation_rate: total ? Number((avoidable / total).toFixed(3)) : null,
|
|
531
|
+
top_screens: [...byScreen.entries()]
|
|
532
|
+
.sort((a, b) => b[1] - a[1])
|
|
533
|
+
.slice(0, 10)
|
|
534
|
+
.map(([fingerprint, count]) => ({ fingerprint, count })),
|
|
535
|
+
// `kept`, not `records` — a filtered breakdown that reports the whole
|
|
536
|
+
// log's model turns is the same class of mistake as pooling two sessions.
|
|
537
|
+
model_turns_spent: kept.reduce((acc, r) => acc + (r.model_turns_spent ?? 0), 0),
|
|
538
|
+
};
|
|
539
|
+
}
|
|
540
|
+
|
|
541
|
+
/**
|
|
542
|
+
* How much slower than the committed baseline fails the build.
|
|
543
|
+
*
|
|
544
|
+
* 25%, not the 10% research §1 proposed, and the number is traceable to a
|
|
545
|
+
* measurement rather than to taste. Three HPI measurements of *identical code*
|
|
546
|
+
* on the same device the same afternoon gave HPI_time 0.475, 0.413 and 0.371,
|
|
547
|
+
* and per flow the medians moved up to 18% between runs — a 10% gate would
|
|
548
|
+
* have failed on noise roughly half the time, and a gate that cries wolf gets
|
|
549
|
+
* ignored, which costs more than no gate. Two things narrow the band instead
|
|
550
|
+
* of loosening it further: the gate reads the median of three passes rather
|
|
551
|
+
* than one, and accuracy stays strict at any drop at all. Numbers in
|
|
552
|
+
* docs/BENCHMARKS.md under "What the gate is set to, and why".
|
|
553
|
+
*/
|
|
554
|
+
export const TIME_REGRESSION = 0.25;
|
|
555
|
+
|
|
556
|
+
/**
|
|
557
|
+
* The HPI_time a gate should compare: the median across passes when a report
|
|
558
|
+
* has them, and the single measurement when it does not — so a baseline
|
|
559
|
+
* committed before passes existed still gates.
|
|
560
|
+
*/
|
|
561
|
+
export const gateTime = (o) => o?.hpi_time_median_of_passes ?? o?.hpi_time ?? null;
|
|
562
|
+
|
|
563
|
+
/**
|
|
564
|
+
* Compare a measurement against the committed baseline.
|
|
565
|
+
*
|
|
566
|
+
* A pure function rather than a few lines inside the CI script, because an
|
|
567
|
+
* untested gate is this project's recurring failure: the packaging check and a
|
|
568
|
+
* fingerprint eval both shipped unable to fail, and both looked exactly like
|
|
569
|
+
* this. Returns the reasons it should fail — empty means pass.
|
|
570
|
+
*/
|
|
571
|
+
export function gateAgainst(baseline, measured, { timeRegression = TIME_REGRESSION } = {}) {
|
|
572
|
+
const base = baseline?.overall ?? {};
|
|
573
|
+
const now = measured?.overall ?? measured ?? {};
|
|
574
|
+
const baseTime = gateTime(base);
|
|
575
|
+
const nowTime = gateTime(now);
|
|
576
|
+
const failures = [];
|
|
577
|
+
if (base.hpi_accuracy != null && now.hpi_accuracy != null && now.hpi_accuracy < base.hpi_accuracy) {
|
|
578
|
+
failures.push(`HPI_accuracy dropped: ${now.hpi_accuracy} < ${base.hpi_accuracy} (any drop fails)`);
|
|
579
|
+
}
|
|
580
|
+
if (baseTime != null && nowTime != null && nowTime < baseTime * (1 - timeRegression)) {
|
|
581
|
+
failures.push(
|
|
582
|
+
`HPI_time regressed >${timeRegression * 100}%: ${nowTime} < ${(baseTime * (1 - timeRegression)).toFixed(3)}`,
|
|
583
|
+
);
|
|
584
|
+
}
|
|
585
|
+
// A checkout missing the human baseline the committed number was computed
|
|
586
|
+
// against would otherwise pass by having nothing to compare.
|
|
587
|
+
if (baseTime != null && nowTime == null) {
|
|
588
|
+
failures.push('HPI_time is null but the baseline has one — the human baseline it needs is missing from this checkout');
|
|
589
|
+
}
|
|
590
|
+
return failures;
|
|
591
|
+
}
|
|
592
|
+
|
|
593
|
+
/** A flow id that sorts by time and is short enough to read in a log. */
|
|
594
|
+
export function newFlowId() {
|
|
595
|
+
return `${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 8)}`;
|
|
596
|
+
}
|
package/src/navigate.js
CHANGED
|
@@ -6,6 +6,8 @@
|
|
|
6
6
|
// somebody already walked, saved so it can be walked again.
|
|
7
7
|
import { runScript } from './actions.js';
|
|
8
8
|
import * as graph from './graph.js';
|
|
9
|
+
import * as metrics from './metrics.js';
|
|
10
|
+
import * as screenmap from './screenmap.js';
|
|
9
11
|
import * as api from './index.js';
|
|
10
12
|
import * as store from './store.js';
|
|
11
13
|
import fs from 'node:fs';
|
|
@@ -31,6 +33,34 @@ export function stepFor(edge) {
|
|
|
31
33
|
return null;
|
|
32
34
|
}
|
|
33
35
|
|
|
36
|
+
/**
|
|
37
|
+
* A refusal to act, written down.
|
|
38
|
+
*
|
|
39
|
+
* `goto` and `flow run` refuse rather than guess, and a refusal is exactly
|
|
40
|
+
* "the tool handed the decision back" — the thing the escalation log exists to
|
|
41
|
+
* count. The reason mapping lives in metrics.PLAN_REASONS so the five reasons
|
|
42
|
+
* have one owner.
|
|
43
|
+
*/
|
|
44
|
+
function refuse(udid, result, { detail = null, flowName = null } = {}) {
|
|
45
|
+
try {
|
|
46
|
+
const reason = metrics.PLAN_REASONS[result.reason];
|
|
47
|
+
if (reason) {
|
|
48
|
+
metrics.recordEscalation(udid, {
|
|
49
|
+
reason,
|
|
50
|
+
// A refusal by `goto` is about a destination and one by `flow run` is
|
|
51
|
+
// about a named flow. Either is what a breakdown wants to group by.
|
|
52
|
+
flowName,
|
|
53
|
+
fingerprint: metrics.fingerprintNow(udid, screenmap),
|
|
54
|
+
outcome: 'escalated_to_model',
|
|
55
|
+
detail: detail ?? result.reason,
|
|
56
|
+
});
|
|
57
|
+
}
|
|
58
|
+
} catch {
|
|
59
|
+
/* a log that cannot be written must not change what is returned */
|
|
60
|
+
}
|
|
61
|
+
return result;
|
|
62
|
+
}
|
|
63
|
+
|
|
34
64
|
/**
|
|
35
65
|
* Walk to a known screen.
|
|
36
66
|
*
|
|
@@ -43,8 +73,8 @@ export async function goto(deviceQuery, target, { options, ...runOptions } = {})
|
|
|
43
73
|
const udid = device.udid;
|
|
44
74
|
|
|
45
75
|
const found = graph.findScreen(udid, target);
|
|
46
|
-
if (!found) return { ok: false, reason: 'unknown-screen', known: knownScreens(udid) };
|
|
47
|
-
if (found.ambiguous) return { ok: false, reason: 'ambiguous', candidates: found.ambiguous };
|
|
76
|
+
if (!found) return refuse(udid, { ok: false, reason: 'unknown-screen', known: knownScreens(udid) }, { detail: `no screen matches "${target}"`, flowName: `goto:${target}` });
|
|
77
|
+
if (found.ambiguous) return refuse(udid, { ok: false, reason: 'ambiguous', candidates: found.ambiguous }, { detail: `"${target}" fits ${found.ambiguous.length} screens`, flowName: `goto:${target}` });
|
|
48
78
|
|
|
49
79
|
const here = await api.screenIdentity(udid, {});
|
|
50
80
|
if (here.hash === found.node.hash) {
|
|
@@ -57,12 +87,12 @@ export async function goto(deviceQuery, target, { options, ...runOptions } = {})
|
|
|
57
87
|
// `Cannot read properties of null (reading 'slice')` instead of answering.
|
|
58
88
|
// Not hypothetical on Android, where README's own table puts the launcher at
|
|
59
89
|
// one token.
|
|
60
|
-
if (!here.hash) return { ok: false, reason: 'no-identity', to: found.name };
|
|
90
|
+
if (!here.hash) return refuse(udid, { ok: false, reason: 'no-identity', to: found.name }, { flowName: `goto:${target}` });
|
|
61
91
|
const path_ = graph.route(udid, { hash: here.hash, tokens: here.tokens }, found.node.hash);
|
|
62
|
-
if (!path_) return { ok: false, reason: 'no-route', from: here.hash.slice(0, 8), to: found.name };
|
|
92
|
+
if (!path_) return refuse(udid, { ok: false, reason: 'no-route', from: here.hash.slice(0, 8), to: found.name }, { flowName: `goto:${target}` });
|
|
63
93
|
|
|
64
94
|
const steps = path_.map(stepFor);
|
|
65
|
-
if (steps.some((s) => !s)) return { ok: false, reason: 'unreplayable-edge', to: found.name };
|
|
95
|
+
if (steps.some((s) => !s)) return refuse(udid, { ok: false, reason: 'unreplayable-edge', to: found.name }, { flowName: `goto:${target}` });
|
|
66
96
|
|
|
67
97
|
const result = await runScript(udid, { steps, stopOnUnexpected: true, ...runOptions });
|
|
68
98
|
const arrived = await api.screenIdentity(udid, {});
|
|
@@ -121,7 +151,17 @@ export function listFlows(udid) {
|
|
|
121
151
|
export async function runFlow(deviceQuery, name, { options, ...runOptions } = {}) {
|
|
122
152
|
const { device } = await api.ensureDaemon(deviceQuery, options);
|
|
123
153
|
const flow = loadFlow(device.udid, name);
|
|
124
|
-
if (!flow) return { ok: false, reason: 'unknown-flow', known: listFlows(device.udid).map((f) => f.name) };
|
|
125
|
-
|
|
154
|
+
if (!flow) return refuse(device.udid, { ok: false, reason: 'unknown-flow', known: listFlows(device.udid).map((f) => f.name) }, { detail: `no saved flow "${name}"`, flowName: name });
|
|
155
|
+
// A replayed flow knows its own name, so its record can be compared against
|
|
156
|
+
// a human doing the same thing. `minSteps` comes from the flow definition or
|
|
157
|
+
// stays null — the step count of a recorded route is not a claim about the
|
|
158
|
+
// shortest one.
|
|
159
|
+
const result = await runScript(device.udid, {
|
|
160
|
+
steps: flow.steps,
|
|
161
|
+
stopOnUnexpected: true,
|
|
162
|
+
flowName: name,
|
|
163
|
+
minSteps: flow.minSteps ?? null,
|
|
164
|
+
...runOptions,
|
|
165
|
+
});
|
|
126
166
|
return { ok: result.ranSteps === flow.steps.length, name, ...result };
|
|
127
167
|
}
|