simframe 0.12.2 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +138 -9
- package/native/simframed/Sources/SimframeCore/Motion.swift +33 -2
- package/native/supervise.swift +63 -4
- package/package.json +1 -1
- package/scripts/article-md.mjs +185 -0
- package/scripts/ci-device-guard.mjs +82 -0
- package/scripts/ci-integration-local.sh +33 -9
- package/scripts/ci-memory.mjs +117 -15
- package/scripts/eval-fingerprint.mjs +145 -3
- package/scripts/replay-rulings.mjs +60 -1
- package/src/actions.js +213 -13
- package/src/analyze.js +56 -0
- package/src/cli.js +205 -15
- package/src/fingerprint.js +10 -1
- package/src/graph.js +15 -2
- package/src/index.js +302 -16
- package/src/input.js +4 -0
- package/src/matching.js +17 -1
- package/src/mcp.js +148 -13
- package/src/metrics.js +49 -6
- package/src/navigate.js +4 -1
- package/src/ollama.js +37 -9
- package/src/platform/android.js +34 -0
- package/src/platform/index.js +7 -0
- package/src/platform/ios.js +128 -5
- package/src/platform/plist.js +156 -0
- package/src/refs.js +12 -1
- package/src/regions.js +54 -0
- package/src/screenmap.js +85 -6
- package/src/storage.js +201 -0
- package/src/store.js +53 -0
- package/src/supervisor.js +20 -2
- package/src/view.js +84 -9
package/src/mcp.js
CHANGED
|
@@ -15,7 +15,8 @@ import * as api from './index.js';
|
|
|
15
15
|
import * as input from './input.js';
|
|
16
16
|
import * as metrics from './metrics.js';
|
|
17
17
|
import * as navigate from './navigate.js';
|
|
18
|
-
import { bootedDevices, permissionServices } from './platform/index.js';
|
|
18
|
+
import { bootedDevices, listDevices, permissionServices, resolveDevice } from './platform/index.js';
|
|
19
|
+
import * as storage from './storage.js';
|
|
19
20
|
import * as store from './store.js';
|
|
20
21
|
import * as view from './view.js';
|
|
21
22
|
|
|
@@ -132,7 +133,7 @@ const TOOLS = [
|
|
|
132
133
|
steps: {
|
|
133
134
|
type: 'array',
|
|
134
135
|
description:
|
|
135
|
-
'Ordered steps. Every selector below accepts "Save" | "#3" | "@120,400", in that order of preference. Act: {"tap":"Save"} (add "index" if a label is ambiguous), {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"clear":"Notes"} to empty a field and "clear":true on a type/paste to replace rather than append (drop "into" to type into whatever already has focus, which is how you follow a browser next-field chevron — nothing can be read back then, and the step says so), {"scroll":"down"}, {"scrollTo":"Delete account"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"key":"return"} (the keyboard return/enter key, which is how a mobile search field submits — also escape, tab, space, backspace, and the arrows), {"launch":{"value":"com.example.app","relaunch":true,"args":["-uiTest","1"]}}, {"openUrl":"myapp://x"}, {"permission":{"value":"photos","grant":"grant","bundleId":"com.example.app"}}. Check: {"assert":{"value":"Saved","is":"visible"}} (also gone | enabled | disabled | value with "equals"), {"waitFor":{"value":"Saved","timeoutMs":5000}}, {"settle":{"stableMs":600}}, {"pause":300}. Recover without a round trip: add "or" to any step for fallback selectors tried locally — {"tap":"Save","or":["Done","Confirm"]} — and {"seek":"change username","budget":6} explores for something not on this screen: it OPENS containers (a real action — state changes), checks, and returns to where it started, refusing to open anything that commits, abandons or answers. It does not tap the target; it leaves you on the screen where the target resolves so you tap it next. Do not point it into a flow whose progress you cannot afford to lose. A long screen is only knowable a viewport at a time, so {"sweep":"all","fill":{"Last Name":"Asadi","Email":"a@b.c"}} goes to the top, then reads and fills section by section to the bottom — filling each field while it is on screen, which beats finding one and scrolling back. Add "from":"here" to sweep down from where you are. It reports which section each element was in, what it filled, and what it never found at any scroll position. Prefer it to scrollTo on forms and long lists. Brief the supervisor from the plan: top-level "supervise" is standing guidance for the whole batch ("lists here render a count header before rows; REVIEW stays disabled until a provider is chosen") and per-step "expect" adds to it. When it stops a run the result names the steps it did not attempt — re-issue them with a corrected "supervise" note if the judgement was wrong.',
|
|
136
|
+
'Ordered steps. Every selector below accepts "Save" | "#3" | "@120,400", in that order of preference. Act: {"tap":"Save"} (add "index" if a label is ambiguous), {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"clear":"Notes"} to empty a field and "clear":true on a type/paste to replace rather than append (drop "into" to type into whatever already has focus, which is how you follow a browser next-field chevron — nothing can be read back then, and the step says so), {"scroll":"down"}, {"scrollTo":"Delete account"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"key":"return"} (the keyboard return/enter key, which is how a mobile search field submits — also escape, tab, space, backspace, and the arrows), {"launch":{"value":"com.example.app","relaunch":true,"args":["-uiTest","1"]}}, {"openUrl":"myapp://x"}, {"permission":{"value":"photos","grant":"grant","bundleId":"com.example.app"}}. Check: {"assert":{"value":"Saved","is":"visible"}} (also gone | enabled | disabled | value with "equals"), {"waitFor":{"value":"Saved","timeoutMs":5000}} (add "failIfStillFor":15000 to stop early once the screen has plainly stopped changing — a 180s wait once burned three minutes on an app that had logged itself out; without it a timeout still reports how long the screen had been still), {"settle":{"stableMs":600}}, {"pause":300}. Recover without a round trip: add "or" to any step for fallback selectors tried locally — {"tap":"Save","or":["Done","Confirm"]} — and {"seek":"change username","budget":6} explores for something not on this screen: it OPENS containers (a real action — state changes), checks, and returns to where it started, refusing to open anything that commits, abandons or answers. It does not tap the target; it leaves you on the screen where the target resolves so you tap it next. Do not point it into a flow whose progress you cannot afford to lose. A long screen is only knowable a viewport at a time, so {"sweep":"all","fill":{"Last Name":"Asadi","Email":"a@b.c"}} goes to the top, then reads and fills section by section to the bottom — filling each field while it is on screen, which beats finding one and scrolling back. Add "from":"here" to sweep down from where you are. It reports which section each element was in, what it filled, and what it never found at any scroll position. Prefer it to scrollTo on forms and long lists. Brief the supervisor from the plan: top-level "supervise" is standing guidance for the whole batch ("lists here render a count header before rows; REVIEW stays disabled until a provider is chosen") and per-step "expect" adds to it. When it stops a run the result names the steps it did not attempt — re-issue them with a corrected "supervise" note if the judgement was wrong.',
|
|
136
137
|
items: { type: 'object' },
|
|
137
138
|
},
|
|
138
139
|
autoSettle: {
|
|
@@ -192,7 +193,19 @@ const TOOLS = [
|
|
|
192
193
|
description: 'Block until something appears on screen, then return the screen map. Use this instead of pausing and re-reading.',
|
|
193
194
|
inputSchema: {
|
|
194
195
|
type: 'object',
|
|
195
|
-
properties: {
|
|
196
|
+
properties: {
|
|
197
|
+
...deviceProp,
|
|
198
|
+
...modeProps,
|
|
199
|
+
...selectorProp('What to wait for'),
|
|
200
|
+
timeoutMs: { type: 'number', description: 'Default 8000.' },
|
|
201
|
+
failIfStillFor: {
|
|
202
|
+
type: 'number',
|
|
203
|
+
description: 'Give up early once the screen has not moved for this long and the target is still absent.'
|
|
204
|
+
+ ' Off by default, because a still screen is also what a pending network call looks like —'
|
|
205
|
+
+ ' use it when the thing you await would arrive with a visible change or not at all.'
|
|
206
|
+
+ ' A timeout reports the stillness either way.',
|
|
207
|
+
},
|
|
208
|
+
},
|
|
196
209
|
required: ['sel'],
|
|
197
210
|
},
|
|
198
211
|
},
|
|
@@ -386,10 +399,42 @@ const TOOLS = [
|
|
|
386
399
|
required: ['action'],
|
|
387
400
|
},
|
|
388
401
|
},
|
|
402
|
+
{
|
|
403
|
+
name: 'sim_storage',
|
|
404
|
+
description:
|
|
405
|
+
'What the app saved, as text: its UserDefaults and (for React Native) its AsyncStorage, read straight out of'
|
|
406
|
+
+ ' the data container. sim_ui says what is drawn; sim_storage says what the app believes — use it when the'
|
|
407
|
+
+ ' screen and the behaviour disagree, or to check a value without driving the UI to it.'
|
|
408
|
+
+ ' Works on a device that is NOT running, so it can answer before anything is booted.'
|
|
409
|
+
+ ' Call with no bundleId to list the apps that have a container (match filters that list).',
|
|
410
|
+
inputSchema: {
|
|
411
|
+
type: 'object',
|
|
412
|
+
properties: {
|
|
413
|
+
bundleId: { type: 'string', description: 'The app to read, e.g. com.example.myapp. Omit to list apps instead.' },
|
|
414
|
+
match: { type: 'string', description: 'When listing, show only bundle ids containing this string.' },
|
|
415
|
+
...deviceProp,
|
|
416
|
+
...modeProps,
|
|
417
|
+
},
|
|
418
|
+
},
|
|
419
|
+
},
|
|
389
420
|
{
|
|
390
421
|
name: 'sim_devices',
|
|
391
|
-
description: 'List the
|
|
392
|
-
|
|
422
|
+
description: 'List the devices simframe can drive — iOS simulators and Android emulators.'
|
|
423
|
+
+ ' Booted ones by default; pass all to see every device on the host and its state.'
|
|
424
|
+
+ ' Also reports which simframe build is answering.',
|
|
425
|
+
inputSchema: {
|
|
426
|
+
type: 'object',
|
|
427
|
+
properties: {
|
|
428
|
+
all: {
|
|
429
|
+
type: 'boolean',
|
|
430
|
+
description: 'Also account for devices that are shut down, grouped by runtime.',
|
|
431
|
+
},
|
|
432
|
+
match: {
|
|
433
|
+
type: 'string',
|
|
434
|
+
description: 'With all, list shut-down devices whose name or runtime contains this, in full.',
|
|
435
|
+
},
|
|
436
|
+
},
|
|
437
|
+
},
|
|
393
438
|
},
|
|
394
439
|
];
|
|
395
440
|
|
|
@@ -459,7 +504,13 @@ function header(device, state, ageMs, extra = '') {
|
|
|
459
504
|
}
|
|
460
505
|
|
|
461
506
|
function livenessLine(live) {
|
|
462
|
-
|
|
507
|
+
// `ok` is not the only thing worth saying. A surface that has stopped
|
|
508
|
+
// updating leaves every hard signal green — the loop is alive, no read
|
|
509
|
+
// failed, the frame is milliseconds old — and reporting nothing is how
|
|
510
|
+
// `sim_look` served a three-minute-old login screen while announcing it as
|
|
511
|
+
// 66ms old. A note that exists must reach the caller.
|
|
512
|
+
if (live?.note) return `WARNING: ${live.note}`;
|
|
513
|
+
return null;
|
|
463
514
|
}
|
|
464
515
|
|
|
465
516
|
function sinceLine(since) {
|
|
@@ -554,7 +605,7 @@ export async function serve({ device: defaultDevice, options: baseOptions = {} }
|
|
|
554
605
|
options,
|
|
555
606
|
);
|
|
556
607
|
case 'sim_wait_for':
|
|
557
|
-
return await oneStep(target, { waitFor: args.sel, timeoutMs: args.timeoutMs }, args, options);
|
|
608
|
+
return await oneStep(target, { waitFor: args.sel, timeoutMs: args.timeoutMs, failIfStillFor: args.failIfStillFor }, args, options);
|
|
558
609
|
case 'sim_assert':
|
|
559
610
|
return await oneStep(target, { assert: args.sel, is: args.is, equals: args.equals }, args, options);
|
|
560
611
|
case 'sim_launch':
|
|
@@ -579,8 +630,10 @@ export async function serve({ device: defaultDevice, options: baseOptions = {} }
|
|
|
579
630
|
return await flowRun(target, args, options);
|
|
580
631
|
case 'sim_capture':
|
|
581
632
|
return await capture(target, args, options);
|
|
633
|
+
case 'sim_storage':
|
|
634
|
+
return await appStorage(args);
|
|
582
635
|
case 'sim_devices':
|
|
583
|
-
return await devices();
|
|
636
|
+
return await devices(args);
|
|
584
637
|
default:
|
|
585
638
|
throw new Error(`unknown tool ${req.params.name}`);
|
|
586
639
|
}
|
|
@@ -929,7 +982,7 @@ function verdictLineFor(results) {
|
|
|
929
982
|
|
|
930
983
|
function stepLines(res) {
|
|
931
984
|
const lines = [
|
|
932
|
-
|
|
985
|
+
actions.flowSummary(res),
|
|
933
986
|
];
|
|
934
987
|
for (const r of res.results) {
|
|
935
988
|
const settle = r.settled
|
|
@@ -1136,12 +1189,94 @@ function listStateDirs() {
|
|
|
1136
1189
|
}
|
|
1137
1190
|
}
|
|
1138
1191
|
|
|
1139
|
-
|
|
1192
|
+
/**
|
|
1193
|
+
* What an app has persisted.
|
|
1194
|
+
*
|
|
1195
|
+
* Deliberately not a daemon call and deliberately not a `simctl` call. Measured
|
|
1196
|
+
* on this Xcode, `simctl get_app_container` and `simctl listapps` both refuse on
|
|
1197
|
+
* a device that is not running — so the one property that made the field
|
|
1198
|
+
* reporter rate this the highest-leverage thing in their session, answering
|
|
1199
|
+
* *before the device is booted*, is only reachable by reading the container off
|
|
1200
|
+
* the host filesystem. That is what the backend does.
|
|
1201
|
+
*/
|
|
1202
|
+
async function appStorage({ bundleId, match: query, device } = {}) {
|
|
1203
|
+
const resolved = await resolveDevice(device);
|
|
1204
|
+
if (!bundleId) {
|
|
1205
|
+
const list = await storage.apps(resolved.udid);
|
|
1206
|
+
const needle = query ? String(query).toLowerCase() : null;
|
|
1207
|
+
const shown = needle ? list.filter((a) => a.bundleId.toLowerCase().includes(needle)) : list;
|
|
1208
|
+
return { content: [text(storage.formatApps(shown))] };
|
|
1209
|
+
}
|
|
1210
|
+
const result = await storage.read(resolved.udid, bundleId);
|
|
1211
|
+
return { content: [text(storage.format(result))] };
|
|
1212
|
+
}
|
|
1213
|
+
|
|
1214
|
+
async function devices({ all = false, match: query } = {}) {
|
|
1140
1215
|
const booted = await bootedDevices();
|
|
1141
|
-
if (!booted.length) return { content: [text('no booted devices')] };
|
|
1142
1216
|
// Noticing a name collision here is what lets every later header disambiguate
|
|
1143
1217
|
// itself, and it costs nothing: this listing is already being made.
|
|
1144
1218
|
const clash = noteBooted(booted);
|
|
1145
|
-
|
|
1146
|
-
|
|
1219
|
+
// The build that is answering, on the one call every session starts with.
|
|
1220
|
+
//
|
|
1221
|
+
// Reported from the field: a session told to test 0.13.0 could not find out
|
|
1222
|
+
// what it was running. The globally installed CLI said 0.12.2 while the MCP
|
|
1223
|
+
// server ran from a checkout, and answering "am I on the build under test?"
|
|
1224
|
+
// took three shell calls and a read of `~/.claude.json`. A tool being field
|
|
1225
|
+
// tested should be able to state its own build, and this is the cheapest
|
|
1226
|
+
// place to put it.
|
|
1227
|
+
const head = `simframe ${packageVersion()}`;
|
|
1228
|
+
if (!all) {
|
|
1229
|
+
if (!booted.length) {
|
|
1230
|
+
// Never a bare "no devices". The host almost always has some, they are
|
|
1231
|
+
// just off, and the reporter who hit this fell out of the tool entirely
|
|
1232
|
+
// and went to `xcrun simctl` — for a tool whose whole job is driving
|
|
1233
|
+
// simulators, that is the conspicuous hole.
|
|
1234
|
+
const every = await listDevices().catch(() => []);
|
|
1235
|
+
return { content: [text(`${head}\n\nno booted devices`
|
|
1236
|
+
+ (every.length ? ` — the host has ${every.length}, all shut down. Pass all:true to see them.` : ''))] };
|
|
1237
|
+
}
|
|
1238
|
+
const list = booted.map((d) => `● ${d.name} · ${d.runtime} · ${d.udid}`).join('\n');
|
|
1239
|
+
return { content: [text([head, clash, list].filter(Boolean).join('\n\n'))] };
|
|
1240
|
+
}
|
|
1241
|
+
const every = await listDevices();
|
|
1242
|
+
if (!every.length) return { content: [text(`${head}\n\nno devices on this host`)] };
|
|
1243
|
+
|
|
1244
|
+
// Summarised, not dumped.
|
|
1245
|
+
//
|
|
1246
|
+
// The first version of this listed every device in full and produced **126
|
|
1247
|
+
// rows** on this host — two thousand tokens to answer "what else is here",
|
|
1248
|
+
// from a tool whose entire argument is that text beats a screenshot because
|
|
1249
|
+
// it is cheaper. A listing that costs more than the screenshot it replaces has
|
|
1250
|
+
// lost the plot.
|
|
1251
|
+
//
|
|
1252
|
+
// So: booted devices in full, because those are the ones a caller can act on,
|
|
1253
|
+
// and the rest grouped by runtime with counts. `match` lists in full, because
|
|
1254
|
+
// a caller who names what they are looking for has already narrowed it.
|
|
1255
|
+
const wanted = String(query ?? '').trim().toLowerCase();
|
|
1256
|
+
const off = every.filter((d) => d.state !== 'Booted');
|
|
1257
|
+
const lines = [head, clash].filter(Boolean);
|
|
1258
|
+
lines.push(booted.length
|
|
1259
|
+
? booted.map((d) => `● ${d.name} · ${d.runtime} · ${d.udid}`).join('\n')
|
|
1260
|
+
: 'no booted devices');
|
|
1261
|
+
|
|
1262
|
+
const hits = wanted
|
|
1263
|
+
? off.filter((d) => `${d.name} ${d.runtime}`.toLowerCase().includes(wanted))
|
|
1264
|
+
: [];
|
|
1265
|
+
if (wanted) {
|
|
1266
|
+
lines.push(hits.length
|
|
1267
|
+
? `shut down, matching "${query}":\n`
|
|
1268
|
+
+ hits.map((d) => `○ ${d.name} · ${d.runtime} · ${d.udid}`).join('\n')
|
|
1269
|
+
: `no shut-down device matches "${query}" (${off.length} are shut down)`);
|
|
1270
|
+
} else if (off.length) {
|
|
1271
|
+
const byRuntime = new Map();
|
|
1272
|
+
for (const d of off) byRuntime.set(d.runtime, (byRuntime.get(d.runtime) ?? 0) + 1);
|
|
1273
|
+
const summary = [...byRuntime.entries()]
|
|
1274
|
+
.sort((a, b) => b[1] - a[1])
|
|
1275
|
+
.map(([runtime, n]) => ` ${runtime} — ${n}`)
|
|
1276
|
+
.join('\n');
|
|
1277
|
+
lines.push(`${off.length} device(s) shut down, by runtime:\n${summary}\n`
|
|
1278
|
+
+ 'pass match to list the ones you mean, e.g. match:"iPhone 17 Pro".');
|
|
1279
|
+
}
|
|
1280
|
+
lines.push('simframe cannot drive a device until it is booted.');
|
|
1281
|
+
return { content: [text(lines.join('\n\n'))] };
|
|
1147
1282
|
}
|
package/src/metrics.js
CHANGED
|
@@ -300,11 +300,30 @@ const VERIFYING_STEPS = new Set([
|
|
|
300
300
|
*/
|
|
301
301
|
export function reasonForStepError(step, err) {
|
|
302
302
|
const tagged = escalationOf(err);
|
|
303
|
-
if (tagged) return tagged;
|
|
303
|
+
if (tagged) return { ...tagged, classified: true };
|
|
304
304
|
const action = step?.action;
|
|
305
|
-
if (action === 'confirm' || action === 'chooseAny')
|
|
306
|
-
|
|
307
|
-
|
|
305
|
+
if (action === 'confirm' || action === 'chooseAny') {
|
|
306
|
+
return { reason: 'novel_dialog', candidates: [], tried: [], classified: true };
|
|
307
|
+
}
|
|
308
|
+
// Everything else is a *fallback*, and it now says so.
|
|
309
|
+
//
|
|
310
|
+
// This had two branches that returned the same value, which made it look
|
|
311
|
+
// like it discriminated. It does not: any step that threw without a site
|
|
312
|
+
// tagging it lands here. In a real field session that was **90% of all
|
|
313
|
+
// escalations** — and `FACULTY` then reported every one of them as evidence
|
|
314
|
+
// against "sense of time (Phase 11)", a claim nothing in the record supports.
|
|
315
|
+
//
|
|
316
|
+
// The tester's own first call failed with `unknown step "wait_for"` — a typo
|
|
317
|
+
// — and that too would be filed as evidence about which faculty to build
|
|
318
|
+
// next. CLAUDE.md calls this log the steering wheel; a steering wheel that
|
|
319
|
+
// pools typos, inert controls and slow lists into one reason is pointing
|
|
320
|
+
// somewhere nobody chose.
|
|
321
|
+
//
|
|
322
|
+
// No sixth reason: "unknown is not a reason" stays, and a vocabulary that
|
|
323
|
+
// admits "other" collects a pile of "other". What changes is that the record
|
|
324
|
+
// carries whether the reason was *read off the failure* or *assumed*, and
|
|
325
|
+
// the report declines to recommend a faculty for the assumed ones.
|
|
326
|
+
return { reason: 'verification_failed', candidates: [], tried: [], classified: false };
|
|
308
327
|
}
|
|
309
328
|
|
|
310
329
|
/**
|
|
@@ -440,6 +459,10 @@ export function recordEscalation(udid, {
|
|
|
440
459
|
modelTurns = 1,
|
|
441
460
|
wallMs = null,
|
|
442
461
|
detail = null,
|
|
462
|
+
// Was this reason read off the failure, or assumed because nothing said?
|
|
463
|
+
// Default `false`, so a caller that does not think about it cannot
|
|
464
|
+
// accidentally claim precision it does not have.
|
|
465
|
+
classified = false,
|
|
443
466
|
} = {}) {
|
|
444
467
|
if (!REASONS.includes(reason)) throw new Error(`not an escalation reason: ${reason}`);
|
|
445
468
|
if (!OUTCOMES.includes(outcome)) throw new Error(`not an escalation outcome: ${outcome}`);
|
|
@@ -455,6 +478,7 @@ export function recordEscalation(udid, {
|
|
|
455
478
|
// What was asked for, in the caller's words. Ground truth for Phase 17's
|
|
456
479
|
// go/no-go, and on its own it answers "what kind of decision is costing us".
|
|
457
480
|
intent: intent ? String(intent).slice(0, 120) : null,
|
|
481
|
+
classified: Boolean(classified),
|
|
458
482
|
step_index: stepIndex,
|
|
459
483
|
screen_fingerprint: fingerprint,
|
|
460
484
|
reason,
|
|
@@ -635,7 +659,9 @@ export function hpi({ flows, baselines = {} }) {
|
|
|
635
659
|
*/
|
|
636
660
|
export function breakdown(records, { session = null, flow = null } = {}) {
|
|
637
661
|
const byReason = {};
|
|
638
|
-
|
|
662
|
+
const classifiedByReason = {};
|
|
663
|
+
const assumedByReason = {};
|
|
664
|
+
for (const r of REASONS) { byReason[r] = 0; classifiedByReason[r] = 0; assumedByReason[r] = 0; }
|
|
639
665
|
const byScreen = new Map();
|
|
640
666
|
const byOutcome = {};
|
|
641
667
|
const bySession = new Map();
|
|
@@ -659,6 +685,12 @@ export function breakdown(records, { session = null, flow = null } = {}) {
|
|
|
659
685
|
}
|
|
660
686
|
if (r.flow_name) byFlow.set(r.flow_name, (byFlow.get(r.flow_name) ?? 0) + 1);
|
|
661
687
|
byReason[r.reason] += 1;
|
|
688
|
+
// Three states, not two. A record written before this field existed makes
|
|
689
|
+
// no claim either way, and folding it in with "assumed" would make an old
|
|
690
|
+
// log look like a diagnosis failure — a warning that cries wolf is how a
|
|
691
|
+
// real one gets ignored, which this file already knows in another place.
|
|
692
|
+
if (r.classified === true) classifiedByReason[r.reason] += 1;
|
|
693
|
+
else if (r.classified === false) assumedByReason[r.reason] += 1;
|
|
662
694
|
byOutcome[r.outcome] = (byOutcome[r.outcome] ?? 0) + 1;
|
|
663
695
|
// Already avoided locally, so not avoidable by anything unbuilt.
|
|
664
696
|
if (r.outcome !== 'resolved_locally') avoidable += 1;
|
|
@@ -685,9 +717,20 @@ export function breakdown(records, { session = null, flow = null } = {}) {
|
|
|
685
717
|
pooled: sessions.length > 1 || unattributed > 0,
|
|
686
718
|
by_flow: Object.fromEntries([...byFlow.entries()].sort((a, b) => b[1] - a[1])),
|
|
687
719
|
by_reason: byReason,
|
|
720
|
+
// How many of each reason were *read off the failure* rather than assumed.
|
|
721
|
+
//
|
|
722
|
+
// The breakdown above picks the next phase, so its precision has to be
|
|
723
|
+
// visible in it. A reason that is mostly assumed is not a finding about an
|
|
724
|
+
// app; it is a count of things nothing could classify, and reading it as a
|
|
725
|
+
// verdict on a faculty is how the instrument came to disagree with a
|
|
726
|
+
// tester who was right.
|
|
727
|
+
classified_by_reason: classifiedByReason,
|
|
728
|
+
assumed_by_reason: assumedByReason,
|
|
688
729
|
by_outcome: byOutcome,
|
|
730
|
+
// Only where the reason was actually read. A faculty named against a pile
|
|
731
|
+
// of assumptions is advice with nothing behind it.
|
|
689
732
|
faculty: Object.fromEntries(
|
|
690
|
-
REASONS.filter((r) =>
|
|
733
|
+
REASONS.filter((r) => classifiedByReason[r]).map((r) => [r, `${FACULTY[r]}${BUILT_FACULTIES.has(FACULTY[r]) ? ' [built]' : ''}`]),
|
|
691
734
|
),
|
|
692
735
|
avoidable,
|
|
693
736
|
avoidable_escalation_rate: total ? Number((avoidable / total).toFixed(3)) : null,
|
package/src/navigate.js
CHANGED
|
@@ -133,7 +133,10 @@ export async function goto(deviceQuery, target, { options, ...runOptions } = {})
|
|
|
133
133
|
}
|
|
134
134
|
|
|
135
135
|
export function knownScreens(udid) {
|
|
136
|
-
|
|
136
|
+
// A listing needs a handle for every row, so an unnamed screen falls back to
|
|
137
|
+
// its own short hash here — where it is plainly the hash column's value and
|
|
138
|
+
// not a title in quotes.
|
|
139
|
+
return graph.allNodes(udid).map((n) => ({ name: graph.describe(n) ?? n.hash.slice(0, 8), hash: n.hash.slice(0, 8), edges: n.edges.length }));
|
|
137
140
|
}
|
|
138
141
|
|
|
139
142
|
/**
|
package/src/ollama.js
CHANGED
|
@@ -49,13 +49,23 @@ export const DEFAULT_HOST = 'http://127.0.0.1:11434';
|
|
|
49
49
|
* would produce exactly the invisible unfairness this function exists to
|
|
50
50
|
* prevent.
|
|
51
51
|
*/
|
|
52
|
-
export function readBrief(file = SWIFT) {
|
|
52
|
+
export function readBrief(file = SWIFT, { mayAbstain = false } = {}) {
|
|
53
53
|
const src = fs.readFileSync(file, 'utf8');
|
|
54
|
-
const
|
|
55
|
-
if (
|
|
54
|
+
const base = block(src, 'let instructions = """', file);
|
|
55
|
+
if (!mayAbstain) return base;
|
|
56
|
+
// The addendum, from the same file and joined the way the Swift joins it.
|
|
57
|
+
// The fourth word has to reach both arms identically or the comparison is
|
|
58
|
+
// measuring two different briefs, which is the failure this whole function
|
|
59
|
+
// exists to prevent.
|
|
60
|
+
return `${base}\n\n${block(src, 'let abstainInstructions = instructions + "\\n\\n" + """', file)}`;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
function block(src, opener, file) {
|
|
64
|
+
const open = src.indexOf(opener);
|
|
65
|
+
if (open === -1) throw new Error(`no ${JSON.stringify(opener)} block in ${file}`);
|
|
56
66
|
const bodyStart = src.indexOf('\n', open) + 1;
|
|
57
67
|
const close = src.indexOf('"""', bodyStart);
|
|
58
|
-
if (close === -1) throw new Error(`unterminated
|
|
68
|
+
if (close === -1) throw new Error(`unterminated block in ${file}`);
|
|
59
69
|
return src
|
|
60
70
|
.slice(bodyStart, close)
|
|
61
71
|
.split('\n')
|
|
@@ -106,11 +116,28 @@ export function promptFor(s = {}) {
|
|
|
106
116
|
* against would be a second difference between the arms, and the rationale is
|
|
107
117
|
* the one thing a supervisor is explicitly not trusted for.
|
|
108
118
|
*/
|
|
109
|
-
export const
|
|
119
|
+
export const DECISION_WORDS = ['wait', 'retry', 'stop'];
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* The fourth word, available on request.
|
|
123
|
+
*
|
|
124
|
+
* Not a new capability and it cannot become one: `abstain` means "behave as if
|
|
125
|
+
* there is no supervisor", which is the `null` every caller already handles on
|
|
126
|
+
* every failure path. It strictly *shrinks* what the model can cause to happen,
|
|
127
|
+
* which is why a component whose answer space is its safety property can afford
|
|
128
|
+
* to grow one.
|
|
129
|
+
*/
|
|
130
|
+
export const ABSTAIN = 'abstain';
|
|
131
|
+
|
|
132
|
+
export const schemaFor = ({ mayAbstain = false } = {}) => ({
|
|
110
133
|
type: 'object',
|
|
111
|
-
properties: {
|
|
134
|
+
properties: {
|
|
135
|
+
decision: { type: 'string', enum: mayAbstain ? [...DECISION_WORDS, ABSTAIN] : DECISION_WORDS },
|
|
136
|
+
},
|
|
112
137
|
required: ['decision'],
|
|
113
|
-
};
|
|
138
|
+
});
|
|
139
|
+
|
|
140
|
+
export const SCHEMA = schemaFor();
|
|
114
141
|
|
|
115
142
|
/** Parse `ollama`, `ollama:qwen3:14b`, `ollama:qwen3:8b@http://host:port`. */
|
|
116
143
|
export function parseTarget(raw) {
|
|
@@ -147,15 +174,16 @@ async function post(host, route, body, timeoutMs) {
|
|
|
147
174
|
*/
|
|
148
175
|
export async function ask(target, situation, timeoutMs = 2500) {
|
|
149
176
|
const { model, host } = target;
|
|
177
|
+
const mayAbstain = Boolean(situation?.mayAbstain);
|
|
150
178
|
const started = Date.now();
|
|
151
179
|
const out = await post(host, '/api/chat', {
|
|
152
180
|
model,
|
|
153
181
|
messages: [
|
|
154
|
-
{ role: 'system', content: readBrief() },
|
|
182
|
+
{ role: 'system', content: readBrief(SWIFT, { mayAbstain }) },
|
|
155
183
|
{ role: 'user', content: promptFor(situation) },
|
|
156
184
|
],
|
|
157
185
|
stream: false,
|
|
158
|
-
format:
|
|
186
|
+
format: schemaFor({ mayAbstain }),
|
|
159
187
|
// Qwen3 reasons out loud by default. Turned off for two reasons and both
|
|
160
188
|
// are about fairness rather than speed: the Apple arm does not deliberate
|
|
161
189
|
// either, and a supervisor that takes twenty seconds to answer has already
|
package/src/platform/android.js
CHANGED
|
@@ -1018,6 +1018,37 @@ async function restartDevice(serial) {
|
|
|
1018
1018
|
);
|
|
1019
1019
|
}
|
|
1020
1020
|
|
|
1021
|
+
/**
|
|
1022
|
+
* Reading an app's own storage is not implemented for Android, and says so.
|
|
1023
|
+
*
|
|
1024
|
+
* Not a stub and not a borrowed answer. The iOS version reads a CoreSimulator
|
|
1025
|
+
* data container straight off the host filesystem, which an emulator has no
|
|
1026
|
+
* equivalent of: an app's files live inside the emulator's own userdata image,
|
|
1027
|
+
* and the way in is `adb shell run-as <package>` — which works only for a
|
|
1028
|
+
* debuggable build, needs the emulator running, and would be a different
|
|
1029
|
+
* feature with different guarantees rather than the same one.
|
|
1030
|
+
*
|
|
1031
|
+
* The standing rule is that a layer a platform does not have is declined with a
|
|
1032
|
+
* reason, never described in the other platform's vocabulary. Claiming a data
|
|
1033
|
+
* container here is how `doctor` once told an emulator its input driver was
|
|
1034
|
+
* idb.
|
|
1035
|
+
*/
|
|
1036
|
+
const noStorage = (serial, what) => {
|
|
1037
|
+
throw new Error(
|
|
1038
|
+
`simframe cannot read ${what} on an emulator (${serial}) yet. The iOS version reads a`
|
|
1039
|
+
+ ' simulator data container off the host filesystem and an emulator has no such thing —'
|
|
1040
|
+
+ ' its app data lives inside the userdata image, reachable only through'
|
|
1041
|
+
+ ' `adb shell run-as <package>` on a debuggable build, with the emulator running.'
|
|
1042
|
+
+ ' That is a different feature and it has not been built.',
|
|
1043
|
+
);
|
|
1044
|
+
};
|
|
1045
|
+
|
|
1046
|
+
async function listApps(serial) { return noStorage(serial, 'the list of installed apps'); }
|
|
1047
|
+
async function appContainer(serial) { return noStorage(serial, "an app's data container"); }
|
|
1048
|
+
async function readPropertyList() {
|
|
1049
|
+
throw new Error('property lists are an iOS format; Android has no equivalent to read');
|
|
1050
|
+
}
|
|
1051
|
+
|
|
1021
1052
|
/** @type {import('./index.js').Platform} */
|
|
1022
1053
|
export const platform = {
|
|
1023
1054
|
id: 'android',
|
|
@@ -1034,6 +1065,9 @@ export const platform = {
|
|
|
1034
1065
|
terminateApp,
|
|
1035
1066
|
openUrl,
|
|
1036
1067
|
restartDevice,
|
|
1068
|
+
listApps,
|
|
1069
|
+
appContainer,
|
|
1070
|
+
readPropertyList,
|
|
1037
1071
|
setPermission,
|
|
1038
1072
|
setPasteboard,
|
|
1039
1073
|
getPasteboard,
|
package/src/platform/index.js
CHANGED
|
@@ -63,6 +63,7 @@ export const PLATFORM_SURFACE = Object.freeze([
|
|
|
63
63
|
'geometry', 'inputDriver',
|
|
64
64
|
'screenshot', 'launchApp', 'terminateApp', 'openUrl', 'restartDevice',
|
|
65
65
|
'setPermission', 'setPasteboard', 'permissionServices', 'capabilities', 'toolchain',
|
|
66
|
+
'listApps', 'appContainer', 'readPropertyList',
|
|
66
67
|
'bootedAt',
|
|
67
68
|
]);
|
|
68
69
|
|
|
@@ -209,6 +210,12 @@ export const openUrl = (udid, ...args) => platformFor(udid).openUrl(udid, ...arg
|
|
|
209
210
|
export const setPermission = (udid, ...args) => platformFor(udid).setPermission(udid, ...args);
|
|
210
211
|
export const setPasteboard = (udid, ...args) => platformFor(udid).setPasteboard(udid, ...args);
|
|
211
212
|
|
|
213
|
+
// Reading what an app persisted. Routed like everything else, and declined by a
|
|
214
|
+
// backend that has no equivalent rather than answered in the other's terms.
|
|
215
|
+
export const listApps = (udid, ...args) => platformFor(udid).listApps(udid, ...args);
|
|
216
|
+
export const appContainer = (udid, ...args) => platformFor(udid).appContainer(udid, ...args);
|
|
217
|
+
export const readPropertyList = (udid, file) => platformFor(udid).readPropertyList(file);
|
|
218
|
+
|
|
212
219
|
/**
|
|
213
220
|
* The permission services a device understands, or every service any backend
|
|
214
221
|
* understands when no device is named.
|
package/src/platform/ios.js
CHANGED
|
@@ -9,6 +9,7 @@ import fs from 'node:fs';
|
|
|
9
9
|
import os from 'node:os';
|
|
10
10
|
import path from 'node:path';
|
|
11
11
|
import { promisify } from 'node:util';
|
|
12
|
+
import * as plist from './plist.js';
|
|
12
13
|
|
|
13
14
|
const run = promisify(execFile);
|
|
14
15
|
|
|
@@ -178,17 +179,42 @@ function isBootedSync(udid) {
|
|
|
178
179
|
return false;
|
|
179
180
|
}
|
|
180
181
|
|
|
182
|
+
/**
|
|
183
|
+
* What went wrong with a `simctl io screenshot`, in one sentence.
|
|
184
|
+
*
|
|
185
|
+
* Extracted so it can be *tested* rather than reasoned about, for the same
|
|
186
|
+
* reason `pickDevice` and `decisionOf` were: this is the line where a wrong
|
|
187
|
+
* answer was expensive, and it was wrong for a day.
|
|
188
|
+
*/
|
|
189
|
+
export function screenshotFailure(err) {
|
|
190
|
+
const killed = err.killed || err.signal === 'SIGTERM';
|
|
191
|
+
// `simctl` opens with `Note: No display specified …` on every run, success or
|
|
192
|
+
// failure. When the display surface is dead the command does not fail, it
|
|
193
|
+
// *hangs* — so at kill time that Note is the only thing on stderr, and the
|
|
194
|
+
// tool reported a benign informational line as the reason a capture failed.
|
|
195
|
+
// That is how this wedge stayed nameless through five CI failures.
|
|
196
|
+
const lines = String(err.stderr || '').trim().split('\n').map((l) => l.trim()).filter(Boolean);
|
|
197
|
+
const real = lines.filter((l) => !/^Note:/i.test(l)).pop();
|
|
198
|
+
if (killed && !real) {
|
|
199
|
+
// Run to completion the device names it exactly:
|
|
200
|
+
// NSPOSIXErrorDomain code 60 — Timeout waiting for screen surfaces
|
|
201
|
+
// which is CoreSimulator saying the surface is gone, and the closest thing
|
|
202
|
+
// to a positive test for the wedge that exists.
|
|
203
|
+
return 'simctl screenshot did not return within 10s. The display surface is not answering'
|
|
204
|
+
+ ' — run to completion it reports "Timeout waiting for screen surfaces" (NSPOSIXErrorDomain 60).'
|
|
205
|
+
+ ' This is the device, not the capture loop: `simframe revive` restarts it.';
|
|
206
|
+
}
|
|
207
|
+
const detail = real ?? lines.pop();
|
|
208
|
+
return detail ? `simctl screenshot failed: ${detail}` : `simctl screenshot failed: ${err.message}`;
|
|
209
|
+
}
|
|
210
|
+
|
|
181
211
|
async function screenshot(udid, outFile, { mask = 'ignored' } = {}) {
|
|
182
212
|
try {
|
|
183
213
|
await run('xcrun', ['simctl', 'io', udid, 'screenshot', '--type=png', `--mask=${mask}`, outFile], {
|
|
184
214
|
timeout: 10_000,
|
|
185
215
|
});
|
|
186
216
|
} catch (err) {
|
|
187
|
-
|
|
188
|
-
// whole command>" and simctl's actual complaint is in stderr. A CI failure
|
|
189
|
-
// here reported the command and nothing about why it did not work.
|
|
190
|
-
const detail = (err.stderr || '').trim().split('\n').filter(Boolean).pop();
|
|
191
|
-
throw new Error(detail ? `simctl screenshot failed: ${detail}` : `simctl screenshot failed: ${err.message}`);
|
|
217
|
+
throw new Error(screenshotFailure(err));
|
|
192
218
|
}
|
|
193
219
|
}
|
|
194
220
|
|
|
@@ -309,6 +335,100 @@ function ownsUdid(udid) {
|
|
|
309
335
|
return /^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i.test(String(udid ?? ''));
|
|
310
336
|
}
|
|
311
337
|
|
|
338
|
+
/**
|
|
339
|
+
* Where CoreSimulator keeps a device's data.
|
|
340
|
+
*
|
|
341
|
+
* Read straight off disk, and that is the whole point of this family rather
|
|
342
|
+
* than an optimisation. Measured on this Xcode: `simctl get_app_container` and
|
|
343
|
+
* `simctl listapps` **both** fail on a device that is not running —
|
|
344
|
+
* `Unable to lookup in current state: Shutdown`. The field report that asked
|
|
345
|
+
* for this feature rated it highest-leverage precisely because it answered
|
|
346
|
+
* "what did the app save?" *before the device was booted*, and simctl cannot do
|
|
347
|
+
* that. The filesystem can, so this reads the filesystem.
|
|
348
|
+
*/
|
|
349
|
+
const deviceRoot = (udid) =>
|
|
350
|
+
path.join(os.homedir(), 'Library/Developer/CoreSimulator/Devices', String(udid));
|
|
351
|
+
|
|
352
|
+
const containerRoot = (udid) => path.join(deviceRoot(udid), 'data/Containers/Data/Application');
|
|
353
|
+
|
|
354
|
+
/** The per-container metadata file that says which app owns it. */
|
|
355
|
+
const METADATA = '.com.apple.mobile_container_manager.metadata.plist';
|
|
356
|
+
|
|
357
|
+
/**
|
|
358
|
+
* Every app with a data container on this device, booted or not.
|
|
359
|
+
*
|
|
360
|
+
* The bundle id lives in `MCMMetadataIdentifier` in each container's metadata
|
|
361
|
+
* plist. It is *not* recoverable by grepping the file — the binary plist
|
|
362
|
+
* encodes strings in a way that does not leave the id as a plain substring, and
|
|
363
|
+
* an early version of this that tried to pre-filter that way matched nothing.
|
|
364
|
+
* So each metadata file is asked properly. Measured at **0.52s for 150
|
|
365
|
+
* containers**, which is a listing cost rather than a per-read one.
|
|
366
|
+
*/
|
|
367
|
+
async function listApps(udid) {
|
|
368
|
+
const root = containerRoot(udid);
|
|
369
|
+
let entries;
|
|
370
|
+
try {
|
|
371
|
+
entries = await fs.promises.readdir(root, { withFileTypes: true });
|
|
372
|
+
} catch (err) {
|
|
373
|
+
if (err.code === 'ENOENT') {
|
|
374
|
+
const exists = fs.existsSync(deviceRoot(udid));
|
|
375
|
+
throw new Error(exists
|
|
376
|
+
? `device ${udid} has no app data containers yet — nothing has been installed on it`
|
|
377
|
+
: `no simulator data directory for ${udid} (looked in ${root})`);
|
|
378
|
+
}
|
|
379
|
+
throw err;
|
|
380
|
+
}
|
|
381
|
+
const apps = [];
|
|
382
|
+
await Promise.all(entries.filter((e) => e.isDirectory()).map(async (e) => {
|
|
383
|
+
const dir = path.join(root, e.name);
|
|
384
|
+
try {
|
|
385
|
+
const { stdout } = await run('plutil',
|
|
386
|
+
['-extract', 'MCMMetadataIdentifier', 'raw', '-o', '-', path.join(dir, METADATA)],
|
|
387
|
+
{ timeout: 10_000 });
|
|
388
|
+
const bundleId = stdout.trim();
|
|
389
|
+
if (bundleId) apps.push({ bundleId, container: dir });
|
|
390
|
+
} catch {
|
|
391
|
+
// A container without readable metadata is not an app we can name, and
|
|
392
|
+
// naming it by its UUID would be offering an id nobody can use.
|
|
393
|
+
}
|
|
394
|
+
}));
|
|
395
|
+
return apps.sort((a, b) => a.bundleId.localeCompare(b.bundleId));
|
|
396
|
+
}
|
|
397
|
+
|
|
398
|
+
/** The data container for one app, or a listing of what is there instead. */
|
|
399
|
+
async function appContainer(udid, bundleId) {
|
|
400
|
+
const apps = await listApps(udid);
|
|
401
|
+
const hit = apps.find((a) => a.bundleId === bundleId);
|
|
402
|
+
if (hit) return hit.container;
|
|
403
|
+
// Near misses first: the id is the thing people get wrong, and a bare "not
|
|
404
|
+
// installed" on a device with the app under a slightly different id is the
|
|
405
|
+
// least useful true sentence available.
|
|
406
|
+
const needle = String(bundleId).toLowerCase();
|
|
407
|
+
const near = apps.filter((a) => a.bundleId.toLowerCase().includes(needle)
|
|
408
|
+
|| needle.includes(a.bundleId.toLowerCase())).slice(0, 5);
|
|
409
|
+
throw new Error(
|
|
410
|
+
`"${bundleId}" has no data container on ${udid}`
|
|
411
|
+
+ (near.length ? ` — did you mean ${near.map((a) => a.bundleId).join(', ')}?` : '')
|
|
412
|
+
+ ` (${apps.length} app(s) have one)`,
|
|
413
|
+
);
|
|
414
|
+
}
|
|
415
|
+
|
|
416
|
+
/**
|
|
417
|
+
* Read a property list, whatever it contains.
|
|
418
|
+
*
|
|
419
|
+
* `-convert xml1` and not `json`: six of the twenty real preference plists on
|
|
420
|
+
* the bench device cannot be represented as JSON at all, because `<data>` and
|
|
421
|
+
* `<date>` have no JSON form and plutil refuses rather than inventing one. See
|
|
422
|
+
* the note at the top of plist.js.
|
|
423
|
+
*/
|
|
424
|
+
async function readPropertyList(file) {
|
|
425
|
+
const { stdout } = await run('plutil', ['-convert', 'xml1', '-o', '-', file], {
|
|
426
|
+
timeout: 20_000,
|
|
427
|
+
maxBuffer: 64 * 1024 * 1024,
|
|
428
|
+
});
|
|
429
|
+
return plist.parse(stdout);
|
|
430
|
+
}
|
|
431
|
+
|
|
312
432
|
/**
|
|
313
433
|
* The prerequisites `simframe doctor` reports for this backend. Returned rather
|
|
314
434
|
* than printed so doctor stays one renderer: a backend says what it needs, and
|
|
@@ -415,6 +535,9 @@ export const platform = {
|
|
|
415
535
|
terminateApp,
|
|
416
536
|
openUrl,
|
|
417
537
|
restartDevice,
|
|
538
|
+
listApps,
|
|
539
|
+
appContainer,
|
|
540
|
+
readPropertyList,
|
|
418
541
|
setPermission,
|
|
419
542
|
setPasteboard,
|
|
420
543
|
permissionServices: () => PERMISSION_SERVICES,
|