simframe 0.11.0 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/metrics.js CHANGED
@@ -24,6 +24,21 @@ export const REASONS = [
24
24
 
25
25
  export const OUTCOMES = ['resolved_locally', 'escalated_to_model', 'failed'];
26
26
 
27
+ /**
28
+ * What the executor observed after a supervisor ruling — the field that makes a
29
+ * ruling scoreable rather than merely recorded.
30
+ *
31
+ * A ruling on its own says what the supervisor thought. Three items on the
32
+ * deferred list (101's p95 replay, 106's cost of the steps a `stop` skipped,
33
+ * 96's response variable) need what happened *next*, and until now nothing
34
+ * persisted a ruling at all: they went into a `supervisions` array on the
35
+ * result and died with the process. Three rulings had ever existed anywhere.
36
+ *
37
+ * Closed, and mandatory, for the same reason `REASONS` is: a vocabulary that
38
+ * admits "other" collects a pile of "other".
39
+ */
40
+ export const RULING_OUTCOMES = ['recovered', 'still_failed', 'stopped', 'no_ruling'];
41
+
27
42
  /**
28
43
  * Which faculty would have removed this escalation.
29
44
  *
@@ -57,6 +72,7 @@ function metricPaths(udid) {
57
72
  dir,
58
73
  escalations: path.join(dir, 'escalations.jsonl'),
59
74
  flows: path.join(dir, 'flows.jsonl'),
75
+ supervisions: path.join(dir, 'supervisions.jsonl'),
60
76
  baselines: path.join(dir, 'baselines'),
61
77
  };
62
78
  }
@@ -109,6 +125,106 @@ export function readJsonl(file, { limit } = {}) {
109
125
 
110
126
  export const readEscalations = (udid, opts) => readJsonl(metricPaths(udid).escalations, opts);
111
127
  export const readFlows = (udid, opts) => readJsonl(metricPaths(udid).flows, opts);
128
+ export const readSupervisions = (udid, opts) => readJsonl(metricPaths(udid).supervisions, opts);
129
+
130
+ /**
131
+ * Write down one supervisor ruling and what came of it.
132
+ *
133
+ * The fields are chosen so the three items waiting on this can be answered
134
+ * **offline, from the log**, rather than by another device run:
135
+ *
136
+ * - `screen`/`edge` — which `(screen_hash, action)` the ruling was about, so
137
+ * rulings can be grouped by edge the way the graph groups timings.
138
+ * - `p95`/`samples` — what the graph knew about that edge **at the moment of
139
+ * the ruling**. Recording it now rather than looking it up later is the
140
+ * difference between 101 being arithmetic and 101 being archaeology: the
141
+ * graph keeps learning, so a p95 read next week is not the p95 the
142
+ * supervisor was implicitly competing with.
143
+ * - `stillMs` — the same input the supervisor got, so a replay sees what it saw.
144
+ * - `expect` — the plan's own note, because a ruling made with a briefing and
145
+ * one made without are not the same measurement (96's critical arm).
146
+ * - `outcome` — what the executor observed afterwards. Mandatory.
147
+ *
148
+ * `from` separates a deterministic rule from a model answer. Both are rulings
149
+ * and both have outcomes, but a comparison that mixed them would credit the
150
+ * model for what a two-line rule decided.
151
+ */
152
+ export function recordSupervision(udid, {
153
+ session, index, step, edge, screen, decision, from, reason, ms,
154
+ stillMs, p95, samples, expect, failure, outcome,
155
+ }) {
156
+ // Swallowed rather than thrown, unlike `recordEscalation`'s guard, and the
157
+ // difference is deliberate: this is called from inside a flow's failure
158
+ // handler, where a throw would turn a recoverable step failure into a crash.
159
+ // It still lands in `lastWriteError`, which `simframe escalations` prints.
160
+ if (!RULING_OUTCOMES.includes(outcome)) {
161
+ lastWriteError = `supervision outcome must be one of ${RULING_OUTCOMES.join('/')}, got ${JSON.stringify(outcome)}`;
162
+ return false;
163
+ }
164
+ return appendJsonl(metricPaths(udid).supervisions, {
165
+ timestamp: new Date().toISOString(),
166
+ session_id: session ?? SESSION_ID,
167
+ client: CLIENT,
168
+ step_index: index ?? null,
169
+ step: step ?? null,
170
+ edge: edge ?? null,
171
+ screen_fingerprint: screen ?? null,
172
+ decision,
173
+ from: from ?? 'model',
174
+ // Recorded, never presented as the ground for what happened: the supervisor
175
+ // has returned a correct decision with a reason citing a rule that did not
176
+ // apply. Keeping it is how that stays measurable instead of anecdotal.
177
+ reason: reason ? String(reason).slice(0, 200) : null,
178
+ latency_ms: Number.isFinite(ms) ? Math.round(ms) : null,
179
+ still_ms: Number.isFinite(stillMs) ? Math.round(stillMs) : null,
180
+ edge_p95_ms: Number.isFinite(p95) ? Math.round(p95) : null,
181
+ edge_samples: Number.isFinite(samples) ? samples : null,
182
+ expect: expect ? String(expect).slice(0, 200) : null,
183
+ failure: failure ? String(failure).slice(0, 300) : null,
184
+ outcome,
185
+ });
186
+ }
187
+
188
+ /**
189
+ * What the ruling log currently says, and whether it can yet answer 101.
190
+ *
191
+ * Deliberately counts rather than scores. Item 101 asks which `wait`/`retry`
192
+ * rulings a p95-per-edge lookup would have got right, and that is a separate
193
+ * piece of work; what this answers is the question that comes first and was
194
+ * embarrassing to get wrong once already — **is there a population to measure
195
+ * at all, and do its rows carry the fields the measurement needs.** `p95_known`
196
+ * is that readiness check: a ruling recorded on an edge the graph had never
197
+ * timed cannot take part in the comparison, however many of them there are.
198
+ */
199
+ export function supervisionBreakdown(records) {
200
+ const by = (key) => {
201
+ const out = {};
202
+ for (const r of records) {
203
+ const k = r[key] ?? 'unknown';
204
+ out[k] = (out[k] ?? 0) + 1;
205
+ }
206
+ return out;
207
+ };
208
+ const matrix = {};
209
+ for (const r of records) {
210
+ const k = `${r.decision ?? 'unknown'} -> ${r.outcome ?? 'unknown'}`;
211
+ matrix[k] = (matrix[k] ?? 0) + 1;
212
+ }
213
+ const timed = records.filter((r) => Number.isFinite(r.edge_p95_ms));
214
+ const latencies = records.map((r) => r.latency_ms).filter((n) => Number.isFinite(n)).sort((a, b) => a - b);
215
+ return {
216
+ total: records.length,
217
+ by_decision: by('decision'),
218
+ by_outcome: by('outcome'),
219
+ by_from: by('from'),
220
+ decision_to_outcome: matrix,
221
+ // Readiness for 101, not a result for it.
222
+ p95_known: timed.length,
223
+ p95_unknown: records.length - timed.length,
224
+ sessions: [...new Set(records.map((r) => r.session_id).filter(Boolean))],
225
+ median_latency_ms: latencies.length ? latencies[latencies.length >> 1] : null,
226
+ };
227
+ }
112
228
 
113
229
  /**
114
230
  * Mark an error as an escalation with a reason, at the site that knows why.
@@ -996,6 +996,28 @@ function capabilities() {
996
996
  };
997
997
  }
998
998
 
999
+ /**
1000
+ * Not implemented here, and it says so in its own vocabulary.
1001
+ *
1002
+ * `revive` exists for a display that has stopped rendering — a CoreSimulator
1003
+ * pathology. An emulator's failure modes are its own (the console port going
1004
+ * away, adb losing the device) and the remedy is not the same sequence, so
1005
+ * borrowing the other platform's answer would be the mistake `doctor` made when
1006
+ * it reported "input driver: idb" for an emulator: a claim about a tool that has
1007
+ * never spoken to an Android device.
1008
+ *
1009
+ * When an emulator wedge is characterised rather than guessed at, this becomes
1010
+ * a real implementation. Until then the honest answer is that this backend does
1011
+ * not have the layer.
1012
+ */
1013
+ async function restartDevice(serial) {
1014
+ throw new Error(
1015
+ `simframe cannot yet restart an emulator (${serial}) — that remedy is written for a`
1016
+ + ' simulator display that stopped rendering, and an emulator fails differently.'
1017
+ + ' Restart it from Android Studio, or `adb -s <serial> emu kill` and relaunch.',
1018
+ );
1019
+ }
1020
+
999
1021
  /** @type {import('./index.js').Platform} */
1000
1022
  export const platform = {
1001
1023
  id: 'android',
@@ -1011,6 +1033,7 @@ export const platform = {
1011
1033
  launchApp,
1012
1034
  terminateApp,
1013
1035
  openUrl,
1036
+ restartDevice,
1014
1037
  setPermission,
1015
1038
  setPasteboard,
1016
1039
  getPasteboard,
@@ -61,7 +61,7 @@ export const PLATFORM_SURFACE = Object.freeze([
61
61
  'id', 'deviceNoun',
62
62
  'listDevices', 'bootedDevices', 'resolveDevice', 'isBootedSync', 'ownsUdid',
63
63
  'geometry', 'inputDriver',
64
- 'screenshot', 'launchApp', 'terminateApp', 'openUrl',
64
+ 'screenshot', 'launchApp', 'terminateApp', 'openUrl', 'restartDevice',
65
65
  'setPermission', 'setPasteboard', 'permissionServices', 'capabilities', 'toolchain',
66
66
  'bootedAt',
67
67
  ]);
@@ -260,6 +260,16 @@ export const inputDriverFor = (udid) => platformFor(udid).inputDriver(udid);
260
260
  */
261
261
  export const capabilitiesFor = (udid) => platformFor(udid).capabilities();
262
262
 
263
+ /**
264
+ * Power-cycle a device, on the backend that owns it.
265
+ *
266
+ * Reached only from `simframe revive`, never from the capture loop: the loop
267
+ * detects a stalled display and reports it, and restarting is the operator's
268
+ * call. A backend that does not have this remedy throws in its own terms rather
269
+ * than borrowing the other's.
270
+ */
271
+ export const restartDevice = (udid) => platformFor(udid).restartDevice(udid);
272
+
263
273
  /** What `simframe doctor` should check: each registered backend's own toolchain. */
264
274
  export function toolchainChecks() {
265
275
  return backends().flatMap((backend) => backend.toolchain());
@@ -189,6 +189,28 @@ async function openUrl(udid, url) {
189
189
  await run('xcrun', ['simctl', 'openurl', udid, url], { timeout: 20_000 });
190
190
  }
191
191
 
192
+ /**
193
+ * Shut a device down and bring it back, waiting for the boot to finish.
194
+ *
195
+ * The remedy for a display that has stopped rendering, which the capture loop
196
+ * can detect and must not perform: it reports `stalled` and stops, because a
197
+ * capture loop that rebooted the device it was watching would be a tool
198
+ * reaching for the mains when a reading looks wrong. This is the operator's
199
+ * decision, reached by `simframe revive`.
200
+ *
201
+ * `bootstatus -b` and not `boot`, for the reason it is used in CI: `boot`
202
+ * returns before the device is usable, and everything downstream then races the
203
+ * boot. Timeboxed generously — a cold boot on a busy machine is slow, and a
204
+ * boot that never finishes should fail here with a reason rather than as a
205
+ * puzzle further down.
206
+ */
207
+ async function restartDevice(udid) {
208
+ // Tolerated: a device that is already off cannot be shut down, and that is
209
+ // the state this command is most often reached from.
210
+ await run('xcrun', ['simctl', 'shutdown', udid], { timeout: 60_000 }).catch(() => null);
211
+ await run('xcrun', ['simctl', 'bootstatus', udid, '-b'], { timeout: 240_000 });
212
+ }
213
+
192
214
  const PERMISSION_SERVICES = [
193
215
  'all', 'calendar', 'contacts-limited', 'contacts', 'location', 'location-always',
194
216
  'photos-add', 'photos', 'media-library', 'microphone', 'motion', 'reminders', 'siri',
@@ -347,6 +369,7 @@ export const platform = {
347
369
  launchApp,
348
370
  terminateApp,
349
371
  openUrl,
372
+ restartDevice,
350
373
  setPermission,
351
374
  setPasteboard,
352
375
  permissionServices: () => PERMISSION_SERVICES,
package/src/regions.js CHANGED
@@ -24,6 +24,27 @@ export const REGIONS = [
24
24
  'content',
25
25
  ];
26
26
 
27
+ /**
28
+ * Regions the screen map will not offer as something to act on.
29
+ *
30
+ * The status bar says the time and the battery level. It is on every screen, it
31
+ * is never what anybody wants to tap, and it costs a row every time — so
32
+ * `sim_ui` hides it.
33
+ *
34
+ * It lives here rather than in `view.js`, where it started, because it is not
35
+ * only a presentation rule. A target the map refuses to *show* must also be a
36
+ * target nothing may resolve onto *behind the caller's back*, and the case that
37
+ * proved it was exactly that: a stale `#1` numbered "Reminders" in Reminders,
38
+ * re-resolved in Contacts onto the status-bar back-to-app breadcrumb
39
+ * "• Reminders", scored 0.64 and handed back a tap point at (47,40) — a place
40
+ * the map would never have put in front of anybody. One rule, one home, both
41
+ * readers.
42
+ */
43
+ const UNOFFERED_REGIONS = new Set(['status-bar']);
44
+
45
+ /** Would the screen map offer a target in this region? */
46
+ export const offerable = (region) => !UNOFFERED_REGIONS.has(region);
47
+
27
48
  /**
28
49
  * The status bar stays positional, and deliberately.
29
50
  *
@@ -63,6 +84,46 @@ const BOTTOM_CHROME_LIMIT = 0.82;
63
84
  * screen sets its own scale.
64
85
  */
65
86
  const MIN_BOUNDARY_GAP_PT = 10;
87
+ /**
88
+ * How wide a lone row may be and still be a screen's title rather than its
89
+ * first paragraph.
90
+ *
91
+ * 0.6 of the screen, measured: the Settings root's large title is 133 pt of
92
+ * 402 (0.33), while Reminders' empty-state heading "Welcome to Reminders" is
93
+ * 326 pt (0.81) and is content — it describes the screen instead of naming it.
94
+ * A title is a name and names are short.
95
+ */
96
+ const LARGE_TITLE_MAX_WIDTH_FRACTION = 0.6;
97
+ /**
98
+ * How far below the status bar a large title can start.
99
+ *
100
+ * iOS draws one at a **system** offset, not an app-chosen one, so this is a
101
+ * bound on a platform constant rather than a tuned threshold. Measured on the
102
+ * bench device: the Settings root's title starts 63-79 pt below the status bar
103
+ * depending on which sensor reports its box, while example.com's `<h1>` — page
104
+ * *content* that merely happens to be the first row, because Safari on iOS puts
105
+ * its chrome at the bottom — starts **122 pt** down.
106
+ *
107
+ * Without this bound the rule promoted that `<h1>` to chrome and "example
108
+ * domain" entered the screen's identity. Pulling page content into identity is
109
+ * the exact failure this module has been bitten by twice (a phantom keyboard,
110
+ * and content that merely fell into a band), so the bound is not optional.
111
+ */
112
+ const LARGE_TITLE_MAX_INSET_PT = 96;
113
+ /**
114
+ * And the least it can be, before it is just the next row.
115
+ *
116
+ * Absolute, like the maximum, and for the same reason: the inset is drawn by
117
+ * the system, so it is not a function of what the screen contains. The first
118
+ * version tested it against the screen's *median row gap* — and the testbed
119
+ * caught that on its first day, with two screens of the same app. A list of 24
120
+ * rows has a median gap of 0 and the rule fired; a list of 4 rows above a tab
121
+ * bar has a median gap of **414**, because the empty area counts as a gap, and
122
+ * the rule did not. Same title, same inset of 62.9pt, opposite answers — so one
123
+ * screen had a name in its identity and the other did not, and the graph then
124
+ * called them the same screen at 0.50 similarity.
125
+ */
126
+ const LARGE_TITLE_MIN_INSET_PT = 24;
66
127
  const BOUNDARY_GAP_FACTOR = 1.9;
67
128
 
68
129
  /** A tab bar is several things spread across the width, not one thing at the bottom. */
@@ -177,6 +238,50 @@ export function bands(elements, screen) {
177
238
  }
178
239
  }
179
240
 
241
+ // --- a large title, which has no gap under it to be found by.
242
+ //
243
+ // The loop above identifies top chrome by the whitespace *beneath* it, and an
244
+ // iOS large title is drawn tight against the content it heads: measured on
245
+ // the Settings root, 79 pt of inset above it and **5.3 pt** below, against a
246
+ // bar of `max(10, typical * 1.9)` = 66.5. No threshold reaches that, so the
247
+ // title fell into `content` and was discarded as content — leaving the screen
248
+ // with **no name at all** in its fingerprint, in either sensor mode.
249
+ //
250
+ // That is not cosmetic. Chrome labels are the only text identity keeps, and
251
+ // `fingerprint.js` names the consequence: two list screens with identical
252
+ // structure differ by their title and nothing else says so. A nameless screen
253
+ // is pure geometry, and on a hosted runner two sparse nameless readings
254
+ // matched exactly — one screen's hash for two screens.
255
+ //
256
+ // So it is found by the inset *above* it instead, which is the half iOS does
257
+ // provide. A large title sits alone, narrow, high, under a generous gap; a
258
+ // compact bar is the mirror image of that (20-33 pt above, 118-134 below) and
259
+ // is already caught by the loop. Deliberately not keyed on the `Heading`
260
+ // role: OCR has no roles, and the reading that actually collided was
261
+ // OCR-only, so a role test would work only in the case that does not fail.
262
+ //
263
+ // Measured across all 17 perception fixtures before being written here: it
264
+ // changes exactly one of them, the Settings root.
265
+ if (!navBarBottom) {
266
+ const first = rows.findIndex((r) => r.top >= statusBarBottom - 1);
267
+ const row = first >= 0 ? rows[first] : null;
268
+ if (
269
+ row
270
+ && first + 1 < rows.length
271
+ && row.items.length === 1
272
+ && (row.items[0].frame?.width ?? 0) <= screen.width * LARGE_TITLE_MAX_WIDTH_FRACTION
273
+ && row.bottom <= screen.height * TOP_CHROME_LIMIT
274
+ // A real separation from the status bar, but a system-sized one: far
275
+ // enough to be an inset, near enough to still be the app's own title.
276
+ // Both bounds absolute — see LARGE_TITLE_MIN_INSET_PT for what keying the
277
+ // lower one on the screen's own row rhythm cost.
278
+ && row.top - statusBarBottom >= LARGE_TITLE_MIN_INSET_PT
279
+ && row.top - statusBarBottom <= LARGE_TITLE_MAX_INSET_PT
280
+ ) {
281
+ navBarBottom = row.bottom;
282
+ }
283
+ }
284
+
180
285
  // --- bottom chrome. One row: a tab bar is one row by construction, and the
181
286
  // gap above it is what separates it from the list it floats over.
182
287
  let tabBarTop = Infinity;
package/src/supervisor.js CHANGED
@@ -45,6 +45,30 @@ const helper = lineServer({
45
45
 
46
46
  export const DECISIONS = new Set(['wait', 'retry', 'stop']);
47
47
 
48
+ /**
49
+ * The vocabulary gate, as a function so it can be *tested* rather than grepped.
50
+ *
51
+ * This one line is the safety property: the supervisor cannot invent a step,
52
+ * skip one, substitute a target or continue past an unexpected screen, because
53
+ * those are not words it can say. Anything outside the three is not a decision
54
+ * and becomes `null`, which means "behave as if there is no supervisor".
55
+ *
56
+ * It was previously inline, and the test that guarded it matched the source
57
+ * text — so it broke when the branch grew an else, with nothing actually wrong.
58
+ * A property this important deserves an assertion that runs it.
59
+ */
60
+ export function decisionOf(answer) {
61
+ // A string, checked rather than coerced. `String(["wait"])` is `"wait"`, so a
62
+ // `String(...)` coercion here let `{decision: ["wait"]}` through the one gate
63
+ // that defines this component's answer space. Found the first time this
64
+ // property was *run* instead of grepped for in the source — the old test
65
+ // matched the source text of the branch and could never have caught it.
66
+ const raw = answer?.decision;
67
+ if (typeof raw !== 'string') return null;
68
+ const decision = raw.toLowerCase();
69
+ return DECISIONS.has(decision) ? decision : null;
70
+ }
71
+
48
72
  /** Which backend the caller asked for, per call first and environment second. */
49
73
  export function requested(options) {
50
74
  const raw = String(options?.supervisor ?? process.env.SIMFRAME_SUPERVISOR ?? '').trim().toLowerCase();
@@ -62,6 +86,7 @@ export function requested(options) {
62
86
  */
63
87
  export async function judge({
64
88
  goal, step, expected, failure, screen, stillMs, note, options, timeoutMs = 2500,
89
+ detail,
65
90
  } = {}) {
66
91
  if (!requested(options)) return null;
67
92
  if (!step || !failure) return null;
@@ -74,10 +99,24 @@ export async function judge({
74
99
  stillMs: Number.isFinite(stillMs) ? Math.round(stillMs) : null,
75
100
  note: note ? String(note).slice(0, 200) : null,
76
101
  }, timeoutMs);
77
- const decision = String(answer?.decision ?? '').toLowerCase();
78
- // An answer outside the vocabulary is not a decision. Refusing it here is
79
- // what makes the three-word constraint real rather than merely documented.
80
- if (!DECISIONS.has(decision)) return null;
102
+ const decision = decisionOf(answer);
103
+ if (decision == null) {
104
+ // Why it did not answer, for the caller's log — through an out-parameter
105
+ // rather than the return value, because returning anything truthy here
106
+ // would change what the executor does. `null` means "behave as if there is
107
+ // no supervisor" and that safety property is the one thing in this file
108
+ // that must not become conditional.
109
+ //
110
+ // Before this, every failure reached the supervision log as "the
111
+ // supervisor did not answer": a timeout, a guardrail refusal and a model
112
+ // that was never installed were one indistinguishable line.
113
+ if (detail && typeof detail === 'object') {
114
+ detail.kind = answer?.kind
115
+ ?? (answer == null ? 'no answer' : answer.decision ? 'outside the vocabulary' : 'unparseable');
116
+ if (answer?.error) detail.error = String(answer.error).slice(0, 200);
117
+ }
118
+ return null;
119
+ }
81
120
  return { decision, reason: String(answer.reason ?? '').slice(0, 120), ms: answer.ms ?? null };
82
121
  }
83
122
 
@@ -101,16 +140,21 @@ export async function status(options) {
101
140
  screen: ['Probe'],
102
141
  stillMs: 5000,
103
142
  }, 6000);
104
- const decision = String(probe?.decision ?? '').toLowerCase();
105
- if (!DECISIONS.has(decision)) {
143
+ if (decisionOf(probe) == null) {
106
144
  return {
107
145
  supervisor: 'none',
108
146
  detail: 'the model loaded but did not answer a probe — it is present and not working',
109
147
  };
110
148
  }
149
+ // The window, read from the model rather than repeated from documentation.
150
+ // Worth printing: the worst case our own clipping allows measures 1,918
151
+ // tokens against it, and that ratio is the reason there is no per-call token
152
+ // budget check — see DEFERRED 99.
153
+ const ctx = live.hello?.contextSize;
111
154
  return {
112
155
  supervisor: 'apple',
113
- detail: `Apple Foundation Models, on-device; answered a probe in ${probe.ms ?? '?'}ms; may only answer wait/retry/stop`,
156
+ detail: `Apple Foundation Models, on-device; answered a probe in ${probe.ms ?? '?'}ms;`
157
+ + `${ctx ? ` ${ctx}-token window;` : ''} may only answer wait/retry/stop`,
114
158
  };
115
159
  }
116
160
 
package/src/view.js CHANGED
@@ -22,10 +22,10 @@ import * as matching from './matching.js';
22
22
  const REGION_ORDER = ['nav-bar', 'content', 'tab-bar', 'keyboard', 'status-bar'];
23
23
 
24
24
  /**
25
- * The status bar says the time and the battery level. It is on every screen,
26
- * it is never what anybody wants to tap, and it costs a row every time.
25
+ * Which regions the map will not offer now lives in `regions.js`, because it is
26
+ * not only a presentation rule — see `regions.offerable`. A target this hides
27
+ * must also be one nothing resolves onto behind the caller's back.
27
28
  */
28
- const HIDDEN_REGIONS = new Set(['status-bar']);
29
29
 
30
30
  /** A keyboard is 30-odd keys nobody refers to by name. One line says it. */
31
31
  const COLLAPSE_REGIONS = new Set(['keyboard']);
@@ -249,7 +249,7 @@ export function rowsFor(entry, { screen, filter, interactive, all = false, limit
249
249
  if (!isNum(t.x) || !isNum(t.y)) return false;
250
250
  // Off-screen elements are real in the tree and untappable in fact.
251
251
  if (regions.offViewport(t, screen)) return false;
252
- if (!all && HIDDEN_REGIONS.has(t.region)) return false;
252
+ if (!all && !regions.offerable(t.region)) return false;
253
253
  if (!all && isNoise(t)) return false;
254
254
  return true;
255
255
  });
@@ -329,6 +329,28 @@ export function recalledNote(identity, now = Date.now()) {
329
329
  return `elements recalled from ${ago} ago — pass refresh for what is there now`;
330
330
  }
331
331
 
332
+ /**
333
+ * How old the frame behind this reading is, and whether that is a problem.
334
+ *
335
+ * Two thresholds, because "slightly behind" and "possibly a different screen"
336
+ * are different messages. Under `FRAME_FRESH_MS` nothing is said: a map that
337
+ * announced "42ms old" on every call would train a reader to skip the line that
338
+ * matters. Over `FRAME_STALE_MS` it shouts, because at that age the app may have
339
+ * moved on entirely and the whole element list is then a description of the past.
340
+ */
341
+ export const FRAME_FRESH_MS = 1500;
342
+ export const FRAME_STALE_MS = 4000;
343
+
344
+ export function frameAgeNote(identity, now = Date.now()) {
345
+ const at = identity?.state?.capturedAt;
346
+ if (!Number.isFinite(at)) return null;
347
+ const age = Math.max(0, now - at);
348
+ if (age < FRAME_FRESH_MS) return null;
349
+ if (age < FRAME_STALE_MS) return `frame ${(age / 1000).toFixed(1)}s old`;
350
+ return `WARNING this frame is ${(age / 1000).toFixed(1)}s old — the screen may have moved on `
351
+ + 'since, so treat the elements below as a description of the past and pass refresh';
352
+ }
353
+
332
354
  /**
333
355
  * What a control *contains*, from the sensor that actually knows.
334
356
  *
@@ -702,6 +724,20 @@ export function render({ device, identity, rows, truncated, collapsed, screen, n
702
724
  identity?.settled === false ? 'STILL MOVING' : null,
703
725
  // Still and finished are not the same thing.
704
726
  identity?.loading === true ? 'STILL LOADING' : null,
727
+ // How old the *frame* this map was read from is.
728
+ //
729
+ // `sim_look` and `sim_state` have printed this since they existed, and this
730
+ // map never has — so the one tool an agent is told to start with was the one
731
+ // that could not say how old its evidence was. Reported from the field: a
732
+ // complete 20-element map of a screen the app was not on, and *"a wrong
733
+ // answer is worse than an error here, because nothing downstream knows to
734
+ // doubt it"*. `ensureDaemon` will hand back a frame up to 30s old, so this
735
+ // was reachable without anything being broken.
736
+ //
737
+ // `recalledNote` below is a different claim — that the *element map* came
738
+ // from memory — and having one was what made the absence of the other easy
739
+ // to miss.
740
+ frameAgeNote(identity),
705
741
  recalledNote(identity),
706
742
  ].filter(Boolean).join(' · ');
707
743