simframe 0.9.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +165 -5
- package/data/vocabulary/en.json +148 -0
- package/native/ocr.swift +13 -1
- package/native/rank.swift +87 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +43 -3
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
- package/native/simframed/Sources/simframed/main.swift +13 -1
- package/native/supervise.swift +181 -0
- package/package.json +4 -1
- package/scripts/check-package.mjs +22 -2
- package/scripts/check-private.mjs +143 -0
- package/scripts/ci-memory.mjs +104 -20
- package/scripts/eval-perception.mjs +281 -0
- package/scripts/phase17-corpus.mjs +176 -0
- package/skills/simframe/SKILL.md +237 -5
- package/src/actions.js +1825 -38
- package/src/analyze.js +70 -0
- package/src/cli.js +214 -15
- package/src/control.js +1 -0
- package/src/fingerprint.js +19 -1
- package/src/graph.js +193 -11
- package/src/index.js +428 -16
- package/src/input.js +115 -8
- package/src/localhelper.js +155 -0
- package/src/matching.js +119 -3
- package/src/mcp.js +319 -27
- package/src/metrics.js +134 -8
- package/src/navigate.js +10 -7
- package/src/ocr.js +18 -1
- package/src/planner.js +195 -0
- package/src/platform/android.js +3 -2
- package/src/platform/ios.js +2 -1
- package/src/png.js +26 -0
- package/src/refs.js +51 -8
- package/src/regions.js +110 -1
- package/src/screenmap.js +109 -10
- package/src/supervisor.js +117 -0
- package/src/view.js +396 -7
- package/src/vocabulary.js +134 -0
- package/src/wrote.js +136 -0
package/src/view.js
CHANGED
|
@@ -12,6 +12,8 @@
|
|
|
12
12
|
// app published and one OCR read off the pixels deserve different amounts of
|
|
13
13
|
// trust.
|
|
14
14
|
import * as api from './index.js';
|
|
15
|
+
import * as regions from './regions.js';
|
|
16
|
+
import * as wrote from './wrote.js';
|
|
15
17
|
import * as graph from './graph.js';
|
|
16
18
|
import { writeRefs } from './refs.js';
|
|
17
19
|
import * as matching from './matching.js';
|
|
@@ -160,12 +162,93 @@ const trim = (text) => {
|
|
|
160
162
|
return one.length > MAX_LABEL ? `${one.slice(0, MAX_LABEL - 1)}…` : one;
|
|
161
163
|
};
|
|
162
164
|
|
|
163
|
-
/**
|
|
165
|
+
/**
|
|
166
|
+
* Why no row is dropped for being long — reverted 2026-09-10, same day.
|
|
167
|
+
*
|
|
168
|
+
* There was a rule here that dropped non-interactive rows whose label ran past
|
|
169
|
+
* 45 characters, on the evidence that four of sixteen rows on a Settings screen
|
|
170
|
+
* were the explanatory paragraph under each switch: 31% of that map's
|
|
171
|
+
* characters describing things nobody can tap. Across fourteen recorded screens
|
|
172
|
+
* it cut the map 15%.
|
|
173
|
+
*
|
|
174
|
+
* It was wrong, and the counter-example is decisive. A React Native list card
|
|
175
|
+
* exposes all of its children as one concatenated accessibility label —
|
|
176
|
+
* `Anaheim | Store # 1020, , 1234 Main St, … | Quick Casual Restaurant`, 105
|
|
177
|
+
* characters, type `GenericElement`, region `content`. Every property my rule
|
|
178
|
+
* tested is identical to the Settings caption's, and that card is *the only
|
|
179
|
+
* tappable thing on the screen*.
|
|
180
|
+
*
|
|
181
|
+
* Nothing became untappable — `locate`, `assert` and `waitFor` read
|
|
182
|
+
* `entry.targets`, so the rows are a view and the data survived. What was lost
|
|
183
|
+
* is **discovery**: the map stopped saying what was on screen, so an agent
|
|
184
|
+
* could only tap labels it already knew, and the fallback on a screen of
|
|
185
|
+
* unknown data is a ~1600-token screenshot. It saved characters on
|
|
186
|
+
* settings-shaped screens and spent an image on list-shaped ones, which is the
|
|
187
|
+
* exact cost the phase existed to remove. Blast radius: every data list in
|
|
188
|
+
* every RN app.
|
|
189
|
+
*
|
|
190
|
+
* The lesson is not "find a better discriminator". It is that this was a
|
|
191
|
+
* threshold shipped with no harness case that could catch its failure, in the
|
|
192
|
+
* same session as building the harness. So `expect.discoverable` now exists,
|
|
193
|
+
* and there is a fixture of the reported shape — a rule like this may return
|
|
194
|
+
* only when it can be gated.
|
|
195
|
+
*
|
|
196
|
+
* What survives is what the reporter suggested instead: truncate, do not drop.
|
|
197
|
+
* Position and tappability are the valuable parts of a row, not the full text.
|
|
198
|
+
* See `MAX_LABEL`.
|
|
199
|
+
*/
|
|
200
|
+
|
|
201
|
+
/**
|
|
202
|
+
* Whether a target in the keyboard band really is a key.
|
|
203
|
+
*
|
|
204
|
+
* Region bands are positional, and this was the fourth and fifth bug they
|
|
205
|
+
* produced. With the keyboard up, the bottom band is collapsed to one line —
|
|
206
|
+
* thirty keys nobody names — and a primary action pinned above the keyboard was
|
|
207
|
+
* collapsed with them: four reads running printed `keyboard: 6 keys` and **no
|
|
208
|
+
* forward control**, while the hint said "nothing ambiguous — chain the next
|
|
209
|
+
* steps without looking again". That it was an emission bug and not a
|
|
210
|
+
* perception one was proved by the next call, which hit the button instantly at
|
|
211
|
+
* a coordinate the map had never printed.
|
|
212
|
+
*
|
|
213
|
+
* Delegates to `regions.looksLikeKey`, which is the canonical test. These two
|
|
214
|
+
* having separate copies is what let a phantom keyboard survive in the
|
|
215
|
+
* fingerprint after it had already been fixed in the map — and there it was
|
|
216
|
+
* deleting screens' content from their own identity.
|
|
217
|
+
*/
|
|
218
|
+
export const isKey = regions.looksLikeKey;
|
|
219
|
+
|
|
220
|
+
/**
|
|
221
|
+
* Whether a target can be acted on, by role *or* by evidence.
|
|
222
|
+
*
|
|
223
|
+
* The role alone was wrong twice on real forms. A React Native composite select
|
|
224
|
+
* surfaces as a generic element, and a text input shows only its placeholder as
|
|
225
|
+
* `StaticText` — so `--interactive` answered "1 element" on a form with two
|
|
226
|
+
* visible, bordered, *required* inputs and an agent concluded there was nothing
|
|
227
|
+
* to fill in.
|
|
228
|
+
*
|
|
229
|
+
* Evidence is used rather than a longer list of role names, because the roles
|
|
230
|
+
* are what the tree got wrong. Only a control carries a `value`, only a
|
|
231
|
+
* focusable thing is `focused`, and `enabled` is a state a caption never
|
|
232
|
+
* declares. A generic element that is none of those really is a container.
|
|
233
|
+
*
|
|
234
|
+
* The case this still cannot see: an empty, unfocused input whose placeholder is
|
|
235
|
+
* its only text. Nothing in the tree distinguishes it from a caption, which is
|
|
236
|
+
* why a filtered view now says it is filtered rather than implying it is the
|
|
237
|
+
* whole screen.
|
|
238
|
+
*/
|
|
239
|
+
export function actsInteractive(t) {
|
|
240
|
+
if (INTERACTIVE.test(t?.type || '')) return true;
|
|
241
|
+
if (t?.value != null && t.value !== '') return true;
|
|
242
|
+
if (t?.focused) return true;
|
|
243
|
+
if (t?.enabled === false) return true;
|
|
244
|
+
return false;
|
|
245
|
+
}
|
|
246
|
+
|
|
164
247
|
export function rowsFor(entry, { screen, filter, interactive, all = false, limit = DEFAULT_LIMIT } = {}) {
|
|
165
248
|
let kept = (entry?.targets ?? []).map((t) => ({ ...t })).filter((t) => {
|
|
166
249
|
if (!isNum(t.x) || !isNum(t.y)) return false;
|
|
167
250
|
// Off-screen elements are real in the tree and untappable in fact.
|
|
168
|
-
if (
|
|
251
|
+
if (regions.offViewport(t, screen)) return false;
|
|
169
252
|
if (!all && HIDDEN_REGIONS.has(t.region)) return false;
|
|
170
253
|
if (!all && isNoise(t)) return false;
|
|
171
254
|
return true;
|
|
@@ -181,7 +264,7 @@ export function rowsFor(entry, { screen, filter, interactive, all = false, limit
|
|
|
181
264
|
kept = kept.filter((t) =>
|
|
182
265
|
[t.label, ...(t.aliases ?? [])].filter(Boolean).join(' ').toLowerCase().includes(q));
|
|
183
266
|
}
|
|
184
|
-
if (interactive) kept = kept.filter(
|
|
267
|
+
if (interactive) kept = kept.filter(actsInteractive);
|
|
185
268
|
|
|
186
269
|
const order = (t) => {
|
|
187
270
|
const i = REGION_ORDER.indexOf(t.region ?? 'content');
|
|
@@ -189,11 +272,25 @@ export function rowsFor(entry, { screen, filter, interactive, all = false, limit
|
|
|
189
272
|
};
|
|
190
273
|
kept.sort((a, b) => order(a) - order(b) || a.y - b.y || a.x - b.x);
|
|
191
274
|
|
|
275
|
+
// A band is only the keyboard if there is a keyboard in it.
|
|
276
|
+
//
|
|
277
|
+
// Region bands are positional, so on a screen with no keyboard at all the
|
|
278
|
+
// bottom band was still called `keyboard` and review-summary rows were filed
|
|
279
|
+
// under it, followed by `keyboard: 1 keys (tap by label or type directly)` —
|
|
280
|
+
// advice that is actively wrong about page content. Reported as noise on
|
|
281
|
+
// every map of two screens. Relabelled from the contents rather than the
|
|
282
|
+
// position, which is the only evidence available here.
|
|
283
|
+
const keysPresent = kept.some((t) => COLLAPSE_REGIONS.has(t.region ?? '') && isKey(t));
|
|
284
|
+
const bandOf = (t) => {
|
|
285
|
+
const region = t.region ?? 'content';
|
|
286
|
+
return COLLAPSE_REGIONS.has(region) && !keysPresent ? 'content' : region;
|
|
287
|
+
};
|
|
288
|
+
|
|
192
289
|
const rows = [];
|
|
193
290
|
const collapsed = new Map();
|
|
194
291
|
for (const t of kept) {
|
|
195
|
-
const region = t
|
|
196
|
-
if (COLLAPSE_REGIONS.has(region)) {
|
|
292
|
+
const region = bandOf(t);
|
|
293
|
+
if (COLLAPSE_REGIONS.has(region) && isKey(t)) {
|
|
197
294
|
collapsed.set(region, (collapsed.get(region) ?? 0) + 1);
|
|
198
295
|
continue;
|
|
199
296
|
}
|
|
@@ -202,9 +299,69 @@ export function rowsFor(entry, { screen, filter, interactive, all = false, limit
|
|
|
202
299
|
return { rows: rows.slice(0, limit), truncated: Math.max(0, rows.length - limit), collapsed };
|
|
203
300
|
}
|
|
204
301
|
|
|
302
|
+
/**
|
|
303
|
+
* Say when the rows below were remembered rather than looked at.
|
|
304
|
+
*
|
|
305
|
+
* Screen memory is deliberately keyed on the pixel layout hash, because a list
|
|
306
|
+
* with new rows is the same screen and re-perceiving it per step is the cost
|
|
307
|
+
* Phase 13 exists to remove. That is right for *identity* and wrong for
|
|
308
|
+
* *contents*, and the map made no distinction: a field's text reaches a row as
|
|
309
|
+
* an OCR alias, so a recalled map reports the text the field held when the map
|
|
310
|
+
* was built. Reported from a real session — a picker described the previous
|
|
311
|
+
* sheet's options and did it in 23 ms, which is the giveaway, because 23 ms is
|
|
312
|
+
* not enough time to have looked.
|
|
313
|
+
*
|
|
314
|
+
* This does not fix that. It stops it being invisible, which is the part that
|
|
315
|
+
* cost two wrong conclusions about an app.
|
|
316
|
+
*
|
|
317
|
+
* Only past a second, because a map built by this very call is not a
|
|
318
|
+
* recollection and saying so on every screen is how a real warning gets
|
|
319
|
+
* skimmed.
|
|
320
|
+
*/
|
|
321
|
+
export const RECALL_NOTE_FLOOR_MS = 1000;
|
|
322
|
+
|
|
323
|
+
export function recalledNote(identity, now = Date.now()) {
|
|
324
|
+
const at = identity?.entry?.at;
|
|
325
|
+
if (!Number.isFinite(at)) return null;
|
|
326
|
+
const age = now - at;
|
|
327
|
+
if (age < RECALL_NOTE_FLOOR_MS) return null;
|
|
328
|
+
const ago = age < 60_000 ? `${Math.round(age / 1000)}s` : `${Math.round(age / 60_000)}m`;
|
|
329
|
+
return `elements recalled from ${ago} ago — pass refresh for what is there now`;
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
/**
|
|
333
|
+
* What a control *contains*, from the sensor that actually knows.
|
|
334
|
+
*
|
|
335
|
+
* A field's text reached a row only as the OCR alias, which means it was as old
|
|
336
|
+
* as the map and had no authoritative source at all. Reported from a real
|
|
337
|
+
* session: an `assert` on a field's contents failed against a field that did
|
|
338
|
+
* contain the string, the operator retyped, and the field ended up with a
|
|
339
|
+
* doubled value and a validation error. A character counter read `0/1000` in
|
|
340
|
+
* the map and `56/1000` in a screenshot of the same frame.
|
|
341
|
+
*
|
|
342
|
+
* The accessibility tree carries `value` and always has —
|
|
343
|
+
* `input.elementToNode` sets it on every node — and the renderer simply never
|
|
344
|
+
* printed it. Printed as `= <value>` and *alongside* the OCR alias rather than
|
|
345
|
+
* instead of it, so when the two disagree that is visible instead of resolved
|
|
346
|
+
* by whichever one the renderer preferred. Disagreement is the signal.
|
|
347
|
+
*
|
|
348
|
+
* Skipped when the label already says it, which is most switches and rows: iOS
|
|
349
|
+
* labels a settings row "Larger Text, Off" and printing `= Off` after that is
|
|
350
|
+
* noise.
|
|
351
|
+
*/
|
|
352
|
+
function valueNote(r) {
|
|
353
|
+
if (r.value == null || r.value === '') return null;
|
|
354
|
+
const v = trim(String(r.value));
|
|
355
|
+
if (!v) return null;
|
|
356
|
+
const said = alnum(r.label);
|
|
357
|
+
if (said && alnum(v) && said.includes(alnum(v))) return null;
|
|
358
|
+
return `= ${v}`;
|
|
359
|
+
}
|
|
360
|
+
|
|
205
361
|
function renderRow(r) {
|
|
206
362
|
const name = [
|
|
207
363
|
trim(r.label) || (matching.isAxTarget(r) ? '(unlabelled)' : '(no text)'),
|
|
364
|
+
valueNote(r),
|
|
208
365
|
aliasNote(r),
|
|
209
366
|
].filter(Boolean).join(' ');
|
|
210
367
|
const state = [
|
|
@@ -296,6 +453,41 @@ export async function screenMap(deviceQuery, {
|
|
|
296
453
|
// screen is what the agent can actually use: it says how much of this screen
|
|
297
454
|
// the graph can navigate from without being told.
|
|
298
455
|
const exits = node ? node.edges.length : null;
|
|
456
|
+
// Not just how many — which. The graph has always known what worked here and
|
|
457
|
+
// only ever reported a count, so an agent on a screen simframe had driven ten
|
|
458
|
+
// times still read it to learn what was tappable.
|
|
459
|
+
const remembered = node ? graph.exitsOf(node) : [];
|
|
460
|
+
// Memory intersected with what is actually here, never memory alone.
|
|
461
|
+
//
|
|
462
|
+
// This is the correction to the feature above, and it was reported with the
|
|
463
|
+
// consequence spelled out. A wizard's read-only *review* screen had been given
|
|
464
|
+
// the same identity as its step 1, so it inherited step 1's entire vocabulary:
|
|
465
|
+
// the map offered `tap "APPLY"`, `tap "No Power"`, `tap "PLACE A SERVICE
|
|
466
|
+
// REQUEST"` — **not one of which exists on it** — while the hint said "nothing
|
|
467
|
+
// ambiguous, chain the next steps without looking again". The only control on
|
|
468
|
+
// that screen files a real work order. Confident advice pointed at a
|
|
469
|
+
// destructive button on a screen it had misidentified.
|
|
470
|
+
//
|
|
471
|
+
// The line was right and the identity was wrong, so the line now checks. A
|
|
472
|
+
// remembered action is only offered when its label is on the screen in front
|
|
473
|
+
// of us; the rest are counted and reported as a disagreement, because memory
|
|
474
|
+
// that does not match what is here is itself the most useful thing to say.
|
|
475
|
+
const { exitList, stale: staleExits } = presentOnly(remembered, rows);
|
|
476
|
+
// Values simframe wrote itself and can no longer see. Computed from the same
|
|
477
|
+
// rows the map is about to print, so it costs nothing, and it is the only
|
|
478
|
+
// thing that can notice a form being cleared underneath a caller — six `ok`
|
|
479
|
+
// calls in a row hid exactly that.
|
|
480
|
+
const cleared = wrote.missingLine(wrote.missing(device?.udid, rows));
|
|
481
|
+
// One layer covering another, detected where it shows: an ax element and an
|
|
482
|
+
// OCR word at one coordinate disagreeing about what is there. Reported by a
|
|
483
|
+
// peer who had to fall back to a screenshot to count five radio options
|
|
484
|
+
// through a sheet, which is the case the text map exists to remove.
|
|
485
|
+
const layered = (identity?.entry?.occluded ?? []).length;
|
|
486
|
+
const overlay = layered
|
|
487
|
+
? `${layered} element(s) on this screen overlap and disagree about what is there`
|
|
488
|
+
+ ' — a sheet or overlay is probably covering the screen behind it, so treat anything'
|
|
489
|
+
+ ' you did not expect to see as belonging to the layer underneath'
|
|
490
|
+
: null;
|
|
299
491
|
|
|
300
492
|
return {
|
|
301
493
|
device,
|
|
@@ -303,14 +495,202 @@ export async function screenMap(deviceQuery, {
|
|
|
303
495
|
rows,
|
|
304
496
|
truncated,
|
|
305
497
|
collapsed,
|
|
498
|
+
// A filtered view is not the screen, and the hint used to report its count
|
|
499
|
+
// as though it were: `--interactive` on a form said "1 element; nothing
|
|
500
|
+
// ambiguous — chain the next steps" while two required inputs sat unseen
|
|
501
|
+
// below it. The over-claim was the harmful half, not the filter.
|
|
502
|
+
filtered: Boolean(filter || interactive),
|
|
306
503
|
screen,
|
|
307
504
|
name,
|
|
308
505
|
exits,
|
|
309
|
-
|
|
506
|
+
exitList,
|
|
507
|
+
staleExits,
|
|
508
|
+
cleared,
|
|
509
|
+
overlay,
|
|
510
|
+
text: render({ device, identity, rows, truncated, collapsed, screen, name, exits, exitList, staleExits, cleared, overlay }),
|
|
511
|
+
};
|
|
512
|
+
}
|
|
513
|
+
|
|
514
|
+
/**
|
|
515
|
+
* One line saying whether the model needs to stop and think.
|
|
516
|
+
*
|
|
517
|
+
* The measured loop in a real session is observe → think → tap → observe →
|
|
518
|
+
* think, and the thinking dominates wall time. Phases 11–16 reduce how *often*
|
|
519
|
+
* a decision has to reach the model; this reduces how often the model *believes*
|
|
520
|
+
* it has to decide. Measured: 48 of 62 real calls were three steps or fewer, so
|
|
521
|
+
* a twelve-step flow arrived as four or five calls and every boundary was a
|
|
522
|
+
* think — not because anything was ambiguous, but because nothing said it was
|
|
523
|
+
* not.
|
|
524
|
+
*
|
|
525
|
+
* Everything here is already in hand when the result is assembled: whether the
|
|
526
|
+
* flow stopped, whether the screen settled, whether the graph recognises it,
|
|
527
|
+
* how many elements there are, and whether any two of them answer to the same
|
|
528
|
+
* label. No perception pass, no model call, no new state.
|
|
529
|
+
*
|
|
530
|
+
* The order is deliberate. It reports the *strongest reason to think* first,
|
|
531
|
+
* and only says "keep going" when it can rule all of them out — a hint that
|
|
532
|
+
* cheerfully says "carry on" into an unknown screen would be worse than no hint
|
|
533
|
+
* at all.
|
|
534
|
+
*/
|
|
535
|
+
export function nextHint({ ok, escalated, settled, loading, known, hash, exits, elements, ambiguous, filtered, exitList, staleExits } = {}) {
|
|
536
|
+
if (ok === false) {
|
|
537
|
+
return 'next: the flow stopped here — this is the moment to think. sim_recall shows how you got here; sim_ui re-reads the screen.';
|
|
538
|
+
}
|
|
539
|
+
// A stopped flow and a completed flow carrying a soft verdict are different
|
|
540
|
+
// things, and conflating them printed `flow completed — 16/16 steps` directly
|
|
541
|
+
// above `next: the flow stopped here` on a run where nothing stopped.
|
|
542
|
+
//
|
|
543
|
+
// The conflation was load-bearing, not cosmetic. `no-visible-change` is an
|
|
544
|
+
// escalating verdict and it fires falsely — reported three rounds running —
|
|
545
|
+
// so **one** wrong verdict anywhere in a clean flow told the agent to abandon
|
|
546
|
+
// batching and re-read. That is the exact failure this hint exists to
|
|
547
|
+
// prevent, caused by the hint.
|
|
548
|
+
//
|
|
549
|
+
// A completed flow with an unconfirmed step is worth one targeted look, not a
|
|
550
|
+
// re-plan, so the hint says which and keeps the horizon open.
|
|
551
|
+
if (escalated) {
|
|
552
|
+
return 'next: every step ran, but at least one could not be confirmed — check that one thing landed (re-read the field, or assert it) rather than re-planning the flow.';
|
|
553
|
+
}
|
|
554
|
+
// A screen awaiting a network call is *settled* — nothing is moving — and
|
|
555
|
+
// incomplete. Reported: a settle returned satisfied while a list was still
|
|
556
|
+
// loading, the map showed an empty content region, and an empty region and a
|
|
557
|
+
// still-loading one produced identical output. A person sees a spinner.
|
|
558
|
+
if (loading) {
|
|
559
|
+
return 'next: settled, but the transition classifier still sees loading — an empty-looking region may be a list that has not arrived. waitFor a string you expect rather than acting on this.';
|
|
560
|
+
}
|
|
561
|
+
if (settled === false) {
|
|
562
|
+
return 'next: the screen is still moving. sim_state polls it for a fraction of a map; do not act on this reading yet.';
|
|
563
|
+
}
|
|
564
|
+
if (known === false) {
|
|
565
|
+
return 'next: new screen, nothing predicted here yet — read it before acting on a label you have not seen on it.';
|
|
566
|
+
}
|
|
567
|
+
if (ambiguous > 0) {
|
|
568
|
+
return ambiguous === 1
|
|
569
|
+
? 'next: one label repeats on this screen — address that one by #ref, and the rest can go in one sim_do.'
|
|
570
|
+
: `next: ${ambiguous} labels repeat on this screen — address those by #ref, and the rest can go in one sim_do.`;
|
|
571
|
+
}
|
|
572
|
+
const known_ = hash ? `known (${hash.slice(0, 8)}${exits ? `, ${exits} known exit${exits === 1 ? '' : 's'}` : ''})` : 'known';
|
|
573
|
+
if (filtered) {
|
|
574
|
+
// The count belongs to the filter, not to the screen. An agent that reads
|
|
575
|
+
// it as the screen concludes a form has nothing to fill in.
|
|
576
|
+
return `next: settled; screen ${known_}; ${elements} element${elements === 1 ? '' : 's'} **matching your filter** — this is not the whole screen, and an empty text input can look like a caption. Read it unfiltered before concluding something is absent.`;
|
|
577
|
+
}
|
|
578
|
+
// Memory that contradicts the screen outranks "carry on", because the reason
|
|
579
|
+
// it contradicts is usually that this screen has been confused with another —
|
|
580
|
+
// and a confident "chain without looking again" on a misidentified screen is
|
|
581
|
+
// how remembered advice ends up pointing at a control that files a work order.
|
|
582
|
+
const offerable = (exitList ?? []).filter((e) => e.label);
|
|
583
|
+
if (!offerable.length && staleExits) {
|
|
584
|
+
return `next: this screen is recognised but ${staleExits} remembered control${staleExits === 1 ? ' is' : 's are'} not on it,`
|
|
585
|
+
+ ' so the identity is probably wrong — two screens sharing one hash. Act only on the element list, and re-read before anything irreversible.';
|
|
586
|
+
}
|
|
587
|
+
// Naming the vocabulary is what makes "chain" actionable. A hint that says
|
|
588
|
+
// "chain the next steps" without saying what the steps could be is asking the
|
|
589
|
+
// agent to plan from a map it has to keep re-reading.
|
|
590
|
+
const vocab = offerable.slice(0, 6)
|
|
591
|
+
.map((e) => `${e.action} ${JSON.stringify(String(e.label).slice(0, 28))}`).join(', ');
|
|
592
|
+
return `next: settled; screen ${known_}; ${elements} element${elements === 1 ? '' : 's'}; nothing ambiguous — chain the next steps in one sim_do without looking again.`
|
|
593
|
+
+ (vocab ? ` Known to work here: ${vocab}.` : '')
|
|
594
|
+
+ (staleExits ? ` (${staleExits} other remembered control${staleExits === 1 ? '' : 's'} not on this screen — the graph may be conflating it with another.)` : '');
|
|
595
|
+
}
|
|
596
|
+
|
|
597
|
+
/**
|
|
598
|
+
* The hint for a rendered map, from the map itself.
|
|
599
|
+
*
|
|
600
|
+
* Lives here rather than in the MCP server because it was only reachable from
|
|
601
|
+
* there, and the MCP server is a long-lived process: a session that started
|
|
602
|
+
* before a change is running the old code, so the headline change of a phase
|
|
603
|
+
* could not be exercised at all. Reported, correctly, as the first finding
|
|
604
|
+
* against Phase 11.5. In `view.js` both front ends share one implementation and
|
|
605
|
+
* a unit test can reach it.
|
|
606
|
+
*/
|
|
607
|
+
export function hintFor(map, { flowOk = true, escalated = false } = {}) {
|
|
608
|
+
return nextHint({
|
|
609
|
+
ok: flowOk !== false,
|
|
610
|
+
escalated: Boolean(escalated),
|
|
611
|
+
filtered: map?.filtered === true,
|
|
612
|
+
settled: map?.identity?.settled !== false,
|
|
613
|
+
loading: map?.identity?.loading === true,
|
|
614
|
+
known: map?.exits != null,
|
|
615
|
+
hash: map?.identity?.hash ?? null,
|
|
616
|
+
exits: map?.exits ?? 0,
|
|
617
|
+
exitList: map?.exitList ?? [],
|
|
618
|
+
staleExits: map?.staleExits ?? 0,
|
|
619
|
+
elements: map?.rows?.length ?? 0,
|
|
620
|
+
ambiguous: ambiguousLabels(map?.rows),
|
|
621
|
+
});
|
|
622
|
+
}
|
|
623
|
+
|
|
624
|
+
/**
|
|
625
|
+
* Keep only the remembered actions whose control is actually on this screen.
|
|
626
|
+
*
|
|
627
|
+
* @returns {{exitList: Array, stale: number}} what can be offered, and how many
|
|
628
|
+
* remembered actions found nothing here — which is evidence the screen has
|
|
629
|
+
* been misidentified, and worth saying out loud.
|
|
630
|
+
*/
|
|
631
|
+
export function presentOnly(remembered, rows) {
|
|
632
|
+
const here = new Set();
|
|
633
|
+
for (const r of rows ?? []) {
|
|
634
|
+
for (const name of [r.label, ...(r.aliases ?? [])]) {
|
|
635
|
+
const k = alnum(name);
|
|
636
|
+
if (k) here.add(k);
|
|
637
|
+
}
|
|
638
|
+
}
|
|
639
|
+
const has = (label) => {
|
|
640
|
+
const k = alnum(label);
|
|
641
|
+
if (!k) return false;
|
|
642
|
+
if (here.has(k)) return true;
|
|
643
|
+
// A row may carry the label inside a longer one — a list card concatenates
|
|
644
|
+
// its children, and truncation adds an ellipsis.
|
|
645
|
+
for (const seen of here) if (seen.includes(k) || k.includes(seen)) return true;
|
|
646
|
+
return false;
|
|
310
647
|
};
|
|
648
|
+
const exitList = (remembered ?? []).filter((e) => has(e.label));
|
|
649
|
+
return { exitList, stale: (remembered ?? []).length - exitList.length };
|
|
650
|
+
}
|
|
651
|
+
|
|
652
|
+
/**
|
|
653
|
+
* What has worked from this screen before, printed rather than counted.
|
|
654
|
+
*
|
|
655
|
+
* The map said `(known, 3 known exits)` and stopped there, so the graph's own
|
|
656
|
+
* vocabulary never reached the caller. A flow whose labels were known in
|
|
657
|
+
* advance ran 16 steps in one call; the same agent on screens the graph also
|
|
658
|
+
* knew, but whose labels it had to rediscover, spent 25 calls on 31 steps.
|
|
659
|
+
*
|
|
660
|
+
* Deliberately terse and deliberately *not* a promise. These are actions that
|
|
661
|
+
* previously worked here, with how often — evidence for a plan, not a
|
|
662
|
+
* guarantee, and the destructive-label rules apply to them exactly as before.
|
|
663
|
+
*/
|
|
664
|
+
export function exitsLine(exitList, { limit = 6, stale = 0 } = {}) {
|
|
665
|
+
const list = (exitList ?? []).filter((e) => e.label).slice(0, limit);
|
|
666
|
+
if (!list.length) {
|
|
667
|
+
// Everything remembered here is missing. That is not "no memory" — it is
|
|
668
|
+
// memory that contradicts the screen, which usually means two screens have
|
|
669
|
+
// collapsed into one identity, and it is the most useful thing to say.
|
|
670
|
+
return stale
|
|
671
|
+
? `memory disagrees with this screen: ${stale} remembered control${stale === 1 ? '' : 's'} not present`
|
|
672
|
+
+ ' — this screen has probably been confused with another. Trust the element list, not the graph.'
|
|
673
|
+
: null;
|
|
674
|
+
}
|
|
675
|
+
const parts = list.map((e) => {
|
|
676
|
+
const label = String(e.label).length > 28 ? `${String(e.label).slice(0, 28)}…` : String(e.label);
|
|
677
|
+
return `${e.action} ${JSON.stringify(label)}${e.count > 1 ? ` (${e.count}x)` : ''}`;
|
|
678
|
+
});
|
|
679
|
+
return `worked here before: ${parts.join(', ')}`;
|
|
680
|
+
}
|
|
681
|
+
|
|
682
|
+
/** How many labels are worn by more than one element a caller could act on. */
|
|
683
|
+
export function ambiguousLabels(rows) {
|
|
684
|
+
const seen = new Map();
|
|
685
|
+
for (const r of rows ?? []) {
|
|
686
|
+
const key = alnum(r.label);
|
|
687
|
+
if (!key) continue;
|
|
688
|
+
seen.set(key, (seen.get(key) ?? 0) + 1);
|
|
689
|
+
}
|
|
690
|
+
return [...seen.values()].filter((n) => n > 1).length;
|
|
311
691
|
}
|
|
312
692
|
|
|
313
|
-
export function render({ device, identity, rows, truncated, collapsed, screen, name, exits, verdictLine, ambiguities }) {
|
|
693
|
+
export function render({ device, identity, rows, truncated, collapsed, screen, name, exits, exitList, staleExits, verdictLine, ambiguities, cleared, overlay }) {
|
|
314
694
|
const head = [
|
|
315
695
|
device?.name,
|
|
316
696
|
screen?.width ? `${screen.width}x${screen.height}pt` : null,
|
|
@@ -320,10 +700,19 @@ export function render({ device, identity, rows, truncated, collapsed, screen, n
|
|
|
320
700
|
: 'screen unidentified',
|
|
321
701
|
identity?.keyboard ? 'keyboard up' : null,
|
|
322
702
|
identity?.settled === false ? 'STILL MOVING' : null,
|
|
703
|
+
// Still and finished are not the same thing.
|
|
704
|
+
identity?.loading === true ? 'STILL LOADING' : null,
|
|
705
|
+
recalledNote(identity),
|
|
323
706
|
].filter(Boolean).join(' · ');
|
|
324
707
|
|
|
325
708
|
const lines = [head];
|
|
326
709
|
if (verdictLine) lines.push(verdictLine);
|
|
710
|
+
// Above the element list, not below it: it contradicts something the caller
|
|
711
|
+
// already believes, which is the one kind of news that must not be scrolled to.
|
|
712
|
+
if (cleared) lines.push(cleared);
|
|
713
|
+
if (overlay) lines.push(overlay);
|
|
714
|
+
const worked = exitsLine(exitList, { stale: staleExits });
|
|
715
|
+
if (worked) lines.push(worked);
|
|
327
716
|
|
|
328
717
|
let region = null;
|
|
329
718
|
for (const r of rows) {
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The words simframe may not act on by itself.
|
|
3
|
+
*
|
|
4
|
+
* CLAUDE.md has required this since the human-parity series was written — *"a
|
|
5
|
+
* reflex never taps anything whose label matches the destructive vocabulary
|
|
6
|
+
* (Delete, Remove, Pay, Send, Sign out, Reset…). Add the list to the same data
|
|
7
|
+
* file."* — and several of my own comments already talk about "the destructive
|
|
8
|
+
* vocabulary" as though it existed. It did not. This is it.
|
|
9
|
+
*
|
|
10
|
+
* **What it gates, precisely.** This restricts what simframe does *on its own
|
|
11
|
+
* initiative*: a retry, an alternative selector, an exploration step, a reflex,
|
|
12
|
+
* a speculative tap. It never restricts what the caller explicitly asked for.
|
|
13
|
+
* `{"tap": "DELETE ACCOUNT"}` is a request and is honoured; substituting
|
|
14
|
+
* "DELETE ACCOUNT" for a "Done" that did not resolve is not, and that is the
|
|
15
|
+
* whole distinction. Getting it backwards would make the tool refuse the thing
|
|
16
|
+
* a tester most needs to test.
|
|
17
|
+
*
|
|
18
|
+
* Data, not code, and locale-keyed, so another language is a file rather than a
|
|
19
|
+
* release.
|
|
20
|
+
*/
|
|
21
|
+
import fs from 'node:fs';
|
|
22
|
+
import path from 'node:path';
|
|
23
|
+
import { fileURLToPath } from 'node:url';
|
|
24
|
+
|
|
25
|
+
const HERE = path.dirname(fileURLToPath(import.meta.url));
|
|
26
|
+
const DATA = path.join(HERE, '..', 'data', 'vocabulary');
|
|
27
|
+
|
|
28
|
+
const cache = new Map();
|
|
29
|
+
|
|
30
|
+
/** The vocabulary for a locale, falling back to English. */
|
|
31
|
+
export function load(locale = process.env.SIMFRAME_LOCALE || 'en') {
|
|
32
|
+
const key = String(locale).toLowerCase().split(/[-_]/)[0];
|
|
33
|
+
if (cache.has(key)) return cache.get(key);
|
|
34
|
+
let data = null;
|
|
35
|
+
for (const candidate of [key, 'en']) {
|
|
36
|
+
try {
|
|
37
|
+
data = JSON.parse(fs.readFileSync(path.join(DATA, `${candidate}.json`), 'utf8'));
|
|
38
|
+
break;
|
|
39
|
+
} catch { /* try the fallback */ }
|
|
40
|
+
}
|
|
41
|
+
// A missing file must not silently disable the barrier. An empty vocabulary
|
|
42
|
+
// would make every label safe, which is the wrong direction to fail in, so
|
|
43
|
+
// this throws rather than returning nothing.
|
|
44
|
+
if (!data) throw new Error(`no vocabulary for "${locale}" and no en fallback in ${DATA}`);
|
|
45
|
+
cache.set(key, data);
|
|
46
|
+
return data;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
const alnum = (s) => String(s ?? '').toLowerCase().replace(/[^\p{L}\p{N}]+/gu, ' ').trim();
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Does this phrase occur in the label as whole words?
|
|
53
|
+
*
|
|
54
|
+
* Word boundaries, not substrings, and the reason is a real screen: an app's
|
|
55
|
+
* "Work Orders" tab contains the letters of "order", and a substring match would
|
|
56
|
+
* make its main navigation untouchable by anything local. Matching whole words
|
|
57
|
+
* means "order" does not match "orders", which is the behaviour wanted here —
|
|
58
|
+
* an exact tappable word is the signal, and a longer word is a different word.
|
|
59
|
+
*/
|
|
60
|
+
function saysPhrase(label, phrase) {
|
|
61
|
+
const l = ` ${alnum(label)} `;
|
|
62
|
+
const p = alnum(phrase);
|
|
63
|
+
if (!p) return false;
|
|
64
|
+
return l.includes(` ${p} `);
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* May simframe act on this label on its own initiative?
|
|
69
|
+
*
|
|
70
|
+
* `purpose` matters, and conflating two purposes cost a real run. The default,
|
|
71
|
+
* `substitute`, answers *"may a retry aim at this instead?"* and permits
|
|
72
|
+
* "Cancel", deliberately, so a local tier can decline a dialog rather than
|
|
73
|
+
* stranding on every confirmation it meets.
|
|
74
|
+
*
|
|
75
|
+
* `explore` answers a different question — *"may I open this as a door and see
|
|
76
|
+
* what is behind it?"* — and there "Cancel" is the abandon-this-task control.
|
|
77
|
+
* `seek` asked the first question and got the first answer: it opened CANCEL
|
|
78
|
+
* first, then AI TROUBLESHOOTING, then pressed "YES, THIS FIXED MY PROBLEM",
|
|
79
|
+
* ending five screens deep in a live support chat with a half-built service
|
|
80
|
+
* request destroyed. One label further along was SUBMIT SERVICE REQUEST.
|
|
81
|
+
*
|
|
82
|
+
* So exploration has its own list and it errs toward refusing. A door missed
|
|
83
|
+
* costs one step of a bounded budget; a door taken wrongly costs the run, and
|
|
84
|
+
* can cost the thing being tested.
|
|
85
|
+
*
|
|
86
|
+
* @returns {{allowed: boolean, reason?: string, matched?: string}}
|
|
87
|
+
*/
|
|
88
|
+
export function mayActLocally(label, { locale, purpose = 'substitute' } = {}) {
|
|
89
|
+
const text = String(label ?? '').trim();
|
|
90
|
+
if (!text) return { allowed: purpose !== 'explore' };
|
|
91
|
+
const vocab = load(locale);
|
|
92
|
+
|
|
93
|
+
if (purpose === 'explore') {
|
|
94
|
+
// Checked before the "listed as safe" exemption below, which exists for
|
|
95
|
+
// declining dialogs and must never make something a door.
|
|
96
|
+
for (const word of vocab.exploration?.neverOpen ?? []) {
|
|
97
|
+
if (saysPhrase(text, word)) {
|
|
98
|
+
return { allowed: false, reason: 'not a door — it commits, abandons or answers', matched: word };
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
for (const pattern of vocab.exploration?.neverOpenPatterns ?? []) {
|
|
102
|
+
if (new RegExp(pattern, 'i').test(alnum(text)) || new RegExp(pattern, 'i').test(text.toLowerCase())) {
|
|
103
|
+
return { allowed: false, reason: 'not a door — it reads as an instruction or an answer', matched: pattern };
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
// Exceptions first, and matched against the **whole** label rather than as a
|
|
109
|
+
// phrase inside it. "Cancel" is how you *decline* a dialog and a barrier that
|
|
110
|
+
// refused it would strand a local tier on every confirmation it met — but
|
|
111
|
+
// "Cancel order" is a different act, and a phrase match would have waved it
|
|
112
|
+
// through on the strength of its first word.
|
|
113
|
+
const whole = alnum(text);
|
|
114
|
+
for (const ok of vocab.destructive?.notWords?.words ?? []) {
|
|
115
|
+
if (whole === alnum(ok)) return { allowed: true, reason: 'listed as safe', matched: ok };
|
|
116
|
+
}
|
|
117
|
+
for (const word of vocab.destructive?.words ?? []) {
|
|
118
|
+
if (saysPhrase(text, word)) {
|
|
119
|
+
return { allowed: false, reason: 'destructive vocabulary', matched: word };
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
for (const word of vocab.leavesTheApp?.words ?? []) {
|
|
123
|
+
if (saysPhrase(text, word)) {
|
|
124
|
+
return { allowed: false, reason: 'leaves the app', matched: word };
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
return { allowed: true };
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
/** Convenience for a filter: keep only what a local tier may act on. */
|
|
131
|
+
export const actableLocally = (label, options) => mayActLocally(label, options).allowed;
|
|
132
|
+
|
|
133
|
+
/** May exploration open this as a door? Stricter than substitution, on purpose. */
|
|
134
|
+
export const openableAsDoor = (label, options) => mayActLocally(label, { ...options, purpose: 'explore' }).allowed;
|