simframe 0.10.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +137 -2
- package/data/vocabulary/en.json +148 -0
- package/native/ocr.swift +13 -1
- package/native/rank.swift +87 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +43 -3
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
- package/native/simframed/Sources/simframed/main.swift +13 -1
- package/native/supervise.swift +181 -0
- package/package.json +4 -1
- package/scripts/check-package.mjs +22 -2
- package/scripts/ci-memory.mjs +104 -20
- package/scripts/eval-perception.mjs +33 -0
- package/scripts/phase17-corpus.mjs +176 -0
- package/skills/simframe/SKILL.md +237 -5
- package/src/actions.js +1692 -44
- package/src/cli.js +131 -9
- package/src/control.js +1 -0
- package/src/fingerprint.js +19 -1
- package/src/graph.js +89 -7
- package/src/index.js +211 -14
- package/src/input.js +66 -3
- package/src/localhelper.js +155 -0
- package/src/matching.js +64 -1
- package/src/mcp.js +319 -27
- package/src/metrics.js +32 -3
- package/src/ocr.js +18 -1
- package/src/planner.js +195 -0
- package/src/platform/android.js +2 -1
- package/src/platform/ios.js +2 -1
- package/src/png.js +26 -0
- package/src/refs.js +51 -8
- package/src/regions.js +110 -1
- package/src/screenmap.js +89 -9
- package/src/supervisor.js +117 -0
- package/src/view.js +335 -7
- package/src/vocabulary.js +134 -0
- package/src/wrote.js +136 -0
package/src/view.js
CHANGED
|
@@ -12,6 +12,8 @@
|
|
|
12
12
|
// app published and one OCR read off the pixels deserve different amounts of
|
|
13
13
|
// trust.
|
|
14
14
|
import * as api from './index.js';
|
|
15
|
+
import * as regions from './regions.js';
|
|
16
|
+
import * as wrote from './wrote.js';
|
|
15
17
|
import * as graph from './graph.js';
|
|
16
18
|
import { writeRefs } from './refs.js';
|
|
17
19
|
import * as matching from './matching.js';
|
|
@@ -160,12 +162,93 @@ const trim = (text) => {
|
|
|
160
162
|
return one.length > MAX_LABEL ? `${one.slice(0, MAX_LABEL - 1)}…` : one;
|
|
161
163
|
};
|
|
162
164
|
|
|
163
|
-
/**
|
|
165
|
+
/**
|
|
166
|
+
* Why no row is dropped for being long — reverted 2026-09-10, same day.
|
|
167
|
+
*
|
|
168
|
+
* There was a rule here that dropped non-interactive rows whose label ran past
|
|
169
|
+
* 45 characters, on the evidence that four of sixteen rows on a Settings screen
|
|
170
|
+
* were the explanatory paragraph under each switch: 31% of that map's
|
|
171
|
+
* characters describing things nobody can tap. Across fourteen recorded screens
|
|
172
|
+
* it cut the map 15%.
|
|
173
|
+
*
|
|
174
|
+
* It was wrong, and the counter-example is decisive. A React Native list card
|
|
175
|
+
* exposes all of its children as one concatenated accessibility label —
|
|
176
|
+
* `Anaheim | Store # 1020, , 1234 Main St, … | Quick Casual Restaurant`, 105
|
|
177
|
+
* characters, type `GenericElement`, region `content`. Every property my rule
|
|
178
|
+
* tested is identical to the Settings caption's, and that card is *the only
|
|
179
|
+
* tappable thing on the screen*.
|
|
180
|
+
*
|
|
181
|
+
* Nothing became untappable — `locate`, `assert` and `waitFor` read
|
|
182
|
+
* `entry.targets`, so the rows are a view and the data survived. What was lost
|
|
183
|
+
* is **discovery**: the map stopped saying what was on screen, so an agent
|
|
184
|
+
* could only tap labels it already knew, and the fallback on a screen of
|
|
185
|
+
* unknown data is a ~1600-token screenshot. It saved characters on
|
|
186
|
+
* settings-shaped screens and spent an image on list-shaped ones, which is the
|
|
187
|
+
* exact cost the phase existed to remove. Blast radius: every data list in
|
|
188
|
+
* every RN app.
|
|
189
|
+
*
|
|
190
|
+
* The lesson is not "find a better discriminator". It is that this was a
|
|
191
|
+
* threshold shipped with no harness case that could catch its failure, in the
|
|
192
|
+
* same session as building the harness. So `expect.discoverable` now exists,
|
|
193
|
+
* and there is a fixture of the reported shape — a rule like this may return
|
|
194
|
+
* only when it can be gated.
|
|
195
|
+
*
|
|
196
|
+
* What survives is what the reporter suggested instead: truncate, do not drop.
|
|
197
|
+
* Position and tappability are the valuable parts of a row, not the full text.
|
|
198
|
+
* See `MAX_LABEL`.
|
|
199
|
+
*/
|
|
200
|
+
|
|
201
|
+
/**
|
|
202
|
+
* Whether a target in the keyboard band really is a key.
|
|
203
|
+
*
|
|
204
|
+
* Region bands are positional, and this was the fourth and fifth bug they
|
|
205
|
+
* produced. With the keyboard up, the bottom band is collapsed to one line —
|
|
206
|
+
* thirty keys nobody names — and a primary action pinned above the keyboard was
|
|
207
|
+
* collapsed with them: four reads running printed `keyboard: 6 keys` and **no
|
|
208
|
+
* forward control**, while the hint said "nothing ambiguous — chain the next
|
|
209
|
+
* steps without looking again". That it was an emission bug and not a
|
|
210
|
+
* perception one was proved by the next call, which hit the button instantly at
|
|
211
|
+
* a coordinate the map had never printed.
|
|
212
|
+
*
|
|
213
|
+
* Delegates to `regions.looksLikeKey`, which is the canonical test. These two
|
|
214
|
+
* having separate copies is what let a phantom keyboard survive in the
|
|
215
|
+
* fingerprint after it had already been fixed in the map — and there it was
|
|
216
|
+
* deleting screens' content from their own identity.
|
|
217
|
+
*/
|
|
218
|
+
export const isKey = regions.looksLikeKey;
|
|
219
|
+
|
|
220
|
+
/**
|
|
221
|
+
* Whether a target can be acted on, by role *or* by evidence.
|
|
222
|
+
*
|
|
223
|
+
* The role alone was wrong twice on real forms. A React Native composite select
|
|
224
|
+
* surfaces as a generic element, and a text input shows only its placeholder as
|
|
225
|
+
* `StaticText` — so `--interactive` answered "1 element" on a form with two
|
|
226
|
+
* visible, bordered, *required* inputs and an agent concluded there was nothing
|
|
227
|
+
* to fill in.
|
|
228
|
+
*
|
|
229
|
+
* Evidence is used rather than a longer list of role names, because the roles
|
|
230
|
+
* are what the tree got wrong. Only a control carries a `value`, only a
|
|
231
|
+
* focusable thing is `focused`, and `enabled` is a state a caption never
|
|
232
|
+
* declares. A generic element that is none of those really is a container.
|
|
233
|
+
*
|
|
234
|
+
* The case this still cannot see: an empty, unfocused input whose placeholder is
|
|
235
|
+
* its only text. Nothing in the tree distinguishes it from a caption, which is
|
|
236
|
+
* why a filtered view now says it is filtered rather than implying it is the
|
|
237
|
+
* whole screen.
|
|
238
|
+
*/
|
|
239
|
+
export function actsInteractive(t) {
|
|
240
|
+
if (INTERACTIVE.test(t?.type || '')) return true;
|
|
241
|
+
if (t?.value != null && t.value !== '') return true;
|
|
242
|
+
if (t?.focused) return true;
|
|
243
|
+
if (t?.enabled === false) return true;
|
|
244
|
+
return false;
|
|
245
|
+
}
|
|
246
|
+
|
|
164
247
|
export function rowsFor(entry, { screen, filter, interactive, all = false, limit = DEFAULT_LIMIT } = {}) {
|
|
165
248
|
let kept = (entry?.targets ?? []).map((t) => ({ ...t })).filter((t) => {
|
|
166
249
|
if (!isNum(t.x) || !isNum(t.y)) return false;
|
|
167
250
|
// Off-screen elements are real in the tree and untappable in fact.
|
|
168
|
-
if (
|
|
251
|
+
if (regions.offViewport(t, screen)) return false;
|
|
169
252
|
if (!all && HIDDEN_REGIONS.has(t.region)) return false;
|
|
170
253
|
if (!all && isNoise(t)) return false;
|
|
171
254
|
return true;
|
|
@@ -181,7 +264,7 @@ export function rowsFor(entry, { screen, filter, interactive, all = false, limit
|
|
|
181
264
|
kept = kept.filter((t) =>
|
|
182
265
|
[t.label, ...(t.aliases ?? [])].filter(Boolean).join(' ').toLowerCase().includes(q));
|
|
183
266
|
}
|
|
184
|
-
if (interactive) kept = kept.filter(
|
|
267
|
+
if (interactive) kept = kept.filter(actsInteractive);
|
|
185
268
|
|
|
186
269
|
const order = (t) => {
|
|
187
270
|
const i = REGION_ORDER.indexOf(t.region ?? 'content');
|
|
@@ -189,11 +272,25 @@ export function rowsFor(entry, { screen, filter, interactive, all = false, limit
|
|
|
189
272
|
};
|
|
190
273
|
kept.sort((a, b) => order(a) - order(b) || a.y - b.y || a.x - b.x);
|
|
191
274
|
|
|
275
|
+
// A band is only the keyboard if there is a keyboard in it.
|
|
276
|
+
//
|
|
277
|
+
// Region bands are positional, so on a screen with no keyboard at all the
|
|
278
|
+
// bottom band was still called `keyboard` and review-summary rows were filed
|
|
279
|
+
// under it, followed by `keyboard: 1 keys (tap by label or type directly)` —
|
|
280
|
+
// advice that is actively wrong about page content. Reported as noise on
|
|
281
|
+
// every map of two screens. Relabelled from the contents rather than the
|
|
282
|
+
// position, which is the only evidence available here.
|
|
283
|
+
const keysPresent = kept.some((t) => COLLAPSE_REGIONS.has(t.region ?? '') && isKey(t));
|
|
284
|
+
const bandOf = (t) => {
|
|
285
|
+
const region = t.region ?? 'content';
|
|
286
|
+
return COLLAPSE_REGIONS.has(region) && !keysPresent ? 'content' : region;
|
|
287
|
+
};
|
|
288
|
+
|
|
192
289
|
const rows = [];
|
|
193
290
|
const collapsed = new Map();
|
|
194
291
|
for (const t of kept) {
|
|
195
|
-
const region = t
|
|
196
|
-
if (COLLAPSE_REGIONS.has(region)) {
|
|
292
|
+
const region = bandOf(t);
|
|
293
|
+
if (COLLAPSE_REGIONS.has(region) && isKey(t)) {
|
|
197
294
|
collapsed.set(region, (collapsed.get(region) ?? 0) + 1);
|
|
198
295
|
continue;
|
|
199
296
|
}
|
|
@@ -356,6 +453,41 @@ export async function screenMap(deviceQuery, {
|
|
|
356
453
|
// screen is what the agent can actually use: it says how much of this screen
|
|
357
454
|
// the graph can navigate from without being told.
|
|
358
455
|
const exits = node ? node.edges.length : null;
|
|
456
|
+
// Not just how many — which. The graph has always known what worked here and
|
|
457
|
+
// only ever reported a count, so an agent on a screen simframe had driven ten
|
|
458
|
+
// times still read it to learn what was tappable.
|
|
459
|
+
const remembered = node ? graph.exitsOf(node) : [];
|
|
460
|
+
// Memory intersected with what is actually here, never memory alone.
|
|
461
|
+
//
|
|
462
|
+
// This is the correction to the feature above, and it was reported with the
|
|
463
|
+
// consequence spelled out. A wizard's read-only *review* screen had been given
|
|
464
|
+
// the same identity as its step 1, so it inherited step 1's entire vocabulary:
|
|
465
|
+
// the map offered `tap "APPLY"`, `tap "No Power"`, `tap "PLACE A SERVICE
|
|
466
|
+
// REQUEST"` — **not one of which exists on it** — while the hint said "nothing
|
|
467
|
+
// ambiguous, chain the next steps without looking again". The only control on
|
|
468
|
+
// that screen files a real work order. Confident advice pointed at a
|
|
469
|
+
// destructive button on a screen it had misidentified.
|
|
470
|
+
//
|
|
471
|
+
// The line was right and the identity was wrong, so the line now checks. A
|
|
472
|
+
// remembered action is only offered when its label is on the screen in front
|
|
473
|
+
// of us; the rest are counted and reported as a disagreement, because memory
|
|
474
|
+
// that does not match what is here is itself the most useful thing to say.
|
|
475
|
+
const { exitList, stale: staleExits } = presentOnly(remembered, rows);
|
|
476
|
+
// Values simframe wrote itself and can no longer see. Computed from the same
|
|
477
|
+
// rows the map is about to print, so it costs nothing, and it is the only
|
|
478
|
+
// thing that can notice a form being cleared underneath a caller — six `ok`
|
|
479
|
+
// calls in a row hid exactly that.
|
|
480
|
+
const cleared = wrote.missingLine(wrote.missing(device?.udid, rows));
|
|
481
|
+
// One layer covering another, detected where it shows: an ax element and an
|
|
482
|
+
// OCR word at one coordinate disagreeing about what is there. Reported by a
|
|
483
|
+
// peer who had to fall back to a screenshot to count five radio options
|
|
484
|
+
// through a sheet, which is the case the text map exists to remove.
|
|
485
|
+
const layered = (identity?.entry?.occluded ?? []).length;
|
|
486
|
+
const overlay = layered
|
|
487
|
+
? `${layered} element(s) on this screen overlap and disagree about what is there`
|
|
488
|
+
+ ' — a sheet or overlay is probably covering the screen behind it, so treat anything'
|
|
489
|
+
+ ' you did not expect to see as belonging to the layer underneath'
|
|
490
|
+
: null;
|
|
359
491
|
|
|
360
492
|
return {
|
|
361
493
|
device,
|
|
@@ -363,14 +495,202 @@ export async function screenMap(deviceQuery, {
|
|
|
363
495
|
rows,
|
|
364
496
|
truncated,
|
|
365
497
|
collapsed,
|
|
498
|
+
// A filtered view is not the screen, and the hint used to report its count
|
|
499
|
+
// as though it were: `--interactive` on a form said "1 element; nothing
|
|
500
|
+
// ambiguous — chain the next steps" while two required inputs sat unseen
|
|
501
|
+
// below it. The over-claim was the harmful half, not the filter.
|
|
502
|
+
filtered: Boolean(filter || interactive),
|
|
366
503
|
screen,
|
|
367
504
|
name,
|
|
368
505
|
exits,
|
|
369
|
-
|
|
506
|
+
exitList,
|
|
507
|
+
staleExits,
|
|
508
|
+
cleared,
|
|
509
|
+
overlay,
|
|
510
|
+
text: render({ device, identity, rows, truncated, collapsed, screen, name, exits, exitList, staleExits, cleared, overlay }),
|
|
511
|
+
};
|
|
512
|
+
}
|
|
513
|
+
|
|
514
|
+
/**
|
|
515
|
+
* One line saying whether the model needs to stop and think.
|
|
516
|
+
*
|
|
517
|
+
* The measured loop in a real session is observe → think → tap → observe →
|
|
518
|
+
* think, and the thinking dominates wall time. Phases 11–16 reduce how *often*
|
|
519
|
+
* a decision has to reach the model; this reduces how often the model *believes*
|
|
520
|
+
* it has to decide. Measured: 48 of 62 real calls were three steps or fewer, so
|
|
521
|
+
* a twelve-step flow arrived as four or five calls and every boundary was a
|
|
522
|
+
* think — not because anything was ambiguous, but because nothing said it was
|
|
523
|
+
* not.
|
|
524
|
+
*
|
|
525
|
+
* Everything here is already in hand when the result is assembled: whether the
|
|
526
|
+
* flow stopped, whether the screen settled, whether the graph recognises it,
|
|
527
|
+
* how many elements there are, and whether any two of them answer to the same
|
|
528
|
+
* label. No perception pass, no model call, no new state.
|
|
529
|
+
*
|
|
530
|
+
* The order is deliberate. It reports the *strongest reason to think* first,
|
|
531
|
+
* and only says "keep going" when it can rule all of them out — a hint that
|
|
532
|
+
* cheerfully says "carry on" into an unknown screen would be worse than no hint
|
|
533
|
+
* at all.
|
|
534
|
+
*/
|
|
535
|
+
export function nextHint({ ok, escalated, settled, loading, known, hash, exits, elements, ambiguous, filtered, exitList, staleExits } = {}) {
|
|
536
|
+
if (ok === false) {
|
|
537
|
+
return 'next: the flow stopped here — this is the moment to think. sim_recall shows how you got here; sim_ui re-reads the screen.';
|
|
538
|
+
}
|
|
539
|
+
// A stopped flow and a completed flow carrying a soft verdict are different
|
|
540
|
+
// things, and conflating them printed `flow completed — 16/16 steps` directly
|
|
541
|
+
// above `next: the flow stopped here` on a run where nothing stopped.
|
|
542
|
+
//
|
|
543
|
+
// The conflation was load-bearing, not cosmetic. `no-visible-change` is an
|
|
544
|
+
// escalating verdict and it fires falsely — reported three rounds running —
|
|
545
|
+
// so **one** wrong verdict anywhere in a clean flow told the agent to abandon
|
|
546
|
+
// batching and re-read. That is the exact failure this hint exists to
|
|
547
|
+
// prevent, caused by the hint.
|
|
548
|
+
//
|
|
549
|
+
// A completed flow with an unconfirmed step is worth one targeted look, not a
|
|
550
|
+
// re-plan, so the hint says which and keeps the horizon open.
|
|
551
|
+
if (escalated) {
|
|
552
|
+
return 'next: every step ran, but at least one could not be confirmed — check that one thing landed (re-read the field, or assert it) rather than re-planning the flow.';
|
|
553
|
+
}
|
|
554
|
+
// A screen awaiting a network call is *settled* — nothing is moving — and
|
|
555
|
+
// incomplete. Reported: a settle returned satisfied while a list was still
|
|
556
|
+
// loading, the map showed an empty content region, and an empty region and a
|
|
557
|
+
// still-loading one produced identical output. A person sees a spinner.
|
|
558
|
+
if (loading) {
|
|
559
|
+
return 'next: settled, but the transition classifier still sees loading — an empty-looking region may be a list that has not arrived. waitFor a string you expect rather than acting on this.';
|
|
560
|
+
}
|
|
561
|
+
if (settled === false) {
|
|
562
|
+
return 'next: the screen is still moving. sim_state polls it for a fraction of a map; do not act on this reading yet.';
|
|
563
|
+
}
|
|
564
|
+
if (known === false) {
|
|
565
|
+
return 'next: new screen, nothing predicted here yet — read it before acting on a label you have not seen on it.';
|
|
566
|
+
}
|
|
567
|
+
if (ambiguous > 0) {
|
|
568
|
+
return ambiguous === 1
|
|
569
|
+
? 'next: one label repeats on this screen — address that one by #ref, and the rest can go in one sim_do.'
|
|
570
|
+
: `next: ${ambiguous} labels repeat on this screen — address those by #ref, and the rest can go in one sim_do.`;
|
|
571
|
+
}
|
|
572
|
+
const known_ = hash ? `known (${hash.slice(0, 8)}${exits ? `, ${exits} known exit${exits === 1 ? '' : 's'}` : ''})` : 'known';
|
|
573
|
+
if (filtered) {
|
|
574
|
+
// The count belongs to the filter, not to the screen. An agent that reads
|
|
575
|
+
// it as the screen concludes a form has nothing to fill in.
|
|
576
|
+
return `next: settled; screen ${known_}; ${elements} element${elements === 1 ? '' : 's'} **matching your filter** — this is not the whole screen, and an empty text input can look like a caption. Read it unfiltered before concluding something is absent.`;
|
|
577
|
+
}
|
|
578
|
+
// Memory that contradicts the screen outranks "carry on", because the reason
|
|
579
|
+
// it contradicts is usually that this screen has been confused with another —
|
|
580
|
+
// and a confident "chain without looking again" on a misidentified screen is
|
|
581
|
+
// how remembered advice ends up pointing at a control that files a work order.
|
|
582
|
+
const offerable = (exitList ?? []).filter((e) => e.label);
|
|
583
|
+
if (!offerable.length && staleExits) {
|
|
584
|
+
return `next: this screen is recognised but ${staleExits} remembered control${staleExits === 1 ? ' is' : 's are'} not on it,`
|
|
585
|
+
+ ' so the identity is probably wrong — two screens sharing one hash. Act only on the element list, and re-read before anything irreversible.';
|
|
586
|
+
}
|
|
587
|
+
// Naming the vocabulary is what makes "chain" actionable. A hint that says
|
|
588
|
+
// "chain the next steps" without saying what the steps could be is asking the
|
|
589
|
+
// agent to plan from a map it has to keep re-reading.
|
|
590
|
+
const vocab = offerable.slice(0, 6)
|
|
591
|
+
.map((e) => `${e.action} ${JSON.stringify(String(e.label).slice(0, 28))}`).join(', ');
|
|
592
|
+
return `next: settled; screen ${known_}; ${elements} element${elements === 1 ? '' : 's'}; nothing ambiguous — chain the next steps in one sim_do without looking again.`
|
|
593
|
+
+ (vocab ? ` Known to work here: ${vocab}.` : '')
|
|
594
|
+
+ (staleExits ? ` (${staleExits} other remembered control${staleExits === 1 ? '' : 's'} not on this screen — the graph may be conflating it with another.)` : '');
|
|
595
|
+
}
|
|
596
|
+
|
|
597
|
+
/**
|
|
598
|
+
* The hint for a rendered map, from the map itself.
|
|
599
|
+
*
|
|
600
|
+
* Lives here rather than in the MCP server because it was only reachable from
|
|
601
|
+
* there, and the MCP server is a long-lived process: a session that started
|
|
602
|
+
* before a change is running the old code, so the headline change of a phase
|
|
603
|
+
* could not be exercised at all. Reported, correctly, as the first finding
|
|
604
|
+
* against Phase 11.5. In `view.js` both front ends share one implementation and
|
|
605
|
+
* a unit test can reach it.
|
|
606
|
+
*/
|
|
607
|
+
export function hintFor(map, { flowOk = true, escalated = false } = {}) {
|
|
608
|
+
return nextHint({
|
|
609
|
+
ok: flowOk !== false,
|
|
610
|
+
escalated: Boolean(escalated),
|
|
611
|
+
filtered: map?.filtered === true,
|
|
612
|
+
settled: map?.identity?.settled !== false,
|
|
613
|
+
loading: map?.identity?.loading === true,
|
|
614
|
+
known: map?.exits != null,
|
|
615
|
+
hash: map?.identity?.hash ?? null,
|
|
616
|
+
exits: map?.exits ?? 0,
|
|
617
|
+
exitList: map?.exitList ?? [],
|
|
618
|
+
staleExits: map?.staleExits ?? 0,
|
|
619
|
+
elements: map?.rows?.length ?? 0,
|
|
620
|
+
ambiguous: ambiguousLabels(map?.rows),
|
|
621
|
+
});
|
|
622
|
+
}
|
|
623
|
+
|
|
624
|
+
/**
|
|
625
|
+
* Keep only the remembered actions whose control is actually on this screen.
|
|
626
|
+
*
|
|
627
|
+
* @returns {{exitList: Array, stale: number}} what can be offered, and how many
|
|
628
|
+
* remembered actions found nothing here — which is evidence the screen has
|
|
629
|
+
* been misidentified, and worth saying out loud.
|
|
630
|
+
*/
|
|
631
|
+
export function presentOnly(remembered, rows) {
|
|
632
|
+
const here = new Set();
|
|
633
|
+
for (const r of rows ?? []) {
|
|
634
|
+
for (const name of [r.label, ...(r.aliases ?? [])]) {
|
|
635
|
+
const k = alnum(name);
|
|
636
|
+
if (k) here.add(k);
|
|
637
|
+
}
|
|
638
|
+
}
|
|
639
|
+
const has = (label) => {
|
|
640
|
+
const k = alnum(label);
|
|
641
|
+
if (!k) return false;
|
|
642
|
+
if (here.has(k)) return true;
|
|
643
|
+
// A row may carry the label inside a longer one — a list card concatenates
|
|
644
|
+
// its children, and truncation adds an ellipsis.
|
|
645
|
+
for (const seen of here) if (seen.includes(k) || k.includes(seen)) return true;
|
|
646
|
+
return false;
|
|
370
647
|
};
|
|
648
|
+
const exitList = (remembered ?? []).filter((e) => has(e.label));
|
|
649
|
+
return { exitList, stale: (remembered ?? []).length - exitList.length };
|
|
650
|
+
}
|
|
651
|
+
|
|
652
|
+
/**
|
|
653
|
+
* What has worked from this screen before, printed rather than counted.
|
|
654
|
+
*
|
|
655
|
+
* The map said `(known, 3 known exits)` and stopped there, so the graph's own
|
|
656
|
+
* vocabulary never reached the caller. A flow whose labels were known in
|
|
657
|
+
* advance ran 16 steps in one call; the same agent on screens the graph also
|
|
658
|
+
* knew, but whose labels it had to rediscover, spent 25 calls on 31 steps.
|
|
659
|
+
*
|
|
660
|
+
* Deliberately terse and deliberately *not* a promise. These are actions that
|
|
661
|
+
* previously worked here, with how often — evidence for a plan, not a
|
|
662
|
+
* guarantee, and the destructive-label rules apply to them exactly as before.
|
|
663
|
+
*/
|
|
664
|
+
export function exitsLine(exitList, { limit = 6, stale = 0 } = {}) {
|
|
665
|
+
const list = (exitList ?? []).filter((e) => e.label).slice(0, limit);
|
|
666
|
+
if (!list.length) {
|
|
667
|
+
// Everything remembered here is missing. That is not "no memory" — it is
|
|
668
|
+
// memory that contradicts the screen, which usually means two screens have
|
|
669
|
+
// collapsed into one identity, and it is the most useful thing to say.
|
|
670
|
+
return stale
|
|
671
|
+
? `memory disagrees with this screen: ${stale} remembered control${stale === 1 ? '' : 's'} not present`
|
|
672
|
+
+ ' — this screen has probably been confused with another. Trust the element list, not the graph.'
|
|
673
|
+
: null;
|
|
674
|
+
}
|
|
675
|
+
const parts = list.map((e) => {
|
|
676
|
+
const label = String(e.label).length > 28 ? `${String(e.label).slice(0, 28)}…` : String(e.label);
|
|
677
|
+
return `${e.action} ${JSON.stringify(label)}${e.count > 1 ? ` (${e.count}x)` : ''}`;
|
|
678
|
+
});
|
|
679
|
+
return `worked here before: ${parts.join(', ')}`;
|
|
680
|
+
}
|
|
681
|
+
|
|
682
|
+
/** How many labels are worn by more than one element a caller could act on. */
|
|
683
|
+
export function ambiguousLabels(rows) {
|
|
684
|
+
const seen = new Map();
|
|
685
|
+
for (const r of rows ?? []) {
|
|
686
|
+
const key = alnum(r.label);
|
|
687
|
+
if (!key) continue;
|
|
688
|
+
seen.set(key, (seen.get(key) ?? 0) + 1);
|
|
689
|
+
}
|
|
690
|
+
return [...seen.values()].filter((n) => n > 1).length;
|
|
371
691
|
}
|
|
372
692
|
|
|
373
|
-
export function render({ device, identity, rows, truncated, collapsed, screen, name, exits, verdictLine, ambiguities }) {
|
|
693
|
+
export function render({ device, identity, rows, truncated, collapsed, screen, name, exits, exitList, staleExits, verdictLine, ambiguities, cleared, overlay }) {
|
|
374
694
|
const head = [
|
|
375
695
|
device?.name,
|
|
376
696
|
screen?.width ? `${screen.width}x${screen.height}pt` : null,
|
|
@@ -380,11 +700,19 @@ export function render({ device, identity, rows, truncated, collapsed, screen, n
|
|
|
380
700
|
: 'screen unidentified',
|
|
381
701
|
identity?.keyboard ? 'keyboard up' : null,
|
|
382
702
|
identity?.settled === false ? 'STILL MOVING' : null,
|
|
703
|
+
// Still and finished are not the same thing.
|
|
704
|
+
identity?.loading === true ? 'STILL LOADING' : null,
|
|
383
705
|
recalledNote(identity),
|
|
384
706
|
].filter(Boolean).join(' · ');
|
|
385
707
|
|
|
386
708
|
const lines = [head];
|
|
387
709
|
if (verdictLine) lines.push(verdictLine);
|
|
710
|
+
// Above the element list, not below it: it contradicts something the caller
|
|
711
|
+
// already believes, which is the one kind of news that must not be scrolled to.
|
|
712
|
+
if (cleared) lines.push(cleared);
|
|
713
|
+
if (overlay) lines.push(overlay);
|
|
714
|
+
const worked = exitsLine(exitList, { stale: staleExits });
|
|
715
|
+
if (worked) lines.push(worked);
|
|
388
716
|
|
|
389
717
|
let region = null;
|
|
390
718
|
for (const r of rows) {
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The words simframe may not act on by itself.
|
|
3
|
+
*
|
|
4
|
+
* CLAUDE.md has required this since the human-parity series was written — *"a
|
|
5
|
+
* reflex never taps anything whose label matches the destructive vocabulary
|
|
6
|
+
* (Delete, Remove, Pay, Send, Sign out, Reset…). Add the list to the same data
|
|
7
|
+
* file."* — and several of my own comments already talk about "the destructive
|
|
8
|
+
* vocabulary" as though it existed. It did not. This is it.
|
|
9
|
+
*
|
|
10
|
+
* **What it gates, precisely.** This restricts what simframe does *on its own
|
|
11
|
+
* initiative*: a retry, an alternative selector, an exploration step, a reflex,
|
|
12
|
+
* a speculative tap. It never restricts what the caller explicitly asked for.
|
|
13
|
+
* `{"tap": "DELETE ACCOUNT"}` is a request and is honoured; substituting
|
|
14
|
+
* "DELETE ACCOUNT" for a "Done" that did not resolve is not, and that is the
|
|
15
|
+
* whole distinction. Getting it backwards would make the tool refuse the thing
|
|
16
|
+
* a tester most needs to test.
|
|
17
|
+
*
|
|
18
|
+
* Data, not code, and locale-keyed, so another language is a file rather than a
|
|
19
|
+
* release.
|
|
20
|
+
*/
|
|
21
|
+
import fs from 'node:fs';
|
|
22
|
+
import path from 'node:path';
|
|
23
|
+
import { fileURLToPath } from 'node:url';
|
|
24
|
+
|
|
25
|
+
const HERE = path.dirname(fileURLToPath(import.meta.url));
|
|
26
|
+
const DATA = path.join(HERE, '..', 'data', 'vocabulary');
|
|
27
|
+
|
|
28
|
+
const cache = new Map();
|
|
29
|
+
|
|
30
|
+
/** The vocabulary for a locale, falling back to English. */
|
|
31
|
+
export function load(locale = process.env.SIMFRAME_LOCALE || 'en') {
|
|
32
|
+
const key = String(locale).toLowerCase().split(/[-_]/)[0];
|
|
33
|
+
if (cache.has(key)) return cache.get(key);
|
|
34
|
+
let data = null;
|
|
35
|
+
for (const candidate of [key, 'en']) {
|
|
36
|
+
try {
|
|
37
|
+
data = JSON.parse(fs.readFileSync(path.join(DATA, `${candidate}.json`), 'utf8'));
|
|
38
|
+
break;
|
|
39
|
+
} catch { /* try the fallback */ }
|
|
40
|
+
}
|
|
41
|
+
// A missing file must not silently disable the barrier. An empty vocabulary
|
|
42
|
+
// would make every label safe, which is the wrong direction to fail in, so
|
|
43
|
+
// this throws rather than returning nothing.
|
|
44
|
+
if (!data) throw new Error(`no vocabulary for "${locale}" and no en fallback in ${DATA}`);
|
|
45
|
+
cache.set(key, data);
|
|
46
|
+
return data;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
const alnum = (s) => String(s ?? '').toLowerCase().replace(/[^\p{L}\p{N}]+/gu, ' ').trim();
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Does this phrase occur in the label as whole words?
|
|
53
|
+
*
|
|
54
|
+
* Word boundaries, not substrings, and the reason is a real screen: an app's
|
|
55
|
+
* "Work Orders" tab contains the letters of "order", and a substring match would
|
|
56
|
+
* make its main navigation untouchable by anything local. Matching whole words
|
|
57
|
+
* means "order" does not match "orders", which is the behaviour wanted here —
|
|
58
|
+
* an exact tappable word is the signal, and a longer word is a different word.
|
|
59
|
+
*/
|
|
60
|
+
function saysPhrase(label, phrase) {
|
|
61
|
+
const l = ` ${alnum(label)} `;
|
|
62
|
+
const p = alnum(phrase);
|
|
63
|
+
if (!p) return false;
|
|
64
|
+
return l.includes(` ${p} `);
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* May simframe act on this label on its own initiative?
|
|
69
|
+
*
|
|
70
|
+
* `purpose` matters, and conflating two purposes cost a real run. The default,
|
|
71
|
+
* `substitute`, answers *"may a retry aim at this instead?"* and permits
|
|
72
|
+
* "Cancel", deliberately, so a local tier can decline a dialog rather than
|
|
73
|
+
* stranding on every confirmation it meets.
|
|
74
|
+
*
|
|
75
|
+
* `explore` answers a different question — *"may I open this as a door and see
|
|
76
|
+
* what is behind it?"* — and there "Cancel" is the abandon-this-task control.
|
|
77
|
+
* `seek` asked the first question and got the first answer: it opened CANCEL
|
|
78
|
+
* first, then AI TROUBLESHOOTING, then pressed "YES, THIS FIXED MY PROBLEM",
|
|
79
|
+
* ending five screens deep in a live support chat with a half-built service
|
|
80
|
+
* request destroyed. One label further along was SUBMIT SERVICE REQUEST.
|
|
81
|
+
*
|
|
82
|
+
* So exploration has its own list and it errs toward refusing. A door missed
|
|
83
|
+
* costs one step of a bounded budget; a door taken wrongly costs the run, and
|
|
84
|
+
* can cost the thing being tested.
|
|
85
|
+
*
|
|
86
|
+
* @returns {{allowed: boolean, reason?: string, matched?: string}}
|
|
87
|
+
*/
|
|
88
|
+
export function mayActLocally(label, { locale, purpose = 'substitute' } = {}) {
|
|
89
|
+
const text = String(label ?? '').trim();
|
|
90
|
+
if (!text) return { allowed: purpose !== 'explore' };
|
|
91
|
+
const vocab = load(locale);
|
|
92
|
+
|
|
93
|
+
if (purpose === 'explore') {
|
|
94
|
+
// Checked before the "listed as safe" exemption below, which exists for
|
|
95
|
+
// declining dialogs and must never make something a door.
|
|
96
|
+
for (const word of vocab.exploration?.neverOpen ?? []) {
|
|
97
|
+
if (saysPhrase(text, word)) {
|
|
98
|
+
return { allowed: false, reason: 'not a door — it commits, abandons or answers', matched: word };
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
for (const pattern of vocab.exploration?.neverOpenPatterns ?? []) {
|
|
102
|
+
if (new RegExp(pattern, 'i').test(alnum(text)) || new RegExp(pattern, 'i').test(text.toLowerCase())) {
|
|
103
|
+
return { allowed: false, reason: 'not a door — it reads as an instruction or an answer', matched: pattern };
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
// Exceptions first, and matched against the **whole** label rather than as a
|
|
109
|
+
// phrase inside it. "Cancel" is how you *decline* a dialog and a barrier that
|
|
110
|
+
// refused it would strand a local tier on every confirmation it met — but
|
|
111
|
+
// "Cancel order" is a different act, and a phrase match would have waved it
|
|
112
|
+
// through on the strength of its first word.
|
|
113
|
+
const whole = alnum(text);
|
|
114
|
+
for (const ok of vocab.destructive?.notWords?.words ?? []) {
|
|
115
|
+
if (whole === alnum(ok)) return { allowed: true, reason: 'listed as safe', matched: ok };
|
|
116
|
+
}
|
|
117
|
+
for (const word of vocab.destructive?.words ?? []) {
|
|
118
|
+
if (saysPhrase(text, word)) {
|
|
119
|
+
return { allowed: false, reason: 'destructive vocabulary', matched: word };
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
for (const word of vocab.leavesTheApp?.words ?? []) {
|
|
123
|
+
if (saysPhrase(text, word)) {
|
|
124
|
+
return { allowed: false, reason: 'leaves the app', matched: word };
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
return { allowed: true };
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
/** Convenience for a filter: keep only what a local tier may act on. */
|
|
131
|
+
export const actableLocally = (label, options) => mayActLocally(label, options).allowed;
|
|
132
|
+
|
|
133
|
+
/** May exploration open this as a door? Stricter than substitution, on purpose. */
|
|
134
|
+
export const openableAsDoor = (label, options) => mayActLocally(label, { ...options, purpose: 'explore' }).allowed;
|
package/src/wrote.js
ADDED
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* What simframe itself typed into this device, and whether it is still there.
|
|
3
|
+
*
|
|
4
|
+
* A peer filled three fields on a web form, and six calls later they were
|
|
5
|
+
* empty. Every one of those six calls returned `ok`. simframe had *recorded*
|
|
6
|
+
* all three values two calls earlier and said nothing when they vanished — so
|
|
7
|
+
* the agent only found out by paying for a screenshot, and an agent that had
|
|
8
|
+
* trusted the verdicts would have submitted an empty order.
|
|
9
|
+
*
|
|
10
|
+
* Two design decisions, both from how that failure actually presented.
|
|
11
|
+
*
|
|
12
|
+
* **Not keyed on screen identity.** The obvious home for this is the screen
|
|
13
|
+
* map, and it would have missed the whole thing: the wipe was observed under a
|
|
14
|
+
* different screen hash than the one the values were written under, because a
|
|
15
|
+
* pinch-zoom and a horizontal scroll had both moved the identity. What stayed
|
|
16
|
+
* constant was the field's own label. So this is a small per-device journal,
|
|
17
|
+
* and a screen is only relevant in that its labels are what we match against.
|
|
18
|
+
*
|
|
19
|
+
* **Only confirmed writes are journalled.** An unconfirmed write is not
|
|
20
|
+
* evidence a value was ever there, and journalling one would manufacture a
|
|
21
|
+
* "your text disappeared" warning for text that never arrived — trading a
|
|
22
|
+
* silent failure for a confident wrong answer, which is the worse of the two.
|
|
23
|
+
*/
|
|
24
|
+
import path from 'node:path';
|
|
25
|
+
import * as store from './store.js';
|
|
26
|
+
|
|
27
|
+
const FILE = 'wrote.json';
|
|
28
|
+
|
|
29
|
+
/** Keep the journal small: this is a recency signal, not a history. */
|
|
30
|
+
export const KEEP = 24;
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* How long a journalled value stays interesting.
|
|
34
|
+
*
|
|
35
|
+
* Long enough to span the kind of sequence that lost the peer's three fields
|
|
36
|
+
* (six calls, a little over two minutes), short enough that yesterday's form
|
|
37
|
+
* never comments on today's.
|
|
38
|
+
*/
|
|
39
|
+
export const MAX_AGE_MS = 15 * 60 * 1000;
|
|
40
|
+
|
|
41
|
+
const file = (udid) => path.join(store.deviceDir(udid), FILE);
|
|
42
|
+
|
|
43
|
+
/** Comparison that ignores what OCR adds — the same rule the readback uses. */
|
|
44
|
+
const alnum = (v) => String(v ?? '').toLowerCase().replace(/[^\p{L}\p{N}]+/gu, '');
|
|
45
|
+
|
|
46
|
+
/** Does this read as a label at all, rather than as a coordinate or a ref? */
|
|
47
|
+
const hasLetters = (v) => /\p{L}/u.test(String(v ?? ''));
|
|
48
|
+
|
|
49
|
+
export function read(udid) {
|
|
50
|
+
const got = store.readJson(file(udid));
|
|
51
|
+
return Array.isArray(got) ? got : [];
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* Record a value that was *confirmed* to have landed.
|
|
56
|
+
*
|
|
57
|
+
* Re-writing the same field replaces its entry rather than adding one, so a
|
|
58
|
+
* field filled twice cannot warn about its own earlier contents.
|
|
59
|
+
*/
|
|
60
|
+
export function record(udid, { selector, value, screen }) {
|
|
61
|
+
if (!udid || !selector || !alnum(value)) return null;
|
|
62
|
+
// A selector with no letters in it is not a label: "(125,325)", "@120,400",
|
|
63
|
+
// "#3". There is nothing to look for on a later screen, and its digits would
|
|
64
|
+
// happily match some unrelated number — so it is never journalled at all,
|
|
65
|
+
// rather than journalled and then skipped.
|
|
66
|
+
if (!hasLetters(selector)) return null;
|
|
67
|
+
const now = Date.now();
|
|
68
|
+
const key = alnum(selector);
|
|
69
|
+
const kept = read(udid).filter((e) => alnum(e.selector) !== key);
|
|
70
|
+
const next = [{ selector: String(selector), value: String(value), screen: screen ?? null, at: now }, ...kept]
|
|
71
|
+
.slice(0, KEEP);
|
|
72
|
+
try {
|
|
73
|
+
store.ensureDirs(udid);
|
|
74
|
+
store.writeAtomic(file(udid), JSON.stringify(next));
|
|
75
|
+
} catch {
|
|
76
|
+
// A journal that cannot be written must never fail the step that wrote the
|
|
77
|
+
// field. Losing the warning is a smaller cost than losing the write.
|
|
78
|
+
return null;
|
|
79
|
+
}
|
|
80
|
+
return next[0];
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
/**
|
|
84
|
+
* Which journalled values look as though they have been cleared.
|
|
85
|
+
*
|
|
86
|
+
* Deliberately narrow, because a false alarm here would teach an agent to
|
|
87
|
+
* ignore the line. An entry only counts as missing when its own **label is on
|
|
88
|
+
* this screen** — so we are plausibly looking at the same form — and its value
|
|
89
|
+
* appears nowhere in the map. Anything else (the label absent, the value
|
|
90
|
+
* present, the screen empty) says nothing and is reported as nothing.
|
|
91
|
+
*/
|
|
92
|
+
export function missing(udid, targets, { now = Date.now() } = {}) {
|
|
93
|
+
const rows = targets ?? [];
|
|
94
|
+
if (!rows.length) return [];
|
|
95
|
+
const haystack = rows.map((t) => alnum(`${t.label ?? ''} ${t.value ?? ''}`)).filter(Boolean);
|
|
96
|
+
if (!haystack.length) return [];
|
|
97
|
+
const gone = [];
|
|
98
|
+
for (const e of read(udid)) {
|
|
99
|
+
if (!Number.isFinite(e.at) || now - e.at > MAX_AGE_MS) continue;
|
|
100
|
+
const label = alnum(e.selector);
|
|
101
|
+
const value = alnum(e.value);
|
|
102
|
+
if (!label || !value) continue;
|
|
103
|
+
// `startsWith`, not `includes`. A field's row begins with its own label —
|
|
104
|
+
// and OCR fuses the value onto the end of it ("Telephone: 5551234567"),
|
|
105
|
+
// which is why this cannot be an equality test. A substring test looked
|
|
106
|
+
// equivalent and was not: on the very first field run after this shipped, a
|
|
107
|
+
// journalled "Email" matched the page footer's newsletter box, "Enter your
|
|
108
|
+
// email address", and announced a value gone that was merely on a different
|
|
109
|
+
// part of the page. A false alarm here teaches an agent to ignore the line,
|
|
110
|
+
// which costs more than the line is worth.
|
|
111
|
+
if (!haystack.some((h) => h.startsWith(label))) continue;
|
|
112
|
+
if (haystack.some((h) => h.includes(value))) continue;
|
|
113
|
+
gone.push(e);
|
|
114
|
+
}
|
|
115
|
+
return gone;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/** The line the map prints, or null when there is nothing to say. */
|
|
119
|
+
export function missingLine(gone) {
|
|
120
|
+
const list = (gone ?? []).slice(0, 3);
|
|
121
|
+
if (!list.length) return null;
|
|
122
|
+
const parts = list.map((e) => {
|
|
123
|
+
const v = String(e.value).length > 24 ? `${String(e.value).slice(0, 24)}…` : String(e.value);
|
|
124
|
+
return `${JSON.stringify(v)} in ${JSON.stringify(String(e.selector))}`;
|
|
125
|
+
});
|
|
126
|
+
const more = (gone?.length ?? 0) - list.length;
|
|
127
|
+
return `a value simframe wrote here is gone: ${parts.join(', ')}${more > 0 ? `, and ${more} more` : ''}`
|
|
128
|
+
+ ' — the field is on this screen and its contents are not, so something cleared it. Re-fill before continuing.';
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/** Forget the journal for a device — used when its memory is reset. */
|
|
132
|
+
export function forget(udid) {
|
|
133
|
+
try {
|
|
134
|
+
store.writeAtomic(file(udid), JSON.stringify([]));
|
|
135
|
+
} catch { /* nothing to forget */ }
|
|
136
|
+
}
|