simframe 0.10.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +181 -2
- package/data/vocabulary/en.json +148 -0
- package/native/ocr.swift +13 -1
- package/native/rank.swift +87 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +72 -4
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
- package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +19 -0
- package/native/simframed/Sources/simframed/main.swift +33 -3
- package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +41 -0
- package/native/supervise.swift +216 -0
- package/package.json +4 -1
- package/scripts/check-package.mjs +22 -2
- package/scripts/check-private.mjs +9 -0
- package/scripts/ci-integration-local.sh +79 -0
- package/scripts/ci-memory.mjs +104 -20
- package/scripts/collect-rulings.mjs +312 -0
- package/scripts/eval-fingerprint.mjs +100 -23
- package/scripts/eval-perception.mjs +33 -0
- package/scripts/phase17-corpus.mjs +176 -0
- package/scripts/probe-network.mjs +118 -0
- package/scripts/soak-capture.mjs +72 -0
- package/skills/simframe/SKILL.md +257 -7
- package/src/actions.js +1791 -44
- package/src/cli.js +216 -14
- package/src/control.js +1 -0
- package/src/fingerprint.js +43 -1
- package/src/graph.js +136 -7
- package/src/index.js +252 -14
- package/src/input.js +66 -3
- package/src/localhelper.js +161 -0
- package/src/matching.js +64 -1
- package/src/mcp.js +333 -32
- package/src/metrics.js +148 -3
- package/src/ocr.js +18 -1
- package/src/planner.js +195 -0
- package/src/platform/android.js +25 -1
- package/src/platform/index.js +11 -1
- package/src/platform/ios.js +25 -1
- package/src/png.js +26 -0
- package/src/refs.js +51 -8
- package/src/regions.js +215 -1
- package/src/screenmap.js +89 -9
- package/src/supervisor.js +161 -0
- package/src/view.js +375 -11
- package/src/vocabulary.js +134 -0
- package/src/wrote.js +136 -0
package/src/view.js
CHANGED
|
@@ -12,6 +12,8 @@
|
|
|
12
12
|
// app published and one OCR read off the pixels deserve different amounts of
|
|
13
13
|
// trust.
|
|
14
14
|
import * as api from './index.js';
|
|
15
|
+
import * as regions from './regions.js';
|
|
16
|
+
import * as wrote from './wrote.js';
|
|
15
17
|
import * as graph from './graph.js';
|
|
16
18
|
import { writeRefs } from './refs.js';
|
|
17
19
|
import * as matching from './matching.js';
|
|
@@ -20,10 +22,10 @@ import * as matching from './matching.js';
|
|
|
20
22
|
const REGION_ORDER = ['nav-bar', 'content', 'tab-bar', 'keyboard', 'status-bar'];
|
|
21
23
|
|
|
22
24
|
/**
|
|
23
|
-
*
|
|
24
|
-
*
|
|
25
|
+
* Which regions the map will not offer now lives in `regions.js`, because it is
|
|
26
|
+
* not only a presentation rule — see `regions.offerable`. A target this hides
|
|
27
|
+
* must also be one nothing resolves onto behind the caller's back.
|
|
25
28
|
*/
|
|
26
|
-
const HIDDEN_REGIONS = new Set(['status-bar']);
|
|
27
29
|
|
|
28
30
|
/** A keyboard is 30-odd keys nobody refers to by name. One line says it. */
|
|
29
31
|
const COLLAPSE_REGIONS = new Set(['keyboard']);
|
|
@@ -160,13 +162,94 @@ const trim = (text) => {
|
|
|
160
162
|
return one.length > MAX_LABEL ? `${one.slice(0, MAX_LABEL - 1)}…` : one;
|
|
161
163
|
};
|
|
162
164
|
|
|
163
|
-
/**
|
|
165
|
+
/**
|
|
166
|
+
* Why no row is dropped for being long — reverted 2026-09-10, same day.
|
|
167
|
+
*
|
|
168
|
+
* There was a rule here that dropped non-interactive rows whose label ran past
|
|
169
|
+
* 45 characters, on the evidence that four of sixteen rows on a Settings screen
|
|
170
|
+
* were the explanatory paragraph under each switch: 31% of that map's
|
|
171
|
+
* characters describing things nobody can tap. Across fourteen recorded screens
|
|
172
|
+
* it cut the map 15%.
|
|
173
|
+
*
|
|
174
|
+
* It was wrong, and the counter-example is decisive. A React Native list card
|
|
175
|
+
* exposes all of its children as one concatenated accessibility label —
|
|
176
|
+
* `Anaheim | Store # 1020, , 1234 Main St, … | Quick Casual Restaurant`, 105
|
|
177
|
+
* characters, type `GenericElement`, region `content`. Every property my rule
|
|
178
|
+
* tested is identical to the Settings caption's, and that card is *the only
|
|
179
|
+
* tappable thing on the screen*.
|
|
180
|
+
*
|
|
181
|
+
* Nothing became untappable — `locate`, `assert` and `waitFor` read
|
|
182
|
+
* `entry.targets`, so the rows are a view and the data survived. What was lost
|
|
183
|
+
* is **discovery**: the map stopped saying what was on screen, so an agent
|
|
184
|
+
* could only tap labels it already knew, and the fallback on a screen of
|
|
185
|
+
* unknown data is a ~1600-token screenshot. It saved characters on
|
|
186
|
+
* settings-shaped screens and spent an image on list-shaped ones, which is the
|
|
187
|
+
* exact cost the phase existed to remove. Blast radius: every data list in
|
|
188
|
+
* every RN app.
|
|
189
|
+
*
|
|
190
|
+
* The lesson is not "find a better discriminator". It is that this was a
|
|
191
|
+
* threshold shipped with no harness case that could catch its failure, in the
|
|
192
|
+
* same session as building the harness. So `expect.discoverable` now exists,
|
|
193
|
+
* and there is a fixture of the reported shape — a rule like this may return
|
|
194
|
+
* only when it can be gated.
|
|
195
|
+
*
|
|
196
|
+
* What survives is what the reporter suggested instead: truncate, do not drop.
|
|
197
|
+
* Position and tappability are the valuable parts of a row, not the full text.
|
|
198
|
+
* See `MAX_LABEL`.
|
|
199
|
+
*/
|
|
200
|
+
|
|
201
|
+
/**
|
|
202
|
+
* Whether a target in the keyboard band really is a key.
|
|
203
|
+
*
|
|
204
|
+
* Region bands are positional, and this was the fourth and fifth bug they
|
|
205
|
+
* produced. With the keyboard up, the bottom band is collapsed to one line —
|
|
206
|
+
* thirty keys nobody names — and a primary action pinned above the keyboard was
|
|
207
|
+
* collapsed with them: four reads running printed `keyboard: 6 keys` and **no
|
|
208
|
+
* forward control**, while the hint said "nothing ambiguous — chain the next
|
|
209
|
+
* steps without looking again". That it was an emission bug and not a
|
|
210
|
+
* perception one was proved by the next call, which hit the button instantly at
|
|
211
|
+
* a coordinate the map had never printed.
|
|
212
|
+
*
|
|
213
|
+
* Delegates to `regions.looksLikeKey`, which is the canonical test. These two
|
|
214
|
+
* having separate copies is what let a phantom keyboard survive in the
|
|
215
|
+
* fingerprint after it had already been fixed in the map — and there it was
|
|
216
|
+
* deleting screens' content from their own identity.
|
|
217
|
+
*/
|
|
218
|
+
export const isKey = regions.looksLikeKey;
|
|
219
|
+
|
|
220
|
+
/**
|
|
221
|
+
* Whether a target can be acted on, by role *or* by evidence.
|
|
222
|
+
*
|
|
223
|
+
* The role alone was wrong twice on real forms. A React Native composite select
|
|
224
|
+
* surfaces as a generic element, and a text input shows only its placeholder as
|
|
225
|
+
* `StaticText` — so `--interactive` answered "1 element" on a form with two
|
|
226
|
+
* visible, bordered, *required* inputs and an agent concluded there was nothing
|
|
227
|
+
* to fill in.
|
|
228
|
+
*
|
|
229
|
+
* Evidence is used rather than a longer list of role names, because the roles
|
|
230
|
+
* are what the tree got wrong. Only a control carries a `value`, only a
|
|
231
|
+
* focusable thing is `focused`, and `enabled` is a state a caption never
|
|
232
|
+
* declares. A generic element that is none of those really is a container.
|
|
233
|
+
*
|
|
234
|
+
* The case this still cannot see: an empty, unfocused input whose placeholder is
|
|
235
|
+
* its only text. Nothing in the tree distinguishes it from a caption, which is
|
|
236
|
+
* why a filtered view now says it is filtered rather than implying it is the
|
|
237
|
+
* whole screen.
|
|
238
|
+
*/
|
|
239
|
+
export function actsInteractive(t) {
|
|
240
|
+
if (INTERACTIVE.test(t?.type || '')) return true;
|
|
241
|
+
if (t?.value != null && t.value !== '') return true;
|
|
242
|
+
if (t?.focused) return true;
|
|
243
|
+
if (t?.enabled === false) return true;
|
|
244
|
+
return false;
|
|
245
|
+
}
|
|
246
|
+
|
|
164
247
|
export function rowsFor(entry, { screen, filter, interactive, all = false, limit = DEFAULT_LIMIT } = {}) {
|
|
165
248
|
let kept = (entry?.targets ?? []).map((t) => ({ ...t })).filter((t) => {
|
|
166
249
|
if (!isNum(t.x) || !isNum(t.y)) return false;
|
|
167
250
|
// Off-screen elements are real in the tree and untappable in fact.
|
|
168
|
-
if (
|
|
169
|
-
if (!all &&
|
|
251
|
+
if (regions.offViewport(t, screen)) return false;
|
|
252
|
+
if (!all && !regions.offerable(t.region)) return false;
|
|
170
253
|
if (!all && isNoise(t)) return false;
|
|
171
254
|
return true;
|
|
172
255
|
});
|
|
@@ -181,7 +264,7 @@ export function rowsFor(entry, { screen, filter, interactive, all = false, limit
|
|
|
181
264
|
kept = kept.filter((t) =>
|
|
182
265
|
[t.label, ...(t.aliases ?? [])].filter(Boolean).join(' ').toLowerCase().includes(q));
|
|
183
266
|
}
|
|
184
|
-
if (interactive) kept = kept.filter(
|
|
267
|
+
if (interactive) kept = kept.filter(actsInteractive);
|
|
185
268
|
|
|
186
269
|
const order = (t) => {
|
|
187
270
|
const i = REGION_ORDER.indexOf(t.region ?? 'content');
|
|
@@ -189,11 +272,25 @@ export function rowsFor(entry, { screen, filter, interactive, all = false, limit
|
|
|
189
272
|
};
|
|
190
273
|
kept.sort((a, b) => order(a) - order(b) || a.y - b.y || a.x - b.x);
|
|
191
274
|
|
|
275
|
+
// A band is only the keyboard if there is a keyboard in it.
|
|
276
|
+
//
|
|
277
|
+
// Region bands are positional, so on a screen with no keyboard at all the
|
|
278
|
+
// bottom band was still called `keyboard` and review-summary rows were filed
|
|
279
|
+
// under it, followed by `keyboard: 1 keys (tap by label or type directly)` —
|
|
280
|
+
// advice that is actively wrong about page content. Reported as noise on
|
|
281
|
+
// every map of two screens. Relabelled from the contents rather than the
|
|
282
|
+
// position, which is the only evidence available here.
|
|
283
|
+
const keysPresent = kept.some((t) => COLLAPSE_REGIONS.has(t.region ?? '') && isKey(t));
|
|
284
|
+
const bandOf = (t) => {
|
|
285
|
+
const region = t.region ?? 'content';
|
|
286
|
+
return COLLAPSE_REGIONS.has(region) && !keysPresent ? 'content' : region;
|
|
287
|
+
};
|
|
288
|
+
|
|
192
289
|
const rows = [];
|
|
193
290
|
const collapsed = new Map();
|
|
194
291
|
for (const t of kept) {
|
|
195
|
-
const region = t
|
|
196
|
-
if (COLLAPSE_REGIONS.has(region)) {
|
|
292
|
+
const region = bandOf(t);
|
|
293
|
+
if (COLLAPSE_REGIONS.has(region) && isKey(t)) {
|
|
197
294
|
collapsed.set(region, (collapsed.get(region) ?? 0) + 1);
|
|
198
295
|
continue;
|
|
199
296
|
}
|
|
@@ -232,6 +329,28 @@ export function recalledNote(identity, now = Date.now()) {
|
|
|
232
329
|
return `elements recalled from ${ago} ago — pass refresh for what is there now`;
|
|
233
330
|
}
|
|
234
331
|
|
|
332
|
+
/**
|
|
333
|
+
* How old the frame behind this reading is, and whether that is a problem.
|
|
334
|
+
*
|
|
335
|
+
* Two thresholds, because "slightly behind" and "possibly a different screen"
|
|
336
|
+
* are different messages. Under `FRAME_FRESH_MS` nothing is said: a map that
|
|
337
|
+
* announced "42ms old" on every call would train a reader to skip the line that
|
|
338
|
+
* matters. Over `FRAME_STALE_MS` it shouts, because at that age the app may have
|
|
339
|
+
* moved on entirely and the whole element list is then a description of the past.
|
|
340
|
+
*/
|
|
341
|
+
export const FRAME_FRESH_MS = 1500;
|
|
342
|
+
export const FRAME_STALE_MS = 4000;
|
|
343
|
+
|
|
344
|
+
export function frameAgeNote(identity, now = Date.now()) {
|
|
345
|
+
const at = identity?.state?.capturedAt;
|
|
346
|
+
if (!Number.isFinite(at)) return null;
|
|
347
|
+
const age = Math.max(0, now - at);
|
|
348
|
+
if (age < FRAME_FRESH_MS) return null;
|
|
349
|
+
if (age < FRAME_STALE_MS) return `frame ${(age / 1000).toFixed(1)}s old`;
|
|
350
|
+
return `WARNING this frame is ${(age / 1000).toFixed(1)}s old — the screen may have moved on `
|
|
351
|
+
+ 'since, so treat the elements below as a description of the past and pass refresh';
|
|
352
|
+
}
|
|
353
|
+
|
|
235
354
|
/**
|
|
236
355
|
* What a control *contains*, from the sensor that actually knows.
|
|
237
356
|
*
|
|
@@ -356,6 +475,41 @@ export async function screenMap(deviceQuery, {
|
|
|
356
475
|
// screen is what the agent can actually use: it says how much of this screen
|
|
357
476
|
// the graph can navigate from without being told.
|
|
358
477
|
const exits = node ? node.edges.length : null;
|
|
478
|
+
// Not just how many — which. The graph has always known what worked here and
|
|
479
|
+
// only ever reported a count, so an agent on a screen simframe had driven ten
|
|
480
|
+
// times still read it to learn what was tappable.
|
|
481
|
+
const remembered = node ? graph.exitsOf(node) : [];
|
|
482
|
+
// Memory intersected with what is actually here, never memory alone.
|
|
483
|
+
//
|
|
484
|
+
// This is the correction to the feature above, and it was reported with the
|
|
485
|
+
// consequence spelled out. A wizard's read-only *review* screen had been given
|
|
486
|
+
// the same identity as its step 1, so it inherited step 1's entire vocabulary:
|
|
487
|
+
// the map offered `tap "APPLY"`, `tap "No Power"`, `tap "PLACE A SERVICE
|
|
488
|
+
// REQUEST"` — **not one of which exists on it** — while the hint said "nothing
|
|
489
|
+
// ambiguous, chain the next steps without looking again". The only control on
|
|
490
|
+
// that screen files a real work order. Confident advice pointed at a
|
|
491
|
+
// destructive button on a screen it had misidentified.
|
|
492
|
+
//
|
|
493
|
+
// The line was right and the identity was wrong, so the line now checks. A
|
|
494
|
+
// remembered action is only offered when its label is on the screen in front
|
|
495
|
+
// of us; the rest are counted and reported as a disagreement, because memory
|
|
496
|
+
// that does not match what is here is itself the most useful thing to say.
|
|
497
|
+
const { exitList, stale: staleExits } = presentOnly(remembered, rows);
|
|
498
|
+
// Values simframe wrote itself and can no longer see. Computed from the same
|
|
499
|
+
// rows the map is about to print, so it costs nothing, and it is the only
|
|
500
|
+
// thing that can notice a form being cleared underneath a caller — six `ok`
|
|
501
|
+
// calls in a row hid exactly that.
|
|
502
|
+
const cleared = wrote.missingLine(wrote.missing(device?.udid, rows));
|
|
503
|
+
// One layer covering another, detected where it shows: an ax element and an
|
|
504
|
+
// OCR word at one coordinate disagreeing about what is there. Reported by a
|
|
505
|
+
// peer who had to fall back to a screenshot to count five radio options
|
|
506
|
+
// through a sheet, which is the case the text map exists to remove.
|
|
507
|
+
const layered = (identity?.entry?.occluded ?? []).length;
|
|
508
|
+
const overlay = layered
|
|
509
|
+
? `${layered} element(s) on this screen overlap and disagree about what is there`
|
|
510
|
+
+ ' — a sheet or overlay is probably covering the screen behind it, so treat anything'
|
|
511
|
+
+ ' you did not expect to see as belonging to the layer underneath'
|
|
512
|
+
: null;
|
|
359
513
|
|
|
360
514
|
return {
|
|
361
515
|
device,
|
|
@@ -363,14 +517,202 @@ export async function screenMap(deviceQuery, {
|
|
|
363
517
|
rows,
|
|
364
518
|
truncated,
|
|
365
519
|
collapsed,
|
|
520
|
+
// A filtered view is not the screen, and the hint used to report its count
|
|
521
|
+
// as though it were: `--interactive` on a form said "1 element; nothing
|
|
522
|
+
// ambiguous — chain the next steps" while two required inputs sat unseen
|
|
523
|
+
// below it. The over-claim was the harmful half, not the filter.
|
|
524
|
+
filtered: Boolean(filter || interactive),
|
|
366
525
|
screen,
|
|
367
526
|
name,
|
|
368
527
|
exits,
|
|
369
|
-
|
|
528
|
+
exitList,
|
|
529
|
+
staleExits,
|
|
530
|
+
cleared,
|
|
531
|
+
overlay,
|
|
532
|
+
text: render({ device, identity, rows, truncated, collapsed, screen, name, exits, exitList, staleExits, cleared, overlay }),
|
|
370
533
|
};
|
|
371
534
|
}
|
|
372
535
|
|
|
373
|
-
|
|
536
|
+
/**
|
|
537
|
+
* One line saying whether the model needs to stop and think.
|
|
538
|
+
*
|
|
539
|
+
* The measured loop in a real session is observe → think → tap → observe →
|
|
540
|
+
* think, and the thinking dominates wall time. Phases 11–16 reduce how *often*
|
|
541
|
+
* a decision has to reach the model; this reduces how often the model *believes*
|
|
542
|
+
* it has to decide. Measured: 48 of 62 real calls were three steps or fewer, so
|
|
543
|
+
* a twelve-step flow arrived as four or five calls and every boundary was a
|
|
544
|
+
* think — not because anything was ambiguous, but because nothing said it was
|
|
545
|
+
* not.
|
|
546
|
+
*
|
|
547
|
+
* Everything here is already in hand when the result is assembled: whether the
|
|
548
|
+
* flow stopped, whether the screen settled, whether the graph recognises it,
|
|
549
|
+
* how many elements there are, and whether any two of them answer to the same
|
|
550
|
+
* label. No perception pass, no model call, no new state.
|
|
551
|
+
*
|
|
552
|
+
* The order is deliberate. It reports the *strongest reason to think* first,
|
|
553
|
+
* and only says "keep going" when it can rule all of them out — a hint that
|
|
554
|
+
* cheerfully says "carry on" into an unknown screen would be worse than no hint
|
|
555
|
+
* at all.
|
|
556
|
+
*/
|
|
557
|
+
export function nextHint({ ok, escalated, settled, loading, known, hash, exits, elements, ambiguous, filtered, exitList, staleExits } = {}) {
|
|
558
|
+
if (ok === false) {
|
|
559
|
+
return 'next: the flow stopped here — this is the moment to think. sim_recall shows how you got here; sim_ui re-reads the screen.';
|
|
560
|
+
}
|
|
561
|
+
// A stopped flow and a completed flow carrying a soft verdict are different
|
|
562
|
+
// things, and conflating them printed `flow completed — 16/16 steps` directly
|
|
563
|
+
// above `next: the flow stopped here` on a run where nothing stopped.
|
|
564
|
+
//
|
|
565
|
+
// The conflation was load-bearing, not cosmetic. `no-visible-change` is an
|
|
566
|
+
// escalating verdict and it fires falsely — reported three rounds running —
|
|
567
|
+
// so **one** wrong verdict anywhere in a clean flow told the agent to abandon
|
|
568
|
+
// batching and re-read. That is the exact failure this hint exists to
|
|
569
|
+
// prevent, caused by the hint.
|
|
570
|
+
//
|
|
571
|
+
// A completed flow with an unconfirmed step is worth one targeted look, not a
|
|
572
|
+
// re-plan, so the hint says which and keeps the horizon open.
|
|
573
|
+
if (escalated) {
|
|
574
|
+
return 'next: every step ran, but at least one could not be confirmed — check that one thing landed (re-read the field, or assert it) rather than re-planning the flow.';
|
|
575
|
+
}
|
|
576
|
+
// A screen awaiting a network call is *settled* — nothing is moving — and
|
|
577
|
+
// incomplete. Reported: a settle returned satisfied while a list was still
|
|
578
|
+
// loading, the map showed an empty content region, and an empty region and a
|
|
579
|
+
// still-loading one produced identical output. A person sees a spinner.
|
|
580
|
+
if (loading) {
|
|
581
|
+
return 'next: settled, but the transition classifier still sees loading — an empty-looking region may be a list that has not arrived. waitFor a string you expect rather than acting on this.';
|
|
582
|
+
}
|
|
583
|
+
if (settled === false) {
|
|
584
|
+
return 'next: the screen is still moving. sim_state polls it for a fraction of a map; do not act on this reading yet.';
|
|
585
|
+
}
|
|
586
|
+
if (known === false) {
|
|
587
|
+
return 'next: new screen, nothing predicted here yet — read it before acting on a label you have not seen on it.';
|
|
588
|
+
}
|
|
589
|
+
if (ambiguous > 0) {
|
|
590
|
+
return ambiguous === 1
|
|
591
|
+
? 'next: one label repeats on this screen — address that one by #ref, and the rest can go in one sim_do.'
|
|
592
|
+
: `next: ${ambiguous} labels repeat on this screen — address those by #ref, and the rest can go in one sim_do.`;
|
|
593
|
+
}
|
|
594
|
+
const known_ = hash ? `known (${hash.slice(0, 8)}${exits ? `, ${exits} known exit${exits === 1 ? '' : 's'}` : ''})` : 'known';
|
|
595
|
+
if (filtered) {
|
|
596
|
+
// The count belongs to the filter, not to the screen. An agent that reads
|
|
597
|
+
// it as the screen concludes a form has nothing to fill in.
|
|
598
|
+
return `next: settled; screen ${known_}; ${elements} element${elements === 1 ? '' : 's'} **matching your filter** — this is not the whole screen, and an empty text input can look like a caption. Read it unfiltered before concluding something is absent.`;
|
|
599
|
+
}
|
|
600
|
+
// Memory that contradicts the screen outranks "carry on", because the reason
|
|
601
|
+
// it contradicts is usually that this screen has been confused with another —
|
|
602
|
+
// and a confident "chain without looking again" on a misidentified screen is
|
|
603
|
+
// how remembered advice ends up pointing at a control that files a work order.
|
|
604
|
+
const offerable = (exitList ?? []).filter((e) => e.label);
|
|
605
|
+
if (!offerable.length && staleExits) {
|
|
606
|
+
return `next: this screen is recognised but ${staleExits} remembered control${staleExits === 1 ? ' is' : 's are'} not on it,`
|
|
607
|
+
+ ' so the identity is probably wrong — two screens sharing one hash. Act only on the element list, and re-read before anything irreversible.';
|
|
608
|
+
}
|
|
609
|
+
// Naming the vocabulary is what makes "chain" actionable. A hint that says
|
|
610
|
+
// "chain the next steps" without saying what the steps could be is asking the
|
|
611
|
+
// agent to plan from a map it has to keep re-reading.
|
|
612
|
+
const vocab = offerable.slice(0, 6)
|
|
613
|
+
.map((e) => `${e.action} ${JSON.stringify(String(e.label).slice(0, 28))}`).join(', ');
|
|
614
|
+
return `next: settled; screen ${known_}; ${elements} element${elements === 1 ? '' : 's'}; nothing ambiguous — chain the next steps in one sim_do without looking again.`
|
|
615
|
+
+ (vocab ? ` Known to work here: ${vocab}.` : '')
|
|
616
|
+
+ (staleExits ? ` (${staleExits} other remembered control${staleExits === 1 ? '' : 's'} not on this screen — the graph may be conflating it with another.)` : '');
|
|
617
|
+
}
|
|
618
|
+
|
|
619
|
+
/**
|
|
620
|
+
* The hint for a rendered map, from the map itself.
|
|
621
|
+
*
|
|
622
|
+
* Lives here rather than in the MCP server because it was only reachable from
|
|
623
|
+
* there, and the MCP server is a long-lived process: a session that started
|
|
624
|
+
* before a change is running the old code, so the headline change of a phase
|
|
625
|
+
* could not be exercised at all. Reported, correctly, as the first finding
|
|
626
|
+
* against Phase 11.5. In `view.js` both front ends share one implementation and
|
|
627
|
+
* a unit test can reach it.
|
|
628
|
+
*/
|
|
629
|
+
export function hintFor(map, { flowOk = true, escalated = false } = {}) {
|
|
630
|
+
return nextHint({
|
|
631
|
+
ok: flowOk !== false,
|
|
632
|
+
escalated: Boolean(escalated),
|
|
633
|
+
filtered: map?.filtered === true,
|
|
634
|
+
settled: map?.identity?.settled !== false,
|
|
635
|
+
loading: map?.identity?.loading === true,
|
|
636
|
+
known: map?.exits != null,
|
|
637
|
+
hash: map?.identity?.hash ?? null,
|
|
638
|
+
exits: map?.exits ?? 0,
|
|
639
|
+
exitList: map?.exitList ?? [],
|
|
640
|
+
staleExits: map?.staleExits ?? 0,
|
|
641
|
+
elements: map?.rows?.length ?? 0,
|
|
642
|
+
ambiguous: ambiguousLabels(map?.rows),
|
|
643
|
+
});
|
|
644
|
+
}
|
|
645
|
+
|
|
646
|
+
/**
|
|
647
|
+
* Keep only the remembered actions whose control is actually on this screen.
|
|
648
|
+
*
|
|
649
|
+
* @returns {{exitList: Array, stale: number}} what can be offered, and how many
|
|
650
|
+
* remembered actions found nothing here — which is evidence the screen has
|
|
651
|
+
* been misidentified, and worth saying out loud.
|
|
652
|
+
*/
|
|
653
|
+
export function presentOnly(remembered, rows) {
|
|
654
|
+
const here = new Set();
|
|
655
|
+
for (const r of rows ?? []) {
|
|
656
|
+
for (const name of [r.label, ...(r.aliases ?? [])]) {
|
|
657
|
+
const k = alnum(name);
|
|
658
|
+
if (k) here.add(k);
|
|
659
|
+
}
|
|
660
|
+
}
|
|
661
|
+
const has = (label) => {
|
|
662
|
+
const k = alnum(label);
|
|
663
|
+
if (!k) return false;
|
|
664
|
+
if (here.has(k)) return true;
|
|
665
|
+
// A row may carry the label inside a longer one — a list card concatenates
|
|
666
|
+
// its children, and truncation adds an ellipsis.
|
|
667
|
+
for (const seen of here) if (seen.includes(k) || k.includes(seen)) return true;
|
|
668
|
+
return false;
|
|
669
|
+
};
|
|
670
|
+
const exitList = (remembered ?? []).filter((e) => has(e.label));
|
|
671
|
+
return { exitList, stale: (remembered ?? []).length - exitList.length };
|
|
672
|
+
}
|
|
673
|
+
|
|
674
|
+
/**
|
|
675
|
+
* What has worked from this screen before, printed rather than counted.
|
|
676
|
+
*
|
|
677
|
+
* The map said `(known, 3 known exits)` and stopped there, so the graph's own
|
|
678
|
+
* vocabulary never reached the caller. A flow whose labels were known in
|
|
679
|
+
* advance ran 16 steps in one call; the same agent on screens the graph also
|
|
680
|
+
* knew, but whose labels it had to rediscover, spent 25 calls on 31 steps.
|
|
681
|
+
*
|
|
682
|
+
* Deliberately terse and deliberately *not* a promise. These are actions that
|
|
683
|
+
* previously worked here, with how often — evidence for a plan, not a
|
|
684
|
+
* guarantee, and the destructive-label rules apply to them exactly as before.
|
|
685
|
+
*/
|
|
686
|
+
export function exitsLine(exitList, { limit = 6, stale = 0 } = {}) {
|
|
687
|
+
const list = (exitList ?? []).filter((e) => e.label).slice(0, limit);
|
|
688
|
+
if (!list.length) {
|
|
689
|
+
// Everything remembered here is missing. That is not "no memory" — it is
|
|
690
|
+
// memory that contradicts the screen, which usually means two screens have
|
|
691
|
+
// collapsed into one identity, and it is the most useful thing to say.
|
|
692
|
+
return stale
|
|
693
|
+
? `memory disagrees with this screen: ${stale} remembered control${stale === 1 ? '' : 's'} not present`
|
|
694
|
+
+ ' — this screen has probably been confused with another. Trust the element list, not the graph.'
|
|
695
|
+
: null;
|
|
696
|
+
}
|
|
697
|
+
const parts = list.map((e) => {
|
|
698
|
+
const label = String(e.label).length > 28 ? `${String(e.label).slice(0, 28)}…` : String(e.label);
|
|
699
|
+
return `${e.action} ${JSON.stringify(label)}${e.count > 1 ? ` (${e.count}x)` : ''}`;
|
|
700
|
+
});
|
|
701
|
+
return `worked here before: ${parts.join(', ')}`;
|
|
702
|
+
}
|
|
703
|
+
|
|
704
|
+
/** How many labels are worn by more than one element a caller could act on. */
|
|
705
|
+
export function ambiguousLabels(rows) {
|
|
706
|
+
const seen = new Map();
|
|
707
|
+
for (const r of rows ?? []) {
|
|
708
|
+
const key = alnum(r.label);
|
|
709
|
+
if (!key) continue;
|
|
710
|
+
seen.set(key, (seen.get(key) ?? 0) + 1);
|
|
711
|
+
}
|
|
712
|
+
return [...seen.values()].filter((n) => n > 1).length;
|
|
713
|
+
}
|
|
714
|
+
|
|
715
|
+
export function render({ device, identity, rows, truncated, collapsed, screen, name, exits, exitList, staleExits, verdictLine, ambiguities, cleared, overlay }) {
|
|
374
716
|
const head = [
|
|
375
717
|
device?.name,
|
|
376
718
|
screen?.width ? `${screen.width}x${screen.height}pt` : null,
|
|
@@ -380,11 +722,33 @@ export function render({ device, identity, rows, truncated, collapsed, screen, n
|
|
|
380
722
|
: 'screen unidentified',
|
|
381
723
|
identity?.keyboard ? 'keyboard up' : null,
|
|
382
724
|
identity?.settled === false ? 'STILL MOVING' : null,
|
|
725
|
+
// Still and finished are not the same thing.
|
|
726
|
+
identity?.loading === true ? 'STILL LOADING' : null,
|
|
727
|
+
// How old the *frame* this map was read from is.
|
|
728
|
+
//
|
|
729
|
+
// `sim_look` and `sim_state` have printed this since they existed, and this
|
|
730
|
+
// map never has — so the one tool an agent is told to start with was the one
|
|
731
|
+
// that could not say how old its evidence was. Reported from the field: a
|
|
732
|
+
// complete 20-element map of a screen the app was not on, and *"a wrong
|
|
733
|
+
// answer is worse than an error here, because nothing downstream knows to
|
|
734
|
+
// doubt it"*. `ensureDaemon` will hand back a frame up to 30s old, so this
|
|
735
|
+
// was reachable without anything being broken.
|
|
736
|
+
//
|
|
737
|
+
// `recalledNote` below is a different claim — that the *element map* came
|
|
738
|
+
// from memory — and having one was what made the absence of the other easy
|
|
739
|
+
// to miss.
|
|
740
|
+
frameAgeNote(identity),
|
|
383
741
|
recalledNote(identity),
|
|
384
742
|
].filter(Boolean).join(' · ');
|
|
385
743
|
|
|
386
744
|
const lines = [head];
|
|
387
745
|
if (verdictLine) lines.push(verdictLine);
|
|
746
|
+
// Above the element list, not below it: it contradicts something the caller
|
|
747
|
+
// already believes, which is the one kind of news that must not be scrolled to.
|
|
748
|
+
if (cleared) lines.push(cleared);
|
|
749
|
+
if (overlay) lines.push(overlay);
|
|
750
|
+
const worked = exitsLine(exitList, { stale: staleExits });
|
|
751
|
+
if (worked) lines.push(worked);
|
|
388
752
|
|
|
389
753
|
let region = null;
|
|
390
754
|
for (const r of rows) {
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The words simframe may not act on by itself.
|
|
3
|
+
*
|
|
4
|
+
* CLAUDE.md has required this since the human-parity series was written — *"a
|
|
5
|
+
* reflex never taps anything whose label matches the destructive vocabulary
|
|
6
|
+
* (Delete, Remove, Pay, Send, Sign out, Reset…). Add the list to the same data
|
|
7
|
+
* file."* — and several of my own comments already talk about "the destructive
|
|
8
|
+
* vocabulary" as though it existed. It did not. This is it.
|
|
9
|
+
*
|
|
10
|
+
* **What it gates, precisely.** This restricts what simframe does *on its own
|
|
11
|
+
* initiative*: a retry, an alternative selector, an exploration step, a reflex,
|
|
12
|
+
* a speculative tap. It never restricts what the caller explicitly asked for.
|
|
13
|
+
* `{"tap": "DELETE ACCOUNT"}` is a request and is honoured; substituting
|
|
14
|
+
* "DELETE ACCOUNT" for a "Done" that did not resolve is not, and that is the
|
|
15
|
+
* whole distinction. Getting it backwards would make the tool refuse the thing
|
|
16
|
+
* a tester most needs to test.
|
|
17
|
+
*
|
|
18
|
+
* Data, not code, and locale-keyed, so another language is a file rather than a
|
|
19
|
+
* release.
|
|
20
|
+
*/
|
|
21
|
+
import fs from 'node:fs';
|
|
22
|
+
import path from 'node:path';
|
|
23
|
+
import { fileURLToPath } from 'node:url';
|
|
24
|
+
|
|
25
|
+
const HERE = path.dirname(fileURLToPath(import.meta.url));
|
|
26
|
+
const DATA = path.join(HERE, '..', 'data', 'vocabulary');
|
|
27
|
+
|
|
28
|
+
const cache = new Map();
|
|
29
|
+
|
|
30
|
+
/** The vocabulary for a locale, falling back to English. */
|
|
31
|
+
export function load(locale = process.env.SIMFRAME_LOCALE || 'en') {
|
|
32
|
+
const key = String(locale).toLowerCase().split(/[-_]/)[0];
|
|
33
|
+
if (cache.has(key)) return cache.get(key);
|
|
34
|
+
let data = null;
|
|
35
|
+
for (const candidate of [key, 'en']) {
|
|
36
|
+
try {
|
|
37
|
+
data = JSON.parse(fs.readFileSync(path.join(DATA, `${candidate}.json`), 'utf8'));
|
|
38
|
+
break;
|
|
39
|
+
} catch { /* try the fallback */ }
|
|
40
|
+
}
|
|
41
|
+
// A missing file must not silently disable the barrier. An empty vocabulary
|
|
42
|
+
// would make every label safe, which is the wrong direction to fail in, so
|
|
43
|
+
// this throws rather than returning nothing.
|
|
44
|
+
if (!data) throw new Error(`no vocabulary for "${locale}" and no en fallback in ${DATA}`);
|
|
45
|
+
cache.set(key, data);
|
|
46
|
+
return data;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
const alnum = (s) => String(s ?? '').toLowerCase().replace(/[^\p{L}\p{N}]+/gu, ' ').trim();
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Does this phrase occur in the label as whole words?
|
|
53
|
+
*
|
|
54
|
+
* Word boundaries, not substrings, and the reason is a real screen: an app's
|
|
55
|
+
* "Work Orders" tab contains the letters of "order", and a substring match would
|
|
56
|
+
* make its main navigation untouchable by anything local. Matching whole words
|
|
57
|
+
* means "order" does not match "orders", which is the behaviour wanted here —
|
|
58
|
+
* an exact tappable word is the signal, and a longer word is a different word.
|
|
59
|
+
*/
|
|
60
|
+
function saysPhrase(label, phrase) {
|
|
61
|
+
const l = ` ${alnum(label)} `;
|
|
62
|
+
const p = alnum(phrase);
|
|
63
|
+
if (!p) return false;
|
|
64
|
+
return l.includes(` ${p} `);
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* May simframe act on this label on its own initiative?
|
|
69
|
+
*
|
|
70
|
+
* `purpose` matters, and conflating two purposes cost a real run. The default,
|
|
71
|
+
* `substitute`, answers *"may a retry aim at this instead?"* and permits
|
|
72
|
+
* "Cancel", deliberately, so a local tier can decline a dialog rather than
|
|
73
|
+
* stranding on every confirmation it meets.
|
|
74
|
+
*
|
|
75
|
+
* `explore` answers a different question — *"may I open this as a door and see
|
|
76
|
+
* what is behind it?"* — and there "Cancel" is the abandon-this-task control.
|
|
77
|
+
* `seek` asked the first question and got the first answer: it opened CANCEL
|
|
78
|
+
* first, then AI TROUBLESHOOTING, then pressed "YES, THIS FIXED MY PROBLEM",
|
|
79
|
+
* ending five screens deep in a live support chat with a half-built service
|
|
80
|
+
* request destroyed. One label further along was SUBMIT SERVICE REQUEST.
|
|
81
|
+
*
|
|
82
|
+
* So exploration has its own list and it errs toward refusing. A door missed
|
|
83
|
+
* costs one step of a bounded budget; a door taken wrongly costs the run, and
|
|
84
|
+
* can cost the thing being tested.
|
|
85
|
+
*
|
|
86
|
+
* @returns {{allowed: boolean, reason?: string, matched?: string}}
|
|
87
|
+
*/
|
|
88
|
+
export function mayActLocally(label, { locale, purpose = 'substitute' } = {}) {
|
|
89
|
+
const text = String(label ?? '').trim();
|
|
90
|
+
if (!text) return { allowed: purpose !== 'explore' };
|
|
91
|
+
const vocab = load(locale);
|
|
92
|
+
|
|
93
|
+
if (purpose === 'explore') {
|
|
94
|
+
// Checked before the "listed as safe" exemption below, which exists for
|
|
95
|
+
// declining dialogs and must never make something a door.
|
|
96
|
+
for (const word of vocab.exploration?.neverOpen ?? []) {
|
|
97
|
+
if (saysPhrase(text, word)) {
|
|
98
|
+
return { allowed: false, reason: 'not a door — it commits, abandons or answers', matched: word };
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
for (const pattern of vocab.exploration?.neverOpenPatterns ?? []) {
|
|
102
|
+
if (new RegExp(pattern, 'i').test(alnum(text)) || new RegExp(pattern, 'i').test(text.toLowerCase())) {
|
|
103
|
+
return { allowed: false, reason: 'not a door — it reads as an instruction or an answer', matched: pattern };
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
// Exceptions first, and matched against the **whole** label rather than as a
|
|
109
|
+
// phrase inside it. "Cancel" is how you *decline* a dialog and a barrier that
|
|
110
|
+
// refused it would strand a local tier on every confirmation it met — but
|
|
111
|
+
// "Cancel order" is a different act, and a phrase match would have waved it
|
|
112
|
+
// through on the strength of its first word.
|
|
113
|
+
const whole = alnum(text);
|
|
114
|
+
for (const ok of vocab.destructive?.notWords?.words ?? []) {
|
|
115
|
+
if (whole === alnum(ok)) return { allowed: true, reason: 'listed as safe', matched: ok };
|
|
116
|
+
}
|
|
117
|
+
for (const word of vocab.destructive?.words ?? []) {
|
|
118
|
+
if (saysPhrase(text, word)) {
|
|
119
|
+
return { allowed: false, reason: 'destructive vocabulary', matched: word };
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
for (const word of vocab.leavesTheApp?.words ?? []) {
|
|
123
|
+
if (saysPhrase(text, word)) {
|
|
124
|
+
return { allowed: false, reason: 'leaves the app', matched: word };
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
return { allowed: true };
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
/** Convenience for a filter: keep only what a local tier may act on. */
|
|
131
|
+
export const actableLocally = (label, options) => mayActLocally(label, options).allowed;
|
|
132
|
+
|
|
133
|
+
/** May exploration open this as a door? Stricter than substitution, on purpose. */
|
|
134
|
+
export const openableAsDoor = (label, options) => mayActLocally(label, { ...options, purpose: 'explore' }).allowed;
|