verikun 0.9.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -9
- package/dist/agent/claude.js +6 -0
- package/dist/agent/cli-provider.js +69 -1
- package/dist/agent/cost.js +6 -4
- package/dist/agent/engine.js +332 -42
- package/dist/agent/grammar.js +73 -6
- package/dist/agent/ir.js +220 -46
- package/dist/agent/lint.js +77 -0
- package/dist/agent/openai.js +6 -0
- package/dist/bin/verikun.js +0 -0
- package/dist/cli.js +104 -40
- package/dist/ui/selector.js +17 -1
- package/dist/version.js +1 -1
- package/package.json +1 -1
- package/dist/drivers/simctl.js +0 -156
package/dist/agent/engine.js
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
-
exports.DEFAULT_RUN_TIMEOUT_MS = void 0;
|
|
3
|
+
exports.DEFAULT_GUARD_SETTLE_MS = exports.DEFAULT_RUN_TIMEOUT_MS = void 0;
|
|
4
4
|
exports.runPlan = runPlan;
|
|
5
|
+
const node_crypto_1 = require("node:crypto");
|
|
5
6
|
const selector_1 = require("../ui/selector");
|
|
6
7
|
const errors_1 = require("../errors");
|
|
7
8
|
const ir_1 = require("./ir");
|
|
@@ -13,23 +14,120 @@ const describe = (leaf) => [leaf.command, ...leaf.positionals, ...leaf.flags.map
|
|
|
13
14
|
* fails (a device hiccup on screencap) must never turn a green run red — see
|
|
14
15
|
* the guard in execLeaf. */
|
|
15
16
|
const isScreenshotLeaf = (leaf) => leaf.command === 'screenshot' || leaf.command === 'shot';
|
|
16
|
-
/** A structural fingerprint of the screen: sorted id+text+type set. Used for the
|
|
17
|
-
* loop no-progress check — deliberately NOT the raw hierarchy (its node ordering
|
|
18
|
-
*
|
|
17
|
+
/** A structural fingerprint of the screen: a sorted id+text+desc+type set. Used for the
|
|
18
|
+
* loop no-progress check — deliberately NOT the raw hierarchy (its node ordering is
|
|
19
|
+
* nondeterministic between identical states, which would false-trip).
|
|
20
|
+
*
|
|
21
|
+
* BOTH content fields are sampled, and that is load-bearing rather than belt-and-braces.
|
|
22
|
+
* Sampling `text` alone made this blind on Flutter apps, which map `Semantics(label:)` to
|
|
23
|
+
* Android's `contentDescription` — i.e. to `desc`, never to `text`. Measured on a live
|
|
24
|
+
* Flutter screen: 14 elements, **0** carrying `text`, 8 carrying `desc`. The fingerprint
|
|
25
|
+
* degenerated to `id||type` for the entire screen, so two completely different questions
|
|
26
|
+
* hashed byte-identically and a loop answering them correctly was declared stalled.
|
|
27
|
+
*
|
|
28
|
+
* Uses the full `id`, not `idShort` (the suffix after the last '/'), so elements from
|
|
29
|
+
* different packages/namespaces cannot collide in what is meant to be a fingerprint.
|
|
30
|
+
*
|
|
31
|
+
* When in doubt, sample MORE: an over-sensitive hash only costs a loop running to its cap;
|
|
32
|
+
* an under-sensitive one fails a passing test, since a stalled loop is now fatal. */
|
|
19
33
|
function structuralHash(els) {
|
|
20
34
|
return els
|
|
21
|
-
.map((e) => `${e.
|
|
35
|
+
.map((e) => `${e.id}|${e.text}|${e.desc}|${e.type}`)
|
|
22
36
|
.sort()
|
|
23
37
|
.join('\n');
|
|
24
38
|
}
|
|
39
|
+
/** An unresolvable {{...}} placeholder. Terminal and never healed: the model cannot
|
|
40
|
+
* repair a missing value, and substituting one would be inventing test data. */
|
|
41
|
+
class CtxError extends Error {
|
|
42
|
+
}
|
|
25
43
|
function isHealable(outcome) {
|
|
26
44
|
return (!!outcome.error &&
|
|
27
45
|
(outcome.error instanceof errors_1.SelectorNotFoundError || outcome.error instanceof errors_1.AmbiguousSelectorError));
|
|
28
46
|
}
|
|
29
47
|
/** Default wall-clock ceiling for a whole `vk ai` run (overridable via --timeout). */
|
|
30
48
|
exports.DEFAULT_RUN_TIMEOUT_MS = 15 * 60 * 1000;
|
|
49
|
+
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
|
50
|
+
/** How long a CONDITIONAL guard (`if-present`) waits for its selector to show up
|
|
51
|
+
* before concluding "absent". Interstitials animate in: a permission dialog or promo
|
|
52
|
+
* panel typically lands a few hundred ms after the transition that triggers it. Every
|
|
53
|
+
* selector-resolving leaf command already auto-waits ~5s (cli.ts resolveOneWaiting),
|
|
54
|
+
* so before this window existed a guard was strictly LESS patient than a bare `tap` —
|
|
55
|
+
* and an optional dialog could be missed by the very construct meant to catch it.
|
|
56
|
+
* Kept far below the leaf 5s because an ABSENT guard pays this window, every time, and
|
|
57
|
+
* skipping fast is the common case.
|
|
58
|
+
*
|
|
59
|
+
* NOTE the floor in present(): any non-zero window buys at least TWO looks at the screen.
|
|
60
|
+
* Wall clock alone is not a usable unit here — a uiautomator dump measured ~2.4s on
|
|
61
|
+
* emulator-5554 and can be ~10x faster on a physical device, so a pure time box gives a
|
|
62
|
+
* fast phone a dozen looks and a slow emulator none. Set 0 to restore the old
|
|
63
|
+
* single-shot probe. Override per run with VERIKUN_GUARD_SETTLE_MS (see cli.ts). */
|
|
64
|
+
exports.DEFAULT_GUARD_SETTLE_MS = 1500;
|
|
65
|
+
/** Re-dump cadence inside a guard's settle window. */
|
|
66
|
+
const GUARD_POLL_MS = 150;
|
|
67
|
+
/** Consecutive identical screen snapshots before a loop is believed to be stuck.
|
|
68
|
+
*
|
|
69
|
+
* This check is a TIME SAVER and nothing more. A loop already fails when its exit
|
|
70
|
+
* selector never appears, so stopping early buys no safety — it only decides how long we
|
|
71
|
+
* wait before reaching the same verdict. That asymmetry is the whole design: a false
|
|
72
|
+
* positive fails a passing test, while a false negative costs `cap` iterations of runtime.
|
|
73
|
+
*
|
|
74
|
+
* Tuned from a measured false positive, twice over. At 2 strikes the same plan answered
|
|
75
|
+
* 17 questions on one run and bailed after 4 on the next — the loop bodies drive screen
|
|
76
|
+
* transitions, and a single dump lands mid-transition often enough to produce a
|
|
77
|
+
* coincidental pair of matching hashes.
|
|
78
|
+
*
|
|
79
|
+
* 4 discriminates well because the two cases differ in character, not just degree: a loop
|
|
80
|
+
* genuinely at rest (a scrolled-to-the-bottom list — the case this exists for) yields
|
|
81
|
+
* identical hashes indefinitely, whereas an animating screen rarely yields four in a row. */
|
|
82
|
+
const NO_PROGRESS_STRIKES = 4;
|
|
31
83
|
async function runPlan(plan, deps) {
|
|
32
84
|
const maxRepairs = deps.maxRepairs ?? 3;
|
|
85
|
+
const guardSettleMs = deps.guardSettleMs ?? exports.DEFAULT_GUARD_SETTLE_MS;
|
|
86
|
+
// The run's mutable state, scoped to THIS runPlan call — never module-level. `vk suite`
|
|
87
|
+
// runs every test in one process, so a shared store would hand two sign-up tests the
|
|
88
|
+
// same {{uuid}} and recreate the collision it exists to prevent. Per-leaf generation
|
|
89
|
+
// would be equally wrong: a sign-up needs the same address in the email field, the
|
|
90
|
+
// confirm field, and a later assert.
|
|
91
|
+
const ctx = new Map(Object.entries(deps.initialCtx ?? {}));
|
|
92
|
+
const generated = new Map();
|
|
93
|
+
const runId = deps.runId ?? 'run';
|
|
94
|
+
/** Resolve {{...}} placeholders. A closed set on purpose — this is a template
|
|
95
|
+
* substitution, not an expression language, so there is nothing to sandbox.
|
|
96
|
+
* Unknown placeholders are left verbatim rather than blanked: a silent empty string
|
|
97
|
+
* is how a typo becomes a false green. */
|
|
98
|
+
function interpolate(s) {
|
|
99
|
+
if (!s.includes('{{'))
|
|
100
|
+
return s;
|
|
101
|
+
return s.replace(/\{\{\s*([A-Za-z_][A-Za-z0-9_.]*)\s*\}\}/g, (whole, name) => {
|
|
102
|
+
if (name.startsWith('ctx.')) {
|
|
103
|
+
const key = name.slice(4);
|
|
104
|
+
const v = ctx.get(key);
|
|
105
|
+
if (v === undefined)
|
|
106
|
+
throw new CtxError(`{{${name}}} is not set — no earlier step stored it`);
|
|
107
|
+
return v;
|
|
108
|
+
}
|
|
109
|
+
if (name.startsWith('env.')) {
|
|
110
|
+
const key = name.slice(4);
|
|
111
|
+
const v = process.env[key];
|
|
112
|
+
// Loud, not empty: a missing CI secret must fail the step, not type "" into a field.
|
|
113
|
+
if (v === undefined || v === '')
|
|
114
|
+
throw new CtxError(`{{${name}}} is not set in the environment`);
|
|
115
|
+
return v;
|
|
116
|
+
}
|
|
117
|
+
if (name === 'run_id')
|
|
118
|
+
return runId;
|
|
119
|
+
// uuid/timestamp are generated ONCE per run and memoized by placeholder name.
|
|
120
|
+
if (name === 'uuid' || name === 'timestamp') {
|
|
121
|
+
const existing = generated.get(name);
|
|
122
|
+
if (existing !== undefined)
|
|
123
|
+
return existing;
|
|
124
|
+
const v = name === 'uuid' ? (0, node_crypto_1.randomUUID)() : String(Date.now());
|
|
125
|
+
generated.set(name, v);
|
|
126
|
+
return v;
|
|
127
|
+
}
|
|
128
|
+
return whole;
|
|
129
|
+
});
|
|
130
|
+
}
|
|
33
131
|
const overDeadline = () => deps.deadline !== undefined && Date.now() >= deps.deadline;
|
|
34
132
|
const improvements = [];
|
|
35
133
|
let modelRepairs = 0;
|
|
@@ -44,24 +142,22 @@ async function runPlan(plan, deps) {
|
|
|
44
142
|
return [];
|
|
45
143
|
}
|
|
46
144
|
};
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
if (els === undefined)
|
|
62
|
-
return false;
|
|
145
|
+
/** Is `selector` on screen? `settleMs` is how long to keep re-dumping while it is
|
|
146
|
+
* absent before answering "no" — see DEFAULT_GUARD_SETTLE_MS.
|
|
147
|
+
*
|
|
148
|
+
* Two retry behaviours here are deliberately ORTHOGONAL, and conflating them is the
|
|
149
|
+
* bug this signature exists to prevent:
|
|
150
|
+
* - a dump that THROWS is always retried once, even at settleMs=0, so a flaky
|
|
151
|
+
* uiautomator call never silently reads as "absent" and skips a body that
|
|
152
|
+
* should run;
|
|
153
|
+
* - a dump that SUCCEEDS but does not match is re-polled only while settleMs
|
|
154
|
+
* remains. At settleMs=0 that means exactly one pass — the fast, single-shot
|
|
155
|
+
* probe a loop-exit check needs.
|
|
156
|
+
* So one dump attempt always happens regardless of the window. */
|
|
157
|
+
const present = async (selector, settleMs) => {
|
|
158
|
+
let sel;
|
|
63
159
|
try {
|
|
64
|
-
|
|
160
|
+
sel = (0, selector_1.parseSelector)(selector);
|
|
65
161
|
}
|
|
66
162
|
catch (e) {
|
|
67
163
|
// A guard selector that won't parse is a compiler/plan bug — surface it (then treat
|
|
@@ -69,14 +165,68 @@ async function runPlan(plan, deps) {
|
|
|
69
165
|
deps.log(`[ai] guard selector '${selector}' did not parse (${e.message}) — treating as not present`);
|
|
70
166
|
return false;
|
|
71
167
|
}
|
|
168
|
+
const deadline = Date.now() + Math.max(0, settleMs);
|
|
169
|
+
// A non-zero window must buy at least one SECOND look, independent of the clock.
|
|
170
|
+
// Measured on emulator-5554: one uiautomator dump costs ~2.4s, which already exceeds
|
|
171
|
+
// a 1.5s window — so a purely time-boxed loop returns after a single dump and the
|
|
172
|
+
// window is a silent no-op on exactly the slow devices that need it most. Dump cost
|
|
173
|
+
// swings ~10x across devices, so "how long to wait" cannot be expressed in wall clock
|
|
174
|
+
// alone without making the guard's patience device-dependent.
|
|
175
|
+
let looks = 0;
|
|
176
|
+
const minLooks = settleMs > 0 ? 2 : 1;
|
|
177
|
+
for (;;) {
|
|
178
|
+
let els;
|
|
179
|
+
for (let i = 0; i < 2; i++) {
|
|
180
|
+
try {
|
|
181
|
+
els = await deps.getElements();
|
|
182
|
+
}
|
|
183
|
+
catch {
|
|
184
|
+
els = undefined; // transient dump failure — retry once before concluding "absent"
|
|
185
|
+
}
|
|
186
|
+
// An EMPTY tree is not a screen, it is a bad read: a live app always has nodes, and
|
|
187
|
+
// this device routinely returns a partial/blank dump mid-transition (measured: `ui`
|
|
188
|
+
// reported zero elements while `tap`'s auto-wait found its target 3.1s later).
|
|
189
|
+
// Trusting it means "absent" — which silently skips a guard, or sends a loop round
|
|
190
|
+
// again to tap something that already went away. Retry once even at settleMs=0,
|
|
191
|
+
// which is what the loop-exit check runs at.
|
|
192
|
+
if (els !== undefined && els.length > 0)
|
|
193
|
+
break;
|
|
194
|
+
}
|
|
195
|
+
looks++;
|
|
196
|
+
if (els !== undefined && (0, selector_1.matchElements)(els, sel).matches.length > 0)
|
|
197
|
+
return true;
|
|
198
|
+
const remaining = deadline - Date.now();
|
|
199
|
+
if (looks >= minLooks && remaining <= 0)
|
|
200
|
+
return false;
|
|
201
|
+
await sleep(Math.max(0, Math.min(GUARD_POLL_MS, remaining)));
|
|
202
|
+
}
|
|
203
|
+
};
|
|
204
|
+
const runLeaf = (leaf) => {
|
|
205
|
+
const flags = (0, ir_1.leafToFlags)(leaf);
|
|
206
|
+
for (const k of Object.keys(flags))
|
|
207
|
+
flags[k] = interpolate(flags[k]);
|
|
208
|
+
return deps.exec(leaf.command, leaf.positionals.map(interpolate), flags);
|
|
72
209
|
};
|
|
73
|
-
const runLeaf = (leaf) => deps.exec(leaf.command, leaf.positionals, (0, ir_1.leafToFlags)(leaf));
|
|
74
210
|
/** Execute one leaf, healing a selector miss/ambiguity via the model up to the cap.
|
|
75
211
|
* `replace` writes a repaired leaf back into the plan so it persists on green. */
|
|
76
|
-
async function execLeaf(leaf, where, replace) {
|
|
212
|
+
async function execLeaf(leaf, where, replace, guard) {
|
|
77
213
|
deps.log(`[ai] ${where}: ${describe(leaf)}`);
|
|
78
214
|
let current = leaf;
|
|
79
215
|
let outcome = await runLeaf(current);
|
|
216
|
+
// Guard raced the body: `if-present X { … tap X … }` checked X, X was there, and by
|
|
217
|
+
// the time the tap ran it had gone. That is not drift and there is nothing to repair —
|
|
218
|
+
// it is exactly the transient the guard exists to tolerate, and on a fast-transitioning
|
|
219
|
+
// app it happens routinely. Checked BEFORE the heal loop, because otherwise it costs
|
|
220
|
+
// three model calls to conclude that a thing which was optional is absent.
|
|
221
|
+
// Deliberately narrow: only the leaf that targets the guard's own selector, and only
|
|
222
|
+
// once the guard is confirmed false again.
|
|
223
|
+
if (guard !== undefined &&
|
|
224
|
+
outcome.error instanceof errors_1.SelectorNotFoundError &&
|
|
225
|
+
current.positionals.some((p) => interpolate(p) === interpolate(guard)) &&
|
|
226
|
+
!(await present(interpolate(guard), 0))) {
|
|
227
|
+
deps.log(`[ai] ${where}: '${interpolate(guard)}' disappeared after the guard matched — ending the guarded body`);
|
|
228
|
+
return { status: 'guard-gone' };
|
|
229
|
+
}
|
|
80
230
|
let attempts = 0;
|
|
81
231
|
while (isHealable(outcome) && attempts < maxRepairs) {
|
|
82
232
|
if (deps.cost.exceeded()) {
|
|
@@ -154,49 +304,189 @@ async function runPlan(plan, deps) {
|
|
|
154
304
|
const reason = outcome.error ? outcome.error.message.split('\n')[0] : `exited ${outcome.code}`;
|
|
155
305
|
return { status: 'fail', where, reason };
|
|
156
306
|
}
|
|
157
|
-
async function walkBody(body, parentWhere) {
|
|
307
|
+
async function walkBody(body, parentWhere, guard) {
|
|
158
308
|
for (let j = 0; j < body.length; j++) {
|
|
159
|
-
const res = await
|
|
309
|
+
const res = await walkNode(body[j], `${parentWhere}[${j}]`, (l) => (body[j] = l), guard);
|
|
310
|
+
// 'guard-gone' ends this body, not the run: the thing the guard matched went away
|
|
311
|
+
// mid-body, which is precisely the situation the guard exists to tolerate.
|
|
312
|
+
if (res.status === 'guard-gone')
|
|
313
|
+
return { status: 'ok' };
|
|
160
314
|
if (res.status !== 'ok')
|
|
161
315
|
return res;
|
|
162
316
|
}
|
|
163
317
|
return { status: 'ok' };
|
|
164
318
|
}
|
|
165
|
-
|
|
319
|
+
/** Capture a value from the live tree into ctx. Not routed through `exec`: the
|
|
320
|
+
* ExecFn contract returns only {code,error}, so a leaf could never hand a value back. */
|
|
321
|
+
async function execRead(node, where) {
|
|
322
|
+
const selector = interpolate(node.selector);
|
|
323
|
+
let sel;
|
|
324
|
+
try {
|
|
325
|
+
sel = (0, selector_1.parseSelector)(selector);
|
|
326
|
+
}
|
|
327
|
+
catch (e) {
|
|
328
|
+
return { status: 'fail', where, reason: `read selector '${selector}' did not parse: ${e.message}` };
|
|
329
|
+
}
|
|
330
|
+
// Same patience as a conditional guard: the value may not have rendered yet.
|
|
331
|
+
const deadline = Date.now() + Math.max(0, guardSettleMs);
|
|
332
|
+
let looks = 0;
|
|
333
|
+
const minLooks = guardSettleMs > 0 ? 2 : 1;
|
|
334
|
+
for (;;) {
|
|
335
|
+
const els = await safeElements();
|
|
336
|
+
const { matches } = (0, selector_1.matchElements)(els, sel);
|
|
337
|
+
looks++;
|
|
338
|
+
if (matches.length > 0) {
|
|
339
|
+
const value = String(matches[0][node.field] ?? '');
|
|
340
|
+
ctx.set(node.into, value);
|
|
341
|
+
deps.log(`[ai] ${where}: read ${node.field} of '${selector}' → ctx.${node.into} = ${JSON.stringify(value)}`);
|
|
342
|
+
return { status: 'ok' };
|
|
343
|
+
}
|
|
344
|
+
const remaining = deadline - Date.now();
|
|
345
|
+
if (looks >= minLooks && remaining <= 0) {
|
|
346
|
+
// Terminal, not healable: a missing source value would silently propagate an
|
|
347
|
+
// empty string into every later {{ctx.*}} use, which is a false green waiting
|
|
348
|
+
// to happen. Fail where the information is.
|
|
349
|
+
return { status: 'fail', where, reason: `read found no element matching '${selector}' (nothing to store in ctx.${node.into})` };
|
|
350
|
+
}
|
|
351
|
+
await sleep(Math.max(0, Math.min(GUARD_POLL_MS, remaining)));
|
|
352
|
+
}
|
|
353
|
+
}
|
|
354
|
+
async function walkNode(node, where, replace, guard) {
|
|
355
|
+
try {
|
|
356
|
+
return await walkNodeInner(node, where, replace, guard);
|
|
357
|
+
}
|
|
358
|
+
catch (e) {
|
|
359
|
+
// An unresolvable placeholder is a plan defect, not a device failure: report it
|
|
360
|
+
// as this step's terminal failure rather than letting it abort the whole run as
|
|
361
|
+
// an "unexpected error" (exit 3).
|
|
362
|
+
if (e instanceof CtxError)
|
|
363
|
+
return { status: 'fail', where, reason: e.message };
|
|
364
|
+
throw e;
|
|
365
|
+
}
|
|
366
|
+
}
|
|
367
|
+
async function walkNodeInner(node, where, replace, guard) {
|
|
166
368
|
switch (node.type) {
|
|
167
369
|
case 'command':
|
|
168
|
-
return execLeaf(node, where, replace);
|
|
370
|
+
return execLeaf(node, where, replace, guard);
|
|
371
|
+
case 'read':
|
|
372
|
+
return execRead(node, where);
|
|
169
373
|
case 'if-present': {
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
374
|
+
const selector = interpolate(node.selector);
|
|
375
|
+
if (await present(selector, guardSettleMs)) {
|
|
376
|
+
deps.log(`[ai] ${where}: if-present '${selector}' → present, running ${node.body.length} step(s)`);
|
|
377
|
+
return walkBody(node.body, `${where}.body`, selector);
|
|
173
378
|
}
|
|
174
|
-
deps.log(`[ai] ${where}: if-present '${
|
|
379
|
+
deps.log(`[ai] ${where}: if-present '${selector}' → absent, skipping`);
|
|
175
380
|
return { status: 'ok' };
|
|
176
381
|
}
|
|
382
|
+
case 'when': {
|
|
383
|
+
for (let b = 0; b < node.branches.length; b++) {
|
|
384
|
+
const selector = interpolate(node.branches[b].selector);
|
|
385
|
+
// Ordered: first present branch wins, and only it runs. The settle window is
|
|
386
|
+
// per NODE, spent on the first branch checked — re-polling every branch would
|
|
387
|
+
// cost cap x branches x window inside a loop, and would let a slower-appearing
|
|
388
|
+
// earlier branch beat an already-present later one (making dispatch depend on
|
|
389
|
+
// timing rather than on order).
|
|
390
|
+
if (await present(selector, b === 0 ? guardSettleMs : 0)) {
|
|
391
|
+
deps.log(`[ai] ${where}: when → branch ${b} '${selector}' matched`);
|
|
392
|
+
return walkBody(node.branches[b].body, `${where}.branches[${b}].body`);
|
|
393
|
+
}
|
|
394
|
+
}
|
|
395
|
+
if (node.else) {
|
|
396
|
+
deps.log(`[ai] ${where}: when → no branch matched, running else (${node.else.length} step(s))`);
|
|
397
|
+
return walkBody(node.else, `${where}.else`);
|
|
398
|
+
}
|
|
399
|
+
// No branch, no else => FAIL. Skipping here is the false-green this node exists
|
|
400
|
+
// to prevent: inside a loop it would spin to the cap doing nothing and pass.
|
|
401
|
+
const tried = node.branches.map((b) => `'${interpolate(b.selector)}'`).join(', ');
|
|
402
|
+
return {
|
|
403
|
+
status: 'fail',
|
|
404
|
+
where,
|
|
405
|
+
reason: `when: no branch matched (tried ${tried}) and no else was given — the screen is one this test does not handle`,
|
|
406
|
+
};
|
|
407
|
+
}
|
|
177
408
|
case 'repeat': {
|
|
178
409
|
let prevHash = '';
|
|
179
|
-
|
|
410
|
+
let stalled = 0;
|
|
411
|
+
let satisfied = false;
|
|
412
|
+
let i = 0;
|
|
413
|
+
for (; i < node.cap; i++) {
|
|
180
414
|
if (overDeadline()) {
|
|
181
415
|
deps.log(`[ai] ${where}: run timeout reached — stopping repeat after ${i} iteration(s)`);
|
|
182
416
|
return { status: 'timeout' };
|
|
183
417
|
}
|
|
184
|
-
|
|
418
|
+
// settleMs=0 on purpose: this guard is absent on EVERY iteration by construction
|
|
419
|
+
// (that is what makes it a loop), so a settle window here would be paid `cap`
|
|
420
|
+
// times — 25 × 1.5s ≈ 37s of dead wall-clock per loop — to discover something
|
|
421
|
+
// we already expect. The interstitial case that needs patience is `if-present`.
|
|
422
|
+
if (await present(interpolate(node.selector), 0)) {
|
|
423
|
+
satisfied = true;
|
|
185
424
|
deps.log(`[ai] ${where}: repeat reached '${node.selector}' after ${i} iteration(s)`);
|
|
186
|
-
|
|
425
|
+
break;
|
|
187
426
|
}
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
427
|
+
// No-progress detection over a SINGLE un-settled dump is unreliable on a real
|
|
428
|
+
// app: mid-transition frames read as empty (measured — `vk ui` reported zero
|
|
429
|
+
// elements on a screen where `tap`'s auto-wait found its target 3.1s later).
|
|
430
|
+
// Two such frames hash identically and look exactly like a stuck loop. Since a
|
|
431
|
+
// stalled loop is now a FAILURE, a false positive here fails a passing test, so
|
|
432
|
+
// this needs two independent guards:
|
|
433
|
+
// 1. an empty dump is UNKNOWN, not "unchanged" — never a progress sample;
|
|
434
|
+
// 2. require consecutive identical NON-EMPTY snapshots before believing it.
|
|
435
|
+
const els = await safeElements();
|
|
436
|
+
if (els.length === 0) {
|
|
437
|
+
deps.log(`[ai] ${where}: screen read as empty (mid-transition?) — not counting it as progress either way`);
|
|
438
|
+
}
|
|
439
|
+
else {
|
|
440
|
+
const hash = structuralHash(els);
|
|
441
|
+
stalled = i > 0 && hash === prevHash ? stalled + 1 : 0;
|
|
442
|
+
prevHash = hash;
|
|
443
|
+
if (stalled >= NO_PROGRESS_STRIKES) {
|
|
444
|
+
deps.log(`[ai] ${where}: repeat made no progress across ${stalled + 1} checks — stopping after ${i} iteration(s)`);
|
|
445
|
+
break;
|
|
446
|
+
}
|
|
192
447
|
}
|
|
193
|
-
prevHash = hash;
|
|
194
448
|
deps.log(`[ai] ${where}: repeat iteration ${i + 1}/${node.cap}`);
|
|
195
|
-
const res = await walkBody(node.body, `${where}#${i + 1}`);
|
|
449
|
+
const res = await walkBody(node.body, `${where}#${i + 1}.body`);
|
|
450
|
+
if (res.status !== 'ok')
|
|
451
|
+
return res;
|
|
452
|
+
}
|
|
453
|
+
// A loop that stopped without ever seeing its target did NOT do its job. Both
|
|
454
|
+
// exits (cap exhausted, and the no-progress bail) used to return ok, which let a
|
|
455
|
+
// loop whose body did nothing report green — the reachable false green.
|
|
456
|
+
if (!satisfied && !(await present(interpolate(node.selector), guardSettleMs))) {
|
|
457
|
+
return {
|
|
458
|
+
status: 'fail',
|
|
459
|
+
where,
|
|
460
|
+
reason: `repeat stopped after ${i} iteration(s) without '${interpolate(node.selector)}' ever appearing`,
|
|
461
|
+
};
|
|
462
|
+
}
|
|
463
|
+
return { status: 'ok' };
|
|
464
|
+
}
|
|
465
|
+
case 'while-present': {
|
|
466
|
+
if (node.bind)
|
|
467
|
+
ctx.set(node.bind, '0');
|
|
468
|
+
let ran = 0;
|
|
469
|
+
for (let i = 0; i < node.cap; i++) {
|
|
470
|
+
if (overDeadline()) {
|
|
471
|
+
deps.log(`[ai] ${where}: run timeout reached — stopping while-present after ${i} iteration(s)`);
|
|
472
|
+
return { status: 'timeout' };
|
|
473
|
+
}
|
|
474
|
+
const selector = interpolate(node.selector);
|
|
475
|
+
// First check gets the settle window (the list may still be rendering); later
|
|
476
|
+
// checks are single-shot, since by then the list is up and we are just walking it.
|
|
477
|
+
if (!(await present(selector, i === 0 ? guardSettleMs : 0))) {
|
|
478
|
+
deps.log(`[ai] ${where}: while-present '${selector}' → absent, loop done after ${ran} iteration(s)`);
|
|
479
|
+
return { status: 'ok' };
|
|
480
|
+
}
|
|
481
|
+
deps.log(`[ai] ${where}: while-present iteration ${i + 1}/${node.cap} ('${selector}')`);
|
|
482
|
+
const res = await walkBody(node.body, `${where}#${i + 1}.body`, selector);
|
|
196
483
|
if (res.status !== 'ok')
|
|
197
484
|
return res;
|
|
485
|
+
ran++;
|
|
486
|
+
if (node.bind)
|
|
487
|
+
ctx.set(node.bind, String(Number(ctx.get(node.bind) ?? '0') + 1));
|
|
198
488
|
}
|
|
199
|
-
deps.log(`[ai] ${where}:
|
|
489
|
+
deps.log(`[ai] ${where}: while-present hit cap ${node.cap}`);
|
|
200
490
|
return { status: 'ok' };
|
|
201
491
|
}
|
|
202
492
|
}
|
package/dist/agent/grammar.js
CHANGED
|
@@ -34,13 +34,75 @@ Each step is one of three node types:
|
|
|
34
34
|
permission dialogs, "rate us" popups, cookie banners, A/B variants. This is how you
|
|
35
35
|
keep a flow from breaking when an extra screen sometimes appears.
|
|
36
36
|
|
|
37
|
-
3. REPEAT — { "type":"repeat", "selector":<sel>, "cap":<n>, "body":[<
|
|
38
|
-
Repeat body
|
|
39
|
-
is visible". Always set a sane cap (e.g.
|
|
40
|
-
stops changing.
|
|
37
|
+
3. REPEAT — { "type":"repeat", "selector":<sel>, "cap":<n>, "body":[<nodes>] }
|
|
38
|
+
Repeat body UNTIL the selector appears, up to cap iterations. Use for "scroll until X
|
|
39
|
+
is visible", or "keep answering until the results screen". Always set a sane cap (e.g.
|
|
40
|
+
10). The engine also stops early if the screen stops changing. A repeat that finishes
|
|
41
|
+
without its selector ever appearing FAILS the test — it did not do its job.
|
|
41
42
|
|
|
42
|
-
|
|
43
|
-
|
|
43
|
+
4. WHEN — { "type":"when", "branches":[{ "selector":<sel>, "body":[<nodes>] }, ...],
|
|
44
|
+
"else":[<nodes>] (optional) }
|
|
45
|
+
Ordered n-way dispatch: the FIRST branch whose selector is on screen runs, and only it.
|
|
46
|
+
Use when a screen is one of several KINDS that each need different handling —
|
|
47
|
+
"the question is multiple-choice, or match-the-pairs, or arrange-the-words".
|
|
48
|
+
If no branch matches and there is no "else", the test FAILS (the app showed something
|
|
49
|
+
this test does not handle — that is a real result, not something to skip past).
|
|
50
|
+
Use "else": [] to say explicitly "if none match, do nothing".
|
|
51
|
+
WHEN vs IF-PRESENT: if-present = "this may or may not be there, carry on either way".
|
|
52
|
+
when = "it is one of these; if it is none of them, that is a failure".
|
|
53
|
+
|
|
54
|
+
5. WHILE-PRESENT — { "type":"while-present", "selector":<sel>, "bind":<name>,
|
|
55
|
+
"cap":<n>, "body":[<nodes>] }
|
|
56
|
+
Repeat body WHILE the selector is present. With "bind", the named counter starts at 0
|
|
57
|
+
and increments after each iteration, and you reference it as {{ctx.<name>}} inside the
|
|
58
|
+
selector and the body. This is how you walk an index-addressed list whose LENGTH you
|
|
59
|
+
cannot know when compiling:
|
|
60
|
+
{ "type":"while-present", "selector":"id:word_bubble_container_id_{{ctx.i}}",
|
|
61
|
+
"bind":"i", "cap":20,
|
|
62
|
+
"body":[ { "type":"command","command":"tap",
|
|
63
|
+
"positionals":["id:word_bubble_container_id_{{ctx.i}}"],"flags":[] } ] }
|
|
64
|
+
|
|
65
|
+
6. READ — { "type":"read", "selector":<sel>, "field":"text"|"desc"|"id"|"idShort",
|
|
66
|
+
"into":<name> }
|
|
67
|
+
Capture a value off the live screen into {{ctx.<name>}} for a later step to use. Use it
|
|
68
|
+
when the test must ACT ON a value it cannot know in advance — e.g. read the correct
|
|
69
|
+
answer's text, then type that text into a field.
|
|
70
|
+
|
|
71
|
+
PLACEHOLDERS — any positional or flag value, and any control-node selector, may contain:
|
|
72
|
+
{{ctx.NAME}} a value stored by read, or a while-present counter
|
|
73
|
+
{{env.NAME}} an environment variable (use for credentials; never inline a secret)
|
|
74
|
+
{{uuid}} a fresh id, generated once per run (same value everywhere in the run)
|
|
75
|
+
{{timestamp}} epoch ms, once per run {{run_id}} this run's id
|
|
76
|
+
Use {{uuid}} when the test needs data that must be unique per run, e.g. a signup email
|
|
77
|
+
like "user-{{uuid}}@example.com" — never a hard-coded literal, which collides on rerun.
|
|
78
|
+
|
|
79
|
+
NESTING: control nodes may nest ONE level — a control node inside a control node, whose
|
|
80
|
+
body is command leaves. A repeat containing a when is the shape for "until the flow ends,
|
|
81
|
+
handle whichever screen is showing", and it is legal. Three levels is not.
|
|
82
|
+
EXCEPT: if-present and while-present may go one level deeper (their bodies are leaves), so
|
|
83
|
+
repeat { when { while-present { tap ... } } } — walk an index-addressed list
|
|
84
|
+
repeat { when { if-present { tap ... } } } — an optional step inside a branch
|
|
85
|
+
are both legal. Use them rather than approximating.
|
|
86
|
+
In particular: "if X appears, tap it" inside a branch is an if-present. Do NOT turn it
|
|
87
|
+
into a bare wait + tap — wait FAILS the test when X never appears, and "if" means it
|
|
88
|
+
might not. That mistake reads as a passing plan and fails on the first run where the app
|
|
89
|
+
skips that step.
|
|
90
|
+
|
|
91
|
+
Inside a "repeat until X" loop, GUARD a tap whose target is the thing that brings X about:
|
|
92
|
+
repeat until <form> { if-present <button> { tap <button> } ; screenshot }
|
|
93
|
+
Once the transition starts, <button> is gone — so on the final iteration an unguarded tap
|
|
94
|
+
misses and fails the whole run, even though the loop did its job. The guard makes that
|
|
95
|
+
last lap a no-op instead of an error. Apply this whenever the prose says "tap ... until"
|
|
96
|
+
or "repeatedly until".
|
|
97
|
+
|
|
98
|
+
Do NOT flatten a branch into an unconditional sequence: emitting the taps for ALL the
|
|
99
|
+
kinds of screen one after another is wrong — on any given iteration most of them are not
|
|
100
|
+
there, and the test will fail on the first one that is missing. Use when.
|
|
101
|
+
|
|
102
|
+
Do NOT hard-code a run of indices you were not told the length of. If the prose says
|
|
103
|
+
"tap each pair", "until every pair is matched", or "tap the bubbles in order", the COUNT
|
|
104
|
+
varies per run — emit a while-present over {{ctx.i}} rather than tap _0, _1, _2, _3.
|
|
105
|
+
A hard-coded list is right only when the prose states the exact count.
|
|
44
106
|
|
|
45
107
|
SELECTORS (the engine auto-heals case/whitespace/partial, so prefer stable identifiers):
|
|
46
108
|
@login resource-id 'login' (shorthand for id:login)
|
|
@@ -51,6 +113,11 @@ SELECTORS (the engine auto-heals case/whitespace/partial, so prefer stable ident
|
|
|
51
113
|
"Sign in" bare string == text:Sign in
|
|
52
114
|
|
|
53
115
|
RULES:
|
|
116
|
+
- --enabled on a tap makes it match only a control that is ACTIONABLE right now, and (with
|
|
117
|
+
auto-wait) wait until it becomes so. Use it for any button that the app disables until
|
|
118
|
+
something else is done — a Check/Submit/Continue that only lights up once an answer is
|
|
119
|
+
selected or a form is valid. Without it the step taps a dead control, does nothing, and
|
|
120
|
+
the failure surfaces later as a confusing timeout on the NEXT step.
|
|
54
121
|
- assert is for VERIFICATION only and is terminal — never use it as a step you expect to
|
|
55
122
|
fail. Put genuinely-optional UI behind if-present.
|
|
56
123
|
- Prefer resource-id / accessibility selectors over visible text where possible.
|