verikun 0.29.0 → 0.30.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/skills/verikun/SKILL.md +5 -0
- package/CHANGELOG.md +28 -0
- package/dist/agent/engine.js +15 -1
- package/dist/agent/grammar.js +12 -2
- package/dist/agent/lint.js +104 -3
- package/dist/agent/remote.js +57 -6
- package/dist/cli.js +103 -6
- package/dist/errors.js +39 -1
- package/dist/report.js +2 -1
- package/dist/rpc.js +13 -0
- package/dist/server-http.js +5 -1
- package/dist/server.js +43 -19
- package/dist/suite.js +269 -15
- package/dist/version.js +1 -1
- package/package.json +1 -1
|
@@ -437,6 +437,8 @@ vk ai onboarding.md --timeout 5m # tighten the run timeout (default 15m)
|
|
|
437
437
|
- **`3` also means the device toolchain is broken**, checked *before* anything is compiled
|
|
438
438
|
(missing `adb`/`idb`, no device, an ambiguous target) and again if it breaks mid-run. Treat
|
|
439
439
|
it as "fix the machine", never as a failing test — the message carries the install hint.
|
|
440
|
+
Exception: over `--server`, a `--json` result with `evicted: true` means the server's device
|
|
441
|
+
left the pool mid-run — nothing is broken; run the test again.
|
|
440
442
|
|
|
441
443
|
## Run a suite of tests (vk suite)
|
|
442
444
|
|
|
@@ -488,6 +490,9 @@ vk suite tests/ --app com.example.app --servers http://a:8391,http://b:8391
|
|
|
488
490
|
- `--concurrency N` caps how many run at once — more devices on one host can thrash it.
|
|
489
491
|
`--max-suite-cost-usd N` stops the suite once total model spend crosses it (exit `1`).
|
|
490
492
|
- A device that breaks retires; its tests move to the others. Exit `3` only when all are gone.
|
|
493
|
+
- Over `--server`, a lane with no free device waits instead of failing tests, and a test whose
|
|
494
|
+
device left the pool re-runs without spending a retry. With no device for any lane the suite
|
|
495
|
+
stops after `VERIKUN_SUITE_DEVICE_WAIT_MIN` (default 10) with exit `3` and a `notRun` list.
|
|
491
496
|
- `--ensure-device` is refused with `--devices` — start the pool with `vk devices start`.
|
|
492
497
|
|
|
493
498
|
## Drive a remote device (--server)
|
package/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,34 @@ All notable changes to this project are documented here. The format is based on
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.30.0] - 2026-09-26
|
|
10
|
+
|
|
11
|
+
### Fixed
|
|
12
|
+
- **`vk suite --server`** waits for a free device instead of failing tests that never got one, and
|
|
13
|
+
a waiting lane resumes when the pool recovers. ([#147])
|
|
14
|
+
|
|
15
|
+
### Added
|
|
16
|
+
- **`vk suite --server`** re-runs a test whose device left the pool mid-run without spending a
|
|
17
|
+
retry, up to twice; needs the matching `vk server`. ([#147])
|
|
18
|
+
- **`VERIKUN_SUITE_DEVICE_WAIT_MIN`** (default `10`) caps how long a pooled suite waits when no lane
|
|
19
|
+
gets a device, busy pools included, then exits `3`; `0` fails fast. ([#147])
|
|
20
|
+
|
|
21
|
+
### Changed
|
|
22
|
+
- **`--json` errors** name a refused `vk server` lease `NoFreeDeviceError` and an evicted run
|
|
23
|
+
`RunEvictedError` in `errorKind`; exit codes are unchanged. ([#147])
|
|
24
|
+
- **`vk ai --json`** reports a run whose device left the pool as an environment abort with
|
|
25
|
+
`evicted: true` and its report, instead of a bare error. ([#147])
|
|
26
|
+
|
|
27
|
+
[#147]: https://github.com/ddikman/verikun/issues/147
|
|
28
|
+
|
|
29
|
+
## [0.29.1] - 2026-09-26
|
|
30
|
+
|
|
31
|
+
### Fixed
|
|
32
|
+
- **`vk ai`** no longer guesses `id:` selectors; an element the test names only by label compiles
|
|
33
|
+
to `text:` — write the id into the test to pin it. ([#148])
|
|
34
|
+
|
|
35
|
+
[#148]: https://github.com/ddikman/verikun/issues/148
|
|
36
|
+
|
|
9
37
|
## [0.29.0] - 2026-09-16
|
|
10
38
|
|
|
11
39
|
### Changed
|
package/dist/agent/engine.js
CHANGED
|
@@ -31,6 +31,17 @@ class GuardBlindError extends Error {
|
|
|
31
31
|
this.name = 'GuardBlindError';
|
|
32
32
|
}
|
|
33
33
|
}
|
|
34
|
+
/**
|
|
35
|
+
* A run the server EVICTED cannot continue: its phone left the pool, and every later call is
|
|
36
|
+
* refused the same way. So the catches below that absorb a failed READ — a guard's retry, the
|
|
37
|
+
* empty-tree fallback, a repair that could not be made — must let it through. Absorbed, a lost
|
|
38
|
+
* phone read as an absent guard, a "read found no element" FAIL or a failed repair, and a
|
|
39
|
+
* parallel suite lost the class it needs to re-run the test as a fresh run (#147).
|
|
40
|
+
*/
|
|
41
|
+
const rethrowIfEvicted = (e) => {
|
|
42
|
+
if (e instanceof errors_1.RunEvictedError)
|
|
43
|
+
throw e;
|
|
44
|
+
};
|
|
34
45
|
const describe = (leaf) => [leaf.command, ...leaf.positionals, ...leaf.flags.map((f) => (f.value === 'true' ? `--${f.name}` : `--${f.name} ${f.value}`))]
|
|
35
46
|
.join(' ')
|
|
36
47
|
.trim();
|
|
@@ -190,7 +201,8 @@ async function runPlan(plan, deps) {
|
|
|
190
201
|
try {
|
|
191
202
|
return await deps.getElements();
|
|
192
203
|
}
|
|
193
|
-
catch {
|
|
204
|
+
catch (e) {
|
|
205
|
+
rethrowIfEvicted(e);
|
|
194
206
|
return [];
|
|
195
207
|
}
|
|
196
208
|
};
|
|
@@ -256,6 +268,7 @@ async function runPlan(plan, deps) {
|
|
|
256
268
|
everRead = true;
|
|
257
269
|
}
|
|
258
270
|
catch (e) {
|
|
271
|
+
rethrowIfEvicted(e);
|
|
259
272
|
els = undefined; // transient dump failure — retry once before concluding "absent"
|
|
260
273
|
lastErr = e;
|
|
261
274
|
}
|
|
@@ -366,6 +379,7 @@ async function runPlan(plan, deps) {
|
|
|
366
379
|
repaired = node;
|
|
367
380
|
}
|
|
368
381
|
catch (e) {
|
|
382
|
+
rethrowIfEvicted(e);
|
|
369
383
|
const msg = e instanceof Error ? e.message : String(e);
|
|
370
384
|
return { status: 'fail', where, reason: `repair failed: ${msg}` };
|
|
371
385
|
}
|
package/dist/agent/grammar.js
CHANGED
|
@@ -118,7 +118,7 @@ Do NOT hard-code a run of indices you were not told the length of. If the prose
|
|
|
118
118
|
varies per run — emit a while-present over {{ctx.i}} rather than tap _0, _1, _2, _3.
|
|
119
119
|
A hard-coded list is right only when the prose states the exact count.
|
|
120
120
|
|
|
121
|
-
SELECTORS (the engine auto-heals case/whitespace/partial, so prefer
|
|
121
|
+
SELECTORS (the engine auto-heals case/whitespace/partial, so prefer an id the test gives you):
|
|
122
122
|
@login resource-id 'login' (shorthand for id:login)
|
|
123
123
|
id:login resource-id (full, suffix, or short)
|
|
124
124
|
text:Sign in visible text (case-insensitive)
|
|
@@ -156,7 +156,17 @@ RULES:
|
|
|
156
156
|
below the fold — that is now redundant. Emit an explicit swipe only when the SCROLLING
|
|
157
157
|
ITSELF is what the test asks for ("scroll the feed three times"), or to reveal content
|
|
158
158
|
that is not in the hierarchy until it is built (an infinite/lazy list).
|
|
159
|
-
-
|
|
159
|
+
- IDS COME FROM THE TEST. Use @x / id:x only with an id the test writes (or one a PRIOR PLAN
|
|
160
|
+
shown to you already uses). You cannot see the app, so never compose an id from a label, a
|
|
161
|
+
description, or the pattern of the app's other ids: a made-up id matches nothing, and a
|
|
162
|
+
wait/assert on it FAILS the test with no repair. An element the test names only by what it
|
|
163
|
+
says ("tap Sign in", "the Home tab") is text:<those words> — forgiving, and it also matches an
|
|
164
|
+
accessibility label.
|
|
165
|
+
- An outcome the test states ("the tab bar (Home, Search, Profile) appears") is checked on what
|
|
166
|
+
it NAMES — the id it gives, or the words it quotes or lists — never on an id you made up. A
|
|
167
|
+
description that names nothing you could select without inventing it ("it shows the
|
|
168
|
+
signed-in user's name") is context, not a step; so is rationale (why a step exists, what a
|
|
169
|
+
field already holds).
|
|
160
170
|
- Translate the test literally and minimally: do not invent ACTION steps (tap/text/swipe/key/assert)
|
|
161
171
|
the prose does not imply. The ONE exception is screenshot — insert screenshot steps liberally as
|
|
162
172
|
post-run review evidence: after each screen transition (launch, a navigation tap, a submit, a
|
package/dist/agent/lint.js
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
// Compile-fidelity lint: does the plan the model produced still say what the prose said?
|
|
3
3
|
//
|
|
4
4
|
// The model is a compiler, and this is the compiler's own sanity check. It exists because
|
|
5
|
-
// compilation is NONDETERMINISTIC in a way that is invisible until a run fails, and
|
|
5
|
+
// compilation is NONDETERMINISTIC in a way that is invisible until a run fails, and four
|
|
6
6
|
// failure modes showed up repeatedly against a real suite:
|
|
7
7
|
//
|
|
8
8
|
// 1. An explicit directive silently vanishes. The same prose ("Launch the app WITH ITS
|
|
@@ -18,8 +18,12 @@
|
|
|
18
18
|
// against later builds. A test exercising none of its subject reporting success is the
|
|
19
19
|
// worst failure mode a testing tool has, which is why the two rules that detect it are
|
|
20
20
|
// the only FATAL findings here.
|
|
21
|
+
// 4. A selector names an id the test never gave (issue #148). The compiler cannot see the
|
|
22
|
+
// app, so it composed one from the app's naming pattern, and 3 of 4 such guesses named
|
|
23
|
+
// nothing in the app. A wait/assert is never repaired, so the test went red — and each
|
|
24
|
+
// recompile guessed differently, so it flaked.
|
|
21
25
|
//
|
|
22
|
-
// 1 and
|
|
26
|
+
// 1, 2 and 4 are cheap to detect and cheap to fix: hand the finding back to the model and let
|
|
23
27
|
// it compile once more. That is far better than the alternative, which is a plan that is
|
|
24
28
|
// quietly wrong and burns a device run to say so. 3 gets the same guided recompile, but a
|
|
25
29
|
// plan that STILL does not cover its test must never run — see `fatal` below.
|
|
@@ -34,9 +38,11 @@ exports.instructionLines = instructionLines;
|
|
|
34
38
|
exports.instructionUnits = instructionUnits;
|
|
35
39
|
exports.actionNodes = actionNodes;
|
|
36
40
|
exports.tailAnchors = tailAnchors;
|
|
41
|
+
exports.ungroundedIds = ungroundedIds;
|
|
37
42
|
exports.lintPlan = lintPlan;
|
|
38
43
|
exports.looksTruncated = looksTruncated;
|
|
39
44
|
const ir_1 = require("./ir");
|
|
45
|
+
const selector_1 = require("../ui/selector");
|
|
40
46
|
/** Walk every node in the plan, including control-node bodies. */
|
|
41
47
|
function* walk(nodes) {
|
|
42
48
|
for (const n of nodes) {
|
|
@@ -257,15 +263,95 @@ function planMentions(tokens, anchor) {
|
|
|
257
263
|
return b.length >= 3 && (b.includes(a) || a.includes(b));
|
|
258
264
|
});
|
|
259
265
|
}
|
|
266
|
+
// --- ids: does every id the plan selects by come from the prose? (issue #148) ------------
|
|
267
|
+
/** Commands whose FIRST positional is a selector. The rest of `text`'s positionals, and every
|
|
268
|
+
* positional of `type`/`launch`/…, is data — `type @handle` types a handle, it selects nothing. */
|
|
269
|
+
const SELECTOR_FIRST = new Set(['tap', 'click', 'text', 'wait', 'assert']);
|
|
270
|
+
/** Every string the engine resolves as a selector — unlike `planTokens`, which also returns data. */
|
|
271
|
+
function selectorsOf(plan) {
|
|
272
|
+
const out = [];
|
|
273
|
+
for (const n of walk(plan.steps)) {
|
|
274
|
+
if (n.type === 'command') {
|
|
275
|
+
if (SELECTOR_FIRST.has(n.command) && n.positionals[0])
|
|
276
|
+
out.push(n.positionals[0]);
|
|
277
|
+
if (n.command === 'swipe' || n.command === 'scroll') {
|
|
278
|
+
for (const f of n.flags)
|
|
279
|
+
if (f.name === 'on')
|
|
280
|
+
out.push(f.value);
|
|
281
|
+
}
|
|
282
|
+
}
|
|
283
|
+
else if (n.type === 'when') {
|
|
284
|
+
out.push(...n.branches.map((b) => b.selector));
|
|
285
|
+
}
|
|
286
|
+
else {
|
|
287
|
+
out.push(n.selector);
|
|
288
|
+
}
|
|
289
|
+
}
|
|
290
|
+
return out;
|
|
291
|
+
}
|
|
292
|
+
/** The id values (`@x` / `id:x`) the plan selects by, parsed by the engine's own parser so the
|
|
293
|
+
* two can never disagree about what an id selector is — trailing state modifiers included. */
|
|
294
|
+
function planIds(plan) {
|
|
295
|
+
const out = [];
|
|
296
|
+
for (const raw of selectorsOf(plan)) {
|
|
297
|
+
let sel;
|
|
298
|
+
try {
|
|
299
|
+
sel = (0, selector_1.parseSelector)(raw);
|
|
300
|
+
}
|
|
301
|
+
catch {
|
|
302
|
+
continue; // an empty value or contradictory modifiers: a runtime error, not a guess
|
|
303
|
+
}
|
|
304
|
+
if (sel.kind === 'id' && !out.includes(sel.value))
|
|
305
|
+
out.push(sel.value);
|
|
306
|
+
}
|
|
307
|
+
return out;
|
|
308
|
+
}
|
|
309
|
+
/** `com.app:id/login` and `login` name the same element. */
|
|
310
|
+
const RESOURCE_PKG_RE = /^[\w.]+:id\//;
|
|
311
|
+
/** `{{ctx.i}}` / `{{env.X}}` are filled at run time, so they are never part of a guess. */
|
|
312
|
+
const PLACEHOLDER_RE = /\{\{[^}]*\}\}/;
|
|
313
|
+
/** The literal parts of an id worth judging. Under three characters a substring matches
|
|
314
|
+
* anything, so a part that short has no opinion. */
|
|
315
|
+
const idParts = (id) => id
|
|
316
|
+
.replace(RESOURCE_PKG_RE, '')
|
|
317
|
+
.split(PLACEHOLDER_RE)
|
|
318
|
+
.map(normToken)
|
|
319
|
+
.filter((p) => p.length >= 3);
|
|
320
|
+
/** Every word of the prose, and every quoted span whole — an iOS identifier may hold spaces. */
|
|
321
|
+
function proseTokens(nl) {
|
|
322
|
+
const out = [];
|
|
323
|
+
for (const m of nl.matchAll(ANCHOR_RE))
|
|
324
|
+
out.push(m[1] ?? m[2] ?? m[3] ?? '');
|
|
325
|
+
out.push(...nl.split(/[\s`"“”'‘’()[\]{}<>,;!?]+/));
|
|
326
|
+
return out.map(normToken).filter((t) => t.length > 0);
|
|
327
|
+
}
|
|
328
|
+
/**
|
|
329
|
+
* The ids the plan selects by that neither the prose nor `seed` gives — each one a guess.
|
|
330
|
+
*
|
|
331
|
+
* LENIENT on purpose, since a false positive costs a recompile: an id is grounded when every
|
|
332
|
+
* literal part of it (package qualifier and placeholders dropped, case and punctuation ignored)
|
|
333
|
+
* sits inside ONE token of the prose, so `@spinner` is grounded by `vk_spinner` and
|
|
334
|
+
* `id:option_{{ctx.i}}` by `option_0`. One token, never two: `settings_tab` built from the words
|
|
335
|
+
* "Settings tab" is precisely how a guess is made.
|
|
336
|
+
*
|
|
337
|
+
* `seed` is a prior plan the compiler was told to reuse. A green run re-persists the HEALED plan
|
|
338
|
+
* and a repair takes its id off a live screen, so an id a seed already uses is grounded too —
|
|
339
|
+
* pass one only where it can hold a repair (see `compileFromSegments` in ../cli.ts).
|
|
340
|
+
*/
|
|
341
|
+
function ungroundedIds(nl, plan, seed) {
|
|
342
|
+
const ground = [...proseTokens(nl), ...(seed ? planIds(seed).flatMap(idParts) : [])];
|
|
343
|
+
return planIds(plan).filter((id) => idParts(id).some((part) => !ground.some((t) => t.includes(part))));
|
|
344
|
+
}
|
|
260
345
|
/**
|
|
261
346
|
* Check a compiled plan against the prose it came from.
|
|
262
347
|
*
|
|
263
348
|
* @param nl the natural-language test, verbatim
|
|
264
349
|
* @param plan the plan the model just produced
|
|
350
|
+
* @param seed the prior plan the model was handed to adapt, if any — its ids are not guesses
|
|
265
351
|
* @returns findings; empty means the plan is consistent with the prose. A finding with
|
|
266
352
|
* `fatal` set means the plan does not cover the test and must not be run.
|
|
267
353
|
*/
|
|
268
|
-
function lintPlan(nl, plan) {
|
|
354
|
+
function lintPlan(nl, plan, seed) {
|
|
269
355
|
const findings = [];
|
|
270
356
|
if (FRESH_START_RE.test(nl) && !hasLeafWithFlag(plan, 'launch', 'clear')) {
|
|
271
357
|
findings.push({
|
|
@@ -281,6 +367,21 @@ function lintPlan(nl, plan) {
|
|
|
281
367
|
'(skip when absent), or behind when (when the screen is one of several known kinds).',
|
|
282
368
|
});
|
|
283
369
|
}
|
|
370
|
+
// Not fatal: a guess that happens to be right still runs, and a wrong one fails its step
|
|
371
|
+
// loudly, so the guided recompile is the remedy. The exception is an ABSENCE check (`--gone`,
|
|
372
|
+
// a guard that skips): it passes on any selector that never matches, a wrong label as much as
|
|
373
|
+
// a guessed id, so making this rule fatal would not close it.
|
|
374
|
+
const guessed = ungroundedIds(nl, plan, seed);
|
|
375
|
+
if (guessed.length > 0) {
|
|
376
|
+
const named = guessed.slice(0, 5).map((id) => JSON.stringify(id)).join(', ');
|
|
377
|
+
findings.push({
|
|
378
|
+
message: `The plan selects by id ${named}${guessed.length > 5 ? ` and ${guessed.length - 5} more` : ''}, but the test ` +
|
|
379
|
+
`never gives ${guessed.length === 1 ? 'that id' : 'those ids'}. You cannot see the app, so an id the test does not state is a guess — it matches ` +
|
|
380
|
+
'nothing, and a wait/assert on it fails the test with no repair. Use an id only when the test writes it; ' +
|
|
381
|
+
'select an element the test names by its label with text:<label>, and do not check something the test ' +
|
|
382
|
+
'names nothing selectable for.',
|
|
383
|
+
});
|
|
384
|
+
}
|
|
284
385
|
if (!coverageChecksEnabled())
|
|
285
386
|
return findings;
|
|
286
387
|
const units = instructionUnits(nl);
|
package/dist/agent/remote.js
CHANGED
|
@@ -12,6 +12,7 @@ Object.defineProperty(exports, "__esModule", { value: true });
|
|
|
12
12
|
exports.describeStatus = describeStatus;
|
|
13
13
|
exports.transportReason = transportReason;
|
|
14
14
|
exports.pingServer = pingServer;
|
|
15
|
+
exports.poolNote = poolNote;
|
|
15
16
|
exports.remoteDeviceOp = remoteDeviceOp;
|
|
16
17
|
exports.remoteDeviceList = remoteDeviceList;
|
|
17
18
|
exports.createRemoteBackend = createRemoteBackend;
|
|
@@ -105,9 +106,18 @@ function requestWithNodeHttp(url, method, headers, body, timeoutMs) {
|
|
|
105
106
|
/**
|
|
106
107
|
* Turn a non-2xx into the error the caller sees.
|
|
107
108
|
*
|
|
108
|
-
* The 401/409/503 arms come FIRST and
|
|
109
|
+
* The 401/409/503 arms come FIRST and never rebuild a driver error: those describe the
|
|
109
110
|
* TRANSPORT (wrong key, device leased, nothing attached), not something a driver threw, so
|
|
110
111
|
* there is no device-error identity to restore and their wording is what a user acts on.
|
|
112
|
+
* Two of them do carry a class of their own, because a parallel suite has to act on the
|
|
113
|
+
* difference (#147) — both still exit 3:
|
|
114
|
+
*
|
|
115
|
+
* - on the LEASE route (`opts.lease`), 409 and 503 are `NoFreeDeviceError`: the run never
|
|
116
|
+
* started. The route decides, not the body — every run mints a fresh token, so a lease can
|
|
117
|
+
* be refused but never evicted.
|
|
118
|
+
* - elsewhere, a 409 the server tagged `RunEvictedError` is one: the run lost its device
|
|
119
|
+
* part-way. That is the ONLY kind a 409 honours. Anything else a body carries is ignored,
|
|
120
|
+
* and an untagged 409 (an older server) stays the plain error it always was.
|
|
111
121
|
*
|
|
112
122
|
* Everything else prefers the server's `errorKind`. That field is what stops a `--server` run
|
|
113
123
|
* reading a mid-launch `NoWindowError` as a fatal environment error: the class survives the
|
|
@@ -115,16 +125,22 @@ function requestWithNodeHttp(url, method, headers, body, timeoutMs) {
|
|
|
115
125
|
* `CliError` (issue #80). No field — an older server, or a failure with no class worth
|
|
116
126
|
* naming — falls through to exactly the previous behaviour.
|
|
117
127
|
*/
|
|
118
|
-
function describeStatus(status, body, url) {
|
|
128
|
+
function describeStatus(status, body, url, opts = {}) {
|
|
119
129
|
const detail = body?.error ? `: ${body.error}` : '';
|
|
120
130
|
if (status === 401) {
|
|
121
131
|
return new errors_1.CliError(`verikun server rejected the auth key (401)${detail}. Check --auth-key / VERIKUN_SERVER_AUTH_KEY.`, 3);
|
|
122
132
|
}
|
|
123
133
|
if (status === 409) {
|
|
124
|
-
|
|
134
|
+
const busy = `verikun server device is busy (409)${detail || ' — another run holds the device; retry when it finishes'}.`;
|
|
135
|
+
if (opts.lease)
|
|
136
|
+
return new errors_1.NoFreeDeviceError(busy);
|
|
137
|
+
if (body?.errorKind === 'RunEvictedError')
|
|
138
|
+
return new errors_1.RunEvictedError(`verikun server ended this run (409)${detail}.`);
|
|
139
|
+
return new errors_1.CliError(busy, 3);
|
|
125
140
|
}
|
|
126
141
|
if (status === 503) {
|
|
127
|
-
|
|
142
|
+
const none = `verikun server has no device attached (503)${detail}.`;
|
|
143
|
+
return opts.lease ? new errors_1.NoFreeDeviceError(none) : new errors_1.CliError(none, 3);
|
|
128
144
|
}
|
|
129
145
|
// The server sends the intended exit code (usage 2 / env 3) in the body; fall
|
|
130
146
|
// back on the HTTP class when it didn't.
|
|
@@ -172,6 +188,9 @@ class RemoteTransport {
|
|
|
172
188
|
base;
|
|
173
189
|
/** One token per backend = one logical run holding the server's device lock. */
|
|
174
190
|
runToken = (0, node_crypto_1.randomUUID)();
|
|
191
|
+
/** The server has said this run was evicted. Latched, because the request that hears it is
|
|
192
|
+
* usually not the step that failed — see ExecBackend.wasEvicted. */
|
|
193
|
+
evicted = false;
|
|
175
194
|
constructor(opts) {
|
|
176
195
|
this.opts = opts;
|
|
177
196
|
this.base = trimUrl(opts.url);
|
|
@@ -214,7 +233,10 @@ class RemoteTransport {
|
|
|
214
233
|
// exactly the case a caller must not miss (an exhausted install, a dead-device read).
|
|
215
234
|
if (body?.deviceChanged)
|
|
216
235
|
this.opts.onDeviceChange?.(body.deviceChanged);
|
|
217
|
-
|
|
236
|
+
const error = describeStatus(res.status, body, url, { lease: path === '/v1/lease' });
|
|
237
|
+
if (error instanceof errors_1.RunEvictedError || body?.evicted)
|
|
238
|
+
this.evicted = true;
|
|
239
|
+
throw error;
|
|
218
240
|
}
|
|
219
241
|
const parsed = await readBody(res);
|
|
220
242
|
if (parsed === null)
|
|
@@ -245,6 +267,21 @@ async function pingServer(opts) {
|
|
|
245
267
|
}
|
|
246
268
|
return health;
|
|
247
269
|
}
|
|
270
|
+
/**
|
|
271
|
+
* What a `vk server` is serving, and what it has ruled out, in one line from one
|
|
272
|
+
* `/v1/health`. Carried on every "no free device" so the reader learns WHICH phone left and
|
|
273
|
+
* the server's own reason, not the lane slot (`host:port#1`) that happened to notice (#147).
|
|
274
|
+
* A device the server shed keeps its quarantine, which is what makes it nameable here.
|
|
275
|
+
*/
|
|
276
|
+
function poolNote(health) {
|
|
277
|
+
const serving = health.devices ?? (health.serial ? [health.serial] : []);
|
|
278
|
+
const count = health.capacity ?? serving.length;
|
|
279
|
+
const out = serving.length ? ` (${serving.join(', ')})` : '';
|
|
280
|
+
const ruledOut = health.quarantined?.length
|
|
281
|
+
? `; ruled out: ${health.quarantined.map((q) => `${q.serial} (${q.reason})`).join(', ')}`
|
|
282
|
+
: '';
|
|
283
|
+
return `pool: ${count} serving${out}${ruledOut}`;
|
|
284
|
+
}
|
|
248
285
|
/**
|
|
249
286
|
* POST /v1/devices/{start,restart,stop}. Standalone beside pingServer rather than on
|
|
250
287
|
* the ExecBackend seam: that seam is the engine's DEVICE WORK (exec/getElements/
|
|
@@ -285,6 +322,10 @@ function createRemoteBackend(opts, health) {
|
|
|
285
322
|
// A failing step is a 200, so this is the ordinary path for a mid-run device death.
|
|
286
323
|
if (res.deviceChanged)
|
|
287
324
|
opts.onDeviceChange?.(res.deviceChanged);
|
|
325
|
+
// …and for the eviction that death caused: the step keeps the phone's own error, and this
|
|
326
|
+
// response is the only one that says the run is over (see ExecResponse.evicted).
|
|
327
|
+
if (res.evicted)
|
|
328
|
+
t.evicted = true;
|
|
288
329
|
return { code: res.code, error: res.error ? (0, rpc_1.rebuildError)(res.error) : undefined };
|
|
289
330
|
};
|
|
290
331
|
return {
|
|
@@ -294,7 +335,16 @@ function createRemoteBackend(opts, health) {
|
|
|
294
335
|
// together, and a client cannot otherwise tell "old server" from "new server".
|
|
295
336
|
if (health.capacity === undefined)
|
|
296
337
|
return null;
|
|
297
|
-
|
|
338
|
+
try {
|
|
339
|
+
return await t.postJson('/v1/lease', {}, HEALTH_TIMEOUT_MS);
|
|
340
|
+
}
|
|
341
|
+
catch (e) {
|
|
342
|
+
// Say what the pool looks like. `health` was read moments ago, on the way here, and
|
|
343
|
+
// it is what names a phone that left — the refusal itself only counts devices.
|
|
344
|
+
if (e instanceof errors_1.NoFreeDeviceError)
|
|
345
|
+
throw new errors_1.NoFreeDeviceError(`${e.message} [${poolNote(health)}]`);
|
|
346
|
+
throw e;
|
|
347
|
+
}
|
|
298
348
|
},
|
|
299
349
|
async getElements() {
|
|
300
350
|
const res = await t.postJson('/v1/elements', {}, ELEMENTS_TIMEOUT_MS);
|
|
@@ -346,6 +396,7 @@ function createRemoteBackend(opts, health) {
|
|
|
346
396
|
if (code !== 0)
|
|
347
397
|
throw error ?? new errors_1.CliError(`reset (${command} ${appId}) failed on the server (exit ${code})`, 3);
|
|
348
398
|
},
|
|
399
|
+
wasEvicted: () => t.evicted,
|
|
349
400
|
async close() {
|
|
350
401
|
// Free the server's device lock so the next command (a fresh run token, e.g.
|
|
351
402
|
// `vk install` then `vk suite` in one CI job) isn't 409'd until the idle
|
package/dist/cli.js
CHANGED
|
@@ -48,6 +48,7 @@ exports.compileFromSegments = compileFromSegments;
|
|
|
48
48
|
exports.obtainPlan = obtainPlan;
|
|
49
49
|
exports.retryAfterDeviceMove = retryAfterDeviceMove;
|
|
50
50
|
exports.terminalFailure = terminalFailure;
|
|
51
|
+
exports.runAiTest = runAiTest;
|
|
51
52
|
exports.laneArgv = laneArgv;
|
|
52
53
|
exports.laneEnv = laneEnv;
|
|
53
54
|
exports.lanesFromFlags = lanesFromFlags;
|
|
@@ -1905,7 +1906,14 @@ async function compileFromSegments(segments, key, opts, cost, provider) {
|
|
|
1905
1906
|
// FREE, and refusing it over a ceiling we were never about to spend against would fail
|
|
1906
1907
|
// a test for somebody else's tokens.
|
|
1907
1908
|
assertBudgetForCompile(cost, opts.maxCostUsd, `compiling ${where} — the test is only partly compiled`);
|
|
1908
|
-
|
|
1909
|
+
// A section's entry is only ever raw compile output — a green run re-persists the WHOLE
|
|
1910
|
+
// test's key, never a section's — so an id in it that its own prose never gives can only be
|
|
1911
|
+
// a guess (issue #148), and handing it back as "reuse this" would keep it alive.
|
|
1912
|
+
let seed = seedPlan(segKey, where);
|
|
1913
|
+
if (seed && (0, lint_1.ungroundedIds)(seg.text, seed.plan).length > 0) {
|
|
1914
|
+
(0, output_1.err)(`[ai] ${where}: ignoring a prior plan that guesses ids its prose never gives — compiling fresh`);
|
|
1915
|
+
seed = null;
|
|
1916
|
+
}
|
|
1909
1917
|
(0, output_1.err)(`[ai] ${where}: compiling with ${opts.model}…${lock.degraded ? ` (${lock.degraded})` : ''}`);
|
|
1910
1918
|
let compiled;
|
|
1911
1919
|
try {
|
|
@@ -1939,6 +1947,14 @@ async function compileFromSegments(segments, key, opts, cost, provider) {
|
|
|
1939
1947
|
(0, output_1.err)(`[ai] ${where}: the compiled section does not cover its prose (${compiled.plan.steps.length} step(s)) — not caching it; compiling the test as one instead`);
|
|
1940
1948
|
return null; // the finally below releases the lock
|
|
1941
1949
|
}
|
|
1950
|
+
// A guessed id in a FRAGMENT is issue #148 spliced into every test that includes it. Same
|
|
1951
|
+
// answer as a short section: keep it out of the cache and compile the test whole, where the
|
|
1952
|
+
// lint hands the guess back for one guided recompile.
|
|
1953
|
+
const guessed = (0, lint_1.ungroundedIds)(seg.text, compiled.plan);
|
|
1954
|
+
if (guessed.length > 0) {
|
|
1955
|
+
(0, output_1.err)(`[ai] ${where}: the compiled section selects by ids its prose never gives (${guessed.join(', ')}) — not caching it; compiling the test as one instead`);
|
|
1956
|
+
return null; // the finally below releases the lock
|
|
1957
|
+
}
|
|
1942
1958
|
// INSIDE the lock and before the release: a waiter re-reads the cache the instant the
|
|
1943
1959
|
// lock disappears, so releasing first would hand it a miss and buy the second compile
|
|
1944
1960
|
// this whole mechanism exists to prevent.
|
|
@@ -2013,7 +2029,10 @@ async function obtainPlan(key, file, opts, cost, provider, segments = []) {
|
|
|
2013
2029
|
// sometimes only the test's opening compiles at all. One guided retry is much cheaper
|
|
2014
2030
|
// than discovering it as a device-run failure several steps later, or (worse, for a
|
|
2015
2031
|
// truncation) as a pass. Budget is re-checked HERE: the first attempt is already billed.
|
|
2016
|
-
|
|
2032
|
+
//
|
|
2033
|
+
// The seed goes in too: a green run re-persists the HEALED plan, whose repaired ids came off a
|
|
2034
|
+
// live screen, so flagging them would push every new build off a selector the device confirmed.
|
|
2035
|
+
let remaining = (0, lint_1.lintPlan)(key.nl, compiled.plan, seed?.plan);
|
|
2017
2036
|
if (remaining.length > 0) {
|
|
2018
2037
|
const feedback = remaining.map((f) => `- ${f.message}`).join('\n');
|
|
2019
2038
|
(0, output_1.err)(`[ai] compiled plan does not match the test — recompiling once:\n${feedback}`);
|
|
@@ -2029,7 +2048,7 @@ async function obtainPlan(key, file, opts, cost, provider, segments = []) {
|
|
|
2029
2048
|
retryFeedback: feedback,
|
|
2030
2049
|
});
|
|
2031
2050
|
cost.add(retry.usage, 'compile');
|
|
2032
|
-
const still = (0, lint_1.lintPlan)(key.nl, retry.plan);
|
|
2051
|
+
const still = (0, lint_1.lintPlan)(key.nl, retry.plan, seed?.plan);
|
|
2033
2052
|
// Keep the retry either way: it was compiled with strictly more information. If it
|
|
2034
2053
|
// still trips the lint, say so rather than pretending the plan is clean — and for a
|
|
2035
2054
|
// coverage finding, do not claim it will run, because the gate below rejects it.
|
|
@@ -2439,6 +2458,7 @@ async function prefetchArchiveLogs(backend, noLogs = false) {
|
|
|
2439
2458
|
* Run one natural-language test through a backend and return DATA — no stdout
|
|
2440
2459
|
* writes (stdout stays the caller's one result; progress streams to stderr).
|
|
2441
2460
|
* `vk ai` wraps it with its --json/report output; `vk suite` calls it per test.
|
|
2461
|
+
* Exported for the unit suite.
|
|
2442
2462
|
*/
|
|
2443
2463
|
async function runAiTest(file, opts, backend, platform, device) {
|
|
2444
2464
|
const { nl, segments } = readAiTest(file);
|
|
@@ -2477,7 +2497,19 @@ async function runAiTest(file, opts, backend, platform, device) {
|
|
|
2477
2497
|
// (exit 3 for a broken device), which is what a caller reads as an environment
|
|
2478
2498
|
// abort — the same verdict `vk suite`'s own reset failure produces.
|
|
2479
2499
|
if (opts.resetApp) {
|
|
2480
|
-
|
|
2500
|
+
try {
|
|
2501
|
+
await backend.reset(opts.resetApp);
|
|
2502
|
+
}
|
|
2503
|
+
catch (e) {
|
|
2504
|
+
// The phone this run was dealt had left the pool while idle, and the reset is the first
|
|
2505
|
+
// request to find out — the one the server marks. That is the server's eviction, not a
|
|
2506
|
+
// broken box: a suite re-runs it as a fresh run (#147). Thrown, because no run exists yet
|
|
2507
|
+
// to carry a result.
|
|
2508
|
+
if ((0, errors_1.isEnvError)(e) && !(e instanceof errors_1.RunEvictedError) && backend.wasEvicted?.()) {
|
|
2509
|
+
throw new errors_1.RunEvictedError(`the device this run was dealt had left the pool: ${e.message.split('\n')[0]}`);
|
|
2510
|
+
}
|
|
2511
|
+
throw e;
|
|
2512
|
+
}
|
|
2481
2513
|
(0, output_1.err)(`[ai] app state reset (${opts.resetApp})`);
|
|
2482
2514
|
}
|
|
2483
2515
|
// One explicit run for the whole flow (so rollover can't split the test).
|
|
@@ -2525,15 +2557,45 @@ async function runAiTest(file, opts, backend, platform, device) {
|
|
|
2525
2557
|
reason: e.message,
|
|
2526
2558
|
kind: (0, errors_1.isEnvError)(e) ? 'env' : 'fail',
|
|
2527
2559
|
});
|
|
2560
|
+
let sealed;
|
|
2528
2561
|
try {
|
|
2529
2562
|
await prefetchArchiveLogs(backend);
|
|
2530
|
-
run_1.Recorder.archive();
|
|
2563
|
+
sealed = run_1.Recorder.archive();
|
|
2531
2564
|
}
|
|
2532
2565
|
catch (sealErr) {
|
|
2533
2566
|
// Best-effort seal in an error path; surface a failure (the run state may itself be
|
|
2534
2567
|
// unreadable) but still throw the ORIGINAL error below.
|
|
2535
2568
|
(0, output_1.err)(`[ai] could not archive the run after a mid-run error (${sealErr.message})`);
|
|
2536
2569
|
}
|
|
2570
|
+
// An EVICTION is the one mid-run throw that is a RESULT, not an error. The server ended
|
|
2571
|
+
// this run because its phone left the pool — the run is not broken, it is over — and the
|
|
2572
|
+
// caller needs what a result carries: the archive (how far it got, which phone) and the
|
|
2573
|
+
// flag. A parallel suite re-runs it as a fresh run without spending a retry (#147); thrown,
|
|
2574
|
+
// it reached the suite as a bare error with no report and no device. Still exit 3.
|
|
2575
|
+
// An environment throw on a run the server evicted counts too: a worker that died mid-step
|
|
2576
|
+
// answers with an error rather than a result, and the server marks that response.
|
|
2577
|
+
if (e instanceof errors_1.RunEvictedError || ((0, errors_1.isEnvError)(e) && backend.wasEvicted?.())) {
|
|
2578
|
+
const reason = e.message.split('\n')[0];
|
|
2579
|
+
(0, output_1.err)(`[ai] ABORTED — environment: ${reason}`);
|
|
2580
|
+
if (sealed)
|
|
2581
|
+
(0, output_1.err)(`[ai] report: ${sealed.htmlPath}`);
|
|
2582
|
+
return {
|
|
2583
|
+
ok: false,
|
|
2584
|
+
cached,
|
|
2585
|
+
costUsd: Number(cost.usd().toFixed(4)),
|
|
2586
|
+
costLine: cost.summaryLine(),
|
|
2587
|
+
modelRepairs: 0,
|
|
2588
|
+
improvements: [],
|
|
2589
|
+
planSteps: plan.steps.length,
|
|
2590
|
+
runDir: sealed?.dir ?? '',
|
|
2591
|
+
reportHtml: sealed?.htmlPath ?? '',
|
|
2592
|
+
junitXml: sealed?.xmlPath ?? '',
|
|
2593
|
+
state: sealed?.state ?? null,
|
|
2594
|
+
failure: { where: 'run', reason },
|
|
2595
|
+
abortedForEnv: true,
|
|
2596
|
+
evicted: true,
|
|
2597
|
+
};
|
|
2598
|
+
}
|
|
2537
2599
|
throw e;
|
|
2538
2600
|
}
|
|
2539
2601
|
finally {
|
|
@@ -2586,6 +2648,11 @@ async function runAiTest(file, opts, backend, platform, device) {
|
|
|
2586
2648
|
...(result.abortedForBudget ? { abortedForBudget: true } : {}),
|
|
2587
2649
|
...(result.abortedForTimeout ? { abortedForTimeout: true } : {}),
|
|
2588
2650
|
...(result.abortedForEnv ? { abortedForEnv: true } : {}),
|
|
2651
|
+
// The step that was running when the phone vanished failed with the phone's own error, and
|
|
2652
|
+
// the server shed the phone while answering it: the eviction is heard only by the requests
|
|
2653
|
+
// after it (the failure evidence and the log fetch above). Asked now, and only of an
|
|
2654
|
+
// environment abort — an assertion that failed first is a regression whatever happened next.
|
|
2655
|
+
...(result.abortedForEnv && backend.wasEvicted?.() ? { evicted: true } : {}),
|
|
2589
2656
|
};
|
|
2590
2657
|
}
|
|
2591
2658
|
async function cmdAi(positionals, flags) {
|
|
@@ -2643,6 +2710,8 @@ async function cmdAi(positionals, flags) {
|
|
|
2643
2710
|
...(result.abortedForBudget ? { abortedForBudget: true } : {}),
|
|
2644
2711
|
...(result.abortedForTimeout ? { abortedForTimeout: true } : {}),
|
|
2645
2712
|
...(result.abortedForEnv ? { abortedForEnv: true } : {}),
|
|
2713
|
+
// The server ended the run because its phone left: start it again as a fresh run.
|
|
2714
|
+
...(result.evicted ? { evicted: true } : {}),
|
|
2646
2715
|
});
|
|
2647
2716
|
}
|
|
2648
2717
|
else if (result.reportHtml) {
|
|
@@ -2938,6 +3007,12 @@ function laneResult(code, parsed, detail, lane) {
|
|
|
2938
3007
|
const str = (k) => (typeof parsed?.[k] === 'string' ? parsed[k] : undefined);
|
|
2939
3008
|
const num = (k) => (typeof parsed?.[k] === 'number' ? parsed[k] : undefined);
|
|
2940
3009
|
const yes = (k) => parsed?.[k] === true;
|
|
3010
|
+
// The two lease outcomes a parallel suite must NOT read as a broken box (#147), both exit 3.
|
|
3011
|
+
// Gated on the code too: the exit code is the verdict, and a kind never overrides it.
|
|
3012
|
+
// - refused at /v1/lease: the child never started a run, so it is no attempt at all;
|
|
3013
|
+
// - evicted part-way: usually a result document (the run archived), or a thrown one.
|
|
3014
|
+
const noDevice = code === 3 && str('errorKind') === 'NoFreeDeviceError';
|
|
3015
|
+
const evicted = code === 3 && (yes('evicted') || str('errorKind') === 'RunEvictedError');
|
|
2941
3016
|
const raw = parsed?.failure;
|
|
2942
3017
|
// A budget / timeout / environment abort comes back as a bare FLAG with no `failure`
|
|
2943
3018
|
// object — the engine returns it that way, and `toSuiteResult` composes its own wording
|
|
@@ -2982,9 +3057,11 @@ function laneResult(code, parsed, detail, lane) {
|
|
|
2982
3057
|
// check a TypeError in a new code path reads as an environment failure: the suite
|
|
2983
3058
|
// probes the lane, finds it healthy, retries, and two in a row retire a perfectly good
|
|
2984
3059
|
// device via ENV_STREAK_LIMIT.
|
|
2985
|
-
...((yes('abortedForEnv') || code === 3 || code === 127) && str('errorKind') !== 'Error'
|
|
3060
|
+
...((yes('abortedForEnv') || code === 3 || code === 127) && str('errorKind') !== 'Error' && !noDevice
|
|
2986
3061
|
? { abortedForEnv: true }
|
|
2987
3062
|
: {}),
|
|
3063
|
+
...(noDevice ? { noDevice: true } : {}),
|
|
3064
|
+
...(evicted ? { evicted: true } : {}),
|
|
2988
3065
|
// Exit 2 is verikun's USAGE code — a flag the child rejected, an unreadable test, a
|
|
2989
3066
|
// payload the server refused. The serial path reaches this verdict from the thrown
|
|
2990
3067
|
// CliError (`isRetryableThrow`); across a process boundary the throw is only an exit
|
|
@@ -3197,6 +3274,21 @@ async function cmdSuiteParallel(dirArg, flags, pool) {
|
|
|
3197
3274
|
claimLanes: (used) => grantLanes(used, platform, pool.elastic, grants),
|
|
3198
3275
|
runTest: (file, lane) => runLaneTest(file, lane, flags, platform, app),
|
|
3199
3276
|
preflight: (lane) => lanePreflight(lane, flags, platform),
|
|
3277
|
+
// Is a phone free for this server lane? Asked of /v1/health, which leases nothing: a
|
|
3278
|
+
// lease request that finds nothing free lets the server take over a sibling lane's
|
|
3279
|
+
// quiet lease (see SuiteDeps.serverSlots). A server that does not answer is left for
|
|
3280
|
+
// the lane's own child to report, exactly as before.
|
|
3281
|
+
serverSlots: async (lane) => {
|
|
3282
|
+
if (!lane.server)
|
|
3283
|
+
return undefined;
|
|
3284
|
+
try {
|
|
3285
|
+
const health = await (0, remote_1.pingServer)(remoteOptsFrom(lane.server, flags));
|
|
3286
|
+
return { capacity: health.capacity ?? 1, note: (0, remote_1.poolNote)(health) };
|
|
3287
|
+
}
|
|
3288
|
+
catch {
|
|
3289
|
+
return undefined;
|
|
3290
|
+
}
|
|
3291
|
+
},
|
|
3200
3292
|
// `reset` is deliberately NOT wired: against a pooled server a reset issued from
|
|
3201
3293
|
// here would take its own lease and could land on a different device than the test
|
|
3202
3294
|
// that follows. `vk ai --reset-app` does it inside the test's own lease instead.
|
|
@@ -3682,6 +3774,11 @@ SUITE (run a directory of natural-language tests as one gated suite)
|
|
|
3682
3774
|
stops the suite once total model spend crosses
|
|
3683
3775
|
it (exit 1). One merged report either way, with
|
|
3684
3776
|
wall-clock reported apart from device time.
|
|
3777
|
+
Over --server a lane with no free device waits
|
|
3778
|
+
for one; a test whose device left the pool
|
|
3779
|
+
re-runs without spending a retry. With no device
|
|
3780
|
+
for any lane it stops after
|
|
3781
|
+
VERIKUN_SUITE_DEVICE_WAIT_MIN (default 10; exit 3).
|
|
3685
3782
|
|
|
3686
3783
|
SERVER (expose a locally-connected device to remote verikun clients)
|
|
3687
3784
|
server [--bind addr] [--port n] [--auth-key k] [--devices all|all-android|all-ios|a,b]
|