verikun 0.4.1 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +145 -27
- package/dist/agent/cost.js +21 -10
- package/dist/agent/engine.js +8 -8
- package/dist/agent/openai.js +243 -0
- package/dist/agent/remote.js +166 -0
- package/dist/args.js +2 -0
- package/dist/bin/verikun.js +0 -0
- package/dist/cli.js +425 -108
- package/dist/drivers/adb.js +12 -0
- package/dist/drivers/index.js +5 -5
- package/dist/drivers/ios.js +358 -0
- package/dist/report.js +93 -3
- package/dist/rpc.js +46 -0
- package/dist/run.js +101 -1
- package/dist/server.js +349 -0
- package/dist/suite.js +152 -0
- package/dist/ui/ios-parse.js +149 -0
- package/dist/version.js +1 -1
- package/package.json +2 -2
package/dist/cli.js
CHANGED
|
@@ -1,5 +1,40 @@
|
|
|
1
1
|
"use strict";
|
|
2
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
3
|
+
if (k2 === undefined) k2 = k;
|
|
4
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
5
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
6
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
7
|
+
}
|
|
8
|
+
Object.defineProperty(o, k2, desc);
|
|
9
|
+
}) : (function(o, m, k, k2) {
|
|
10
|
+
if (k2 === undefined) k2 = k;
|
|
11
|
+
o[k2] = m[k];
|
|
12
|
+
}));
|
|
13
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
14
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
15
|
+
}) : function(o, v) {
|
|
16
|
+
o["default"] = v;
|
|
17
|
+
});
|
|
18
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
19
|
+
var ownKeys = function(o) {
|
|
20
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
21
|
+
var ar = [];
|
|
22
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
23
|
+
return ar;
|
|
24
|
+
};
|
|
25
|
+
return ownKeys(o);
|
|
26
|
+
};
|
|
27
|
+
return function (mod) {
|
|
28
|
+
if (mod && mod.__esModule) return mod;
|
|
29
|
+
var result = {};
|
|
30
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
31
|
+
__setModuleDefault(result, mod);
|
|
32
|
+
return result;
|
|
33
|
+
};
|
|
34
|
+
})();
|
|
2
35
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
36
|
+
exports.platformFromFlags = platformFromFlags;
|
|
37
|
+
exports.deviceFromFlags = deviceFromFlags;
|
|
3
38
|
exports.parsePoint = parsePoint;
|
|
4
39
|
exports.healNote = healNote;
|
|
5
40
|
exports.parseDuration = parseDuration;
|
|
@@ -11,6 +46,7 @@ exports.chooseLogOpts = chooseLogOpts;
|
|
|
11
46
|
exports.evalAssert = evalAssert;
|
|
12
47
|
exports.tokenizeLine = tokenizeLine;
|
|
13
48
|
exports.withBatchGlobals = withBatchGlobals;
|
|
49
|
+
exports.executeForServer = executeForServer;
|
|
14
50
|
exports.run = run;
|
|
15
51
|
const node_fs_1 = require("node:fs");
|
|
16
52
|
const node_path_1 = require("node:path");
|
|
@@ -25,10 +61,14 @@ const run_1 = require("./run");
|
|
|
25
61
|
const image_1 = require("./image");
|
|
26
62
|
const engine_1 = require("./agent/engine");
|
|
27
63
|
const claude_1 = require("./agent/claude");
|
|
64
|
+
const openai_1 = require("./agent/openai");
|
|
28
65
|
const cache_1 = require("./agent/cache");
|
|
29
66
|
const cost_1 = require("./agent/cost");
|
|
67
|
+
const remote_1 = require("./agent/remote");
|
|
68
|
+
const suite_1 = require("./suite");
|
|
30
69
|
const version_1 = require("./version");
|
|
31
70
|
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
|
71
|
+
// Exported for src/server.ts (which resolves its own platform/device at startup).
|
|
32
72
|
function platformFromFlags(flags) {
|
|
33
73
|
if ((0, args_1.flagBool)(flags, 'ios'))
|
|
34
74
|
return 'ios';
|
|
@@ -149,10 +189,10 @@ function cmdDevices(ctx) {
|
|
|
149
189
|
}
|
|
150
190
|
try {
|
|
151
191
|
// Only include booted simulators; always include physical devices (they carry a note)
|
|
152
|
-
allDevices.push(...new drivers_1.
|
|
192
|
+
allDevices.push(...new drivers_1.IdbDriver().listDevices().filter((d) => d.state === 'booted' || d.note));
|
|
153
193
|
}
|
|
154
194
|
catch (e) {
|
|
155
|
-
(0, output_1.err)(`devices:
|
|
195
|
+
(0, output_1.err)(`devices: iOS backend unavailable (${e.message})`);
|
|
156
196
|
}
|
|
157
197
|
if ((0, args_1.flagBool)(ctx.flags, 'json')) {
|
|
158
198
|
(0, output_1.json)(allDevices);
|
|
@@ -171,18 +211,50 @@ function cmdDevices(ctx) {
|
|
|
171
211
|
}
|
|
172
212
|
function cmdDoctor(ctx) {
|
|
173
213
|
if (ctx.platform === 'ios') {
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
214
|
+
try {
|
|
215
|
+
const r = (0, exec_1.runText)('xcrun', ['simctl', 'list', 'devices', 'booted']);
|
|
216
|
+
(0, output_1.out)('xcrun: present');
|
|
217
|
+
(0, output_1.out)(r.stdout.trim() || '(no booted simulators)');
|
|
218
|
+
}
|
|
219
|
+
catch (e) {
|
|
220
|
+
// Not necessarily missing: runText also throws on a spawn timeout or other exec
|
|
221
|
+
// failure, so surface the real reason rather than always claiming "NOT FOUND".
|
|
222
|
+
(0, output_1.err)(`xcrun: ${e.message}`);
|
|
223
|
+
(0, output_1.err)(' (if the Xcode command-line tools are not installed: `xcode-select --install`)');
|
|
224
|
+
return 3;
|
|
225
|
+
}
|
|
226
|
+
// idb (+ its companion) powers everything interactive: ui/tap/text/swipe/key/logs.
|
|
227
|
+
const idb = process.env.IDB || 'idb';
|
|
228
|
+
let idbOk = true;
|
|
229
|
+
try {
|
|
230
|
+
(0, exec_1.runText)(idb, ['--help']); // idb has no --version; --help confirms the binary runs
|
|
231
|
+
(0, output_1.out)('idb: present');
|
|
232
|
+
}
|
|
233
|
+
catch (e) {
|
|
234
|
+
(0, output_1.err)(`idb: ${e.message}`);
|
|
235
|
+
(0, output_1.err)(' needed for ui/tap/text/swipe/key/logs — install: `brew install idb-companion` then `pip install fb-idb`');
|
|
236
|
+
idbOk = false;
|
|
237
|
+
}
|
|
238
|
+
try {
|
|
239
|
+
(0, exec_1.runText)('idb_companion', ['--help']);
|
|
240
|
+
(0, output_1.out)('idb_companion: present');
|
|
241
|
+
}
|
|
242
|
+
catch (e) {
|
|
243
|
+
(0, output_1.err)(`idb_companion: ${e.message}`);
|
|
244
|
+
(0, output_1.err)(' install: `brew install idb-companion`');
|
|
245
|
+
idbOk = false;
|
|
246
|
+
}
|
|
247
|
+
(0, output_1.out)('note: simulator screenshots + launch/stop work via simctl; ui/tap/text/swipe/key/logs use idb.');
|
|
248
|
+
return idbOk ? 0 : 3;
|
|
179
249
|
}
|
|
180
250
|
const adb = process.env.ADB || 'adb';
|
|
181
251
|
try {
|
|
182
252
|
(0, output_1.out)('adb: ' + (0, exec_1.runText)(adb, ['version']).stdout.split('\n')[0]);
|
|
183
253
|
}
|
|
184
|
-
catch {
|
|
185
|
-
|
|
254
|
+
catch (e) {
|
|
255
|
+
// Not necessarily missing: runText also throws on a spawn timeout or other exec
|
|
256
|
+
// failure — surface the real reason rather than always claiming "NOT FOUND".
|
|
257
|
+
(0, output_1.err)(`adb: ${e.message}`);
|
|
186
258
|
return 3;
|
|
187
259
|
}
|
|
188
260
|
const devices = ctx.driver.listDevices();
|
|
@@ -856,99 +928,157 @@ async function cmdBatch(positionals, batchFlags) {
|
|
|
856
928
|
(0, output_1.err)(`[verikun] batch: ${commands.length} command(s) ok`);
|
|
857
929
|
return 0;
|
|
858
930
|
}
|
|
859
|
-
|
|
860
|
-
// ai — compile a natural-language test to a plan IR, then run it (self-healing)
|
|
861
|
-
// ---------------------------------------------------------------------------
|
|
862
|
-
//
|
|
863
|
-
// `vk ai <file>` reads a plain-English test, compiles it ONCE into a deterministic
|
|
864
|
-
// plan IR via the model (cached by NL + app build), then replays it with NO model
|
|
865
|
-
// calls on the happy path. The model is woken only to repair a step that fails to
|
|
866
|
-
// resolve its selector; a green run persists the (possibly repaired) plan so the
|
|
867
|
-
// next run is free again. Cost is bounded by --max-cost-usd. Progress streams to
|
|
868
|
-
// stderr (CI liveness — it never goes quiet); stdout carries the final result.
|
|
869
|
-
async function cmdAi(positionals, flags) {
|
|
870
|
-
const file = positionals[0];
|
|
871
|
-
if (!file) {
|
|
872
|
-
throw new errors_1.CliError('Usage: verikun ai <file> [--model m] [--max-cost-usd n] [--timeout dur] [--show-plan] [--recompile]', 2);
|
|
873
|
-
}
|
|
874
|
-
let nl;
|
|
875
|
-
try {
|
|
876
|
-
nl = (0, node_fs_1.readFileSync)((0, node_path_1.resolve)(process.cwd(), file), 'utf8');
|
|
877
|
-
}
|
|
878
|
-
catch (e) {
|
|
879
|
-
throw new errors_1.CliError(`ai: cannot read '${file}' (${e.message})`, 2);
|
|
880
|
-
}
|
|
881
|
-
if (!nl.trim())
|
|
882
|
-
throw new errors_1.CliError(`ai: '${file}' is empty`, 2);
|
|
883
|
-
const platform = platformFromFlags(flags);
|
|
884
|
-
const device = deviceFromFlags(flags, platform);
|
|
931
|
+
function parseAiOptions(flags) {
|
|
885
932
|
const model = (0, cost_1.resolveModel)((0, args_1.flagStr)(flags, 'model'));
|
|
886
933
|
const overrideRaw = (0, args_1.flagStr)(flags, 'cost-override');
|
|
887
934
|
const override = overrideRaw ? (0, cost_1.parseCostOverride)(overrideRaw) : undefined;
|
|
888
935
|
const maxCostUsd = (0, args_1.flagNum)(flags, 'max-cost-usd') ?? cost_1.DEFAULT_MAX_COST_USD;
|
|
889
936
|
if (maxCostUsd <= 0)
|
|
890
937
|
throw new errors_1.CliError(`--max-cost-usd must be greater than 0 (got ${maxCostUsd}).`, 2);
|
|
891
|
-
const cost = new cost_1.CostTracker((0, cost_1.priceFor)(model, override), maxCostUsd);
|
|
892
938
|
// Whole-run wall-clock ceiling (default 15m) so a runaway loop/repair can't hang the run.
|
|
893
939
|
const timeoutFlag = (0, args_1.flagStr)(flags, 'timeout');
|
|
894
940
|
const timeoutMs = timeoutFlag ? parseDuration(timeoutFlag, 'timeout') : engine_1.DEFAULT_RUN_TIMEOUT_MS;
|
|
895
|
-
|
|
896
|
-
|
|
897
|
-
|
|
898
|
-
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
|
|
902
|
-
|
|
903
|
-
|
|
904
|
-
|
|
905
|
-
|
|
906
|
-
|
|
907
|
-
let
|
|
941
|
+
return {
|
|
942
|
+
model,
|
|
943
|
+
price: (0, cost_1.priceFor)(model, override),
|
|
944
|
+
maxCostUsd,
|
|
945
|
+
timeoutMs,
|
|
946
|
+
effort: (0, args_1.flagStr)(flags, 'effort'),
|
|
947
|
+
pkg: (0, args_1.flagStr)(flags, 'package'),
|
|
948
|
+
build: (0, args_1.flagStr)(flags, 'app-build'),
|
|
949
|
+
recompile: (0, args_1.flagBool)(flags, 'recompile') || (0, args_1.flagBool)(flags, 'no-cache'),
|
|
950
|
+
};
|
|
951
|
+
}
|
|
952
|
+
function readAiTest(file) {
|
|
953
|
+
let nl;
|
|
954
|
+
try {
|
|
955
|
+
nl = (0, node_fs_1.readFileSync)((0, node_path_1.resolve)(process.cwd(), file), 'utf8');
|
|
956
|
+
}
|
|
957
|
+
catch (e) {
|
|
958
|
+
throw new errors_1.CliError(`ai: cannot read '${file}' (${e.message})`, 2);
|
|
959
|
+
}
|
|
960
|
+
if (!nl.trim())
|
|
961
|
+
throw new errors_1.CliError(`ai: '${file}' is empty`, 2);
|
|
962
|
+
return nl;
|
|
963
|
+
}
|
|
964
|
+
/** The env var carrying the API key for a model's provider (per-provider keys). */
|
|
965
|
+
function keyEnvFor(model) {
|
|
966
|
+
return (0, cost_1.providerFor)(model) === 'openai' ? 'OPENAI_API_KEY' : 'ANTHROPIC_API_KEY';
|
|
967
|
+
}
|
|
968
|
+
/** Route the model to its backend; each provider reads its own key. A missing key
|
|
969
|
+
* means no provider (compile/repair unavailable) — the same graceful degradation
|
|
970
|
+
* as before. */
|
|
971
|
+
function makeProvider(opts) {
|
|
972
|
+
const apiKey = process.env[keyEnvFor(opts.model)];
|
|
973
|
+
if (!apiKey)
|
|
974
|
+
return null;
|
|
975
|
+
return (0, cost_1.providerFor)(opts.model) === 'openai'
|
|
976
|
+
? new openai_1.OpenAiProvider({ model: opts.model, apiKey, effort: opts.effort })
|
|
977
|
+
: new claude_1.ClaudeProvider({ model: opts.model, apiKey, effort: opts.effort });
|
|
978
|
+
}
|
|
979
|
+
/** Obtain the plan: a cache hit (free) or a compile (pays tokens; may seed from a
|
|
980
|
+
* prior build's plan to avoid a full recompile). The fresh compile is cached right
|
|
981
|
+
* away, so an unchanged test is never recompiled — even via --show-plan or after a
|
|
982
|
+
* failed run. A green run later re-persists the healed plan (never a half-healed one). */
|
|
983
|
+
async function obtainPlan(key, file, opts, cost, provider) {
|
|
984
|
+
const cached = opts.recompile ? null : (0, cache_1.readPlan)(key);
|
|
908
985
|
if (cached) {
|
|
909
|
-
plan
|
|
910
|
-
|
|
986
|
+
(0, output_1.err)(`[ai] plan cache hit — ${opts.model} not called to compile`);
|
|
987
|
+
return { plan: cached.plan, cached: true };
|
|
911
988
|
}
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
|
|
915
|
-
|
|
916
|
-
|
|
917
|
-
|
|
918
|
-
|
|
919
|
-
|
|
920
|
-
|
|
921
|
-
|
|
922
|
-
|
|
923
|
-
(0,
|
|
924
|
-
// Cache the freshly-compiled plan right away, keyed by the test-text hash, so an
|
|
925
|
-
// unchanged test is never recompiled — even via --show-plan or after a failed run.
|
|
926
|
-
// A green run below re-persists the healed plan; a failed run leaves this clean
|
|
927
|
-
// compile cached (never a half-healed one).
|
|
928
|
-
try {
|
|
929
|
-
(0, cache_1.writePlan)(key, plan);
|
|
930
|
-
}
|
|
931
|
-
catch (e) {
|
|
932
|
-
(0, output_1.err)(`[ai] could not cache compiled plan: ${e.message}`);
|
|
933
|
-
}
|
|
989
|
+
if (!provider) {
|
|
990
|
+
throw new errors_1.CliError(`${keyEnvFor(opts.model)} is not set — \`vk ai\` needs it to compile the test (model ${opts.model}). Set it and retry.`, 3);
|
|
991
|
+
}
|
|
992
|
+
const seed = (0, cache_1.findSeed)(key);
|
|
993
|
+
if (seed)
|
|
994
|
+
(0, output_1.err)(`[ai] no exact cache; seeding from a prior plan (build ${seed.build ?? 'unknown'})`);
|
|
995
|
+
(0, output_1.err)(`[ai] compiling '${file}' with ${opts.model} (effort ${opts.effort ?? 'default'})…`);
|
|
996
|
+
const compiled = await provider.compile({ nl: key.nl, pkg: key.pkg, platform: key.platform, seed: seed?.plan });
|
|
997
|
+
cost.add(compiled.usage, 'compile');
|
|
998
|
+
(0, output_1.err)(`[ai] compiled ${compiled.plan.steps.length} top-level step(s) · ${cost.summaryLine()}`);
|
|
999
|
+
try {
|
|
1000
|
+
(0, cache_1.writePlan)(key, compiled.plan);
|
|
934
1001
|
}
|
|
935
|
-
|
|
936
|
-
|
|
937
|
-
(0, output_1.json)(plan);
|
|
938
|
-
return 0;
|
|
1002
|
+
catch (e) {
|
|
1003
|
+
(0, output_1.err)(`[ai] could not cache compiled plan: ${e.message}`);
|
|
939
1004
|
}
|
|
1005
|
+
return { plan: compiled.plan, cached: false };
|
|
1006
|
+
}
|
|
1007
|
+
async function resolveBackend(platform, device, flags) {
|
|
1008
|
+
const server = (0, args_1.flagStr)(flags, 'server') || process.env.VERIKUN_SERVER || undefined;
|
|
1009
|
+
if (!server) {
|
|
1010
|
+
const driver = (0, drivers_1.getDriver)(platform, device);
|
|
1011
|
+
return {
|
|
1012
|
+
backend: {
|
|
1013
|
+
exec: (command, positionals, f) => executeOutcome(command, positionals, f, driver),
|
|
1014
|
+
getElements: () => driver.getElements(),
|
|
1015
|
+
install: (appPath) => driver.install(appPath),
|
|
1016
|
+
reset: (appId) => {
|
|
1017
|
+
assertSafeAppId(appId);
|
|
1018
|
+
// iOS has no per-app data reset — degrade honestly to a force-stop.
|
|
1019
|
+
if (platform === 'ios')
|
|
1020
|
+
driver.stop(appId);
|
|
1021
|
+
else
|
|
1022
|
+
driver.clearApp(appId);
|
|
1023
|
+
},
|
|
1024
|
+
},
|
|
1025
|
+
platform,
|
|
1026
|
+
device,
|
|
1027
|
+
};
|
|
1028
|
+
}
|
|
1029
|
+
let runCtx = { platform, device };
|
|
1030
|
+
const opts = {
|
|
1031
|
+
url: server,
|
|
1032
|
+
authKey: (0, args_1.flagStr)(flags, 'auth-key') || process.env.VERIKUN_SERVER_AUTH_KEY || undefined,
|
|
1033
|
+
// Each remote step is spliced into the local active run so the archived report
|
|
1034
|
+
// is identical to a local run's.
|
|
1035
|
+
onStep: (step, artifacts) => run_1.Recorder.appendForeignStep(step, artifacts, runCtx),
|
|
1036
|
+
};
|
|
1037
|
+
const health = await (0, remote_1.pingServer)(opts); // fails fast (exit 3) on a bad URL or key
|
|
1038
|
+
runCtx = { platform: health.platform, device: health.serial };
|
|
1039
|
+
(0, output_1.err)(`[verikun] server ${server}: ${health.platform} · device ${health.serial} · verikun ${health.version}`);
|
|
1040
|
+
return {
|
|
1041
|
+
backend: (0, remote_1.createRemoteBackend)(opts, health),
|
|
1042
|
+
platform: health.platform,
|
|
1043
|
+
device: health.serial,
|
|
1044
|
+
remote: { url: server, version: health.version },
|
|
1045
|
+
};
|
|
1046
|
+
}
|
|
1047
|
+
/**
|
|
1048
|
+
* Run one natural-language test through a backend and return DATA — no stdout
|
|
1049
|
+
* writes (stdout stays the caller's one result; progress streams to stderr).
|
|
1050
|
+
* `vk ai` wraps it with its --json/report output; `vk suite` calls it per test.
|
|
1051
|
+
*/
|
|
1052
|
+
async function runAiTest(file, opts, backend, platform, device) {
|
|
1053
|
+
const nl = readAiTest(file);
|
|
1054
|
+
const key = { nl, pkg: opts.pkg, build: opts.build, platform };
|
|
1055
|
+
const cost = new cost_1.CostTracker(opts.price, opts.maxCostUsd);
|
|
1056
|
+
const deadline = Date.now() + opts.timeoutMs;
|
|
1057
|
+
const provider = makeProvider(opts);
|
|
1058
|
+
const { plan, cached } = await obtainPlan(key, file, opts, cost, provider);
|
|
940
1059
|
// Running needs the provider for repair-on-failure; a cache hit with no key can't repair.
|
|
941
1060
|
if (!provider) {
|
|
942
|
-
throw new errors_1.CliError(
|
|
1061
|
+
throw new errors_1.CliError(`${keyEnvFor(opts.model)} is not set — \`vk ai\` needs it to repair a failing step at runtime (model ${opts.model}).`, 3);
|
|
943
1062
|
}
|
|
944
1063
|
// The budget is a TOTAL-run ceiling: if the compile alone already crossed it, abort
|
|
945
1064
|
// before running. A cache hit spends nothing, so a free replay is still allowed.
|
|
946
1065
|
if (!cached && cost.exceeded()) {
|
|
947
|
-
(0, output_1.err)(`[ai] cost ceiling $${maxCostUsd} reached during compile (${cost.summaryLine()}) — not running`);
|
|
948
|
-
return
|
|
949
|
-
|
|
950
|
-
|
|
951
|
-
|
|
1066
|
+
(0, output_1.err)(`[ai] cost ceiling $${opts.maxCostUsd} reached during compile (${cost.summaryLine()}) — not running`);
|
|
1067
|
+
return {
|
|
1068
|
+
ok: false,
|
|
1069
|
+
costUsd: Number(cost.usd().toFixed(4)),
|
|
1070
|
+
costLine: cost.summaryLine(),
|
|
1071
|
+
modelRepairs: 0,
|
|
1072
|
+
improvements: [],
|
|
1073
|
+
runDir: '',
|
|
1074
|
+
reportHtml: '',
|
|
1075
|
+
junitXml: '',
|
|
1076
|
+
state: null,
|
|
1077
|
+
abortedForBudget: true,
|
|
1078
|
+
failure: { where: 'compile', reason: `cost ceiling $${opts.maxCostUsd} reached during compile` },
|
|
1079
|
+
};
|
|
1080
|
+
}
|
|
1081
|
+
// One explicit run for the whole flow (so rollover can't split the test).
|
|
952
1082
|
const existing = run_1.Recorder.status();
|
|
953
1083
|
if (existing && existing.steps.length > 0) {
|
|
954
1084
|
// Seal the pre-existing run into the archive instead of letting start(force=true)
|
|
@@ -957,14 +1087,13 @@ async function cmdAi(positionals, flags) {
|
|
|
957
1087
|
(0, output_1.err)(`[ai] archived the active run ('${existing.name}', ${existing.steps.length} step(s)) → ${sealed.dir}`);
|
|
958
1088
|
}
|
|
959
1089
|
run_1.Recorder.start(`ai: ${(0, node_path_1.basename)(file)}`, platform, device, true);
|
|
960
|
-
const driver = (0, drivers_1.getDriver)(platform, device);
|
|
961
1090
|
// Suppress per-step `out()` so stdout stays the one final result; progress -> stderr.
|
|
962
1091
|
const prevQuiet = (0, output_1.setOutputQuiet)(true);
|
|
963
1092
|
let result;
|
|
964
1093
|
try {
|
|
965
1094
|
result = await (0, engine_1.runPlan)(plan, {
|
|
966
|
-
exec: (command, pos, f) =>
|
|
967
|
-
getElements: () =>
|
|
1095
|
+
exec: (command, pos, f) => backend.exec(command, pos, f),
|
|
1096
|
+
getElements: () => backend.getElements(),
|
|
968
1097
|
provider,
|
|
969
1098
|
cost,
|
|
970
1099
|
log: (m) => (0, output_1.err)(m),
|
|
@@ -991,8 +1120,8 @@ async function cmdAi(positionals, flags) {
|
|
|
991
1120
|
finally {
|
|
992
1121
|
(0, output_1.setOutputQuiet)(prevQuiet);
|
|
993
1122
|
}
|
|
994
|
-
//
|
|
995
|
-
//
|
|
1123
|
+
// Persist the (possibly repaired) plan only on a fully-green run; attach the
|
|
1124
|
+
// cost + improvements summary to the run; archive into the report.
|
|
996
1125
|
const costLine = cost.summaryLine();
|
|
997
1126
|
if (result.ok) {
|
|
998
1127
|
try {
|
|
@@ -1006,13 +1135,13 @@ async function cmdAi(positionals, flags) {
|
|
|
1006
1135
|
run_1.Recorder.annotateRun({
|
|
1007
1136
|
ai: { ok: result.ok, cost: costLine, modelRepairs: result.modelRepairs, improvements: result.improvements },
|
|
1008
1137
|
});
|
|
1009
|
-
const { dir, xmlPath, htmlPath } = run_1.Recorder.archive();
|
|
1138
|
+
const { dir, xmlPath, htmlPath, state } = run_1.Recorder.archive();
|
|
1010
1139
|
const status = result.ok
|
|
1011
1140
|
? 'PASS'
|
|
1012
1141
|
: result.abortedForBudget
|
|
1013
|
-
? `ABORTED — cost ceiling $${maxCostUsd} reached`
|
|
1142
|
+
? `ABORTED — cost ceiling $${opts.maxCostUsd} reached`
|
|
1014
1143
|
: result.abortedForTimeout
|
|
1015
|
-
? `ABORTED — run timeout (${Math.round(timeoutMs / 1000)}s) reached`
|
|
1144
|
+
? `ABORTED — run timeout (${Math.round(opts.timeoutMs / 1000)}s) reached`
|
|
1016
1145
|
: `FAIL at ${result.failure?.where}: ${result.failure?.reason}`;
|
|
1017
1146
|
(0, output_1.err)(`[ai] ${status} · ${costLine}`);
|
|
1018
1147
|
(0, output_1.err)(`[ai] report: ${htmlPath}`);
|
|
@@ -1022,28 +1151,124 @@ async function cmdAi(positionals, flags) {
|
|
|
1022
1151
|
(0, output_1.err)(' - ' + imp);
|
|
1023
1152
|
}
|
|
1024
1153
|
(0, output_1.err)(`[ai] estimated total cost: $${cost.usd().toFixed(4)}`);
|
|
1154
|
+
return {
|
|
1155
|
+
ok: result.ok,
|
|
1156
|
+
costUsd: Number(cost.usd().toFixed(4)),
|
|
1157
|
+
costLine,
|
|
1158
|
+
modelRepairs: result.modelRepairs,
|
|
1159
|
+
improvements: result.improvements,
|
|
1160
|
+
runDir: dir,
|
|
1161
|
+
reportHtml: htmlPath,
|
|
1162
|
+
junitXml: xmlPath,
|
|
1163
|
+
state,
|
|
1164
|
+
...(result.failure ? { failure: result.failure } : {}),
|
|
1165
|
+
...(result.abortedForBudget ? { abortedForBudget: true } : {}),
|
|
1166
|
+
...(result.abortedForTimeout ? { abortedForTimeout: true } : {}),
|
|
1167
|
+
};
|
|
1168
|
+
}
|
|
1169
|
+
async function cmdAi(positionals, flags) {
|
|
1170
|
+
const file = positionals[0];
|
|
1171
|
+
if (!file) {
|
|
1172
|
+
throw new errors_1.CliError('Usage: verikun ai <file> [--model m] [--max-cost-usd n] [--timeout dur] [--server url] [--show-plan] [--recompile]', 2);
|
|
1173
|
+
}
|
|
1174
|
+
const opts = parseAiOptions(flags);
|
|
1175
|
+
// --show-plan: compile (or cache-hit) and print the IR — no device, no backend.
|
|
1176
|
+
if ((0, args_1.flagBool)(flags, 'show-plan')) {
|
|
1177
|
+
const nl = readAiTest(file);
|
|
1178
|
+
const key = { nl, pkg: opts.pkg, build: opts.build, platform: platformFromFlags(flags) };
|
|
1179
|
+
const cost = new cost_1.CostTracker(opts.price, opts.maxCostUsd);
|
|
1180
|
+
const { plan } = await obtainPlan(key, file, opts, cost, makeProvider(opts));
|
|
1181
|
+
(0, output_1.json)(plan);
|
|
1182
|
+
return 0;
|
|
1183
|
+
}
|
|
1184
|
+
const reqPlatform = platformFromFlags(flags);
|
|
1185
|
+
const { backend, platform, device } = await resolveBackend(reqPlatform, deviceFromFlags(flags, reqPlatform), flags);
|
|
1186
|
+
let result;
|
|
1187
|
+
try {
|
|
1188
|
+
result = await runAiTest(file, opts, backend, platform, device);
|
|
1189
|
+
}
|
|
1190
|
+
finally {
|
|
1191
|
+
await backend.close?.(); // frees a remote server's device lock for the next command
|
|
1192
|
+
}
|
|
1025
1193
|
if ((0, args_1.flagBool)(flags, 'json')) {
|
|
1026
1194
|
(0, output_1.json)({
|
|
1027
1195
|
ok: result.ok,
|
|
1028
|
-
model,
|
|
1029
|
-
cost: costLine,
|
|
1030
|
-
costUsd:
|
|
1196
|
+
model: opts.model,
|
|
1197
|
+
cost: result.costLine,
|
|
1198
|
+
costUsd: result.costUsd,
|
|
1031
1199
|
modelRepairs: result.modelRepairs,
|
|
1032
1200
|
improvements: result.improvements,
|
|
1033
|
-
report:
|
|
1034
|
-
junit:
|
|
1035
|
-
runDir:
|
|
1201
|
+
report: result.reportHtml,
|
|
1202
|
+
junit: result.junitXml,
|
|
1203
|
+
runDir: result.runDir,
|
|
1036
1204
|
...(result.failure ? { failure: result.failure } : {}),
|
|
1037
1205
|
...(result.abortedForBudget ? { abortedForBudget: true } : {}),
|
|
1038
1206
|
...(result.abortedForTimeout ? { abortedForTimeout: true } : {}),
|
|
1039
1207
|
});
|
|
1040
1208
|
}
|
|
1041
|
-
else {
|
|
1042
|
-
(0, output_1.out)(
|
|
1209
|
+
else if (result.reportHtml) {
|
|
1210
|
+
(0, output_1.out)(result.reportHtml); // primary machine result: the report path
|
|
1043
1211
|
}
|
|
1044
1212
|
return result.ok ? 0 : 1;
|
|
1045
1213
|
}
|
|
1046
1214
|
// ---------------------------------------------------------------------------
|
|
1215
|
+
// install — put an app build on the device (local driver or remote vk server)
|
|
1216
|
+
// ---------------------------------------------------------------------------
|
|
1217
|
+
async function cmdInstall(positionals, flags) {
|
|
1218
|
+
const appPath = positionals[0];
|
|
1219
|
+
if (!appPath)
|
|
1220
|
+
throw new errors_1.CliError('Usage: verikun install <app.apk|app.ipa> [--server url]', 2);
|
|
1221
|
+
const path = (0, node_path_1.resolve)(process.cwd(), appPath);
|
|
1222
|
+
if (!(0, node_fs_1.existsSync)(path))
|
|
1223
|
+
throw new errors_1.CliError(`install: '${appPath}' does not exist`, 2);
|
|
1224
|
+
const platform = platformFromFlags(flags);
|
|
1225
|
+
const { backend, remote } = await resolveBackend(platform, deviceFromFlags(flags, platform), flags);
|
|
1226
|
+
(0, output_1.err)(`[verikun] installing ${appPath}${remote ? ` via ${remote.url}` : ''}…`);
|
|
1227
|
+
try {
|
|
1228
|
+
await backend.install(path);
|
|
1229
|
+
}
|
|
1230
|
+
finally {
|
|
1231
|
+
await backend.close?.();
|
|
1232
|
+
}
|
|
1233
|
+
if ((0, args_1.flagBool)(flags, 'json'))
|
|
1234
|
+
(0, output_1.json)({ installed: appPath, ...(remote ? { server: remote.url } : {}) });
|
|
1235
|
+
else
|
|
1236
|
+
(0, output_1.out)(`installed ${appPath}`);
|
|
1237
|
+
return 0;
|
|
1238
|
+
}
|
|
1239
|
+
// ---------------------------------------------------------------------------
|
|
1240
|
+
// suite — run a directory of natural-language tests as one gated suite
|
|
1241
|
+
// ---------------------------------------------------------------------------
|
|
1242
|
+
async function cmdSuiteEntry(positionals, flags) {
|
|
1243
|
+
const dirArg = positionals[0];
|
|
1244
|
+
if (!dirArg)
|
|
1245
|
+
throw new errors_1.CliError('Usage: verikun suite <dir> [--app <id>] [--server url] [--name n] [--json]', 2);
|
|
1246
|
+
const opts = parseAiOptions(flags);
|
|
1247
|
+
// Pre-flight the model key BEFORE touching any device/server: every test needs it
|
|
1248
|
+
// to compile (on a cache miss) or to repair at runtime.
|
|
1249
|
+
if (!process.env[keyEnvFor(opts.model)]) {
|
|
1250
|
+
throw new errors_1.CliError(`${keyEnvFor(opts.model)} is not set — \`vk suite\` needs it to compile/repair tests (model ${opts.model}).`, 3);
|
|
1251
|
+
}
|
|
1252
|
+
const reqPlatform = platformFromFlags(flags);
|
|
1253
|
+
const { backend, platform, device } = await resolveBackend(reqPlatform, deviceFromFlags(flags, reqPlatform), flags);
|
|
1254
|
+
const app = (0, args_1.flagStr)(flags, 'app');
|
|
1255
|
+
if (app)
|
|
1256
|
+
assertSafeAppId(app);
|
|
1257
|
+
try {
|
|
1258
|
+
return await (0, suite_1.cmdSuite)(dirArg, flags, {
|
|
1259
|
+
platform,
|
|
1260
|
+
device,
|
|
1261
|
+
runTest: (file) => runAiTest(file, opts, backend, platform, device),
|
|
1262
|
+
// Reset app state between tests only when the app id is known; without --app,
|
|
1263
|
+
// each test is responsible for its own isolation (e.g. `launch --clear`).
|
|
1264
|
+
reset: app ? () => backend.reset(app) : undefined,
|
|
1265
|
+
});
|
|
1266
|
+
}
|
|
1267
|
+
finally {
|
|
1268
|
+
await backend.close?.();
|
|
1269
|
+
}
|
|
1270
|
+
}
|
|
1271
|
+
// ---------------------------------------------------------------------------
|
|
1047
1272
|
// Dispatch
|
|
1048
1273
|
// ---------------------------------------------------------------------------
|
|
1049
1274
|
async function executeCommand(command, ctx) {
|
|
@@ -1149,7 +1374,19 @@ async function executeOutcome(command, positionals, flags, sharedDriver) {
|
|
|
1149
1374
|
}
|
|
1150
1375
|
recorder = run_1.Recorder.beginStep(command, positionals, flags, platform, device, serial, driver);
|
|
1151
1376
|
}
|
|
1152
|
-
|
|
1377
|
+
}
|
|
1378
|
+
catch (e) {
|
|
1379
|
+
return { code: e instanceof errors_1.CliError ? e.exitCode : 3, error: e };
|
|
1380
|
+
}
|
|
1381
|
+
const d = driver; // assigned in the try above, or we already returned
|
|
1382
|
+
const ctx = { driver: d, platform, device, positionals, flags, record: recorder ?? undefined };
|
|
1383
|
+
return runRecorded(command, ctx, recorder, d);
|
|
1384
|
+
}
|
|
1385
|
+
/** The shared middle of executeOutcome / executeForServer: run the handler, close
|
|
1386
|
+
* the step (with failure evidence) whether it returned or threw, map to a raw
|
|
1387
|
+
* outcome (the error NOT printed — callers decide). */
|
|
1388
|
+
async function runRecorded(command, ctx, recorder, driver) {
|
|
1389
|
+
try {
|
|
1153
1390
|
const code = await executeCommand(command, ctx);
|
|
1154
1391
|
recorder?.finish(code, driver);
|
|
1155
1392
|
return { code };
|
|
@@ -1159,6 +1396,28 @@ async function executeOutcome(command, positionals, flags, sharedDriver) {
|
|
|
1159
1396
|
return { code: e instanceof errors_1.CliError ? e.exitCode : 3, error: e };
|
|
1160
1397
|
}
|
|
1161
1398
|
}
|
|
1399
|
+
/**
|
|
1400
|
+
* `vk server`'s per-request executor: run one already-validated leaf against the
|
|
1401
|
+
* server's fixed driver/platform (a client can never repoint the device via flags),
|
|
1402
|
+
* recording into an EPHEMERAL single-step recorder instead of the local run store.
|
|
1403
|
+
* Returns the raw outcome plus the finished step + artifact buffers, which travel
|
|
1404
|
+
* back over the wire and are spliced into the CALLER's run — so `resolved`/`tier`/
|
|
1405
|
+
* failure evidence survive remoting with zero handler changes.
|
|
1406
|
+
*/
|
|
1407
|
+
async function executeForServer(command, positionals, flags, driver, platform) {
|
|
1408
|
+
let serial;
|
|
1409
|
+
try {
|
|
1410
|
+
serial = driver.resolvedSerial();
|
|
1411
|
+
}
|
|
1412
|
+
catch {
|
|
1413
|
+
/* surfaced by the command handler below */
|
|
1414
|
+
}
|
|
1415
|
+
const recorder = run_1.Recorder.beginEphemeralStep(command, positionals, flags, platform, serial);
|
|
1416
|
+
const ctx = { driver, platform, device: serial, positionals, flags, record: recorder };
|
|
1417
|
+
const outcome = await runRecorded(command, ctx, recorder, driver);
|
|
1418
|
+
const { step, artifacts } = recorder.takeEphemeral();
|
|
1419
|
+
return { ...outcome, step, artifacts };
|
|
1420
|
+
}
|
|
1162
1421
|
/**
|
|
1163
1422
|
* Run one already-parsed command for the CLI / `batch`: dispatch meta-commands,
|
|
1164
1423
|
* else execute it and map any failure to a printed exit code. This is the shared
|
|
@@ -1174,9 +1433,9 @@ async function executeParsed(command, positionals, flags) {
|
|
|
1174
1433
|
return cmdRun(positionals, flags, platform, device);
|
|
1175
1434
|
if (command === 'batch')
|
|
1176
1435
|
return cmdBatch(positionals, flags);
|
|
1177
|
-
// `ai`
|
|
1178
|
-
// (usage 2 / env 3 / …) so they honor the exit-code
|
|
1179
|
-
// to the top-level "Fatal" handler (
|
|
1436
|
+
// `ai`/`suite`/`install`/`server` orchestrate their own steps; map their thrown
|
|
1437
|
+
// CliErrors to exit codes here (usage 2 / env 3 / …) so they honor the exit-code
|
|
1438
|
+
// contract instead of escaping to the top-level "Fatal" handler (exit 3).
|
|
1180
1439
|
if (command === 'ai') {
|
|
1181
1440
|
try {
|
|
1182
1441
|
return await cmdAi(positionals, flags);
|
|
@@ -1185,6 +1444,33 @@ async function executeParsed(command, positionals, flags) {
|
|
|
1185
1444
|
return mapError(e, flags);
|
|
1186
1445
|
}
|
|
1187
1446
|
}
|
|
1447
|
+
if (command === 'suite') {
|
|
1448
|
+
try {
|
|
1449
|
+
return await cmdSuiteEntry(positionals, flags);
|
|
1450
|
+
}
|
|
1451
|
+
catch (e) {
|
|
1452
|
+
return mapError(e, flags);
|
|
1453
|
+
}
|
|
1454
|
+
}
|
|
1455
|
+
if (command === 'install') {
|
|
1456
|
+
try {
|
|
1457
|
+
return await cmdInstall(positionals, flags);
|
|
1458
|
+
}
|
|
1459
|
+
catch (e) {
|
|
1460
|
+
return mapError(e, flags);
|
|
1461
|
+
}
|
|
1462
|
+
}
|
|
1463
|
+
if (command === 'server') {
|
|
1464
|
+
// Dynamic import: keeps node:http off the default load path and avoids a
|
|
1465
|
+
// static cli↔server cycle (server.ts imports executeForServer from here).
|
|
1466
|
+
try {
|
|
1467
|
+
const { cmdServer } = await Promise.resolve().then(() => __importStar(require('./server')));
|
|
1468
|
+
return await cmdServer(positionals, flags);
|
|
1469
|
+
}
|
|
1470
|
+
catch (e) {
|
|
1471
|
+
return mapError(e, flags);
|
|
1472
|
+
}
|
|
1473
|
+
}
|
|
1188
1474
|
const { code, error } = await executeOutcome(command, positionals, flags);
|
|
1189
1475
|
return error ? mapError(error, flags) : code;
|
|
1190
1476
|
}
|
|
@@ -1232,6 +1518,9 @@ ACT
|
|
|
1232
1518
|
launch <app> [--clear] [--no-restart] stop <app> App lifecycle (launch restarts by
|
|
1233
1519
|
default — force-stops first; --clear also wipes app data)
|
|
1234
1520
|
clear <app> Wipe app data — login/session, caches (fresh-install state)
|
|
1521
|
+
install <app.apk|.ipa> [--server url] Install a build on the device (adb install -r /
|
|
1522
|
+
idb install). With --server, uploads the file to a
|
|
1523
|
+
remote vk server (which must run --allow-install)
|
|
1235
1524
|
|
|
1236
1525
|
BATCH (script many commands in one process)
|
|
1237
1526
|
batch [--file path] [--quiet] Run newline-separated commands — from --file,
|
|
@@ -1249,11 +1538,37 @@ AI (run a natural-language test — compile once, replay model-free, self-heal)
|
|
|
1249
1538
|
path. The model is woken only to repair a step
|
|
1250
1539
|
that fails to resolve; a green run persists the
|
|
1251
1540
|
(repaired) plan so the next run is free. Needs
|
|
1252
|
-
ANTHROPIC_API_KEY
|
|
1253
|
-
|
|
1254
|
-
|
|
1541
|
+
ANTHROPIC_API_KEY (Claude) or OPENAI_API_KEY
|
|
1542
|
+
(gpt-5.x). Progress -> stderr; the report path ->
|
|
1543
|
+
stdout. --show-plan prints the compiled IR without
|
|
1544
|
+
running; --recompile ignores the cache.
|
|
1255
1545
|
Models: claude-haiku-4-5 | claude-sonnet-4-6
|
|
1256
|
-
(default) | claude-opus-4-8 | claude-fable-5
|
|
1546
|
+
(default) | claude-opus-4-8 | claude-fable-5 |
|
|
1547
|
+
gpt-5.4-mini | gpt-5.4 | gpt-5.5.
|
|
1548
|
+
|
|
1549
|
+
SUITE (run a directory of natural-language tests as one gated suite)
|
|
1550
|
+
suite <dir> [--app <id>] [--name n] [--json] (+ all \`ai\` flags, incl. --server)
|
|
1551
|
+
Run every *.md in <dir> (lexicographic order —
|
|
1552
|
+
prefix 01-, 02- to sequence; README.md skipped)
|
|
1553
|
+
through \`vk ai\`. With --app, app data is reset
|
|
1554
|
+
between tests (iOS: force-stop). Writes a suite
|
|
1555
|
+
overview to ./.verikun/suites/<id>/{index.json,
|
|
1556
|
+
index.html} linking each test's report. Exits 1
|
|
1557
|
+
if any test failed — the CI gate.
|
|
1558
|
+
|
|
1559
|
+
SERVER (expose a locally-connected device to remote verikun clients)
|
|
1560
|
+
server [--bind addr] [--port n] [--auth-key k] [--allow-install]
|
|
1561
|
+
[--allow-unsafe-anonymous] Serve THIS machine's device over HTTP+JSON for
|
|
1562
|
+
\`vk ai/suite/install --server <url>\`. Only
|
|
1563
|
+
verikun's validated action grammar is runnable
|
|
1564
|
+
(never a shell); auth is required (a key is
|
|
1565
|
+
generated if none given; --allow-unsafe-anonymous
|
|
1566
|
+
opts out for trusted networks e.g. Tailscale);
|
|
1567
|
+
binds 127.0.0.1 by default (--bind to expose);
|
|
1568
|
+
one run at a time holds the device lock.
|
|
1569
|
+
Env: VERIKUN_SERVER_AUTH_KEY (keeps it off argv).
|
|
1570
|
+
Clients: pass --server <url> (or VERIKUN_SERVER) + --auth-key (or
|
|
1571
|
+
VERIKUN_SERVER_AUTH_KEY) to ai/suite/install. The server's device+platform apply.
|
|
1257
1572
|
|
|
1258
1573
|
ENVIRONMENT
|
|
1259
1574
|
devices [--json] List attached devices/simulators
|
|
@@ -1294,5 +1609,7 @@ GLOBAL FLAGS
|
|
|
1294
1609
|
EXIT CODES
|
|
1295
1610
|
0 success · 1 not found / assertion failed / timeout · 2 usage or ambiguous selector · 3 environment error
|
|
1296
1611
|
|
|
1297
|
-
iOS:
|
|
1612
|
+
iOS (--ios): full parity via idb — ui/tap/text/swipe/key + screenshot/launch/stop.
|
|
1613
|
+
Needs idb (\`brew install idb-companion\` + \`pip install fb-idb\`); see \`vk doctor --ios\`.
|
|
1614
|
+
Caveats: no \`clear\` (no per-app reset), \`current\` is (unknown), device logs are simulator-only.`;
|
|
1298
1615
|
}
|