verikun 0.5.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +111 -4
- package/dist/agent/cost.js +21 -10
- package/dist/agent/engine.js +8 -8
- package/dist/agent/openai.js +243 -0
- package/dist/agent/remote.js +166 -0
- package/dist/args.js +2 -0
- package/dist/bin/verikun.js +0 -0
- package/dist/cli.js +381 -98
- package/dist/drivers/adb.js +12 -0
- package/dist/drivers/ios.js +7 -0
- package/dist/drivers/simctl.js +156 -0
- package/dist/report.js +93 -3
- package/dist/rpc.js +46 -0
- package/dist/run.js +101 -1
- package/dist/server.js +349 -0
- package/dist/suite.js +152 -0
- package/dist/version.js +1 -1
- package/package.json +1 -1
package/dist/cli.js
CHANGED
|
@@ -1,5 +1,40 @@
|
|
|
1
1
|
"use strict";
|
|
2
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
3
|
+
if (k2 === undefined) k2 = k;
|
|
4
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
5
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
6
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
7
|
+
}
|
|
8
|
+
Object.defineProperty(o, k2, desc);
|
|
9
|
+
}) : (function(o, m, k, k2) {
|
|
10
|
+
if (k2 === undefined) k2 = k;
|
|
11
|
+
o[k2] = m[k];
|
|
12
|
+
}));
|
|
13
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
14
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
15
|
+
}) : function(o, v) {
|
|
16
|
+
o["default"] = v;
|
|
17
|
+
});
|
|
18
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
19
|
+
var ownKeys = function(o) {
|
|
20
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
21
|
+
var ar = [];
|
|
22
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
23
|
+
return ar;
|
|
24
|
+
};
|
|
25
|
+
return ownKeys(o);
|
|
26
|
+
};
|
|
27
|
+
return function (mod) {
|
|
28
|
+
if (mod && mod.__esModule) return mod;
|
|
29
|
+
var result = {};
|
|
30
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
31
|
+
__setModuleDefault(result, mod);
|
|
32
|
+
return result;
|
|
33
|
+
};
|
|
34
|
+
})();
|
|
2
35
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
36
|
+
exports.platformFromFlags = platformFromFlags;
|
|
37
|
+
exports.deviceFromFlags = deviceFromFlags;
|
|
3
38
|
exports.parsePoint = parsePoint;
|
|
4
39
|
exports.healNote = healNote;
|
|
5
40
|
exports.parseDuration = parseDuration;
|
|
@@ -11,6 +46,7 @@ exports.chooseLogOpts = chooseLogOpts;
|
|
|
11
46
|
exports.evalAssert = evalAssert;
|
|
12
47
|
exports.tokenizeLine = tokenizeLine;
|
|
13
48
|
exports.withBatchGlobals = withBatchGlobals;
|
|
49
|
+
exports.executeForServer = executeForServer;
|
|
14
50
|
exports.run = run;
|
|
15
51
|
const node_fs_1 = require("node:fs");
|
|
16
52
|
const node_path_1 = require("node:path");
|
|
@@ -25,10 +61,14 @@ const run_1 = require("./run");
|
|
|
25
61
|
const image_1 = require("./image");
|
|
26
62
|
const engine_1 = require("./agent/engine");
|
|
27
63
|
const claude_1 = require("./agent/claude");
|
|
64
|
+
const openai_1 = require("./agent/openai");
|
|
28
65
|
const cache_1 = require("./agent/cache");
|
|
29
66
|
const cost_1 = require("./agent/cost");
|
|
67
|
+
const remote_1 = require("./agent/remote");
|
|
68
|
+
const suite_1 = require("./suite");
|
|
30
69
|
const version_1 = require("./version");
|
|
31
70
|
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
|
71
|
+
// Exported for src/server.ts (which resolves its own platform/device at startup).
|
|
32
72
|
function platformFromFlags(flags) {
|
|
33
73
|
if ((0, args_1.flagBool)(flags, 'ios'))
|
|
34
74
|
return 'ios';
|
|
@@ -888,99 +928,157 @@ async function cmdBatch(positionals, batchFlags) {
|
|
|
888
928
|
(0, output_1.err)(`[verikun] batch: ${commands.length} command(s) ok`);
|
|
889
929
|
return 0;
|
|
890
930
|
}
|
|
891
|
-
|
|
892
|
-
// ai — compile a natural-language test to a plan IR, then run it (self-healing)
|
|
893
|
-
// ---------------------------------------------------------------------------
|
|
894
|
-
//
|
|
895
|
-
// `vk ai <file>` reads a plain-English test, compiles it ONCE into a deterministic
|
|
896
|
-
// plan IR via the model (cached by NL + app build), then replays it with NO model
|
|
897
|
-
// calls on the happy path. The model is woken only to repair a step that fails to
|
|
898
|
-
// resolve its selector; a green run persists the (possibly repaired) plan so the
|
|
899
|
-
// next run is free again. Cost is bounded by --max-cost-usd. Progress streams to
|
|
900
|
-
// stderr (CI liveness — it never goes quiet); stdout carries the final result.
|
|
901
|
-
async function cmdAi(positionals, flags) {
|
|
902
|
-
const file = positionals[0];
|
|
903
|
-
if (!file) {
|
|
904
|
-
throw new errors_1.CliError('Usage: verikun ai <file> [--model m] [--max-cost-usd n] [--timeout dur] [--show-plan] [--recompile]', 2);
|
|
905
|
-
}
|
|
906
|
-
let nl;
|
|
907
|
-
try {
|
|
908
|
-
nl = (0, node_fs_1.readFileSync)((0, node_path_1.resolve)(process.cwd(), file), 'utf8');
|
|
909
|
-
}
|
|
910
|
-
catch (e) {
|
|
911
|
-
throw new errors_1.CliError(`ai: cannot read '${file}' (${e.message})`, 2);
|
|
912
|
-
}
|
|
913
|
-
if (!nl.trim())
|
|
914
|
-
throw new errors_1.CliError(`ai: '${file}' is empty`, 2);
|
|
915
|
-
const platform = platformFromFlags(flags);
|
|
916
|
-
const device = deviceFromFlags(flags, platform);
|
|
931
|
+
function parseAiOptions(flags) {
|
|
917
932
|
const model = (0, cost_1.resolveModel)((0, args_1.flagStr)(flags, 'model'));
|
|
918
933
|
const overrideRaw = (0, args_1.flagStr)(flags, 'cost-override');
|
|
919
934
|
const override = overrideRaw ? (0, cost_1.parseCostOverride)(overrideRaw) : undefined;
|
|
920
935
|
const maxCostUsd = (0, args_1.flagNum)(flags, 'max-cost-usd') ?? cost_1.DEFAULT_MAX_COST_USD;
|
|
921
936
|
if (maxCostUsd <= 0)
|
|
922
937
|
throw new errors_1.CliError(`--max-cost-usd must be greater than 0 (got ${maxCostUsd}).`, 2);
|
|
923
|
-
const cost = new cost_1.CostTracker((0, cost_1.priceFor)(model, override), maxCostUsd);
|
|
924
938
|
// Whole-run wall-clock ceiling (default 15m) so a runaway loop/repair can't hang the run.
|
|
925
939
|
const timeoutFlag = (0, args_1.flagStr)(flags, 'timeout');
|
|
926
940
|
const timeoutMs = timeoutFlag ? parseDuration(timeoutFlag, 'timeout') : engine_1.DEFAULT_RUN_TIMEOUT_MS;
|
|
927
|
-
|
|
928
|
-
|
|
929
|
-
|
|
930
|
-
|
|
931
|
-
|
|
932
|
-
|
|
933
|
-
|
|
934
|
-
|
|
935
|
-
|
|
936
|
-
|
|
937
|
-
|
|
938
|
-
|
|
939
|
-
let
|
|
941
|
+
return {
|
|
942
|
+
model,
|
|
943
|
+
price: (0, cost_1.priceFor)(model, override),
|
|
944
|
+
maxCostUsd,
|
|
945
|
+
timeoutMs,
|
|
946
|
+
effort: (0, args_1.flagStr)(flags, 'effort'),
|
|
947
|
+
pkg: (0, args_1.flagStr)(flags, 'package'),
|
|
948
|
+
build: (0, args_1.flagStr)(flags, 'app-build'),
|
|
949
|
+
recompile: (0, args_1.flagBool)(flags, 'recompile') || (0, args_1.flagBool)(flags, 'no-cache'),
|
|
950
|
+
};
|
|
951
|
+
}
|
|
952
|
+
function readAiTest(file) {
|
|
953
|
+
let nl;
|
|
954
|
+
try {
|
|
955
|
+
nl = (0, node_fs_1.readFileSync)((0, node_path_1.resolve)(process.cwd(), file), 'utf8');
|
|
956
|
+
}
|
|
957
|
+
catch (e) {
|
|
958
|
+
throw new errors_1.CliError(`ai: cannot read '${file}' (${e.message})`, 2);
|
|
959
|
+
}
|
|
960
|
+
if (!nl.trim())
|
|
961
|
+
throw new errors_1.CliError(`ai: '${file}' is empty`, 2);
|
|
962
|
+
return nl;
|
|
963
|
+
}
|
|
964
|
+
/** The env var carrying the API key for a model's provider (per-provider keys). */
|
|
965
|
+
function keyEnvFor(model) {
|
|
966
|
+
return (0, cost_1.providerFor)(model) === 'openai' ? 'OPENAI_API_KEY' : 'ANTHROPIC_API_KEY';
|
|
967
|
+
}
|
|
968
|
+
/** Route the model to its backend; each provider reads its own key. A missing key
|
|
969
|
+
* means no provider (compile/repair unavailable) — the same graceful degradation
|
|
970
|
+
* as before. */
|
|
971
|
+
function makeProvider(opts) {
|
|
972
|
+
const apiKey = process.env[keyEnvFor(opts.model)];
|
|
973
|
+
if (!apiKey)
|
|
974
|
+
return null;
|
|
975
|
+
return (0, cost_1.providerFor)(opts.model) === 'openai'
|
|
976
|
+
? new openai_1.OpenAiProvider({ model: opts.model, apiKey, effort: opts.effort })
|
|
977
|
+
: new claude_1.ClaudeProvider({ model: opts.model, apiKey, effort: opts.effort });
|
|
978
|
+
}
|
|
979
|
+
/** Obtain the plan: a cache hit (free) or a compile (pays tokens; may seed from a
|
|
980
|
+
* prior build's plan to avoid a full recompile). The fresh compile is cached right
|
|
981
|
+
* away, so an unchanged test is never recompiled — even via --show-plan or after a
|
|
982
|
+
* failed run. A green run later re-persists the healed plan (never a half-healed one). */
|
|
983
|
+
async function obtainPlan(key, file, opts, cost, provider) {
|
|
984
|
+
const cached = opts.recompile ? null : (0, cache_1.readPlan)(key);
|
|
940
985
|
if (cached) {
|
|
941
|
-
plan
|
|
942
|
-
|
|
986
|
+
(0, output_1.err)(`[ai] plan cache hit — ${opts.model} not called to compile`);
|
|
987
|
+
return { plan: cached.plan, cached: true };
|
|
943
988
|
}
|
|
944
|
-
|
|
945
|
-
|
|
946
|
-
|
|
947
|
-
|
|
948
|
-
|
|
949
|
-
|
|
950
|
-
|
|
951
|
-
|
|
952
|
-
|
|
953
|
-
|
|
954
|
-
|
|
955
|
-
(0,
|
|
956
|
-
// Cache the freshly-compiled plan right away, keyed by the test-text hash, so an
|
|
957
|
-
// unchanged test is never recompiled — even via --show-plan or after a failed run.
|
|
958
|
-
// A green run below re-persists the healed plan; a failed run leaves this clean
|
|
959
|
-
// compile cached (never a half-healed one).
|
|
960
|
-
try {
|
|
961
|
-
(0, cache_1.writePlan)(key, plan);
|
|
962
|
-
}
|
|
963
|
-
catch (e) {
|
|
964
|
-
(0, output_1.err)(`[ai] could not cache compiled plan: ${e.message}`);
|
|
965
|
-
}
|
|
989
|
+
if (!provider) {
|
|
990
|
+
throw new errors_1.CliError(`${keyEnvFor(opts.model)} is not set — \`vk ai\` needs it to compile the test (model ${opts.model}). Set it and retry.`, 3);
|
|
991
|
+
}
|
|
992
|
+
const seed = (0, cache_1.findSeed)(key);
|
|
993
|
+
if (seed)
|
|
994
|
+
(0, output_1.err)(`[ai] no exact cache; seeding from a prior plan (build ${seed.build ?? 'unknown'})`);
|
|
995
|
+
(0, output_1.err)(`[ai] compiling '${file}' with ${opts.model} (effort ${opts.effort ?? 'default'})…`);
|
|
996
|
+
const compiled = await provider.compile({ nl: key.nl, pkg: key.pkg, platform: key.platform, seed: seed?.plan });
|
|
997
|
+
cost.add(compiled.usage, 'compile');
|
|
998
|
+
(0, output_1.err)(`[ai] compiled ${compiled.plan.steps.length} top-level step(s) · ${cost.summaryLine()}`);
|
|
999
|
+
try {
|
|
1000
|
+
(0, cache_1.writePlan)(key, compiled.plan);
|
|
966
1001
|
}
|
|
967
|
-
|
|
968
|
-
|
|
969
|
-
(0, output_1.json)(plan);
|
|
970
|
-
return 0;
|
|
1002
|
+
catch (e) {
|
|
1003
|
+
(0, output_1.err)(`[ai] could not cache compiled plan: ${e.message}`);
|
|
971
1004
|
}
|
|
1005
|
+
return { plan: compiled.plan, cached: false };
|
|
1006
|
+
}
|
|
1007
|
+
async function resolveBackend(platform, device, flags) {
|
|
1008
|
+
const server = (0, args_1.flagStr)(flags, 'server') || process.env.VERIKUN_SERVER || undefined;
|
|
1009
|
+
if (!server) {
|
|
1010
|
+
const driver = (0, drivers_1.getDriver)(platform, device);
|
|
1011
|
+
return {
|
|
1012
|
+
backend: {
|
|
1013
|
+
exec: (command, positionals, f) => executeOutcome(command, positionals, f, driver),
|
|
1014
|
+
getElements: () => driver.getElements(),
|
|
1015
|
+
install: (appPath) => driver.install(appPath),
|
|
1016
|
+
reset: (appId) => {
|
|
1017
|
+
assertSafeAppId(appId);
|
|
1018
|
+
// iOS has no per-app data reset — degrade honestly to a force-stop.
|
|
1019
|
+
if (platform === 'ios')
|
|
1020
|
+
driver.stop(appId);
|
|
1021
|
+
else
|
|
1022
|
+
driver.clearApp(appId);
|
|
1023
|
+
},
|
|
1024
|
+
},
|
|
1025
|
+
platform,
|
|
1026
|
+
device,
|
|
1027
|
+
};
|
|
1028
|
+
}
|
|
1029
|
+
let runCtx = { platform, device };
|
|
1030
|
+
const opts = {
|
|
1031
|
+
url: server,
|
|
1032
|
+
authKey: (0, args_1.flagStr)(flags, 'auth-key') || process.env.VERIKUN_SERVER_AUTH_KEY || undefined,
|
|
1033
|
+
// Each remote step is spliced into the local active run so the archived report
|
|
1034
|
+
// is identical to a local run's.
|
|
1035
|
+
onStep: (step, artifacts) => run_1.Recorder.appendForeignStep(step, artifacts, runCtx),
|
|
1036
|
+
};
|
|
1037
|
+
const health = await (0, remote_1.pingServer)(opts); // fails fast (exit 3) on a bad URL or key
|
|
1038
|
+
runCtx = { platform: health.platform, device: health.serial };
|
|
1039
|
+
(0, output_1.err)(`[verikun] server ${server}: ${health.platform} · device ${health.serial} · verikun ${health.version}`);
|
|
1040
|
+
return {
|
|
1041
|
+
backend: (0, remote_1.createRemoteBackend)(opts, health),
|
|
1042
|
+
platform: health.platform,
|
|
1043
|
+
device: health.serial,
|
|
1044
|
+
remote: { url: server, version: health.version },
|
|
1045
|
+
};
|
|
1046
|
+
}
|
|
1047
|
+
/**
|
|
1048
|
+
* Run one natural-language test through a backend and return DATA — no stdout
|
|
1049
|
+
* writes (stdout stays the caller's one result; progress streams to stderr).
|
|
1050
|
+
* `vk ai` wraps it with its --json/report output; `vk suite` calls it per test.
|
|
1051
|
+
*/
|
|
1052
|
+
async function runAiTest(file, opts, backend, platform, device) {
|
|
1053
|
+
const nl = readAiTest(file);
|
|
1054
|
+
const key = { nl, pkg: opts.pkg, build: opts.build, platform };
|
|
1055
|
+
const cost = new cost_1.CostTracker(opts.price, opts.maxCostUsd);
|
|
1056
|
+
const deadline = Date.now() + opts.timeoutMs;
|
|
1057
|
+
const provider = makeProvider(opts);
|
|
1058
|
+
const { plan, cached } = await obtainPlan(key, file, opts, cost, provider);
|
|
972
1059
|
// Running needs the provider for repair-on-failure; a cache hit with no key can't repair.
|
|
973
1060
|
if (!provider) {
|
|
974
|
-
throw new errors_1.CliError(
|
|
1061
|
+
throw new errors_1.CliError(`${keyEnvFor(opts.model)} is not set — \`vk ai\` needs it to repair a failing step at runtime (model ${opts.model}).`, 3);
|
|
975
1062
|
}
|
|
976
1063
|
// The budget is a TOTAL-run ceiling: if the compile alone already crossed it, abort
|
|
977
1064
|
// before running. A cache hit spends nothing, so a free replay is still allowed.
|
|
978
1065
|
if (!cached && cost.exceeded()) {
|
|
979
|
-
(0, output_1.err)(`[ai] cost ceiling $${maxCostUsd} reached during compile (${cost.summaryLine()}) — not running`);
|
|
980
|
-
return
|
|
981
|
-
|
|
982
|
-
|
|
983
|
-
|
|
1066
|
+
(0, output_1.err)(`[ai] cost ceiling $${opts.maxCostUsd} reached during compile (${cost.summaryLine()}) — not running`);
|
|
1067
|
+
return {
|
|
1068
|
+
ok: false,
|
|
1069
|
+
costUsd: Number(cost.usd().toFixed(4)),
|
|
1070
|
+
costLine: cost.summaryLine(),
|
|
1071
|
+
modelRepairs: 0,
|
|
1072
|
+
improvements: [],
|
|
1073
|
+
runDir: '',
|
|
1074
|
+
reportHtml: '',
|
|
1075
|
+
junitXml: '',
|
|
1076
|
+
state: null,
|
|
1077
|
+
abortedForBudget: true,
|
|
1078
|
+
failure: { where: 'compile', reason: `cost ceiling $${opts.maxCostUsd} reached during compile` },
|
|
1079
|
+
};
|
|
1080
|
+
}
|
|
1081
|
+
// One explicit run for the whole flow (so rollover can't split the test).
|
|
984
1082
|
const existing = run_1.Recorder.status();
|
|
985
1083
|
if (existing && existing.steps.length > 0) {
|
|
986
1084
|
// Seal the pre-existing run into the archive instead of letting start(force=true)
|
|
@@ -989,14 +1087,13 @@ async function cmdAi(positionals, flags) {
|
|
|
989
1087
|
(0, output_1.err)(`[ai] archived the active run ('${existing.name}', ${existing.steps.length} step(s)) → ${sealed.dir}`);
|
|
990
1088
|
}
|
|
991
1089
|
run_1.Recorder.start(`ai: ${(0, node_path_1.basename)(file)}`, platform, device, true);
|
|
992
|
-
const driver = (0, drivers_1.getDriver)(platform, device);
|
|
993
1090
|
// Suppress per-step `out()` so stdout stays the one final result; progress -> stderr.
|
|
994
1091
|
const prevQuiet = (0, output_1.setOutputQuiet)(true);
|
|
995
1092
|
let result;
|
|
996
1093
|
try {
|
|
997
1094
|
result = await (0, engine_1.runPlan)(plan, {
|
|
998
|
-
exec: (command, pos, f) =>
|
|
999
|
-
getElements: () =>
|
|
1095
|
+
exec: (command, pos, f) => backend.exec(command, pos, f),
|
|
1096
|
+
getElements: () => backend.getElements(),
|
|
1000
1097
|
provider,
|
|
1001
1098
|
cost,
|
|
1002
1099
|
log: (m) => (0, output_1.err)(m),
|
|
@@ -1023,8 +1120,8 @@ async function cmdAi(positionals, flags) {
|
|
|
1023
1120
|
finally {
|
|
1024
1121
|
(0, output_1.setOutputQuiet)(prevQuiet);
|
|
1025
1122
|
}
|
|
1026
|
-
//
|
|
1027
|
-
//
|
|
1123
|
+
// Persist the (possibly repaired) plan only on a fully-green run; attach the
|
|
1124
|
+
// cost + improvements summary to the run; archive into the report.
|
|
1028
1125
|
const costLine = cost.summaryLine();
|
|
1029
1126
|
if (result.ok) {
|
|
1030
1127
|
try {
|
|
@@ -1038,13 +1135,13 @@ async function cmdAi(positionals, flags) {
|
|
|
1038
1135
|
run_1.Recorder.annotateRun({
|
|
1039
1136
|
ai: { ok: result.ok, cost: costLine, modelRepairs: result.modelRepairs, improvements: result.improvements },
|
|
1040
1137
|
});
|
|
1041
|
-
const { dir, xmlPath, htmlPath } = run_1.Recorder.archive();
|
|
1138
|
+
const { dir, xmlPath, htmlPath, state } = run_1.Recorder.archive();
|
|
1042
1139
|
const status = result.ok
|
|
1043
1140
|
? 'PASS'
|
|
1044
1141
|
: result.abortedForBudget
|
|
1045
|
-
? `ABORTED — cost ceiling $${maxCostUsd} reached`
|
|
1142
|
+
? `ABORTED — cost ceiling $${opts.maxCostUsd} reached`
|
|
1046
1143
|
: result.abortedForTimeout
|
|
1047
|
-
? `ABORTED — run timeout (${Math.round(timeoutMs / 1000)}s) reached`
|
|
1144
|
+
? `ABORTED — run timeout (${Math.round(opts.timeoutMs / 1000)}s) reached`
|
|
1048
1145
|
: `FAIL at ${result.failure?.where}: ${result.failure?.reason}`;
|
|
1049
1146
|
(0, output_1.err)(`[ai] ${status} · ${costLine}`);
|
|
1050
1147
|
(0, output_1.err)(`[ai] report: ${htmlPath}`);
|
|
@@ -1054,28 +1151,124 @@ async function cmdAi(positionals, flags) {
|
|
|
1054
1151
|
(0, output_1.err)(' - ' + imp);
|
|
1055
1152
|
}
|
|
1056
1153
|
(0, output_1.err)(`[ai] estimated total cost: $${cost.usd().toFixed(4)}`);
|
|
1154
|
+
return {
|
|
1155
|
+
ok: result.ok,
|
|
1156
|
+
costUsd: Number(cost.usd().toFixed(4)),
|
|
1157
|
+
costLine,
|
|
1158
|
+
modelRepairs: result.modelRepairs,
|
|
1159
|
+
improvements: result.improvements,
|
|
1160
|
+
runDir: dir,
|
|
1161
|
+
reportHtml: htmlPath,
|
|
1162
|
+
junitXml: xmlPath,
|
|
1163
|
+
state,
|
|
1164
|
+
...(result.failure ? { failure: result.failure } : {}),
|
|
1165
|
+
...(result.abortedForBudget ? { abortedForBudget: true } : {}),
|
|
1166
|
+
...(result.abortedForTimeout ? { abortedForTimeout: true } : {}),
|
|
1167
|
+
};
|
|
1168
|
+
}
|
|
1169
|
+
async function cmdAi(positionals, flags) {
|
|
1170
|
+
const file = positionals[0];
|
|
1171
|
+
if (!file) {
|
|
1172
|
+
throw new errors_1.CliError('Usage: verikun ai <file> [--model m] [--max-cost-usd n] [--timeout dur] [--server url] [--show-plan] [--recompile]', 2);
|
|
1173
|
+
}
|
|
1174
|
+
const opts = parseAiOptions(flags);
|
|
1175
|
+
// --show-plan: compile (or cache-hit) and print the IR — no device, no backend.
|
|
1176
|
+
if ((0, args_1.flagBool)(flags, 'show-plan')) {
|
|
1177
|
+
const nl = readAiTest(file);
|
|
1178
|
+
const key = { nl, pkg: opts.pkg, build: opts.build, platform: platformFromFlags(flags) };
|
|
1179
|
+
const cost = new cost_1.CostTracker(opts.price, opts.maxCostUsd);
|
|
1180
|
+
const { plan } = await obtainPlan(key, file, opts, cost, makeProvider(opts));
|
|
1181
|
+
(0, output_1.json)(plan);
|
|
1182
|
+
return 0;
|
|
1183
|
+
}
|
|
1184
|
+
const reqPlatform = platformFromFlags(flags);
|
|
1185
|
+
const { backend, platform, device } = await resolveBackend(reqPlatform, deviceFromFlags(flags, reqPlatform), flags);
|
|
1186
|
+
let result;
|
|
1187
|
+
try {
|
|
1188
|
+
result = await runAiTest(file, opts, backend, platform, device);
|
|
1189
|
+
}
|
|
1190
|
+
finally {
|
|
1191
|
+
await backend.close?.(); // frees a remote server's device lock for the next command
|
|
1192
|
+
}
|
|
1057
1193
|
if ((0, args_1.flagBool)(flags, 'json')) {
|
|
1058
1194
|
(0, output_1.json)({
|
|
1059
1195
|
ok: result.ok,
|
|
1060
|
-
model,
|
|
1061
|
-
cost: costLine,
|
|
1062
|
-
costUsd:
|
|
1196
|
+
model: opts.model,
|
|
1197
|
+
cost: result.costLine,
|
|
1198
|
+
costUsd: result.costUsd,
|
|
1063
1199
|
modelRepairs: result.modelRepairs,
|
|
1064
1200
|
improvements: result.improvements,
|
|
1065
|
-
report:
|
|
1066
|
-
junit:
|
|
1067
|
-
runDir:
|
|
1201
|
+
report: result.reportHtml,
|
|
1202
|
+
junit: result.junitXml,
|
|
1203
|
+
runDir: result.runDir,
|
|
1068
1204
|
...(result.failure ? { failure: result.failure } : {}),
|
|
1069
1205
|
...(result.abortedForBudget ? { abortedForBudget: true } : {}),
|
|
1070
1206
|
...(result.abortedForTimeout ? { abortedForTimeout: true } : {}),
|
|
1071
1207
|
});
|
|
1072
1208
|
}
|
|
1073
|
-
else {
|
|
1074
|
-
(0, output_1.out)(
|
|
1209
|
+
else if (result.reportHtml) {
|
|
1210
|
+
(0, output_1.out)(result.reportHtml); // primary machine result: the report path
|
|
1075
1211
|
}
|
|
1076
1212
|
return result.ok ? 0 : 1;
|
|
1077
1213
|
}
|
|
1078
1214
|
// ---------------------------------------------------------------------------
|
|
1215
|
+
// install — put an app build on the device (local driver or remote vk server)
|
|
1216
|
+
// ---------------------------------------------------------------------------
|
|
1217
|
+
async function cmdInstall(positionals, flags) {
|
|
1218
|
+
const appPath = positionals[0];
|
|
1219
|
+
if (!appPath)
|
|
1220
|
+
throw new errors_1.CliError('Usage: verikun install <app.apk|app.ipa> [--server url]', 2);
|
|
1221
|
+
const path = (0, node_path_1.resolve)(process.cwd(), appPath);
|
|
1222
|
+
if (!(0, node_fs_1.existsSync)(path))
|
|
1223
|
+
throw new errors_1.CliError(`install: '${appPath}' does not exist`, 2);
|
|
1224
|
+
const platform = platformFromFlags(flags);
|
|
1225
|
+
const { backend, remote } = await resolveBackend(platform, deviceFromFlags(flags, platform), flags);
|
|
1226
|
+
(0, output_1.err)(`[verikun] installing ${appPath}${remote ? ` via ${remote.url}` : ''}…`);
|
|
1227
|
+
try {
|
|
1228
|
+
await backend.install(path);
|
|
1229
|
+
}
|
|
1230
|
+
finally {
|
|
1231
|
+
await backend.close?.();
|
|
1232
|
+
}
|
|
1233
|
+
if ((0, args_1.flagBool)(flags, 'json'))
|
|
1234
|
+
(0, output_1.json)({ installed: appPath, ...(remote ? { server: remote.url } : {}) });
|
|
1235
|
+
else
|
|
1236
|
+
(0, output_1.out)(`installed ${appPath}`);
|
|
1237
|
+
return 0;
|
|
1238
|
+
}
|
|
1239
|
+
// ---------------------------------------------------------------------------
|
|
1240
|
+
// suite — run a directory of natural-language tests as one gated suite
|
|
1241
|
+
// ---------------------------------------------------------------------------
|
|
1242
|
+
async function cmdSuiteEntry(positionals, flags) {
|
|
1243
|
+
const dirArg = positionals[0];
|
|
1244
|
+
if (!dirArg)
|
|
1245
|
+
throw new errors_1.CliError('Usage: verikun suite <dir> [--app <id>] [--server url] [--name n] [--json]', 2);
|
|
1246
|
+
const opts = parseAiOptions(flags);
|
|
1247
|
+
// Pre-flight the model key BEFORE touching any device/server: every test needs it
|
|
1248
|
+
// to compile (on a cache miss) or to repair at runtime.
|
|
1249
|
+
if (!process.env[keyEnvFor(opts.model)]) {
|
|
1250
|
+
throw new errors_1.CliError(`${keyEnvFor(opts.model)} is not set — \`vk suite\` needs it to compile/repair tests (model ${opts.model}).`, 3);
|
|
1251
|
+
}
|
|
1252
|
+
const reqPlatform = platformFromFlags(flags);
|
|
1253
|
+
const { backend, platform, device } = await resolveBackend(reqPlatform, deviceFromFlags(flags, reqPlatform), flags);
|
|
1254
|
+
const app = (0, args_1.flagStr)(flags, 'app');
|
|
1255
|
+
if (app)
|
|
1256
|
+
assertSafeAppId(app);
|
|
1257
|
+
try {
|
|
1258
|
+
return await (0, suite_1.cmdSuite)(dirArg, flags, {
|
|
1259
|
+
platform,
|
|
1260
|
+
device,
|
|
1261
|
+
runTest: (file) => runAiTest(file, opts, backend, platform, device),
|
|
1262
|
+
// Reset app state between tests only when the app id is known; without --app,
|
|
1263
|
+
// each test is responsible for its own isolation (e.g. `launch --clear`).
|
|
1264
|
+
reset: app ? () => backend.reset(app) : undefined,
|
|
1265
|
+
});
|
|
1266
|
+
}
|
|
1267
|
+
finally {
|
|
1268
|
+
await backend.close?.();
|
|
1269
|
+
}
|
|
1270
|
+
}
|
|
1271
|
+
// ---------------------------------------------------------------------------
|
|
1079
1272
|
// Dispatch
|
|
1080
1273
|
// ---------------------------------------------------------------------------
|
|
1081
1274
|
async function executeCommand(command, ctx) {
|
|
@@ -1181,7 +1374,19 @@ async function executeOutcome(command, positionals, flags, sharedDriver) {
|
|
|
1181
1374
|
}
|
|
1182
1375
|
recorder = run_1.Recorder.beginStep(command, positionals, flags, platform, device, serial, driver);
|
|
1183
1376
|
}
|
|
1184
|
-
|
|
1377
|
+
}
|
|
1378
|
+
catch (e) {
|
|
1379
|
+
return { code: e instanceof errors_1.CliError ? e.exitCode : 3, error: e };
|
|
1380
|
+
}
|
|
1381
|
+
const d = driver; // assigned in the try above, or we already returned
|
|
1382
|
+
const ctx = { driver: d, platform, device, positionals, flags, record: recorder ?? undefined };
|
|
1383
|
+
return runRecorded(command, ctx, recorder, d);
|
|
1384
|
+
}
|
|
1385
|
+
/** The shared middle of executeOutcome / executeForServer: run the handler, close
|
|
1386
|
+
* the step (with failure evidence) whether it returned or threw, map to a raw
|
|
1387
|
+
* outcome (the error NOT printed — callers decide). */
|
|
1388
|
+
async function runRecorded(command, ctx, recorder, driver) {
|
|
1389
|
+
try {
|
|
1185
1390
|
const code = await executeCommand(command, ctx);
|
|
1186
1391
|
recorder?.finish(code, driver);
|
|
1187
1392
|
return { code };
|
|
@@ -1191,6 +1396,28 @@ async function executeOutcome(command, positionals, flags, sharedDriver) {
|
|
|
1191
1396
|
return { code: e instanceof errors_1.CliError ? e.exitCode : 3, error: e };
|
|
1192
1397
|
}
|
|
1193
1398
|
}
|
|
1399
|
+
/**
|
|
1400
|
+
* `vk server`'s per-request executor: run one already-validated leaf against the
|
|
1401
|
+
* server's fixed driver/platform (a client can never repoint the device via flags),
|
|
1402
|
+
* recording into an EPHEMERAL single-step recorder instead of the local run store.
|
|
1403
|
+
* Returns the raw outcome plus the finished step + artifact buffers, which travel
|
|
1404
|
+
* back over the wire and are spliced into the CALLER's run — so `resolved`/`tier`/
|
|
1405
|
+
* failure evidence survive remoting with zero handler changes.
|
|
1406
|
+
*/
|
|
1407
|
+
async function executeForServer(command, positionals, flags, driver, platform) {
|
|
1408
|
+
let serial;
|
|
1409
|
+
try {
|
|
1410
|
+
serial = driver.resolvedSerial();
|
|
1411
|
+
}
|
|
1412
|
+
catch {
|
|
1413
|
+
/* surfaced by the command handler below */
|
|
1414
|
+
}
|
|
1415
|
+
const recorder = run_1.Recorder.beginEphemeralStep(command, positionals, flags, platform, serial);
|
|
1416
|
+
const ctx = { driver, platform, device: serial, positionals, flags, record: recorder };
|
|
1417
|
+
const outcome = await runRecorded(command, ctx, recorder, driver);
|
|
1418
|
+
const { step, artifacts } = recorder.takeEphemeral();
|
|
1419
|
+
return { ...outcome, step, artifacts };
|
|
1420
|
+
}
|
|
1194
1421
|
/**
|
|
1195
1422
|
* Run one already-parsed command for the CLI / `batch`: dispatch meta-commands,
|
|
1196
1423
|
* else execute it and map any failure to a printed exit code. This is the shared
|
|
@@ -1206,9 +1433,9 @@ async function executeParsed(command, positionals, flags) {
|
|
|
1206
1433
|
return cmdRun(positionals, flags, platform, device);
|
|
1207
1434
|
if (command === 'batch')
|
|
1208
1435
|
return cmdBatch(positionals, flags);
|
|
1209
|
-
// `ai`
|
|
1210
|
-
// (usage 2 / env 3 / …) so they honor the exit-code
|
|
1211
|
-
// to the top-level "Fatal" handler (
|
|
1436
|
+
// `ai`/`suite`/`install`/`server` orchestrate their own steps; map their thrown
|
|
1437
|
+
// CliErrors to exit codes here (usage 2 / env 3 / …) so they honor the exit-code
|
|
1438
|
+
// contract instead of escaping to the top-level "Fatal" handler (exit 3).
|
|
1212
1439
|
if (command === 'ai') {
|
|
1213
1440
|
try {
|
|
1214
1441
|
return await cmdAi(positionals, flags);
|
|
@@ -1217,6 +1444,33 @@ async function executeParsed(command, positionals, flags) {
|
|
|
1217
1444
|
return mapError(e, flags);
|
|
1218
1445
|
}
|
|
1219
1446
|
}
|
|
1447
|
+
if (command === 'suite') {
|
|
1448
|
+
try {
|
|
1449
|
+
return await cmdSuiteEntry(positionals, flags);
|
|
1450
|
+
}
|
|
1451
|
+
catch (e) {
|
|
1452
|
+
return mapError(e, flags);
|
|
1453
|
+
}
|
|
1454
|
+
}
|
|
1455
|
+
if (command === 'install') {
|
|
1456
|
+
try {
|
|
1457
|
+
return await cmdInstall(positionals, flags);
|
|
1458
|
+
}
|
|
1459
|
+
catch (e) {
|
|
1460
|
+
return mapError(e, flags);
|
|
1461
|
+
}
|
|
1462
|
+
}
|
|
1463
|
+
if (command === 'server') {
|
|
1464
|
+
// Dynamic import: keeps node:http off the default load path and avoids a
|
|
1465
|
+
// static cli↔server cycle (server.ts imports executeForServer from here).
|
|
1466
|
+
try {
|
|
1467
|
+
const { cmdServer } = await Promise.resolve().then(() => __importStar(require('./server')));
|
|
1468
|
+
return await cmdServer(positionals, flags);
|
|
1469
|
+
}
|
|
1470
|
+
catch (e) {
|
|
1471
|
+
return mapError(e, flags);
|
|
1472
|
+
}
|
|
1473
|
+
}
|
|
1220
1474
|
const { code, error } = await executeOutcome(command, positionals, flags);
|
|
1221
1475
|
return error ? mapError(error, flags) : code;
|
|
1222
1476
|
}
|
|
@@ -1264,6 +1518,9 @@ ACT
|
|
|
1264
1518
|
launch <app> [--clear] [--no-restart] stop <app> App lifecycle (launch restarts by
|
|
1265
1519
|
default — force-stops first; --clear also wipes app data)
|
|
1266
1520
|
clear <app> Wipe app data — login/session, caches (fresh-install state)
|
|
1521
|
+
install <app.apk|.ipa> [--server url] Install a build on the device (adb install -r /
|
|
1522
|
+
idb install). With --server, uploads the file to a
|
|
1523
|
+
remote vk server (which must run --allow-install)
|
|
1267
1524
|
|
|
1268
1525
|
BATCH (script many commands in one process)
|
|
1269
1526
|
batch [--file path] [--quiet] Run newline-separated commands — from --file,
|
|
@@ -1281,11 +1538,37 @@ AI (run a natural-language test — compile once, replay model-free, self-heal)
|
|
|
1281
1538
|
path. The model is woken only to repair a step
|
|
1282
1539
|
that fails to resolve; a green run persists the
|
|
1283
1540
|
(repaired) plan so the next run is free. Needs
|
|
1284
|
-
ANTHROPIC_API_KEY
|
|
1285
|
-
|
|
1286
|
-
|
|
1541
|
+
ANTHROPIC_API_KEY (Claude) or OPENAI_API_KEY
|
|
1542
|
+
(gpt-5.x). Progress -> stderr; the report path ->
|
|
1543
|
+
stdout. --show-plan prints the compiled IR without
|
|
1544
|
+
running; --recompile ignores the cache.
|
|
1287
1545
|
Models: claude-haiku-4-5 | claude-sonnet-4-6
|
|
1288
|
-
(default) | claude-opus-4-8 | claude-fable-5
|
|
1546
|
+
(default) | claude-opus-4-8 | claude-fable-5 |
|
|
1547
|
+
gpt-5.4-mini | gpt-5.4 | gpt-5.5.
|
|
1548
|
+
|
|
1549
|
+
SUITE (run a directory of natural-language tests as one gated suite)
|
|
1550
|
+
suite <dir> [--app <id>] [--name n] [--json] (+ all \`ai\` flags, incl. --server)
|
|
1551
|
+
Run every *.md in <dir> (lexicographic order —
|
|
1552
|
+
prefix 01-, 02- to sequence; README.md skipped)
|
|
1553
|
+
through \`vk ai\`. With --app, app data is reset
|
|
1554
|
+
between tests (iOS: force-stop). Writes a suite
|
|
1555
|
+
overview to ./.verikun/suites/<id>/{index.json,
|
|
1556
|
+
index.html} linking each test's report. Exits 1
|
|
1557
|
+
if any test failed — the CI gate.
|
|
1558
|
+
|
|
1559
|
+
SERVER (expose a locally-connected device to remote verikun clients)
|
|
1560
|
+
server [--bind addr] [--port n] [--auth-key k] [--allow-install]
|
|
1561
|
+
[--allow-unsafe-anonymous] Serve THIS machine's device over HTTP+JSON for
|
|
1562
|
+
\`vk ai/suite/install --server <url>\`. Only
|
|
1563
|
+
verikun's validated action grammar is runnable
|
|
1564
|
+
(never a shell); auth is required (a key is
|
|
1565
|
+
generated if none given; --allow-unsafe-anonymous
|
|
1566
|
+
opts out for trusted networks e.g. Tailscale);
|
|
1567
|
+
binds 127.0.0.1 by default (--bind to expose);
|
|
1568
|
+
one run at a time holds the device lock.
|
|
1569
|
+
Env: VERIKUN_SERVER_AUTH_KEY (keeps it off argv).
|
|
1570
|
+
Clients: pass --server <url> (or VERIKUN_SERVER) + --auth-key (or
|
|
1571
|
+
VERIKUN_SERVER_AUTH_KEY) to ai/suite/install. The server's device+platform apply.
|
|
1289
1572
|
|
|
1290
1573
|
ENVIRONMENT
|
|
1291
1574
|
devices [--json] List attached devices/simulators
|
package/dist/drivers/adb.js
CHANGED
|
@@ -233,6 +233,18 @@ class AdbDriver {
|
|
|
233
233
|
throw new errors_1.CliError(`Failed to launch '${appId}': ${r.stderr.trim() || r.stdout.trim() || `exit code ${r.code}`}`, 3);
|
|
234
234
|
}
|
|
235
235
|
}
|
|
236
|
+
install(appPath) {
|
|
237
|
+
// `-r` reinstalls over an existing package keeping its data (the common
|
|
238
|
+
// update-the-build-under-test case). A large APK can legitimately take
|
|
239
|
+
// minutes to stream + install, so the timeout is far above the 30s default.
|
|
240
|
+
// adb reports failures both as a non-zero exit AND as a `Failure [REASON]`
|
|
241
|
+
// line on stdout with exit 0 (varies by adb version) — check both.
|
|
242
|
+
const r = (0, exec_1.runText)(ADB, this.withSerial(['install', '-r', appPath]), { timeout: 10 * 60 * 1000 });
|
|
243
|
+
const combined = `${r.stdout}\n${r.stderr}`;
|
|
244
|
+
if (r.code !== 0 || /^Failure\b/im.test(combined) || !/^Success\b/im.test(combined)) {
|
|
245
|
+
throw new errors_1.CliError(`Failed to install '${appPath}': ${combined.replace(/\s+/g, ' ').trim() || `exit code ${r.code}`}`, 3);
|
|
246
|
+
}
|
|
247
|
+
}
|
|
236
248
|
stop(appId) {
|
|
237
249
|
this.shell(['am', 'force-stop', appId]);
|
|
238
250
|
}
|
package/dist/drivers/ios.js
CHANGED
|
@@ -278,6 +278,13 @@ class IdbDriver {
|
|
|
278
278
|
this.idbText(['launch', appId]);
|
|
279
279
|
}
|
|
280
280
|
}
|
|
281
|
+
install(appPath) {
|
|
282
|
+
// idb install handles both simulators and physical devices, and both .ipa
|
|
283
|
+
// archives and .app bundles locally. (The REMOTE install path accepts only
|
|
284
|
+
// single-file .ipa uploads in v1 — a .app is a directory, which the streamed
|
|
285
|
+
// upload can't carry; see server.ts.) Installs can take minutes.
|
|
286
|
+
this.idbText(['install', appPath], { timeout: 10 * 60 * 1000 });
|
|
287
|
+
}
|
|
281
288
|
stop(appId) {
|
|
282
289
|
// Best-effort force-stop. `terminate` reports a non-zero exit when the app simply
|
|
283
290
|
// wasn't running; that's a no-op success for us (parity with adb `am force-stop`,
|