verikun 0.4.1 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js CHANGED
@@ -1,5 +1,40 @@
1
1
  "use strict";
2
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
3
+ if (k2 === undefined) k2 = k;
4
+ var desc = Object.getOwnPropertyDescriptor(m, k);
5
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
6
+ desc = { enumerable: true, get: function() { return m[k]; } };
7
+ }
8
+ Object.defineProperty(o, k2, desc);
9
+ }) : (function(o, m, k, k2) {
10
+ if (k2 === undefined) k2 = k;
11
+ o[k2] = m[k];
12
+ }));
13
+ var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
14
+ Object.defineProperty(o, "default", { enumerable: true, value: v });
15
+ }) : function(o, v) {
16
+ o["default"] = v;
17
+ });
18
+ var __importStar = (this && this.__importStar) || (function () {
19
+ var ownKeys = function(o) {
20
+ ownKeys = Object.getOwnPropertyNames || function (o) {
21
+ var ar = [];
22
+ for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
23
+ return ar;
24
+ };
25
+ return ownKeys(o);
26
+ };
27
+ return function (mod) {
28
+ if (mod && mod.__esModule) return mod;
29
+ var result = {};
30
+ if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
31
+ __setModuleDefault(result, mod);
32
+ return result;
33
+ };
34
+ })();
2
35
  Object.defineProperty(exports, "__esModule", { value: true });
36
+ exports.platformFromFlags = platformFromFlags;
37
+ exports.deviceFromFlags = deviceFromFlags;
3
38
  exports.parsePoint = parsePoint;
4
39
  exports.healNote = healNote;
5
40
  exports.parseDuration = parseDuration;
@@ -11,6 +46,7 @@ exports.chooseLogOpts = chooseLogOpts;
11
46
  exports.evalAssert = evalAssert;
12
47
  exports.tokenizeLine = tokenizeLine;
13
48
  exports.withBatchGlobals = withBatchGlobals;
49
+ exports.executeForServer = executeForServer;
14
50
  exports.run = run;
15
51
  const node_fs_1 = require("node:fs");
16
52
  const node_path_1 = require("node:path");
@@ -25,10 +61,14 @@ const run_1 = require("./run");
25
61
  const image_1 = require("./image");
26
62
  const engine_1 = require("./agent/engine");
27
63
  const claude_1 = require("./agent/claude");
64
+ const openai_1 = require("./agent/openai");
28
65
  const cache_1 = require("./agent/cache");
29
66
  const cost_1 = require("./agent/cost");
67
+ const remote_1 = require("./agent/remote");
68
+ const suite_1 = require("./suite");
30
69
  const version_1 = require("./version");
31
70
  const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
71
+ // Exported for src/server.ts (which resolves its own platform/device at startup).
32
72
  function platformFromFlags(flags) {
33
73
  if ((0, args_1.flagBool)(flags, 'ios'))
34
74
  return 'ios';
@@ -149,10 +189,10 @@ function cmdDevices(ctx) {
149
189
  }
150
190
  try {
151
191
  // Only include booted simulators; always include physical devices (they carry a note)
152
- allDevices.push(...new drivers_1.SimctlDriver().listDevices().filter((d) => d.state === 'booted' || d.note));
192
+ allDevices.push(...new drivers_1.IdbDriver().listDevices().filter((d) => d.state === 'booted' || d.note));
153
193
  }
154
194
  catch (e) {
155
- (0, output_1.err)(`devices: simctl backend unavailable (${e.message})`);
195
+ (0, output_1.err)(`devices: iOS backend unavailable (${e.message})`);
156
196
  }
157
197
  if ((0, args_1.flagBool)(ctx.flags, 'json')) {
158
198
  (0, output_1.json)(allDevices);
@@ -171,18 +211,50 @@ function cmdDevices(ctx) {
171
211
  }
172
212
  function cmdDoctor(ctx) {
173
213
  if (ctx.platform === 'ios') {
174
- const r = (0, exec_1.runText)('xcrun', ['simctl', 'list', 'devices', 'booted']);
175
- (0, output_1.out)('xcrun: present');
176
- (0, output_1.out)(r.stdout.trim() || '(no booted simulators)');
177
- (0, output_1.out)('note: iOS screenshots + launch/stop work via simctl; tap/text/swipe/hierarchy need idb.');
178
- return 0;
214
+ try {
215
+ const r = (0, exec_1.runText)('xcrun', ['simctl', 'list', 'devices', 'booted']);
216
+ (0, output_1.out)('xcrun: present');
217
+ (0, output_1.out)(r.stdout.trim() || '(no booted simulators)');
218
+ }
219
+ catch (e) {
220
+ // Not necessarily missing: runText also throws on a spawn timeout or other exec
221
+ // failure, so surface the real reason rather than always claiming "NOT FOUND".
222
+ (0, output_1.err)(`xcrun: ${e.message}`);
223
+ (0, output_1.err)(' (if the Xcode command-line tools are not installed: `xcode-select --install`)');
224
+ return 3;
225
+ }
226
+ // idb (+ its companion) powers everything interactive: ui/tap/text/swipe/key/logs.
227
+ const idb = process.env.IDB || 'idb';
228
+ let idbOk = true;
229
+ try {
230
+ (0, exec_1.runText)(idb, ['--help']); // idb has no --version; --help confirms the binary runs
231
+ (0, output_1.out)('idb: present');
232
+ }
233
+ catch (e) {
234
+ (0, output_1.err)(`idb: ${e.message}`);
235
+ (0, output_1.err)(' needed for ui/tap/text/swipe/key/logs — install: `brew install idb-companion` then `pip install fb-idb`');
236
+ idbOk = false;
237
+ }
238
+ try {
239
+ (0, exec_1.runText)('idb_companion', ['--help']);
240
+ (0, output_1.out)('idb_companion: present');
241
+ }
242
+ catch (e) {
243
+ (0, output_1.err)(`idb_companion: ${e.message}`);
244
+ (0, output_1.err)(' install: `brew install idb-companion`');
245
+ idbOk = false;
246
+ }
247
+ (0, output_1.out)('note: simulator screenshots + launch/stop work via simctl; ui/tap/text/swipe/key/logs use idb.');
248
+ return idbOk ? 0 : 3;
179
249
  }
180
250
  const adb = process.env.ADB || 'adb';
181
251
  try {
182
252
  (0, output_1.out)('adb: ' + (0, exec_1.runText)(adb, ['version']).stdout.split('\n')[0]);
183
253
  }
184
- catch {
185
- (0, output_1.err)('adb: NOT FOUND on PATH');
254
+ catch (e) {
255
+ // Not necessarily missing: runText also throws on a spawn timeout or other exec
256
+ // failure — surface the real reason rather than always claiming "NOT FOUND".
257
+ (0, output_1.err)(`adb: ${e.message}`);
186
258
  return 3;
187
259
  }
188
260
  const devices = ctx.driver.listDevices();
@@ -856,99 +928,157 @@ async function cmdBatch(positionals, batchFlags) {
856
928
  (0, output_1.err)(`[verikun] batch: ${commands.length} command(s) ok`);
857
929
  return 0;
858
930
  }
859
- // ---------------------------------------------------------------------------
860
- // ai — compile a natural-language test to a plan IR, then run it (self-healing)
861
- // ---------------------------------------------------------------------------
862
- //
863
- // `vk ai <file>` reads a plain-English test, compiles it ONCE into a deterministic
864
- // plan IR via the model (cached by NL + app build), then replays it with NO model
865
- // calls on the happy path. The model is woken only to repair a step that fails to
866
- // resolve its selector; a green run persists the (possibly repaired) plan so the
867
- // next run is free again. Cost is bounded by --max-cost-usd. Progress streams to
868
- // stderr (CI liveness — it never goes quiet); stdout carries the final result.
869
- async function cmdAi(positionals, flags) {
870
- const file = positionals[0];
871
- if (!file) {
872
- throw new errors_1.CliError('Usage: verikun ai <file> [--model m] [--max-cost-usd n] [--timeout dur] [--show-plan] [--recompile]', 2);
873
- }
874
- let nl;
875
- try {
876
- nl = (0, node_fs_1.readFileSync)((0, node_path_1.resolve)(process.cwd(), file), 'utf8');
877
- }
878
- catch (e) {
879
- throw new errors_1.CliError(`ai: cannot read '${file}' (${e.message})`, 2);
880
- }
881
- if (!nl.trim())
882
- throw new errors_1.CliError(`ai: '${file}' is empty`, 2);
883
- const platform = platformFromFlags(flags);
884
- const device = deviceFromFlags(flags, platform);
931
+ function parseAiOptions(flags) {
885
932
  const model = (0, cost_1.resolveModel)((0, args_1.flagStr)(flags, 'model'));
886
933
  const overrideRaw = (0, args_1.flagStr)(flags, 'cost-override');
887
934
  const override = overrideRaw ? (0, cost_1.parseCostOverride)(overrideRaw) : undefined;
888
935
  const maxCostUsd = (0, args_1.flagNum)(flags, 'max-cost-usd') ?? cost_1.DEFAULT_MAX_COST_USD;
889
936
  if (maxCostUsd <= 0)
890
937
  throw new errors_1.CliError(`--max-cost-usd must be greater than 0 (got ${maxCostUsd}).`, 2);
891
- const cost = new cost_1.CostTracker((0, cost_1.priceFor)(model, override), maxCostUsd);
892
938
  // Whole-run wall-clock ceiling (default 15m) so a runaway loop/repair can't hang the run.
893
939
  const timeoutFlag = (0, args_1.flagStr)(flags, 'timeout');
894
940
  const timeoutMs = timeoutFlag ? parseDuration(timeoutFlag, 'timeout') : engine_1.DEFAULT_RUN_TIMEOUT_MS;
895
- const deadline = Date.now() + timeoutMs;
896
- const effort = (0, args_1.flagStr)(flags, 'effort');
897
- const pkg = (0, args_1.flagStr)(flags, 'package');
898
- const build = (0, args_1.flagStr)(flags, 'app-build');
899
- const key = { nl, pkg, build, platform };
900
- const recompile = (0, args_1.flagBool)(flags, 'recompile') || (0, args_1.flagBool)(flags, 'no-cache');
901
- const showPlan = (0, args_1.flagBool)(flags, 'show-plan');
902
- const apiKey = process.env.ANTHROPIC_API_KEY;
903
- const provider = apiKey ? new claude_1.ClaudeProvider({ model, apiKey, effort }) : null;
904
- // 1. Obtain the plan: a cache hit (free) or a compile (pays tokens; may seed from
905
- // a prior build's plan to avoid a full recompile).
906
- const cached = recompile ? null : (0, cache_1.readPlan)(key);
907
- let plan;
941
+ return {
942
+ model,
943
+ price: (0, cost_1.priceFor)(model, override),
944
+ maxCostUsd,
945
+ timeoutMs,
946
+ effort: (0, args_1.flagStr)(flags, 'effort'),
947
+ pkg: (0, args_1.flagStr)(flags, 'package'),
948
+ build: (0, args_1.flagStr)(flags, 'app-build'),
949
+ recompile: (0, args_1.flagBool)(flags, 'recompile') || (0, args_1.flagBool)(flags, 'no-cache'),
950
+ };
951
+ }
952
+ function readAiTest(file) {
953
+ let nl;
954
+ try {
955
+ nl = (0, node_fs_1.readFileSync)((0, node_path_1.resolve)(process.cwd(), file), 'utf8');
956
+ }
957
+ catch (e) {
958
+ throw new errors_1.CliError(`ai: cannot read '${file}' (${e.message})`, 2);
959
+ }
960
+ if (!nl.trim())
961
+ throw new errors_1.CliError(`ai: '${file}' is empty`, 2);
962
+ return nl;
963
+ }
964
+ /** The env var carrying the API key for a model's provider (per-provider keys). */
965
+ function keyEnvFor(model) {
966
+ return (0, cost_1.providerFor)(model) === 'openai' ? 'OPENAI_API_KEY' : 'ANTHROPIC_API_KEY';
967
+ }
968
+ /** Route the model to its backend; each provider reads its own key. A missing key
969
+ * means no provider (compile/repair unavailable) — the same graceful degradation
970
+ * as before. */
971
+ function makeProvider(opts) {
972
+ const apiKey = process.env[keyEnvFor(opts.model)];
973
+ if (!apiKey)
974
+ return null;
975
+ return (0, cost_1.providerFor)(opts.model) === 'openai'
976
+ ? new openai_1.OpenAiProvider({ model: opts.model, apiKey, effort: opts.effort })
977
+ : new claude_1.ClaudeProvider({ model: opts.model, apiKey, effort: opts.effort });
978
+ }
979
+ /** Obtain the plan: a cache hit (free) or a compile (pays tokens; may seed from a
980
+ * prior build's plan to avoid a full recompile). The fresh compile is cached right
981
+ * away, so an unchanged test is never recompiled — even via --show-plan or after a
982
+ * failed run. A green run later re-persists the healed plan (never a half-healed one). */
983
+ async function obtainPlan(key, file, opts, cost, provider) {
984
+ const cached = opts.recompile ? null : (0, cache_1.readPlan)(key);
908
985
  if (cached) {
909
- plan = cached.plan;
910
- (0, output_1.err)(`[ai] plan cache hit — ${model} not called to compile`);
986
+ (0, output_1.err)(`[ai] plan cache hit — ${opts.model} not called to compile`);
987
+ return { plan: cached.plan, cached: true };
911
988
  }
912
- else {
913
- if (!provider) {
914
- throw new errors_1.CliError('ANTHROPIC_API_KEY is not set — `vk ai` needs it to compile the test. Set it and retry.', 3);
915
- }
916
- const seed = (0, cache_1.findSeed)(key);
917
- if (seed)
918
- (0, output_1.err)(`[ai] no exact cache; seeding from a prior plan (build ${seed.build ?? 'unknown'})`);
919
- (0, output_1.err)(`[ai] compiling '${file}' with ${model} (effort ${effort ?? 'default'})…`);
920
- const compiled = await provider.compile({ nl, pkg, platform, seed: seed?.plan });
921
- cost.add(compiled.usage, 'compile');
922
- plan = compiled.plan;
923
- (0, output_1.err)(`[ai] compiled ${plan.steps.length} top-level step(s) · ${cost.summaryLine()}`);
924
- // Cache the freshly-compiled plan right away, keyed by the test-text hash, so an
925
- // unchanged test is never recompiled — even via --show-plan or after a failed run.
926
- // A green run below re-persists the healed plan; a failed run leaves this clean
927
- // compile cached (never a half-healed one).
928
- try {
929
- (0, cache_1.writePlan)(key, plan);
930
- }
931
- catch (e) {
932
- (0, output_1.err)(`[ai] could not cache compiled plan: ${e.message}`);
933
- }
989
+ if (!provider) {
990
+ throw new errors_1.CliError(`${keyEnvFor(opts.model)} is not set — \`vk ai\` needs it to compile the test (model ${opts.model}). Set it and retry.`, 3);
991
+ }
992
+ const seed = (0, cache_1.findSeed)(key);
993
+ if (seed)
994
+ (0, output_1.err)(`[ai] no exact cache; seeding from a prior plan (build ${seed.build ?? 'unknown'})`);
995
+ (0, output_1.err)(`[ai] compiling '${file}' with ${opts.model} (effort ${opts.effort ?? 'default'})…`);
996
+ const compiled = await provider.compile({ nl: key.nl, pkg: key.pkg, platform: key.platform, seed: seed?.plan });
997
+ cost.add(compiled.usage, 'compile');
998
+ (0, output_1.err)(`[ai] compiled ${compiled.plan.steps.length} top-level step(s) · ${cost.summaryLine()}`);
999
+ try {
1000
+ (0, cache_1.writePlan)(key, compiled.plan);
934
1001
  }
935
- // 2. --show-plan: print the compiled IR and stop (no device run).
936
- if (showPlan) {
937
- (0, output_1.json)(plan);
938
- return 0;
1002
+ catch (e) {
1003
+ (0, output_1.err)(`[ai] could not cache compiled plan: ${e.message}`);
939
1004
  }
1005
+ return { plan: compiled.plan, cached: false };
1006
+ }
1007
+ async function resolveBackend(platform, device, flags) {
1008
+ const server = (0, args_1.flagStr)(flags, 'server') || process.env.VERIKUN_SERVER || undefined;
1009
+ if (!server) {
1010
+ const driver = (0, drivers_1.getDriver)(platform, device);
1011
+ return {
1012
+ backend: {
1013
+ exec: (command, positionals, f) => executeOutcome(command, positionals, f, driver),
1014
+ getElements: () => driver.getElements(),
1015
+ install: (appPath) => driver.install(appPath),
1016
+ reset: (appId) => {
1017
+ assertSafeAppId(appId);
1018
+ // iOS has no per-app data reset — degrade honestly to a force-stop.
1019
+ if (platform === 'ios')
1020
+ driver.stop(appId);
1021
+ else
1022
+ driver.clearApp(appId);
1023
+ },
1024
+ },
1025
+ platform,
1026
+ device,
1027
+ };
1028
+ }
1029
+ let runCtx = { platform, device };
1030
+ const opts = {
1031
+ url: server,
1032
+ authKey: (0, args_1.flagStr)(flags, 'auth-key') || process.env.VERIKUN_SERVER_AUTH_KEY || undefined,
1033
+ // Each remote step is spliced into the local active run so the archived report
1034
+ // is identical to a local run's.
1035
+ onStep: (step, artifacts) => run_1.Recorder.appendForeignStep(step, artifacts, runCtx),
1036
+ };
1037
+ const health = await (0, remote_1.pingServer)(opts); // fails fast (exit 3) on a bad URL or key
1038
+ runCtx = { platform: health.platform, device: health.serial };
1039
+ (0, output_1.err)(`[verikun] server ${server}: ${health.platform} · device ${health.serial} · verikun ${health.version}`);
1040
+ return {
1041
+ backend: (0, remote_1.createRemoteBackend)(opts, health),
1042
+ platform: health.platform,
1043
+ device: health.serial,
1044
+ remote: { url: server, version: health.version },
1045
+ };
1046
+ }
1047
+ /**
1048
+ * Run one natural-language test through a backend and return DATA — no stdout
1049
+ * writes (stdout stays the caller's one result; progress streams to stderr).
1050
+ * `vk ai` wraps it with its --json/report output; `vk suite` calls it per test.
1051
+ */
1052
+ async function runAiTest(file, opts, backend, platform, device) {
1053
+ const nl = readAiTest(file);
1054
+ const key = { nl, pkg: opts.pkg, build: opts.build, platform };
1055
+ const cost = new cost_1.CostTracker(opts.price, opts.maxCostUsd);
1056
+ const deadline = Date.now() + opts.timeoutMs;
1057
+ const provider = makeProvider(opts);
1058
+ const { plan, cached } = await obtainPlan(key, file, opts, cost, provider);
940
1059
  // Running needs the provider for repair-on-failure; a cache hit with no key can't repair.
941
1060
  if (!provider) {
942
- throw new errors_1.CliError('ANTHROPIC_API_KEY is not set — `vk ai` needs it to repair a failing step at runtime.', 3);
1061
+ throw new errors_1.CliError(`${keyEnvFor(opts.model)} is not set — \`vk ai\` needs it to repair a failing step at runtime (model ${opts.model}).`, 3);
943
1062
  }
944
1063
  // The budget is a TOTAL-run ceiling: if the compile alone already crossed it, abort
945
1064
  // before running. A cache hit spends nothing, so a free replay is still allowed.
946
1065
  if (!cached && cost.exceeded()) {
947
- (0, output_1.err)(`[ai] cost ceiling $${maxCostUsd} reached during compile (${cost.summaryLine()}) — not running`);
948
- return 1;
949
- }
950
- // 3. One explicit run + one shared driver for the whole flow (so rollover can't
951
- // split the test, and we don't rebuild a driver per step).
1066
+ (0, output_1.err)(`[ai] cost ceiling $${opts.maxCostUsd} reached during compile (${cost.summaryLine()}) — not running`);
1067
+ return {
1068
+ ok: false,
1069
+ costUsd: Number(cost.usd().toFixed(4)),
1070
+ costLine: cost.summaryLine(),
1071
+ modelRepairs: 0,
1072
+ improvements: [],
1073
+ runDir: '',
1074
+ reportHtml: '',
1075
+ junitXml: '',
1076
+ state: null,
1077
+ abortedForBudget: true,
1078
+ failure: { where: 'compile', reason: `cost ceiling $${opts.maxCostUsd} reached during compile` },
1079
+ };
1080
+ }
1081
+ // One explicit run for the whole flow (so rollover can't split the test).
952
1082
  const existing = run_1.Recorder.status();
953
1083
  if (existing && existing.steps.length > 0) {
954
1084
  // Seal the pre-existing run into the archive instead of letting start(force=true)
@@ -957,14 +1087,13 @@ async function cmdAi(positionals, flags) {
957
1087
  (0, output_1.err)(`[ai] archived the active run ('${existing.name}', ${existing.steps.length} step(s)) → ${sealed.dir}`);
958
1088
  }
959
1089
  run_1.Recorder.start(`ai: ${(0, node_path_1.basename)(file)}`, platform, device, true);
960
- const driver = (0, drivers_1.getDriver)(platform, device);
961
1090
  // Suppress per-step `out()` so stdout stays the one final result; progress -> stderr.
962
1091
  const prevQuiet = (0, output_1.setOutputQuiet)(true);
963
1092
  let result;
964
1093
  try {
965
1094
  result = await (0, engine_1.runPlan)(plan, {
966
- exec: (command, pos, f) => executeOutcome(command, pos, f, driver),
967
- getElements: () => driver.getElements(),
1095
+ exec: (command, pos, f) => backend.exec(command, pos, f),
1096
+ getElements: () => backend.getElements(),
968
1097
  provider,
969
1098
  cost,
970
1099
  log: (m) => (0, output_1.err)(m),
@@ -991,8 +1120,8 @@ async function cmdAi(positionals, flags) {
991
1120
  finally {
992
1121
  (0, output_1.setOutputQuiet)(prevQuiet);
993
1122
  }
994
- // 4. Persist the (possibly repaired) plan only on a fully-green run; attach the
995
- // cost + improvements summary to the run; archive into the report.
1123
+ // Persist the (possibly repaired) plan only on a fully-green run; attach the
1124
+ // cost + improvements summary to the run; archive into the report.
996
1125
  const costLine = cost.summaryLine();
997
1126
  if (result.ok) {
998
1127
  try {
@@ -1006,13 +1135,13 @@ async function cmdAi(positionals, flags) {
1006
1135
  run_1.Recorder.annotateRun({
1007
1136
  ai: { ok: result.ok, cost: costLine, modelRepairs: result.modelRepairs, improvements: result.improvements },
1008
1137
  });
1009
- const { dir, xmlPath, htmlPath } = run_1.Recorder.archive();
1138
+ const { dir, xmlPath, htmlPath, state } = run_1.Recorder.archive();
1010
1139
  const status = result.ok
1011
1140
  ? 'PASS'
1012
1141
  : result.abortedForBudget
1013
- ? `ABORTED — cost ceiling $${maxCostUsd} reached`
1142
+ ? `ABORTED — cost ceiling $${opts.maxCostUsd} reached`
1014
1143
  : result.abortedForTimeout
1015
- ? `ABORTED — run timeout (${Math.round(timeoutMs / 1000)}s) reached`
1144
+ ? `ABORTED — run timeout (${Math.round(opts.timeoutMs / 1000)}s) reached`
1016
1145
  : `FAIL at ${result.failure?.where}: ${result.failure?.reason}`;
1017
1146
  (0, output_1.err)(`[ai] ${status} · ${costLine}`);
1018
1147
  (0, output_1.err)(`[ai] report: ${htmlPath}`);
@@ -1022,28 +1151,124 @@ async function cmdAi(positionals, flags) {
1022
1151
  (0, output_1.err)(' - ' + imp);
1023
1152
  }
1024
1153
  (0, output_1.err)(`[ai] estimated total cost: $${cost.usd().toFixed(4)}`);
1154
+ return {
1155
+ ok: result.ok,
1156
+ costUsd: Number(cost.usd().toFixed(4)),
1157
+ costLine,
1158
+ modelRepairs: result.modelRepairs,
1159
+ improvements: result.improvements,
1160
+ runDir: dir,
1161
+ reportHtml: htmlPath,
1162
+ junitXml: xmlPath,
1163
+ state,
1164
+ ...(result.failure ? { failure: result.failure } : {}),
1165
+ ...(result.abortedForBudget ? { abortedForBudget: true } : {}),
1166
+ ...(result.abortedForTimeout ? { abortedForTimeout: true } : {}),
1167
+ };
1168
+ }
1169
+ async function cmdAi(positionals, flags) {
1170
+ const file = positionals[0];
1171
+ if (!file) {
1172
+ throw new errors_1.CliError('Usage: verikun ai <file> [--model m] [--max-cost-usd n] [--timeout dur] [--server url] [--show-plan] [--recompile]', 2);
1173
+ }
1174
+ const opts = parseAiOptions(flags);
1175
+ // --show-plan: compile (or cache-hit) and print the IR — no device, no backend.
1176
+ if ((0, args_1.flagBool)(flags, 'show-plan')) {
1177
+ const nl = readAiTest(file);
1178
+ const key = { nl, pkg: opts.pkg, build: opts.build, platform: platformFromFlags(flags) };
1179
+ const cost = new cost_1.CostTracker(opts.price, opts.maxCostUsd);
1180
+ const { plan } = await obtainPlan(key, file, opts, cost, makeProvider(opts));
1181
+ (0, output_1.json)(plan);
1182
+ return 0;
1183
+ }
1184
+ const reqPlatform = platformFromFlags(flags);
1185
+ const { backend, platform, device } = await resolveBackend(reqPlatform, deviceFromFlags(flags, reqPlatform), flags);
1186
+ let result;
1187
+ try {
1188
+ result = await runAiTest(file, opts, backend, platform, device);
1189
+ }
1190
+ finally {
1191
+ await backend.close?.(); // frees a remote server's device lock for the next command
1192
+ }
1025
1193
  if ((0, args_1.flagBool)(flags, 'json')) {
1026
1194
  (0, output_1.json)({
1027
1195
  ok: result.ok,
1028
- model,
1029
- cost: costLine,
1030
- costUsd: Number(cost.usd().toFixed(4)),
1196
+ model: opts.model,
1197
+ cost: result.costLine,
1198
+ costUsd: result.costUsd,
1031
1199
  modelRepairs: result.modelRepairs,
1032
1200
  improvements: result.improvements,
1033
- report: htmlPath,
1034
- junit: xmlPath,
1035
- runDir: dir,
1201
+ report: result.reportHtml,
1202
+ junit: result.junitXml,
1203
+ runDir: result.runDir,
1036
1204
  ...(result.failure ? { failure: result.failure } : {}),
1037
1205
  ...(result.abortedForBudget ? { abortedForBudget: true } : {}),
1038
1206
  ...(result.abortedForTimeout ? { abortedForTimeout: true } : {}),
1039
1207
  });
1040
1208
  }
1041
- else {
1042
- (0, output_1.out)(htmlPath); // primary machine result: the report path
1209
+ else if (result.reportHtml) {
1210
+ (0, output_1.out)(result.reportHtml); // primary machine result: the report path
1043
1211
  }
1044
1212
  return result.ok ? 0 : 1;
1045
1213
  }
1046
1214
  // ---------------------------------------------------------------------------
1215
+ // install — put an app build on the device (local driver or remote vk server)
1216
+ // ---------------------------------------------------------------------------
1217
+ async function cmdInstall(positionals, flags) {
1218
+ const appPath = positionals[0];
1219
+ if (!appPath)
1220
+ throw new errors_1.CliError('Usage: verikun install <app.apk|app.ipa> [--server url]', 2);
1221
+ const path = (0, node_path_1.resolve)(process.cwd(), appPath);
1222
+ if (!(0, node_fs_1.existsSync)(path))
1223
+ throw new errors_1.CliError(`install: '${appPath}' does not exist`, 2);
1224
+ const platform = platformFromFlags(flags);
1225
+ const { backend, remote } = await resolveBackend(platform, deviceFromFlags(flags, platform), flags);
1226
+ (0, output_1.err)(`[verikun] installing ${appPath}${remote ? ` via ${remote.url}` : ''}…`);
1227
+ try {
1228
+ await backend.install(path);
1229
+ }
1230
+ finally {
1231
+ await backend.close?.();
1232
+ }
1233
+ if ((0, args_1.flagBool)(flags, 'json'))
1234
+ (0, output_1.json)({ installed: appPath, ...(remote ? { server: remote.url } : {}) });
1235
+ else
1236
+ (0, output_1.out)(`installed ${appPath}`);
1237
+ return 0;
1238
+ }
1239
+ // ---------------------------------------------------------------------------
1240
+ // suite — run a directory of natural-language tests as one gated suite
1241
+ // ---------------------------------------------------------------------------
1242
+ async function cmdSuiteEntry(positionals, flags) {
1243
+ const dirArg = positionals[0];
1244
+ if (!dirArg)
1245
+ throw new errors_1.CliError('Usage: verikun suite <dir> [--app <id>] [--server url] [--name n] [--json]', 2);
1246
+ const opts = parseAiOptions(flags);
1247
+ // Pre-flight the model key BEFORE touching any device/server: every test needs it
1248
+ // to compile (on a cache miss) or to repair at runtime.
1249
+ if (!process.env[keyEnvFor(opts.model)]) {
1250
+ throw new errors_1.CliError(`${keyEnvFor(opts.model)} is not set — \`vk suite\` needs it to compile/repair tests (model ${opts.model}).`, 3);
1251
+ }
1252
+ const reqPlatform = platformFromFlags(flags);
1253
+ const { backend, platform, device } = await resolveBackend(reqPlatform, deviceFromFlags(flags, reqPlatform), flags);
1254
+ const app = (0, args_1.flagStr)(flags, 'app');
1255
+ if (app)
1256
+ assertSafeAppId(app);
1257
+ try {
1258
+ return await (0, suite_1.cmdSuite)(dirArg, flags, {
1259
+ platform,
1260
+ device,
1261
+ runTest: (file) => runAiTest(file, opts, backend, platform, device),
1262
+ // Reset app state between tests only when the app id is known; without --app,
1263
+ // each test is responsible for its own isolation (e.g. `launch --clear`).
1264
+ reset: app ? () => backend.reset(app) : undefined,
1265
+ });
1266
+ }
1267
+ finally {
1268
+ await backend.close?.();
1269
+ }
1270
+ }
1271
+ // ---------------------------------------------------------------------------
1047
1272
  // Dispatch
1048
1273
  // ---------------------------------------------------------------------------
1049
1274
  async function executeCommand(command, ctx) {
@@ -1149,7 +1374,19 @@ async function executeOutcome(command, positionals, flags, sharedDriver) {
1149
1374
  }
1150
1375
  recorder = run_1.Recorder.beginStep(command, positionals, flags, platform, device, serial, driver);
1151
1376
  }
1152
- const ctx = { driver, platform, device, positionals, flags, record: recorder ?? undefined };
1377
+ }
1378
+ catch (e) {
1379
+ return { code: e instanceof errors_1.CliError ? e.exitCode : 3, error: e };
1380
+ }
1381
+ const d = driver; // assigned in the try above, or we already returned
1382
+ const ctx = { driver: d, platform, device, positionals, flags, record: recorder ?? undefined };
1383
+ return runRecorded(command, ctx, recorder, d);
1384
+ }
1385
+ /** The shared middle of executeOutcome / executeForServer: run the handler, close
1386
+ * the step (with failure evidence) whether it returned or threw, map to a raw
1387
+ * outcome (the error NOT printed — callers decide). */
1388
+ async function runRecorded(command, ctx, recorder, driver) {
1389
+ try {
1153
1390
  const code = await executeCommand(command, ctx);
1154
1391
  recorder?.finish(code, driver);
1155
1392
  return { code };
@@ -1159,6 +1396,28 @@ async function executeOutcome(command, positionals, flags, sharedDriver) {
1159
1396
  return { code: e instanceof errors_1.CliError ? e.exitCode : 3, error: e };
1160
1397
  }
1161
1398
  }
1399
+ /**
1400
+ * `vk server`'s per-request executor: run one already-validated leaf against the
1401
+ * server's fixed driver/platform (a client can never repoint the device via flags),
1402
+ * recording into an EPHEMERAL single-step recorder instead of the local run store.
1403
+ * Returns the raw outcome plus the finished step + artifact buffers, which travel
1404
+ * back over the wire and are spliced into the CALLER's run — so `resolved`/`tier`/
1405
+ * failure evidence survive remoting with zero handler changes.
1406
+ */
1407
+ async function executeForServer(command, positionals, flags, driver, platform) {
1408
+ let serial;
1409
+ try {
1410
+ serial = driver.resolvedSerial();
1411
+ }
1412
+ catch {
1413
+ /* surfaced by the command handler below */
1414
+ }
1415
+ const recorder = run_1.Recorder.beginEphemeralStep(command, positionals, flags, platform, serial);
1416
+ const ctx = { driver, platform, device: serial, positionals, flags, record: recorder };
1417
+ const outcome = await runRecorded(command, ctx, recorder, driver);
1418
+ const { step, artifacts } = recorder.takeEphemeral();
1419
+ return { ...outcome, step, artifacts };
1420
+ }
1162
1421
  /**
1163
1422
  * Run one already-parsed command for the CLI / `batch`: dispatch meta-commands,
1164
1423
  * else execute it and map any failure to a printed exit code. This is the shared
@@ -1174,9 +1433,9 @@ async function executeParsed(command, positionals, flags) {
1174
1433
  return cmdRun(positionals, flags, platform, device);
1175
1434
  if (command === 'batch')
1176
1435
  return cmdBatch(positionals, flags);
1177
- // `ai` orchestrates its own steps; map its thrown CliErrors to exit codes here
1178
- // (usage 2 / env 3 / …) so they honor the exit-code contract instead of escaping
1179
- // to the top-level "Fatal" handler (which would force exit 3).
1436
+ // `ai`/`suite`/`install`/`server` orchestrate their own steps; map their thrown
1437
+ // CliErrors to exit codes here (usage 2 / env 3 / …) so they honor the exit-code
1438
+ // contract instead of escaping to the top-level "Fatal" handler (exit 3).
1180
1439
  if (command === 'ai') {
1181
1440
  try {
1182
1441
  return await cmdAi(positionals, flags);
@@ -1185,6 +1444,33 @@ async function executeParsed(command, positionals, flags) {
1185
1444
  return mapError(e, flags);
1186
1445
  }
1187
1446
  }
1447
+ if (command === 'suite') {
1448
+ try {
1449
+ return await cmdSuiteEntry(positionals, flags);
1450
+ }
1451
+ catch (e) {
1452
+ return mapError(e, flags);
1453
+ }
1454
+ }
1455
+ if (command === 'install') {
1456
+ try {
1457
+ return await cmdInstall(positionals, flags);
1458
+ }
1459
+ catch (e) {
1460
+ return mapError(e, flags);
1461
+ }
1462
+ }
1463
+ if (command === 'server') {
1464
+ // Dynamic import: keeps node:http off the default load path and avoids a
1465
+ // static cli↔server cycle (server.ts imports executeForServer from here).
1466
+ try {
1467
+ const { cmdServer } = await Promise.resolve().then(() => __importStar(require('./server')));
1468
+ return await cmdServer(positionals, flags);
1469
+ }
1470
+ catch (e) {
1471
+ return mapError(e, flags);
1472
+ }
1473
+ }
1188
1474
  const { code, error } = await executeOutcome(command, positionals, flags);
1189
1475
  return error ? mapError(error, flags) : code;
1190
1476
  }
@@ -1232,6 +1518,9 @@ ACT
1232
1518
  launch <app> [--clear] [--no-restart] stop <app> App lifecycle (launch restarts by
1233
1519
  default — force-stops first; --clear also wipes app data)
1234
1520
  clear <app> Wipe app data — login/session, caches (fresh-install state)
1521
+ install <app.apk|.ipa> [--server url] Install a build on the device (adb install -r /
1522
+ idb install). With --server, uploads the file to a
1523
+ remote vk server (which must run --allow-install)
1235
1524
 
1236
1525
  BATCH (script many commands in one process)
1237
1526
  batch [--file path] [--quiet] Run newline-separated commands — from --file,
@@ -1249,11 +1538,37 @@ AI (run a natural-language test — compile once, replay model-free, self-heal)
1249
1538
  path. The model is woken only to repair a step
1250
1539
  that fails to resolve; a green run persists the
1251
1540
  (repaired) plan so the next run is free. Needs
1252
- ANTHROPIC_API_KEY. Progress -> stderr; the report
1253
- path -> stdout. --show-plan prints the compiled
1254
- IR without running; --recompile ignores the cache.
1541
+ ANTHROPIC_API_KEY (Claude) or OPENAI_API_KEY
1542
+ (gpt-5.x). Progress -> stderr; the report path ->
1543
+ stdout. --show-plan prints the compiled IR without
1544
+ running; --recompile ignores the cache.
1255
1545
  Models: claude-haiku-4-5 | claude-sonnet-4-6
1256
- (default) | claude-opus-4-8 | claude-fable-5.
1546
+ (default) | claude-opus-4-8 | claude-fable-5 |
1547
+ gpt-5.4-mini | gpt-5.4 | gpt-5.5.
1548
+
1549
+ SUITE (run a directory of natural-language tests as one gated suite)
1550
+ suite <dir> [--app <id>] [--name n] [--json] (+ all \`ai\` flags, incl. --server)
1551
+ Run every *.md in <dir> (lexicographic order —
1552
+ prefix 01-, 02- to sequence; README.md skipped)
1553
+ through \`vk ai\`. With --app, app data is reset
1554
+ between tests (iOS: force-stop). Writes a suite
1555
+ overview to ./.verikun/suites/<id>/{index.json,
1556
+ index.html} linking each test's report. Exits 1
1557
+ if any test failed — the CI gate.
1558
+
1559
+ SERVER (expose a locally-connected device to remote verikun clients)
1560
+ server [--bind addr] [--port n] [--auth-key k] [--allow-install]
1561
+ [--allow-unsafe-anonymous] Serve THIS machine's device over HTTP+JSON for
1562
+ \`vk ai/suite/install --server <url>\`. Only
1563
+ verikun's validated action grammar is runnable
1564
+ (never a shell); auth is required (a key is
1565
+ generated if none given; --allow-unsafe-anonymous
1566
+ opts out for trusted networks e.g. Tailscale);
1567
+ binds 127.0.0.1 by default (--bind to expose);
1568
+ one run at a time holds the device lock.
1569
+ Env: VERIKUN_SERVER_AUTH_KEY (keeps it off argv).
1570
+ Clients: pass --server <url> (or VERIKUN_SERVER) + --auth-key (or
1571
+ VERIKUN_SERVER_AUTH_KEY) to ai/suite/install. The server's device+platform apply.
1257
1572
 
1258
1573
  ENVIRONMENT
1259
1574
  devices [--json] List attached devices/simulators
@@ -1294,5 +1609,7 @@ GLOBAL FLAGS
1294
1609
  EXIT CODES
1295
1610
  0 success · 1 not found / assertion failed / timeout · 2 usage or ambiguous selector · 3 environment error
1296
1611
 
1297
- iOS: screenshots + launch/stop work today via simctl; tap/text/swipe/hierarchy need idb (planned).`;
1612
+ iOS (--ios): full parity via idb — ui/tap/text/swipe/key + screenshot/launch/stop.
1613
+ Needs idb (\`brew install idb-companion\` + \`pip install fb-idb\`); see \`vk doctor --ios\`.
1614
+ Caveats: no \`clear\` (no per-app reset), \`current\` is (unknown), device logs are simulator-only.`;
1298
1615
  }