@wayai/cli 0.3.128 → 0.3.130

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -5004,6 +5004,15 @@ function collectTurnAttachmentHashes(turn) {
5004
5004
  }
5005
5005
  return hashes;
5006
5006
  }
5007
+ function evaluatedRunCount(counts) {
5008
+ return counts.successful_runs + (counts.assertion_failure_runs ?? counts.failed_runs);
5009
+ }
5010
+ function unprovenRunCount(counts) {
5011
+ return (counts.errored_runs ?? 0) + (counts.invalid_runs ?? 0);
5012
+ }
5013
+ function terminalRunCount(counts) {
5014
+ return counts.successful_runs + counts.failed_runs;
5015
+ }
5007
5016
  function isSafeTemplatePath(path31) {
5008
5017
  if (path31.length === 0 || path31.length > 300) return false;
5009
5018
  if (path31.startsWith("/") || path31.includes("\\")) return false;
@@ -6869,7 +6878,17 @@ var init_contracts = __esm({
6869
6878
  total_evals: external_exports.number(),
6870
6879
  total_runs: external_exports.number(),
6871
6880
  successful_runs: external_exports.number(),
6881
+ /**
6882
+ * Every terminal run that did NOT pass, whatever the reason. Unchanged
6883
+ * meaning — deliberately NOT narrowed to behavioral misses, so a released CLI
6884
+ * that predates the split cannot read an infrastructure wipeout as a clean
6885
+ * sweep. Use `assertion_failure_runs` for the narrow count.
6886
+ */
6872
6887
  failed_runs: external_exports.number(),
6888
+ /** The `failed_runs` breakdown by cause (see `EvalRunOutcomeKind`); sums to it. */
6889
+ assertion_failure_runs: external_exports.number(),
6890
+ errored_runs: external_exports.number(),
6891
+ invalid_runs: external_exports.number(),
6873
6892
  created_by: external_exports.string(),
6874
6893
  /** Optional — sessions created pre-PR2 may not have this. */
6875
6894
  scenario_set_id: external_exports.string().nullable().optional(),
@@ -9388,6 +9407,16 @@ function extractApiErrorCode(err) {
9388
9407
  return null;
9389
9408
  }
9390
9409
  }
9410
+ function extractApiRetryAfterSeconds(err) {
9411
+ if (!(err instanceof ApiError)) return null;
9412
+ try {
9413
+ const parsed = JSON.parse(err.body);
9414
+ const seconds = parsed?.details?.retry_after_seconds;
9415
+ return typeof seconds === "number" && Number.isFinite(seconds) && seconds >= 0 ? seconds : null;
9416
+ } catch {
9417
+ return null;
9418
+ }
9419
+ }
9391
9420
  function friendlyHint(err) {
9392
9421
  if (isNetworkError(err)) {
9393
9422
  return "Couldn't reach WayAI \u2014 check your network connection and try again.";
@@ -11890,7 +11919,17 @@ var init_dist = __esm({
11890
11919
  total_evals: external_exports.number(),
11891
11920
  total_runs: external_exports.number(),
11892
11921
  successful_runs: external_exports.number(),
11922
+ /**
11923
+ * Every terminal run that did NOT pass, whatever the reason. Unchanged
11924
+ * meaning — deliberately NOT narrowed to behavioral misses, so a released CLI
11925
+ * that predates the split cannot read an infrastructure wipeout as a clean
11926
+ * sweep. Use `assertion_failure_runs` for the narrow count.
11927
+ */
11893
11928
  failed_runs: external_exports.number(),
11929
+ /** The `failed_runs` breakdown by cause (see `EvalRunOutcomeKind`); sums to it. */
11930
+ assertion_failure_runs: external_exports.number(),
11931
+ errored_runs: external_exports.number(),
11932
+ invalid_runs: external_exports.number(),
11894
11933
  created_by: external_exports.string(),
11895
11934
  /** Optional — sessions created pre-PR2 may not have this. */
11896
11935
  scenario_set_id: external_exports.string().nullable().optional(),
@@ -18789,15 +18828,24 @@ var init_send_message = __esm({
18789
18828
  });
18790
18829
 
18791
18830
  // src/lib/eval-format.ts
18831
+ function runNotes(unproven, inProgress) {
18832
+ return [
18833
+ unproven > 0 ? `${unproven} unproven` : "",
18834
+ inProgress > 0 ? `${inProgress} in progress` : ""
18835
+ ].filter(Boolean);
18836
+ }
18792
18837
  function printResultsTable(results, mode) {
18793
18838
  let totalPassed = 0;
18794
- let totalTerminal = 0;
18839
+ let totalEvaluated = 0;
18840
+ let totalUnproven = 0;
18795
18841
  let totalInProgress = 0;
18796
18842
  for (const r of results) {
18797
18843
  const evalName = r.eval?.eval_name || r.eval_id.slice(0, 8);
18798
- const terminalRuns = r.successful_runs + r.failed_runs;
18844
+ const evaluatedRuns = evaluatedRunCount(r);
18845
+ const unproven = unprovenRunCount(r);
18799
18846
  const inProgress = (r.running_runs ?? 0) + (r.pending_runs ?? 0);
18800
- const passed = inProgress > 0 ? `${r.successful_runs}/${terminalRuns} passed (+${inProgress} in progress)` : `${r.successful_runs}/${terminalRuns} passed`;
18847
+ const notes2 = runNotes(unproven, inProgress);
18848
+ const passed = notes2.length > 0 ? `${r.successful_runs}/${evaluatedRuns} passed (${notes2.join(", ")})` : `${r.successful_runs}/${evaluatedRuns} passed`;
18801
18849
  const avgTime = r.avg_execution_time_ms != null ? `avg ${(r.avg_execution_time_ms / 1e3).toFixed(1)}s` : "";
18802
18850
  let scoresStr = "";
18803
18851
  if (r.aggregated_scores) {
@@ -18808,13 +18856,15 @@ function printResultsTable(results, mode) {
18808
18856
  }
18809
18857
  console.log(` ${evalName.padEnd(20)} ${passed.padEnd(14)} ${avgTime}${scoresStr}`);
18810
18858
  totalPassed += r.successful_runs;
18811
- totalTerminal += terminalRuns;
18859
+ totalEvaluated += evaluatedRuns;
18860
+ totalUnproven += unproven;
18812
18861
  totalInProgress += inProgress;
18813
18862
  }
18814
- const pct = totalTerminal > 0 ? (totalPassed / totalTerminal * 100).toFixed(1) : "0.0";
18815
- const inProgressNote = totalInProgress > 0 ? ` \u2014 ${totalInProgress} in progress` : "";
18863
+ const pct = totalEvaluated > 0 ? (totalPassed / totalEvaluated * 100).toFixed(1) : "0.0";
18864
+ const notes = runNotes(totalUnproven, totalInProgress);
18865
+ const note = notes.length > 0 ? ` \u2014 ${notes.join(", ")}` : "";
18816
18866
  console.log(`
18817
- Overall: ${totalPassed}/${totalTerminal} passed (${pct}%)${inProgressNote}`);
18867
+ Overall: ${totalPassed}/${totalEvaluated} passed (${pct}%)${note}`);
18818
18868
  }
18819
18869
  function truncate(str, maxLen) {
18820
18870
  if (str.length <= maxLen) return str;
@@ -18836,6 +18886,17 @@ function isFailingRun(run) {
18836
18886
  if (run.run_status && !TERMINAL_RUN_STATUSES.has(run.run_status)) return false;
18837
18887
  return run.response_match !== true;
18838
18888
  }
18889
+ function failureTag(run) {
18890
+ switch (run.outcome_kind) {
18891
+ case "execution_error":
18892
+ return "ERROR";
18893
+ case "evaluation_invalid":
18894
+ return "INVALID";
18895
+ // `assertion_failure`, and any pre-taxonomy payload.
18896
+ default:
18897
+ return "FAIL";
18898
+ }
18899
+ }
18839
18900
  function printFailingRuns(runs) {
18840
18901
  const failing = runs.filter(isFailingRun);
18841
18902
  if (failing.length === 0) return;
@@ -18844,7 +18905,7 @@ Failures (${failing.length}):`);
18844
18905
  let anyComment = false;
18845
18906
  for (const run of failing) {
18846
18907
  const evalName = run.eval?.eval_name || run.eval_id?.slice(0, 8) || "eval";
18847
- console.log(` ${evalName} \u2014 Run #${run.run_number} [FAIL]`);
18908
+ console.log(` ${evalName} \u2014 Run #${run.run_number} [${failureTag(run)}]`);
18848
18909
  if (run.error_message) {
18849
18910
  console.log(` Error: ${run.error_message}`);
18850
18911
  }
@@ -18866,6 +18927,7 @@ var NO_EVALUATOR_NOTES_HINT, TERMINAL_RUN_STATUSES;
18866
18927
  var init_eval_format = __esm({
18867
18928
  "src/lib/eval-format.ts"() {
18868
18929
  "use strict";
18930
+ init_contracts();
18869
18931
  NO_EVALUATOR_NOTES_HINT = "No evaluator notes found \u2014 add a `notes` evaluation variable to the message_evaluator agent to surface its reasoning here.";
18870
18932
  TERMINAL_RUN_STATUSES = /* @__PURE__ */ new Set(["completed", "failed"]);
18871
18933
  }
@@ -20002,17 +20064,22 @@ var init_eval_session_control = __esm({
20002
20064
  });
20003
20065
 
20004
20066
  // src/lib/eval-fixture-queue.ts
20067
+ function isQueueableLaunchError(err) {
20068
+ const code = extractApiErrorCode(err);
20069
+ return code !== null && QUEUEABLE_LAUNCH_ERRORS.has(code);
20070
+ }
20005
20071
  async function runWithFixtureQueue(opts) {
20006
20072
  let delay2 = FIXTURE_QUEUE_INITIAL_DELAY_MS;
20007
20073
  for (; ; ) {
20008
20074
  try {
20009
20075
  return await opts.attempt();
20010
20076
  } catch (err) {
20011
- if (extractApiErrorCode(err) !== QUEUEABLE_CONFLICT) throw err;
20077
+ if (!isQueueableLaunchError(err)) throw err;
20012
20078
  if (opts.noQueue) throw err;
20013
20079
  const remaining = opts.deadlineMs - opts.now();
20014
20080
  if (remaining <= 0) throw err;
20015
- const waitMs = Math.min(delay2, remaining);
20081
+ const retryAfterMs = (extractApiRetryAfterSeconds(err) ?? 0) * 1e3;
20082
+ const waitMs = Math.min(Math.max(delay2, retryAfterMs), remaining);
20016
20083
  opts.onWait({ message: extractApiMessage(err), waitMs });
20017
20084
  await opts.sleep(waitMs);
20018
20085
  if (await opts.interrupted?.()) throw err;
@@ -20021,14 +20088,17 @@ async function runWithFixtureQueue(opts) {
20021
20088
  }
20022
20089
  }
20023
20090
  }
20024
- var FIXTURE_QUEUE_INITIAL_DELAY_MS, FIXTURE_QUEUE_MAX_DELAY_MS, QUEUEABLE_CONFLICT;
20091
+ var FIXTURE_QUEUE_INITIAL_DELAY_MS, FIXTURE_QUEUE_MAX_DELAY_MS, QUEUEABLE_LAUNCH_ERRORS;
20025
20092
  var init_eval_fixture_queue = __esm({
20026
20093
  "src/lib/eval-fixture-queue.ts"() {
20027
20094
  "use strict";
20028
20095
  init_errors2();
20029
20096
  FIXTURE_QUEUE_INITIAL_DELAY_MS = 5e3;
20030
20097
  FIXTURE_QUEUE_MAX_DELAY_MS = 3e4;
20031
- QUEUEABLE_CONFLICT = "fixture_target_in_use";
20098
+ QUEUEABLE_LAUNCH_ERRORS = /* @__PURE__ */ new Set([
20099
+ "fixture_target_in_use",
20100
+ "fixture_seed_unavailable"
20101
+ ]);
20032
20102
  }
20033
20103
  });
20034
20104
 
@@ -20204,8 +20274,8 @@ async function pollEvalSession(input) {
20204
20274
  if (await signalController.interrupted()) return;
20205
20275
  console.error(`
20206
20276
  Failed to start session: ${extractApiMessage(err)}`);
20207
- if (extractApiErrorCode(err) === "fixture_target_in_use" && !noQueue) {
20208
- console.error(`Waited ${timeoutSeconds}s for the fixture. Raise --timeout, or use --no-queue to fail immediately.`);
20277
+ if (isQueueableLaunchError(err) && !noQueue) {
20278
+ console.error(`Waited ${timeoutSeconds}s for the fixture to become available. Raise --timeout, or use --no-queue to fail immediately.`);
20209
20279
  }
20210
20280
  console.error(`Session ID: ${sessionId}`);
20211
20281
  console.error(`Stop it if it started anyway: ${manualStopCommand}`);
@@ -20243,7 +20313,7 @@ Lost connection to the eval session: ${extractApiMessage(err)}`);
20243
20313
  const session = details.data.session;
20244
20314
  const results = details.data.results;
20245
20315
  const totalRuns = results.reduce((sum, r) => sum + r.total_runs, 0);
20246
- const completedRuns = results.reduce((sum, r) => sum + r.successful_runs + r.failed_runs, 0);
20316
+ const completedRuns = results.reduce((sum, r) => sum + terminalRunCount(r), 0);
20247
20317
  if (TERMINAL_STATUSES.has(session.session_status)) {
20248
20318
  if (await signalController.interrupted()) return;
20249
20319
  console.log("");
@@ -25186,7 +25256,7 @@ Flags:
25186
25256
  --connection <name> MCP connection to re-sync \u2014 display name or UUID (sync-mcp)
25187
25257
  --enabled/--disabled Filter evals by status (evals)
25188
25258
  --no-wait Don't wait for eval session to complete (run-eval)
25189
- --no-queue Fail immediately if another session holds the fixture (run-eval; implied by --no-wait)
25259
+ --no-queue Fail immediately instead of waiting for the fixture to free up (run-eval; implied by --no-wait)
25190
25260
  --timeout <seconds> Max wait time for eval session (default: 600)
25191
25261
  --pacing <preset|ms> Run pacing: conservative | balanced | fast | <milliseconds> (run-eval; default balanced)
25192
25262
  --fixture <name> Override the suite's declared seed fixture for this run (run-eval)