@wayai/cli 0.3.129 → 0.3.130

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -5004,6 +5004,15 @@ function collectTurnAttachmentHashes(turn) {
5004
5004
  }
5005
5005
  return hashes;
5006
5006
  }
5007
+ function evaluatedRunCount(counts) {
5008
+ return counts.successful_runs + (counts.assertion_failure_runs ?? counts.failed_runs);
5009
+ }
5010
+ function unprovenRunCount(counts) {
5011
+ return (counts.errored_runs ?? 0) + (counts.invalid_runs ?? 0);
5012
+ }
5013
+ function terminalRunCount(counts) {
5014
+ return counts.successful_runs + counts.failed_runs;
5015
+ }
5007
5016
  function isSafeTemplatePath(path31) {
5008
5017
  if (path31.length === 0 || path31.length > 300) return false;
5009
5018
  if (path31.startsWith("/") || path31.includes("\\")) return false;
@@ -6869,7 +6878,17 @@ var init_contracts = __esm({
6869
6878
  total_evals: external_exports.number(),
6870
6879
  total_runs: external_exports.number(),
6871
6880
  successful_runs: external_exports.number(),
6881
+ /**
6882
+ * Every terminal run that did NOT pass, whatever the reason. Unchanged
6883
+ * meaning — deliberately NOT narrowed to behavioral misses, so a released CLI
6884
+ * that predates the split cannot read an infrastructure wipeout as a clean
6885
+ * sweep. Use `assertion_failure_runs` for the narrow count.
6886
+ */
6872
6887
  failed_runs: external_exports.number(),
6888
+ /** The `failed_runs` breakdown by cause (see `EvalRunOutcomeKind`); sums to it. */
6889
+ assertion_failure_runs: external_exports.number(),
6890
+ errored_runs: external_exports.number(),
6891
+ invalid_runs: external_exports.number(),
6873
6892
  created_by: external_exports.string(),
6874
6893
  /** Optional — sessions created pre-PR2 may not have this. */
6875
6894
  scenario_set_id: external_exports.string().nullable().optional(),
@@ -11900,7 +11919,17 @@ var init_dist = __esm({
11900
11919
  total_evals: external_exports.number(),
11901
11920
  total_runs: external_exports.number(),
11902
11921
  successful_runs: external_exports.number(),
11922
+ /**
11923
+ * Every terminal run that did NOT pass, whatever the reason. Unchanged
11924
+ * meaning — deliberately NOT narrowed to behavioral misses, so a released CLI
11925
+ * that predates the split cannot read an infrastructure wipeout as a clean
11926
+ * sweep. Use `assertion_failure_runs` for the narrow count.
11927
+ */
11903
11928
  failed_runs: external_exports.number(),
11929
+ /** The `failed_runs` breakdown by cause (see `EvalRunOutcomeKind`); sums to it. */
11930
+ assertion_failure_runs: external_exports.number(),
11931
+ errored_runs: external_exports.number(),
11932
+ invalid_runs: external_exports.number(),
11904
11933
  created_by: external_exports.string(),
11905
11934
  /** Optional — sessions created pre-PR2 may not have this. */
11906
11935
  scenario_set_id: external_exports.string().nullable().optional(),
@@ -18799,15 +18828,24 @@ var init_send_message = __esm({
18799
18828
  });
18800
18829
 
18801
18830
  // src/lib/eval-format.ts
18831
+ function runNotes(unproven, inProgress) {
18832
+ return [
18833
+ unproven > 0 ? `${unproven} unproven` : "",
18834
+ inProgress > 0 ? `${inProgress} in progress` : ""
18835
+ ].filter(Boolean);
18836
+ }
18802
18837
  function printResultsTable(results, mode) {
18803
18838
  let totalPassed = 0;
18804
- let totalTerminal = 0;
18839
+ let totalEvaluated = 0;
18840
+ let totalUnproven = 0;
18805
18841
  let totalInProgress = 0;
18806
18842
  for (const r of results) {
18807
18843
  const evalName = r.eval?.eval_name || r.eval_id.slice(0, 8);
18808
- const terminalRuns = r.successful_runs + r.failed_runs;
18844
+ const evaluatedRuns = evaluatedRunCount(r);
18845
+ const unproven = unprovenRunCount(r);
18809
18846
  const inProgress = (r.running_runs ?? 0) + (r.pending_runs ?? 0);
18810
- const passed = inProgress > 0 ? `${r.successful_runs}/${terminalRuns} passed (+${inProgress} in progress)` : `${r.successful_runs}/${terminalRuns} passed`;
18847
+ const notes2 = runNotes(unproven, inProgress);
18848
+ const passed = notes2.length > 0 ? `${r.successful_runs}/${evaluatedRuns} passed (${notes2.join(", ")})` : `${r.successful_runs}/${evaluatedRuns} passed`;
18811
18849
  const avgTime = r.avg_execution_time_ms != null ? `avg ${(r.avg_execution_time_ms / 1e3).toFixed(1)}s` : "";
18812
18850
  let scoresStr = "";
18813
18851
  if (r.aggregated_scores) {
@@ -18818,13 +18856,15 @@ function printResultsTable(results, mode) {
18818
18856
  }
18819
18857
  console.log(` ${evalName.padEnd(20)} ${passed.padEnd(14)} ${avgTime}${scoresStr}`);
18820
18858
  totalPassed += r.successful_runs;
18821
- totalTerminal += terminalRuns;
18859
+ totalEvaluated += evaluatedRuns;
18860
+ totalUnproven += unproven;
18822
18861
  totalInProgress += inProgress;
18823
18862
  }
18824
- const pct = totalTerminal > 0 ? (totalPassed / totalTerminal * 100).toFixed(1) : "0.0";
18825
- const inProgressNote = totalInProgress > 0 ? ` \u2014 ${totalInProgress} in progress` : "";
18863
+ const pct = totalEvaluated > 0 ? (totalPassed / totalEvaluated * 100).toFixed(1) : "0.0";
18864
+ const notes = runNotes(totalUnproven, totalInProgress);
18865
+ const note = notes.length > 0 ? ` \u2014 ${notes.join(", ")}` : "";
18826
18866
  console.log(`
18827
- Overall: ${totalPassed}/${totalTerminal} passed (${pct}%)${inProgressNote}`);
18867
+ Overall: ${totalPassed}/${totalEvaluated} passed (${pct}%)${note}`);
18828
18868
  }
18829
18869
  function truncate(str, maxLen) {
18830
18870
  if (str.length <= maxLen) return str;
@@ -18846,6 +18886,17 @@ function isFailingRun(run) {
18846
18886
  if (run.run_status && !TERMINAL_RUN_STATUSES.has(run.run_status)) return false;
18847
18887
  return run.response_match !== true;
18848
18888
  }
18889
+ function failureTag(run) {
18890
+ switch (run.outcome_kind) {
18891
+ case "execution_error":
18892
+ return "ERROR";
18893
+ case "evaluation_invalid":
18894
+ return "INVALID";
18895
+ // `assertion_failure`, and any pre-taxonomy payload.
18896
+ default:
18897
+ return "FAIL";
18898
+ }
18899
+ }
18849
18900
  function printFailingRuns(runs) {
18850
18901
  const failing = runs.filter(isFailingRun);
18851
18902
  if (failing.length === 0) return;
@@ -18854,7 +18905,7 @@ Failures (${failing.length}):`);
18854
18905
  let anyComment = false;
18855
18906
  for (const run of failing) {
18856
18907
  const evalName = run.eval?.eval_name || run.eval_id?.slice(0, 8) || "eval";
18857
- console.log(` ${evalName} \u2014 Run #${run.run_number} [FAIL]`);
18908
+ console.log(` ${evalName} \u2014 Run #${run.run_number} [${failureTag(run)}]`);
18858
18909
  if (run.error_message) {
18859
18910
  console.log(` Error: ${run.error_message}`);
18860
18911
  }
@@ -18876,6 +18927,7 @@ var NO_EVALUATOR_NOTES_HINT, TERMINAL_RUN_STATUSES;
18876
18927
  var init_eval_format = __esm({
18877
18928
  "src/lib/eval-format.ts"() {
18878
18929
  "use strict";
18930
+ init_contracts();
18879
18931
  NO_EVALUATOR_NOTES_HINT = "No evaluator notes found \u2014 add a `notes` evaluation variable to the message_evaluator agent to surface its reasoning here.";
18880
18932
  TERMINAL_RUN_STATUSES = /* @__PURE__ */ new Set(["completed", "failed"]);
18881
18933
  }
@@ -20261,7 +20313,7 @@ Lost connection to the eval session: ${extractApiMessage(err)}`);
20261
20313
  const session = details.data.session;
20262
20314
  const results = details.data.results;
20263
20315
  const totalRuns = results.reduce((sum, r) => sum + r.total_runs, 0);
20264
- const completedRuns = results.reduce((sum, r) => sum + r.successful_runs + r.failed_runs, 0);
20316
+ const completedRuns = results.reduce((sum, r) => sum + terminalRunCount(r), 0);
20265
20317
  if (TERMINAL_STATUSES.has(session.session_status)) {
20266
20318
  if (await signalController.interrupted()) return;
20267
20319
  console.log("");