@wayai/cli 0.3.129 → 0.3.130
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +61 -9
- package/dist/index.js.map +1 -1
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -5004,6 +5004,15 @@ function collectTurnAttachmentHashes(turn) {
|
|
|
5004
5004
|
}
|
|
5005
5005
|
return hashes;
|
|
5006
5006
|
}
|
|
5007
|
+
function evaluatedRunCount(counts) {
|
|
5008
|
+
return counts.successful_runs + (counts.assertion_failure_runs ?? counts.failed_runs);
|
|
5009
|
+
}
|
|
5010
|
+
function unprovenRunCount(counts) {
|
|
5011
|
+
return (counts.errored_runs ?? 0) + (counts.invalid_runs ?? 0);
|
|
5012
|
+
}
|
|
5013
|
+
function terminalRunCount(counts) {
|
|
5014
|
+
return counts.successful_runs + counts.failed_runs;
|
|
5015
|
+
}
|
|
5007
5016
|
function isSafeTemplatePath(path31) {
|
|
5008
5017
|
if (path31.length === 0 || path31.length > 300) return false;
|
|
5009
5018
|
if (path31.startsWith("/") || path31.includes("\\")) return false;
|
|
@@ -6869,7 +6878,17 @@ var init_contracts = __esm({
|
|
|
6869
6878
|
total_evals: external_exports.number(),
|
|
6870
6879
|
total_runs: external_exports.number(),
|
|
6871
6880
|
successful_runs: external_exports.number(),
|
|
6881
|
+
/**
|
|
6882
|
+
* Every terminal run that did NOT pass, whatever the reason. Unchanged
|
|
6883
|
+
* meaning — deliberately NOT narrowed to behavioral misses, so a released CLI
|
|
6884
|
+
* that predates the split cannot read an infrastructure wipeout as a clean
|
|
6885
|
+
* sweep. Use `assertion_failure_runs` for the narrow count.
|
|
6886
|
+
*/
|
|
6872
6887
|
failed_runs: external_exports.number(),
|
|
6888
|
+
/** The `failed_runs` breakdown by cause (see `EvalRunOutcomeKind`); sums to it. */
|
|
6889
|
+
assertion_failure_runs: external_exports.number(),
|
|
6890
|
+
errored_runs: external_exports.number(),
|
|
6891
|
+
invalid_runs: external_exports.number(),
|
|
6873
6892
|
created_by: external_exports.string(),
|
|
6874
6893
|
/** Optional — sessions created pre-PR2 may not have this. */
|
|
6875
6894
|
scenario_set_id: external_exports.string().nullable().optional(),
|
|
@@ -11900,7 +11919,17 @@ var init_dist = __esm({
|
|
|
11900
11919
|
total_evals: external_exports.number(),
|
|
11901
11920
|
total_runs: external_exports.number(),
|
|
11902
11921
|
successful_runs: external_exports.number(),
|
|
11922
|
+
/**
|
|
11923
|
+
* Every terminal run that did NOT pass, whatever the reason. Unchanged
|
|
11924
|
+
* meaning — deliberately NOT narrowed to behavioral misses, so a released CLI
|
|
11925
|
+
* that predates the split cannot read an infrastructure wipeout as a clean
|
|
11926
|
+
* sweep. Use `assertion_failure_runs` for the narrow count.
|
|
11927
|
+
*/
|
|
11903
11928
|
failed_runs: external_exports.number(),
|
|
11929
|
+
/** The `failed_runs` breakdown by cause (see `EvalRunOutcomeKind`); sums to it. */
|
|
11930
|
+
assertion_failure_runs: external_exports.number(),
|
|
11931
|
+
errored_runs: external_exports.number(),
|
|
11932
|
+
invalid_runs: external_exports.number(),
|
|
11904
11933
|
created_by: external_exports.string(),
|
|
11905
11934
|
/** Optional — sessions created pre-PR2 may not have this. */
|
|
11906
11935
|
scenario_set_id: external_exports.string().nullable().optional(),
|
|
@@ -18799,15 +18828,24 @@ var init_send_message = __esm({
|
|
|
18799
18828
|
});
|
|
18800
18829
|
|
|
18801
18830
|
// src/lib/eval-format.ts
|
|
18831
|
+
function runNotes(unproven, inProgress) {
|
|
18832
|
+
return [
|
|
18833
|
+
unproven > 0 ? `${unproven} unproven` : "",
|
|
18834
|
+
inProgress > 0 ? `${inProgress} in progress` : ""
|
|
18835
|
+
].filter(Boolean);
|
|
18836
|
+
}
|
|
18802
18837
|
function printResultsTable(results, mode) {
|
|
18803
18838
|
let totalPassed = 0;
|
|
18804
|
-
let
|
|
18839
|
+
let totalEvaluated = 0;
|
|
18840
|
+
let totalUnproven = 0;
|
|
18805
18841
|
let totalInProgress = 0;
|
|
18806
18842
|
for (const r of results) {
|
|
18807
18843
|
const evalName = r.eval?.eval_name || r.eval_id.slice(0, 8);
|
|
18808
|
-
const
|
|
18844
|
+
const evaluatedRuns = evaluatedRunCount(r);
|
|
18845
|
+
const unproven = unprovenRunCount(r);
|
|
18809
18846
|
const inProgress = (r.running_runs ?? 0) + (r.pending_runs ?? 0);
|
|
18810
|
-
const
|
|
18847
|
+
const notes2 = runNotes(unproven, inProgress);
|
|
18848
|
+
const passed = notes2.length > 0 ? `${r.successful_runs}/${evaluatedRuns} passed (${notes2.join(", ")})` : `${r.successful_runs}/${evaluatedRuns} passed`;
|
|
18811
18849
|
const avgTime = r.avg_execution_time_ms != null ? `avg ${(r.avg_execution_time_ms / 1e3).toFixed(1)}s` : "";
|
|
18812
18850
|
let scoresStr = "";
|
|
18813
18851
|
if (r.aggregated_scores) {
|
|
@@ -18818,13 +18856,15 @@ function printResultsTable(results, mode) {
|
|
|
18818
18856
|
}
|
|
18819
18857
|
console.log(` ${evalName.padEnd(20)} ${passed.padEnd(14)} ${avgTime}${scoresStr}`);
|
|
18820
18858
|
totalPassed += r.successful_runs;
|
|
18821
|
-
|
|
18859
|
+
totalEvaluated += evaluatedRuns;
|
|
18860
|
+
totalUnproven += unproven;
|
|
18822
18861
|
totalInProgress += inProgress;
|
|
18823
18862
|
}
|
|
18824
|
-
const pct =
|
|
18825
|
-
const
|
|
18863
|
+
const pct = totalEvaluated > 0 ? (totalPassed / totalEvaluated * 100).toFixed(1) : "0.0";
|
|
18864
|
+
const notes = runNotes(totalUnproven, totalInProgress);
|
|
18865
|
+
const note = notes.length > 0 ? ` \u2014 ${notes.join(", ")}` : "";
|
|
18826
18866
|
console.log(`
|
|
18827
|
-
Overall: ${totalPassed}/${
|
|
18867
|
+
Overall: ${totalPassed}/${totalEvaluated} passed (${pct}%)${note}`);
|
|
18828
18868
|
}
|
|
18829
18869
|
function truncate(str, maxLen) {
|
|
18830
18870
|
if (str.length <= maxLen) return str;
|
|
@@ -18846,6 +18886,17 @@ function isFailingRun(run) {
|
|
|
18846
18886
|
if (run.run_status && !TERMINAL_RUN_STATUSES.has(run.run_status)) return false;
|
|
18847
18887
|
return run.response_match !== true;
|
|
18848
18888
|
}
|
|
18889
|
+
function failureTag(run) {
|
|
18890
|
+
switch (run.outcome_kind) {
|
|
18891
|
+
case "execution_error":
|
|
18892
|
+
return "ERROR";
|
|
18893
|
+
case "evaluation_invalid":
|
|
18894
|
+
return "INVALID";
|
|
18895
|
+
// `assertion_failure`, and any pre-taxonomy payload.
|
|
18896
|
+
default:
|
|
18897
|
+
return "FAIL";
|
|
18898
|
+
}
|
|
18899
|
+
}
|
|
18849
18900
|
function printFailingRuns(runs) {
|
|
18850
18901
|
const failing = runs.filter(isFailingRun);
|
|
18851
18902
|
if (failing.length === 0) return;
|
|
@@ -18854,7 +18905,7 @@ Failures (${failing.length}):`);
|
|
|
18854
18905
|
let anyComment = false;
|
|
18855
18906
|
for (const run of failing) {
|
|
18856
18907
|
const evalName = run.eval?.eval_name || run.eval_id?.slice(0, 8) || "eval";
|
|
18857
|
-
console.log(` ${evalName} \u2014 Run #${run.run_number} [
|
|
18908
|
+
console.log(` ${evalName} \u2014 Run #${run.run_number} [${failureTag(run)}]`);
|
|
18858
18909
|
if (run.error_message) {
|
|
18859
18910
|
console.log(` Error: ${run.error_message}`);
|
|
18860
18911
|
}
|
|
@@ -18876,6 +18927,7 @@ var NO_EVALUATOR_NOTES_HINT, TERMINAL_RUN_STATUSES;
|
|
|
18876
18927
|
var init_eval_format = __esm({
|
|
18877
18928
|
"src/lib/eval-format.ts"() {
|
|
18878
18929
|
"use strict";
|
|
18930
|
+
init_contracts();
|
|
18879
18931
|
NO_EVALUATOR_NOTES_HINT = "No evaluator notes found \u2014 add a `notes` evaluation variable to the message_evaluator agent to surface its reasoning here.";
|
|
18880
18932
|
TERMINAL_RUN_STATUSES = /* @__PURE__ */ new Set(["completed", "failed"]);
|
|
18881
18933
|
}
|
|
@@ -20261,7 +20313,7 @@ Lost connection to the eval session: ${extractApiMessage(err)}`);
|
|
|
20261
20313
|
const session = details.data.session;
|
|
20262
20314
|
const results = details.data.results;
|
|
20263
20315
|
const totalRuns = results.reduce((sum, r) => sum + r.total_runs, 0);
|
|
20264
|
-
const completedRuns = results.reduce((sum, r) => sum + r
|
|
20316
|
+
const completedRuns = results.reduce((sum, r) => sum + terminalRunCount(r), 0);
|
|
20265
20317
|
if (TERMINAL_STATUSES.has(session.session_status)) {
|
|
20266
20318
|
if (await signalController.interrupted()) return;
|
|
20267
20319
|
console.log("");
|