@wayai/cli 0.3.128 → 0.3.130
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +86 -16
- package/dist/index.js.map +1 -1
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -5004,6 +5004,15 @@ function collectTurnAttachmentHashes(turn) {
|
|
|
5004
5004
|
}
|
|
5005
5005
|
return hashes;
|
|
5006
5006
|
}
|
|
5007
|
+
function evaluatedRunCount(counts) {
|
|
5008
|
+
return counts.successful_runs + (counts.assertion_failure_runs ?? counts.failed_runs);
|
|
5009
|
+
}
|
|
5010
|
+
function unprovenRunCount(counts) {
|
|
5011
|
+
return (counts.errored_runs ?? 0) + (counts.invalid_runs ?? 0);
|
|
5012
|
+
}
|
|
5013
|
+
function terminalRunCount(counts) {
|
|
5014
|
+
return counts.successful_runs + counts.failed_runs;
|
|
5015
|
+
}
|
|
5007
5016
|
function isSafeTemplatePath(path31) {
|
|
5008
5017
|
if (path31.length === 0 || path31.length > 300) return false;
|
|
5009
5018
|
if (path31.startsWith("/") || path31.includes("\\")) return false;
|
|
@@ -6869,7 +6878,17 @@ var init_contracts = __esm({
|
|
|
6869
6878
|
total_evals: external_exports.number(),
|
|
6870
6879
|
total_runs: external_exports.number(),
|
|
6871
6880
|
successful_runs: external_exports.number(),
|
|
6881
|
+
/**
|
|
6882
|
+
* Every terminal run that did NOT pass, whatever the reason. Unchanged
|
|
6883
|
+
* meaning — deliberately NOT narrowed to behavioral misses, so a released CLI
|
|
6884
|
+
* that predates the split cannot read an infrastructure wipeout as a clean
|
|
6885
|
+
* sweep. Use `assertion_failure_runs` for the narrow count.
|
|
6886
|
+
*/
|
|
6872
6887
|
failed_runs: external_exports.number(),
|
|
6888
|
+
/** The `failed_runs` breakdown by cause (see `EvalRunOutcomeKind`); sums to it. */
|
|
6889
|
+
assertion_failure_runs: external_exports.number(),
|
|
6890
|
+
errored_runs: external_exports.number(),
|
|
6891
|
+
invalid_runs: external_exports.number(),
|
|
6873
6892
|
created_by: external_exports.string(),
|
|
6874
6893
|
/** Optional — sessions created pre-PR2 may not have this. */
|
|
6875
6894
|
scenario_set_id: external_exports.string().nullable().optional(),
|
|
@@ -9388,6 +9407,16 @@ function extractApiErrorCode(err) {
|
|
|
9388
9407
|
return null;
|
|
9389
9408
|
}
|
|
9390
9409
|
}
|
|
9410
|
+
function extractApiRetryAfterSeconds(err) {
|
|
9411
|
+
if (!(err instanceof ApiError)) return null;
|
|
9412
|
+
try {
|
|
9413
|
+
const parsed = JSON.parse(err.body);
|
|
9414
|
+
const seconds = parsed?.details?.retry_after_seconds;
|
|
9415
|
+
return typeof seconds === "number" && Number.isFinite(seconds) && seconds >= 0 ? seconds : null;
|
|
9416
|
+
} catch {
|
|
9417
|
+
return null;
|
|
9418
|
+
}
|
|
9419
|
+
}
|
|
9391
9420
|
function friendlyHint(err) {
|
|
9392
9421
|
if (isNetworkError(err)) {
|
|
9393
9422
|
return "Couldn't reach WayAI \u2014 check your network connection and try again.";
|
|
@@ -11890,7 +11919,17 @@ var init_dist = __esm({
|
|
|
11890
11919
|
total_evals: external_exports.number(),
|
|
11891
11920
|
total_runs: external_exports.number(),
|
|
11892
11921
|
successful_runs: external_exports.number(),
|
|
11922
|
+
/**
|
|
11923
|
+
* Every terminal run that did NOT pass, whatever the reason. Unchanged
|
|
11924
|
+
* meaning — deliberately NOT narrowed to behavioral misses, so a released CLI
|
|
11925
|
+
* that predates the split cannot read an infrastructure wipeout as a clean
|
|
11926
|
+
* sweep. Use `assertion_failure_runs` for the narrow count.
|
|
11927
|
+
*/
|
|
11893
11928
|
failed_runs: external_exports.number(),
|
|
11929
|
+
/** The `failed_runs` breakdown by cause (see `EvalRunOutcomeKind`); sums to it. */
|
|
11930
|
+
assertion_failure_runs: external_exports.number(),
|
|
11931
|
+
errored_runs: external_exports.number(),
|
|
11932
|
+
invalid_runs: external_exports.number(),
|
|
11894
11933
|
created_by: external_exports.string(),
|
|
11895
11934
|
/** Optional — sessions created pre-PR2 may not have this. */
|
|
11896
11935
|
scenario_set_id: external_exports.string().nullable().optional(),
|
|
@@ -18789,15 +18828,24 @@ var init_send_message = __esm({
|
|
|
18789
18828
|
});
|
|
18790
18829
|
|
|
18791
18830
|
// src/lib/eval-format.ts
|
|
18831
|
+
function runNotes(unproven, inProgress) {
|
|
18832
|
+
return [
|
|
18833
|
+
unproven > 0 ? `${unproven} unproven` : "",
|
|
18834
|
+
inProgress > 0 ? `${inProgress} in progress` : ""
|
|
18835
|
+
].filter(Boolean);
|
|
18836
|
+
}
|
|
18792
18837
|
function printResultsTable(results, mode) {
|
|
18793
18838
|
let totalPassed = 0;
|
|
18794
|
-
let
|
|
18839
|
+
let totalEvaluated = 0;
|
|
18840
|
+
let totalUnproven = 0;
|
|
18795
18841
|
let totalInProgress = 0;
|
|
18796
18842
|
for (const r of results) {
|
|
18797
18843
|
const evalName = r.eval?.eval_name || r.eval_id.slice(0, 8);
|
|
18798
|
-
const
|
|
18844
|
+
const evaluatedRuns = evaluatedRunCount(r);
|
|
18845
|
+
const unproven = unprovenRunCount(r);
|
|
18799
18846
|
const inProgress = (r.running_runs ?? 0) + (r.pending_runs ?? 0);
|
|
18800
|
-
const
|
|
18847
|
+
const notes2 = runNotes(unproven, inProgress);
|
|
18848
|
+
const passed = notes2.length > 0 ? `${r.successful_runs}/${evaluatedRuns} passed (${notes2.join(", ")})` : `${r.successful_runs}/${evaluatedRuns} passed`;
|
|
18801
18849
|
const avgTime = r.avg_execution_time_ms != null ? `avg ${(r.avg_execution_time_ms / 1e3).toFixed(1)}s` : "";
|
|
18802
18850
|
let scoresStr = "";
|
|
18803
18851
|
if (r.aggregated_scores) {
|
|
@@ -18808,13 +18856,15 @@ function printResultsTable(results, mode) {
|
|
|
18808
18856
|
}
|
|
18809
18857
|
console.log(` ${evalName.padEnd(20)} ${passed.padEnd(14)} ${avgTime}${scoresStr}`);
|
|
18810
18858
|
totalPassed += r.successful_runs;
|
|
18811
|
-
|
|
18859
|
+
totalEvaluated += evaluatedRuns;
|
|
18860
|
+
totalUnproven += unproven;
|
|
18812
18861
|
totalInProgress += inProgress;
|
|
18813
18862
|
}
|
|
18814
|
-
const pct =
|
|
18815
|
-
const
|
|
18863
|
+
const pct = totalEvaluated > 0 ? (totalPassed / totalEvaluated * 100).toFixed(1) : "0.0";
|
|
18864
|
+
const notes = runNotes(totalUnproven, totalInProgress);
|
|
18865
|
+
const note = notes.length > 0 ? ` \u2014 ${notes.join(", ")}` : "";
|
|
18816
18866
|
console.log(`
|
|
18817
|
-
Overall: ${totalPassed}/${
|
|
18867
|
+
Overall: ${totalPassed}/${totalEvaluated} passed (${pct}%)${note}`);
|
|
18818
18868
|
}
|
|
18819
18869
|
function truncate(str, maxLen) {
|
|
18820
18870
|
if (str.length <= maxLen) return str;
|
|
@@ -18836,6 +18886,17 @@ function isFailingRun(run) {
|
|
|
18836
18886
|
if (run.run_status && !TERMINAL_RUN_STATUSES.has(run.run_status)) return false;
|
|
18837
18887
|
return run.response_match !== true;
|
|
18838
18888
|
}
|
|
18889
|
+
function failureTag(run) {
|
|
18890
|
+
switch (run.outcome_kind) {
|
|
18891
|
+
case "execution_error":
|
|
18892
|
+
return "ERROR";
|
|
18893
|
+
case "evaluation_invalid":
|
|
18894
|
+
return "INVALID";
|
|
18895
|
+
// `assertion_failure`, and any pre-taxonomy payload.
|
|
18896
|
+
default:
|
|
18897
|
+
return "FAIL";
|
|
18898
|
+
}
|
|
18899
|
+
}
|
|
18839
18900
|
function printFailingRuns(runs) {
|
|
18840
18901
|
const failing = runs.filter(isFailingRun);
|
|
18841
18902
|
if (failing.length === 0) return;
|
|
@@ -18844,7 +18905,7 @@ Failures (${failing.length}):`);
|
|
|
18844
18905
|
let anyComment = false;
|
|
18845
18906
|
for (const run of failing) {
|
|
18846
18907
|
const evalName = run.eval?.eval_name || run.eval_id?.slice(0, 8) || "eval";
|
|
18847
|
-
console.log(` ${evalName} \u2014 Run #${run.run_number} [
|
|
18908
|
+
console.log(` ${evalName} \u2014 Run #${run.run_number} [${failureTag(run)}]`);
|
|
18848
18909
|
if (run.error_message) {
|
|
18849
18910
|
console.log(` Error: ${run.error_message}`);
|
|
18850
18911
|
}
|
|
@@ -18866,6 +18927,7 @@ var NO_EVALUATOR_NOTES_HINT, TERMINAL_RUN_STATUSES;
|
|
|
18866
18927
|
var init_eval_format = __esm({
|
|
18867
18928
|
"src/lib/eval-format.ts"() {
|
|
18868
18929
|
"use strict";
|
|
18930
|
+
init_contracts();
|
|
18869
18931
|
NO_EVALUATOR_NOTES_HINT = "No evaluator notes found \u2014 add a `notes` evaluation variable to the message_evaluator agent to surface its reasoning here.";
|
|
18870
18932
|
TERMINAL_RUN_STATUSES = /* @__PURE__ */ new Set(["completed", "failed"]);
|
|
18871
18933
|
}
|
|
@@ -20002,17 +20064,22 @@ var init_eval_session_control = __esm({
|
|
|
20002
20064
|
});
|
|
20003
20065
|
|
|
20004
20066
|
// src/lib/eval-fixture-queue.ts
|
|
20067
|
+
function isQueueableLaunchError(err) {
|
|
20068
|
+
const code = extractApiErrorCode(err);
|
|
20069
|
+
return code !== null && QUEUEABLE_LAUNCH_ERRORS.has(code);
|
|
20070
|
+
}
|
|
20005
20071
|
async function runWithFixtureQueue(opts) {
|
|
20006
20072
|
let delay2 = FIXTURE_QUEUE_INITIAL_DELAY_MS;
|
|
20007
20073
|
for (; ; ) {
|
|
20008
20074
|
try {
|
|
20009
20075
|
return await opts.attempt();
|
|
20010
20076
|
} catch (err) {
|
|
20011
|
-
if (
|
|
20077
|
+
if (!isQueueableLaunchError(err)) throw err;
|
|
20012
20078
|
if (opts.noQueue) throw err;
|
|
20013
20079
|
const remaining = opts.deadlineMs - opts.now();
|
|
20014
20080
|
if (remaining <= 0) throw err;
|
|
20015
|
-
const
|
|
20081
|
+
const retryAfterMs = (extractApiRetryAfterSeconds(err) ?? 0) * 1e3;
|
|
20082
|
+
const waitMs = Math.min(Math.max(delay2, retryAfterMs), remaining);
|
|
20016
20083
|
opts.onWait({ message: extractApiMessage(err), waitMs });
|
|
20017
20084
|
await opts.sleep(waitMs);
|
|
20018
20085
|
if (await opts.interrupted?.()) throw err;
|
|
@@ -20021,14 +20088,17 @@ async function runWithFixtureQueue(opts) {
|
|
|
20021
20088
|
}
|
|
20022
20089
|
}
|
|
20023
20090
|
}
|
|
20024
|
-
var FIXTURE_QUEUE_INITIAL_DELAY_MS, FIXTURE_QUEUE_MAX_DELAY_MS,
|
|
20091
|
+
var FIXTURE_QUEUE_INITIAL_DELAY_MS, FIXTURE_QUEUE_MAX_DELAY_MS, QUEUEABLE_LAUNCH_ERRORS;
|
|
20025
20092
|
var init_eval_fixture_queue = __esm({
|
|
20026
20093
|
"src/lib/eval-fixture-queue.ts"() {
|
|
20027
20094
|
"use strict";
|
|
20028
20095
|
init_errors2();
|
|
20029
20096
|
FIXTURE_QUEUE_INITIAL_DELAY_MS = 5e3;
|
|
20030
20097
|
FIXTURE_QUEUE_MAX_DELAY_MS = 3e4;
|
|
20031
|
-
|
|
20098
|
+
QUEUEABLE_LAUNCH_ERRORS = /* @__PURE__ */ new Set([
|
|
20099
|
+
"fixture_target_in_use",
|
|
20100
|
+
"fixture_seed_unavailable"
|
|
20101
|
+
]);
|
|
20032
20102
|
}
|
|
20033
20103
|
});
|
|
20034
20104
|
|
|
@@ -20204,8 +20274,8 @@ async function pollEvalSession(input) {
|
|
|
20204
20274
|
if (await signalController.interrupted()) return;
|
|
20205
20275
|
console.error(`
|
|
20206
20276
|
Failed to start session: ${extractApiMessage(err)}`);
|
|
20207
|
-
if (
|
|
20208
|
-
console.error(`Waited ${timeoutSeconds}s for the fixture. Raise --timeout, or use --no-queue to fail immediately.`);
|
|
20277
|
+
if (isQueueableLaunchError(err) && !noQueue) {
|
|
20278
|
+
console.error(`Waited ${timeoutSeconds}s for the fixture to become available. Raise --timeout, or use --no-queue to fail immediately.`);
|
|
20209
20279
|
}
|
|
20210
20280
|
console.error(`Session ID: ${sessionId}`);
|
|
20211
20281
|
console.error(`Stop it if it started anyway: ${manualStopCommand}`);
|
|
@@ -20243,7 +20313,7 @@ Lost connection to the eval session: ${extractApiMessage(err)}`);
|
|
|
20243
20313
|
const session = details.data.session;
|
|
20244
20314
|
const results = details.data.results;
|
|
20245
20315
|
const totalRuns = results.reduce((sum, r) => sum + r.total_runs, 0);
|
|
20246
|
-
const completedRuns = results.reduce((sum, r) => sum + r
|
|
20316
|
+
const completedRuns = results.reduce((sum, r) => sum + terminalRunCount(r), 0);
|
|
20247
20317
|
if (TERMINAL_STATUSES.has(session.session_status)) {
|
|
20248
20318
|
if (await signalController.interrupted()) return;
|
|
20249
20319
|
console.log("");
|
|
@@ -25186,7 +25256,7 @@ Flags:
|
|
|
25186
25256
|
--connection <name> MCP connection to re-sync \u2014 display name or UUID (sync-mcp)
|
|
25187
25257
|
--enabled/--disabled Filter evals by status (evals)
|
|
25188
25258
|
--no-wait Don't wait for eval session to complete (run-eval)
|
|
25189
|
-
--no-queue Fail immediately
|
|
25259
|
+
--no-queue Fail immediately instead of waiting for the fixture to free up (run-eval; implied by --no-wait)
|
|
25190
25260
|
--timeout <seconds> Max wait time for eval session (default: 600)
|
|
25191
25261
|
--pacing <preset|ms> Run pacing: conservative | balanced | fast | <milliseconds> (run-eval; default balanced)
|
|
25192
25262
|
--fixture <name> Override the suite's declared seed fixture for this run (run-eval)
|