@evolvingmachines/sdk 0.0.52-project-sable.20260730.b88a70e → 0.0.52-project-sable.20260730.c2cb2d2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/hosted/cli.cjs +7 -3
- package/dist/hosted/cli.d.cts +1 -1
- package/dist/hosted/cli.d.ts +1 -1
- package/dist/hosted/cli.js +11 -7
- package/package.json +4 -4
package/dist/hosted/cli.cjs
CHANGED
|
@@ -55,9 +55,13 @@ Regrade options (whole-job regrade only):
|
|
|
55
55
|
--status <s1,s2,...> Only regrade source trials in these statuses
|
|
56
56
|
--task <key> Only regrade source trials of this task
|
|
57
57
|
|
|
58
|
-
Trace options:
|
|
58
|
+
Trace options (--stream and --save are exclusive; --cursor/--limit page the events only):
|
|
59
59
|
--cursor <seq> Resume after this trace seq (a trace cursor IS a seq)
|
|
60
60
|
--limit <n> Max events per page
|
|
61
|
+
--stream <artifact> Print ONE raw artifact instead: verifier |
|
|
62
|
+
trace-stdout | trace-stderr | agent-home
|
|
63
|
+
--save <dir> Save everything the trial recorded into <dir>:
|
|
64
|
+
trace-parsed.jsonl, each raw log, agent-home/
|
|
61
65
|
|
|
62
66
|
Import options (a git source OR a local directory; --name and --version required):
|
|
63
67
|
--git <url> Git repository URL (with --ref)
|
|
@@ -82,11 +86,11 @@ Other options:
|
|
|
82
86
|
--format harbor Export the Harbor job-layout bundle
|
|
83
87
|
--json Machine-readable JSON output
|
|
84
88
|
--api-key <key> API key (default: $EVOLVE_API_KEY)
|
|
85
|
-
--base-url <url> API base URL (default: the Evolve dashboard API)`,g=class extends Error{constructor(t){super(t),this.name="CliUsageError";}},pt={json:"boolean","api-key":"string","base-url":"string"},ft={run:{flags:{benchmark:"string",tasks:"string",agent:"repeat",effort:"string",runs:"number",concurrency:"number","max-trial-spend":"number",provider:"string",watch:"boolean"},required:["benchmark","agent"],minPositionals:0,maxPositionals:0},list:{flags:{limit:"number",cursor:"string"},minPositionals:0,maxPositionals:0},get:{flags:{},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},trials:{flags:{status:"string",limit:"number",cursor:"string"},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},trial:{flags:{},minPositionals:2,maxPositionals:2,positionalUsage:"<id> <trial-id>"},trace:{flags:{cursor:"string",limit:"number"},minPositionals:2,maxPositionals:2,positionalUsage:"<id> <trial-id>"},compare:{flags:{},minPositionals:2,maxPositionals:5,positionalUsage:"<id> <id> [...]"},cancel:{flags:{},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},"rerun-failed":{flags:{},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},regrade:{flags:{status:"string",task:"string"},minPositionals:1,maxPositionals:2,positionalUsage:"<id> [trial-id]"},"regrade-job":{flags:{limit:"number",cursor:"string"},minPositionals:1,maxPositionals:1,positionalUsage:"<job-id>"},export:{flags:{to:"string",format:"string"},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},benchmarks:{flags:{limit:"number",cursor:"string"},minPositionals:0,maxPositionals:2},import:{flags:{git:"string",ref:"string",dir:"string",name:"string",version:"string",watch:"boolean"},minPositionals:0,maxPositionals:2},download:{flags:{to:"string"},minPositionals:1,maxPositionals:1,positionalUsage:"<import-id>"},"custom-harnesses":{flags:{name:"string","install-script":"string",dir:"string",run:"string",env:"repeat",limit:"number",cursor:"string"},minPositionals:0,maxPositionals:2},help:{flags:{},minPositionals:0,maxPositionals:0}},Ie={command:"help",positionals:[],flags:{}};function bt(e){if(e.length===0)throw new g("No command given");if(e[0]==="--help"||e[0]==="-h")return Ie;let t=e[0],r=ft[t];if(!r)throw new g(`Unknown command "${t}"`);let n={},s=[];for(let o=1;o<e.length;o++){let a=e[o];if(a==="--help"||a==="-h")return Ie;if(!a.startsWith("--")){s.push(a);continue}let i=a.slice(2),u,l=i.indexOf("=");l!==-1&&(u=i.slice(l+1),i=i.slice(0,l));let c=r.flags[i]??pt[i];if(!c)throw new g(`Unknown option --${i} for "${t}"`);if(c==="boolean"){if(u!==void 0)throw new g(`Option --${i} takes no value`);n[i]=true;continue}let d=u;if(d===void 0){let m=e[o+1];if(m===void 0||m.startsWith("--"))throw new g(`Option --${i} requires a value`);d=m,o++;}if(c==="number"){let m=Number(d);if(d.trim()===""||!Number.isFinite(m))throw new g(`Option --${i} expects a number, got "${d}"`);n[i]=m;}else if(c==="repeat"){let m=n[i]??[];m.push(d),n[i]=m;}else n[i]=d;}if(s.length<r.minPositionals)throw new g(`"${t}" requires ${r.positionalUsage??`${r.minPositionals} argument(s)`}`);if(s.length>r.maxPositionals)throw new g(`"${t}" got unexpected argument "${s[r.maxPositionals]}"`);for(let o of r.required??[])if(!(o in n))throw new g(`"${t}" requires --${o}`);return {command:t,positionals:s,flags:n}}function ht(e){let t=e.indexOf(":");if(t<=0||t===e.length-1)throw new g(`Invalid --agent "${e}": expected harness:model[:version]`);let r=e.slice(0,t),n=e.slice(t+1),s=n.indexOf(":");if(s===-1)return {harness:r,model:n};let o=n.slice(0,s),a=n.slice(s+1);if(!o||!a)throw new g(`Invalid --agent "${e}": expected harness:model[:version]`);return {harness:r,model:o,harnessVersion:a}}function yt(e){let t=e.flags,r;if(t.tasks!==void 0&&(r=String(t.tasks).split(",").map(n=>n.trim()).filter(Boolean),r.length===0))throw new g("--tasks got an empty task list");return {benchmark:t.benchmark,...r!==void 0?{tasks:r}:{},agents:t.agent.map(n=>{let s=ht(n);return t.effort!==void 0?{...s,reasoningEffort:t.effort}:s}),...t.runs!==void 0?{runsPerTask:t.runs}:{},...t.concurrency!==void 0?{concurrency:t.concurrency}:{},...t["max-trial-spend"]!==void 0?{maxTrialSpendUsd:t["max-trial-spend"]}:{},...t.provider!==void 0?{sandboxProvider:t.provider}:{}}}function kt(e){let t=e.flags,r=typeof t.dir=="string",n=typeof t.git=="string"||typeof t.ref=="string";if(r&&n)throw new g('"import" takes EITHER --dir OR --git/--ref, not both');if(r){for(let s of ["name","version"])if(typeof t[s]!="string")throw new g(`"import" requires --${s}`);return {source:{directory:t.dir},benchmarkName:t.name,version:t.version}}for(let s of ["git","ref","name","version"])if(typeof t[s]!="string"){let o=s==="git"||s==="ref"?" (or --dir for a local corpus directory)":"";throw new g(`"import" requires --${s}${o}`)}return {source:{gitUrl:t.git,ref:t.ref},benchmarkName:t.name,version:t.version}}function Rt(e){let t={};for(let r of e){let n=r.indexOf("=");if(n<=0)throw new g(`Invalid --env "${r}": expected KEY=VALUE`);t[r.slice(0,n)]=r.slice(n+1);}return t}function wt(e,t=r=>fs.readFileSync(r,"utf-8")){let r=e.flags,n=typeof r.dir=="string",s=typeof r["install-script"]=="string";if(n&&s)throw new g('"custom-harnesses add" takes EITHER --dir OR --install-script, not both');if(!n&&!s)throw new g('"custom-harnesses add" requires --install-script (or --dir for a local harness directory)');for(let a of ["name","run"])if(typeof r[a]!="string")throw new g(`"custom-harnesses add" requires --${a}`);let o=Rt(r.env??[]);return {name:r.name,...n?{directory:r.dir}:{installScript:t(r["install-script"])},runCommand:r.run,...Object.keys(o).length>0?{env:o}:{}}}var vt={out:e=>process.stdout.write(e+`
|
|
89
|
+
--base-url <url> API base URL (default: the Evolve dashboard API)`,g=class extends Error{constructor(t){super(t),this.name="CliUsageError";}},pt={json:"boolean","api-key":"string","base-url":"string"},ft={run:{flags:{benchmark:"string",tasks:"string",agent:"repeat",effort:"string",runs:"number",concurrency:"number","max-trial-spend":"number",provider:"string",watch:"boolean"},required:["benchmark","agent"],minPositionals:0,maxPositionals:0},list:{flags:{limit:"number",cursor:"string"},minPositionals:0,maxPositionals:0},get:{flags:{},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},trials:{flags:{status:"string",limit:"number",cursor:"string"},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},trial:{flags:{},minPositionals:2,maxPositionals:2,positionalUsage:"<id> <trial-id>"},trace:{flags:{cursor:"string",limit:"number",stream:"string",save:"string"},minPositionals:2,maxPositionals:2,positionalUsage:"<id> <trial-id>"},compare:{flags:{},minPositionals:2,maxPositionals:5,positionalUsage:"<id> <id> [...]"},cancel:{flags:{},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},"rerun-failed":{flags:{},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},regrade:{flags:{status:"string",task:"string"},minPositionals:1,maxPositionals:2,positionalUsage:"<id> [trial-id]"},"regrade-job":{flags:{limit:"number",cursor:"string"},minPositionals:1,maxPositionals:1,positionalUsage:"<job-id>"},export:{flags:{to:"string",format:"string"},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},benchmarks:{flags:{limit:"number",cursor:"string"},minPositionals:0,maxPositionals:2},import:{flags:{git:"string",ref:"string",dir:"string",name:"string",version:"string",watch:"boolean"},minPositionals:0,maxPositionals:2},download:{flags:{to:"string"},minPositionals:1,maxPositionals:1,positionalUsage:"<import-id>"},"custom-harnesses":{flags:{name:"string","install-script":"string",dir:"string",run:"string",env:"repeat",limit:"number",cursor:"string"},minPositionals:0,maxPositionals:2},help:{flags:{},minPositionals:0,maxPositionals:0}},Ie={command:"help",positionals:[],flags:{}};function bt(e){if(e.length===0)throw new g("No command given");if(e[0]==="--help"||e[0]==="-h")return Ie;let t=e[0],r=ft[t];if(!r)throw new g(`Unknown command "${t}"`);let n={},s=[];for(let o=1;o<e.length;o++){let a=e[o];if(a==="--help"||a==="-h")return Ie;if(!a.startsWith("--")){s.push(a);continue}let i=a.slice(2),u,l=i.indexOf("=");l!==-1&&(u=i.slice(l+1),i=i.slice(0,l));let c=r.flags[i]??pt[i];if(!c)throw new g(`Unknown option --${i} for "${t}"`);if(c==="boolean"){if(u!==void 0)throw new g(`Option --${i} takes no value`);n[i]=true;continue}let d=u;if(d===void 0){let m=e[o+1];if(m===void 0||m.startsWith("--"))throw new g(`Option --${i} requires a value`);d=m,o++;}if(c==="number"){let m=Number(d);if(d.trim()===""||!Number.isFinite(m))throw new g(`Option --${i} expects a number, got "${d}"`);n[i]=m;}else if(c==="repeat"){let m=n[i]??[];m.push(d),n[i]=m;}else n[i]=d;}if(s.length<r.minPositionals)throw new g(`"${t}" requires ${r.positionalUsage??`${r.minPositionals} argument(s)`}`);if(s.length>r.maxPositionals)throw new g(`"${t}" got unexpected argument "${s[r.maxPositionals]}"`);for(let o of r.required??[])if(!(o in n))throw new g(`"${t}" requires --${o}`);return {command:t,positionals:s,flags:n}}function ht(e){let t=e.indexOf(":");if(t<=0||t===e.length-1)throw new g(`Invalid --agent "${e}": expected harness:model[:version]`);let r=e.slice(0,t),n=e.slice(t+1),s=n.indexOf(":");if(s===-1)return {harness:r,model:n};let o=n.slice(0,s),a=n.slice(s+1);if(!o||!a)throw new g(`Invalid --agent "${e}": expected harness:model[:version]`);return {harness:r,model:o,harnessVersion:a}}function yt(e){let t=e.flags,r;if(t.tasks!==void 0&&(r=String(t.tasks).split(",").map(n=>n.trim()).filter(Boolean),r.length===0))throw new g("--tasks got an empty task list");return {benchmark:t.benchmark,...r!==void 0?{tasks:r}:{},agents:t.agent.map(n=>{let s=ht(n);return t.effort!==void 0?{...s,reasoningEffort:t.effort}:s}),...t.runs!==void 0?{runsPerTask:t.runs}:{},...t.concurrency!==void 0?{concurrency:t.concurrency}:{},...t["max-trial-spend"]!==void 0?{maxTrialSpendUsd:t["max-trial-spend"]}:{},...t.provider!==void 0?{sandboxProvider:t.provider}:{}}}function kt(e){let t=e.flags,r=typeof t.dir=="string",n=typeof t.git=="string"||typeof t.ref=="string";if(r&&n)throw new g('"import" takes EITHER --dir OR --git/--ref, not both');if(r){for(let s of ["name","version"])if(typeof t[s]!="string")throw new g(`"import" requires --${s}`);return {source:{directory:t.dir},benchmarkName:t.name,version:t.version}}for(let s of ["git","ref","name","version"])if(typeof t[s]!="string"){let o=s==="git"||s==="ref"?" (or --dir for a local corpus directory)":"";throw new g(`"import" requires --${s}${o}`)}return {source:{gitUrl:t.git,ref:t.ref},benchmarkName:t.name,version:t.version}}function Rt(e){let t={};for(let r of e){let n=r.indexOf("=");if(n<=0)throw new g(`Invalid --env "${r}": expected KEY=VALUE`);t[r.slice(0,n)]=r.slice(n+1);}return t}function wt(e,t=r=>fs.readFileSync(r,"utf-8")){let r=e.flags,n=typeof r.dir=="string",s=typeof r["install-script"]=="string";if(n&&s)throw new g('"custom-harnesses add" takes EITHER --dir OR --install-script, not both');if(!n&&!s)throw new g('"custom-harnesses add" requires --install-script (or --dir for a local harness directory)');for(let a of ["name","run"])if(typeof r[a]!="string")throw new g(`"custom-harnesses add" requires --${a}`);let o=Rt(r.env??[]);return {name:r.name,...n?{directory:r.dir}:{installScript:t(r["install-script"])},runCommand:r.run,...Object.keys(o).length>0?{env:o}:{}}}var vt={out:e=>process.stdout.write(e+`
|
|
86
90
|
`),err:e=>process.stderr.write(e+`
|
|
87
91
|
`)};function k(e){if(e.length===0)return [];let t=[];for(let r of e)r.forEach((n,s)=>{t[s]=Math.max(t[s]??0,n.length);});return e.map(r=>r.map((n,s)=>n.padEnd(t[s])).join(" ").trimEnd())}function C(e){return typeof e=="number"?`$${e.toFixed(2)}`:"-"}function ae(e){let t=`${e.harness}:${e.model}`;return e.harnessVersion?`${t}:${e.harnessVersion}`:t}function W(e,t){return e.length>t?e.slice(0,t-1)+"\u2026":e}function L(e){let t=[["id",e.id],["status",e.status],["benchmark",e.benchmark]];t.push(["agents",e.agents.map(ae).join(", ")]),t.push(["size",`${e.counts.agents} agent(s) x ${e.counts.tasks} task(s) = ${e.trials.total} trial(s)`]),t.push(["runs/task",String(e.runsPerTask)]),t.push(["concurrency",String(e.concurrency)]),t.push(["max spend/trial",C(e.maxTrialSpendUsd)]),t.push(["worst case",C(e.worstCaseSpendUsd)]),t.push(["provider",e.sandboxProvider]),t.push(["spent",C(e.spentUsd)]),t.push(["mean reward",e.meanReward!==null?String(e.meanReward):"-"]);let r=Object.entries(e.trials.byStatus).filter(([,n])=>n>0).map(([n,s])=>`${n} ${s}`).join(" \xB7 ");return r&&t.push(["trials",r]),e.sourceJobId&&t.push(["rerun of",e.sourceJobId]),e.idempotentReplay&&t.push(["note","idempotent replay of an existing job"]),e.failure&&t.push(["failure",`${e.failure.code}: ${e.failure.message}`]),t.push(["created",e.createdAt]),t.push(["updated",e.updatedAt]),k(t)}function St(e){return [e.id,e.status,e.benchmark,String(e.trials.total),D(e.meanReward),C(e.spentUsd),e.createdAt]}function It(e){return [e.taskKey,ae(e.agent),String(e.runNumber),e.status,e.reward!==null?String(e.reward):"-",C(e.spentUsd),e.id]}function xt(e){let t=[["trial id",e.id],["job",e.jobId],["task",e.taskKey],["agent",ae(e.agent)],["run",String(e.runNumber)],["status",e.status],["reward",e.reward!==null?String(e.reward):"-"]];if(e.metrics&&Object.keys(e.metrics).length>0&&t.push(["metrics",Object.entries(e.metrics).map(([r,n])=>`${r}=${n}`).join(" \xB7 ")]),t.push(["spent",C(e.spentUsd)]),(e.status==="RUNNING"||e.status==="SCORING")&&e.liveSpentUsd!==null){let r=e.liveSpendAt?` as of ${e.liveSpendAt}`:"";t.push(["spent (live)",`at least $${e.liveSpentUsd.toFixed(4)}${r}`]);}return e.sandboxProvider&&t.push(["provider",e.sandboxProvider]),e.sandboxId&&t.push(["sandbox",e.sandboxId]),e.verifierMode&&t.push(["verifier",e.verifierMode]),e.verifierSandboxId&&t.push(["verifier sandbox",e.verifierSandboxId]),e.resolvedHarnessVersion&&t.push(["harness version",e.resolvedHarnessVersion]),e.phaseTimingsMs&&Object.keys(e.phaseTimingsMs).length>0&&t.push(["timings",Object.entries(e.phaseTimingsMs).map(([r,n])=>`${r}=${n}ms`).join(" \xB7 ")]),e.failurePhase&&t.push(["failure phase",e.failurePhase]),e.failureDetail&&t.push(["failure detail",e.failureDetail]),e.sessionRef&&t.push(["session",e.sessionRef]),t.push(["created",e.createdAt]),t.push(["updated",e.updatedAt]),k(t)}function Tt(e){if(e===null)return "-";let t=Math.round(e*1e3)/1e3;return t>0?`+${t}`:String(t)}function Ct(e){return [e.taskKey,e.status,D(e.sourceReward),D(e.reward),Tt(e.rewardDelta),e.sourceTrialId]}function Ce(e){let t=[["job id",e.id],["status",e.status],["source job",e.sourceJobId],["provider",e.sandboxProvider],["results",String(e.results.total)]],r=Object.entries(e.results.byStatus).filter(([,s])=>s>0).map(([s,o])=>`${s} ${o}`).join(" \xB7 ");if(r&&t.push(["by status",r]),e.filter&&(e.filter.status?.length||e.filter.taskKey)){let s=[];e.filter.status?.length&&s.push(`status=${e.filter.status.join(",")}`),e.filter.taskKey&&s.push(`task=${e.filter.taskKey}`),t.push(["filter",s.join(" \xB7 ")]);}t.push(["created",e.createdAt]);let n=k(t);if(e.results.items.length>0){n.push("");let s=[["TASK","STATUS","WAS","NOW","\u0394","SOURCE TRIAL ID"]];for(let o of e.results.items)s.push(Ct(o));n.push(...k(s)),e.results.nextCursor&&n.push("",`More: evolve-evals regrade-job ${e.id} --cursor ${e.results.nextCursor}`);}return n}function xe(e){let t=[["name",e.name],["source",e.source],["run command",e.runCommand]],r=Object.keys(e.env??{});return r.length>0&&t.push(["env",r.sort().join(", ")]),t.push(["created",e.createdAt]),t.push(["updated",e.updatedAt]),k(t)}var Pe=["e2b","daytona","modal"];function Pt(e){return Pe.filter(t=>e?.[t]!==void 0).map(t=>`${t} ${e[t].ok?"ok":"NO"}`).join(" \xB7 ")}function D(e){return e!==null?String(Math.round(e*1e3)/1e3):"-"}function Et(e){let t=W(JSON.stringify(e.data??{}),140);return `#${String(e.seq).padStart(4)} ${e.type.padEnd(26)} ${t}`.trimEnd()}function Ee(e){let t=e.failures?.length?` (${e.failures.length} task failure${e.failures.length===1?"":"s"})`:"";return `${e.message}${t}`}function se(e){let t=[["id",e.id],["status",e.status]];if(e.benchmarkName!==void 0&&t.push(["benchmark",e.benchmarkName]),e.version!==void 0&&t.push(["version",e.version]),e.taskCount!==void 0&&t.push(["tasks",String(e.taskCount)]),e.failure){t.push(["failure",Ee(e.failure)]);for(let r of e.failure.failures??[])t.push([` ${r.taskKey}`,r.error]);}return k(t)}function At(e){let t=[];return e.taskCount!==void 0&&t.push(`tasks=${e.taskCount}`),e.failure&&t.push(W(Ee(e.failure),140)),`status ${e.status.padEnd(12)} ${t.join(" ")}`.trimEnd()}function Ot(e){let t={...e.data??{}},r=[];typeof t.trialId=="string"&&r.push(t.trialId),typeof t.taskKey=="string"&&r.push(t.taskKey),typeof t.status=="string"&&r.push(t.status),typeof t.reward=="number"&&r.push(`reward=${t.reward}`);let n=new Set(["trialId","taskKey","status","reward"]);for(let[o,a]of Object.entries(t))n.has(o)||r.push(`${o}=${typeof a=="object"?JSON.stringify(a):String(a)}`);let s=W(r.join(" "),140);return `#${String(e.seq).padStart(4)} ${e.type.padEnd(26)} ${s}`.trimEnd()}function y(e){let t={};return typeof e.flags["api-key"]=="string"&&(t.apiKey=e.flags["api-key"]),typeof e.flags["base-url"]=="string"&&(t.baseUrl=e.flags["base-url"]),t}function G(e){return {...e.flags.limit!==void 0?{limit:e.flags.limit}:{},...e.flags.cursor!==void 0?{cursor:String(e.flags.cursor)}:{}}}function Jt(e){return e.status==="COMPLETED"?0:e.status==="FAILED"||e.status==="CANCELLED"?1:0}async function _t(e,t){let r=yt(e),n=e.flags.json===true,s=e.flags.watch===true,o=R(y(e)),a=await o.run(r);if(!s){if(n)t.out(JSON.stringify(a));else {for(let u of L(a))t.out(u);t.out(""),t.out(`Follow it with: evolve-evals get ${a.id}`);}return 0}n?t.out(JSON.stringify({kind:"job.created",job:a})):t.out(`Job ${a.id} (${a.benchmark}) ${a.status} \u2014 watching\u2026`);let i=await o.watch(a.id,{onEvent:u=>{t.out(n?JSON.stringify({kind:"event",...u}):Ot(u));}});if(n)t.out(JSON.stringify({kind:"job.final",job:i}));else {t.out("");for(let u of L(i))t.out(u);}return Jt(i)}async function Ut(e,t){let n=await R(y(e)).list({...e.flags.limit!==void 0?{limit:e.flags.limit}:{},...e.flags.cursor!==void 0?{cursor:e.flags.cursor}:{}});if(e.flags.json===true)return t.out(JSON.stringify(n)),0;if(n.items.length===0)return t.out("No jobs."),0;let s=[["ID","STATUS","BENCHMARK","TRIALS","MEAN REWARD","SPENT","CREATED"]];for(let o of n.items)s.push(St(o));for(let o of k(s))t.out(o);return n.nextCursor&&t.out(`
|
|
88
92
|
More: evolve-evals list --cursor ${n.nextCursor}`),0}async function $t(e,t){let n=await R(y(e)).get(e.positionals[0]);if(e.flags.json===true)t.out(JSON.stringify(n));else for(let s of L(n))t.out(s);return 0}async function Lt(e,t){let r=R(y(e)),n;if(e.flags.status!==void 0&&(n=String(e.flags.status).split(",").map(a=>a.trim()).filter(Boolean),n.length===0))throw new g("--status got an empty status list");let s=await r.trials(e.positionals[0],{...n!==void 0?{status:n}:{},...e.flags.limit!==void 0?{limit:e.flags.limit}:{},...e.flags.cursor!==void 0?{cursor:e.flags.cursor}:{}});if(e.flags.json===true)return t.out(JSON.stringify(s)),0;if(s.items.length===0)return t.out("No trials."),0;let o=[["TASK","AGENT","RUN","STATUS","REWARD","SPENT","TRIAL ID"]];for(let a of s.items)o.push(It(a));for(let a of k(o))t.out(a);return t.out(`
|
|
89
|
-
${s.items.length} trial(s) shown`),s.nextCursor&&t.out(`More: evolve-evals trials ${e.positionals[0]} --cursor ${s.nextCursor}`),0}async function Dt(e,t){let n=await R(y(e)).trial(e.positionals[0],e.positionals[1]);if(e.flags.json===true)t.out(JSON.stringify(n));else for(let s of xt(n))t.out(s);return 0}async function Bt(e,t){let r=R(y(e)),n=e.flags.json===true,s=e.flags.stream;if(s!==void 0){if(s==="verifier"||s==="trace-stdout"||s==="trace-stderr"){let i=await r.trialArtifact(e.positionals[0],e.positionals[1],s);return i===null?(t.out(n?JSON.stringify({log:null}):`No ${s} log was stored for this trial.`),0):(t.out(n?JSON.stringify({log:i}):i),0)}if(s==="agent-home"){let i=await r.trialArtifact(e.positionals[0],e.positionals[1],s);if(i===null)return t.out(n?JSON.stringify({files:null}):`No ${s} content was stored for this trial.`),0;if(n)t.out(JSON.stringify({files:i}));else for(let[u,l]of Object.entries(i))t.out(`===== ${u} (${Buffer.byteLength(l,"utf8")} bytes) =====`),t.out(l);return 0}
|
|
93
|
+
${s.items.length} trial(s) shown`),s.nextCursor&&t.out(`More: evolve-evals trials ${e.positionals[0]} --cursor ${s.nextCursor}`),0}async function Dt(e,t){let n=await R(y(e)).trial(e.positionals[0],e.positionals[1]);if(e.flags.json===true)t.out(JSON.stringify(n));else for(let s of xt(n))t.out(s);return 0}async function Bt(e,t){let r=R(y(e)),n=e.flags.json===true,s=e.flags.stream,o=e.flags.save;if(s!==void 0&&o!==void 0)throw new g('"trace" takes EITHER --stream OR --save, not both');if((s!==void 0||o!==void 0)&&(e.flags.cursor!==void 0||e.flags.limit!==void 0))throw new g("--cursor/--limit page the parsed events; they do not apply to --stream/--save");if(s!==void 0){if(s==="verifier"||s==="trace-stdout"||s==="trace-stderr"){let i=await r.trialArtifact(e.positionals[0],e.positionals[1],s);return i===null?(t.out(n?JSON.stringify({log:null}):`No ${s} log was stored for this trial.`),0):(t.out(n?JSON.stringify({log:i}):i),0)}if(s==="agent-home"){let i=await r.trialArtifact(e.positionals[0],e.positionals[1],s);if(i===null)return t.out(n?JSON.stringify({files:null}):`No ${s} content was stored for this trial.`),0;if(n)t.out(JSON.stringify({files:i}));else for(let[u,l]of Object.entries(i))t.out(`===== ${u} (${Buffer.byteLength(l,"utf8")} bytes) =====`),t.out(l);return 0}throw new g('--stream must be "verifier", "trace-stdout", "trace-stderr" or "agent-home"')}if(o!==void 0){let{mkdir:i,writeFile:u}=await import('fs/promises'),{join:l,dirname:c}=await import('path');await i(o,{recursive:true});let d=[];for await(let f of r.trialTraceEvents(e.positionals[0],e.positionals[1]))d.push(JSON.stringify(f));await u(l(o,"trace-parsed.jsonl"),d.join(`
|
|
90
94
|
`)+(d.length?`
|
|
91
95
|
`:"")),t.out(`trace-parsed.jsonl (${d.length} events)`);for(let f of ["verifier","trace-stdout","trace-stderr"]){let h=await r.trialArtifact(e.positionals[0],e.positionals[1],f);h!==null&&(await u(l(o,`${f}.log`),h),t.out(`${f}.log (${Buffer.byteLength(h,"utf8")} bytes)`));}let m=await r.trialArtifact(e.positionals[0],e.positionals[1],"agent-home");if(m!==null){for(let[f,h]of Object.entries(m)){let S=l(o,"agent-home",...f.split("/").filter(Boolean));await i(c(S),{recursive:true}),await u(S,h);}t.out(`agent-home/ (${Object.keys(m).length} files)`);}return 0}let a=0;for await(let i of r.trialTraceEvents(e.positionals[0],e.positionals[1],{...e.flags.cursor!==void 0?{cursor:String(e.flags.cursor)}:{},...e.flags.limit!==void 0?{limit:e.flags.limit}:{}}))t.out(n?JSON.stringify(i):Et(i)),a+=1;return !n&&a===0&&t.out("No trace events."),0}function Nt(e){let t=[],r=[["ID","BENCHMARK","STATUS","MEAN REWARD","COVERAGE","SPENT"]];for(let n of e.jobs)r.push([n.id,n.benchmark,n.status,D(n.meanReward),`${n.coverage.scored}/${n.coverage.total}`,C(n.spentUsd)]);if(t.push(...k(r)),e.taskMatrix.length>0){t.push("","Task matrix (disagreements first; columns in the order above):");let n=[["TASK","DIFF",...e.jobs.map((o,a)=>`JOB ${a+1}`)]],s=e.jobs.map(o=>o.id);for(let o of e.taskMatrix){let a=new Map(o.cells.map(i=>[i.jobId,i]));n.push([o.taskKey,o.disagreement?"!":"",...s.map(i=>{let u=a.get(i);return u?u.meanReward!==null?`${u.status} ${D(u.meanReward)}`:u.status:"-"})]);}t.push(...k(n));}return t}async function jt(e,t){let n=await R(y(e)).compare(e.positionals);if(e.flags.json===true)t.out(JSON.stringify(n));else for(let s of Nt(n))t.out(s);return 0}async function Ht(e,t){let n=await R(y(e)).cancel(e.positionals[0]);if(e.flags.json===true)t.out(JSON.stringify(n));else for(let s of L(n))t.out(s);return 0}async function Mt(e,t){let n=await R(y(e)).rerunFailed(e.positionals[0]);if(e.flags.json===true)t.out(JSON.stringify(n));else {for(let s of L(n))t.out(s);t.out(""),t.out(`Follow it with: evolve-evals get ${n.id}`);}return 0}async function Vt(e,t){let r=R(y(e)),[n,s]=e.positionals,o;if(s!==void 0){if(e.flags.status!==void 0||e.flags.task!==void 0)throw new g("--status/--task apply to a whole-job regrade, not a single trial");o=await r.regradeTrial(n,s);}else {let a={};if(e.flags.status!==void 0){let i=String(e.flags.status).split(",").map(u=>u.trim()).filter(Boolean);if(i.length===0)throw new g("--status got an empty status list");a.status=i;}e.flags.task!==void 0&&(a.taskKey=String(e.flags.task)),o=await r.regrade(n,a);}if(e.flags.json===true)t.out(JSON.stringify(o));else {for(let a of Ce(o))t.out(a);t.out(""),t.out(`Follow it with: evolve-evals regrade-job ${o.id}`);}return 0}async function Kt(e,t){let n=await R(y(e)).getRegrade(e.positionals[0],G(e));if(e.flags.json===true)t.out(JSON.stringify(n));else for(let s of Ce(n))t.out(s);return 0}async function Ft(e,t){let r=e.flags.format;if(r!==void 0&&r!=="harbor")throw new g(`Unknown --format "${r}" (supported: harbor)`);let s=await R(y(e)).export(e.positionals[0],{to:e.flags.to??process.cwd(),...r==="harbor"?{format:"harbor"}:{}});return e.flags.json===true?t.out(JSON.stringify({path:s})):t.out(`Saved ${s}`),0}async function qt(e,t){let n=await q(y(e)).downloadPackage(e.positionals[0],{to:e.flags.to??process.cwd()});return e.flags.json===true?t.out(JSON.stringify({path:n})):t.out(`Saved ${n}`),0}function Gt(e){let t=k([["name",e.name],["title",e.title??"-"],["description",e.description??"-"],["active version",e.activeVersion?.version??"-"]]);if(e.versions&&e.versions.length>0){t.push("");let r=[["VERSION","STATE","TASKS","CREATED"]];for(let n of e.versions)r.push([n.version,n.state,String(n.taskCount),n.createdAt??"-"]);t.push(...k(r));}if(e.tasks&&e.tasks.items.length>0){t.push("",`Tasks (version ${e.selectedVersion?.version??"?"}):`);let r=[["TASK","AGENT TIMEOUT","VERIFIER TIMEOUT","PROVIDERS"]];for(let s of e.tasks.items)r.push([s.taskKey,`${s.agentTimeoutSec}s`,`${s.verifierTimeoutSec}s`,Pt(s.providers)]);t.push(...k(r)),e.tasks.nextCursor&&t.push(`More tasks: evolve-evals benchmarks get ${e.name} --cursor ${e.tasks.nextCursor}`);let n=new Map;for(let s of e.tasks.items)for(let o of Pe){let a=s.providers?.[o];a&&!a.ok&&!n.has(`${o}:${a.reason}`)&&n.set(`${o}:${a.reason}`,`${o}: ${a.reason}`);}if(n.size>0){t.push("","Provider limitations:");for(let s of n.values())t.push(` ${s}`);}}return t}async function Wt(e,t){let r=q(y(e)),[n,s]=e.positionals;if(n===void 0){let a=await r.list(G(e));if(e.flags.json===true)return t.out(JSON.stringify(a)),0;if(a.items.length===0)return t.out("No benchmarks."),0;let i=[["NAME","ACTIVE","STATE","TASKS","TITLE"]];for(let u of a.items)i.push([u.name,u.activeVersion?.version??"-",u.activeVersion?.state??"-",u.activeVersion?String(u.activeVersion.taskCount):"-",u.title??"-"]);for(let u of k(i))t.out(u);for(let u of Te(a.items))t.out(u);return 0}if(n!=="get")throw new g(`Unknown benchmarks subcommand "${n}" (did you mean "benchmarks get ${n}"?)`);if(!s)throw new g('"benchmarks get" requires a <name[@version]> ref');let o=await r.get(s,G(e));if(e.flags.json===true)t.out(JSON.stringify(o));else {for(let a of Gt(o))t.out(a);for(let a of Te([o]))t.out(a);}return 0}function Te(e){let t=[];for(let r of e){if(!r.upstream?.moved)continue;let n=r.activeVersion?`@${r.activeVersion.version}`:"";t.push(`${r.name}${n} \xB7 upstream ${r.upstream.ref} moved \u2014 run: evolve-evals import --benchmark ${r.name} --version <new-version> --git-url <url> --ref ${r.upstream.ref}`);}return t}async function Yt(e,t){let[r,n]=e.positionals,s=e.flags.json===true,o=q(y(e));if(r==="status"){if(!n)throw new g('"import status" requires an <id>');let l=await o.getImport(n);if(s)t.out(JSON.stringify(l));else for(let c of se(l))t.out(c);return 0}if(r!==void 0)throw new g(`Unknown import subcommand "${r}" (supported: status)`);let a=kt(e),i=await o.import(a);if(e.flags.watch!==true){if(s)t.out(JSON.stringify(i));else {for(let l of se(i))t.out(l);t.out(""),t.out(`Follow it with: evolve-evals import status ${i.id}`);}return 0}s?t.out(JSON.stringify({kind:"import.created",benchmarkImport:i})):t.out(`Import ${i.id} (${a.benchmarkName}) ${i.status} \u2014 watching\u2026`);let u=await o.watchImport(i.id,{onStatus:l=>{t.out(s?JSON.stringify({kind:"import.status",benchmarkImport:l}):At(l));}});if(s)t.out(JSON.stringify({kind:"import.final",benchmarkImport:u}));else {t.out("");for(let l of se(u))t.out(l);}return u.status==="FAILED"?1:0}async function zt(e,t){let r=Se(y(e)),[n,s]=e.positionals,o=e.flags.json===true;if(n===void 0){let a=await r.list(G(e));if(o)return t.out(JSON.stringify(a)),0;if(a.items.length===0)return t.out("No custom harnesses."),0;let i=[["NAME","SOURCE","RUN COMMAND","UPDATED"]];for(let u of a.items)i.push([u.name,u.source,W(u.runCommand,60),u.updatedAt]);for(let u of k(i))t.out(u);return 0}if(n==="add"){if(s!==void 0)throw new g(`"custom-harnesses add" got unexpected argument "${s}"`);let a=await r.create(wt(e));if(o)t.out(JSON.stringify(a));else {for(let i of xe(a))t.out(i);t.out(""),t.out(`Use it with: evolve-evals run --agent ${a.name}:<model> \u2026`);}return 0}if(n==="get"){if(!s)throw new g('"custom-harnesses get" requires a <name>');let a=await r.get(s);if(o)t.out(JSON.stringify(a));else for(let i of xe(a))t.out(i);return 0}if(n==="remove"){if(!s)throw new g('"custom-harnesses remove" requires a <name>');return await r.delete(s),o?t.out(JSON.stringify({name:s,deleted:true})):t.out(`Deleted custom harness ${s}`),0}throw new g(`Unknown custom-harnesses subcommand "${n}" (supported: add, get, remove)`)}async function Qt(e,t=vt){let r;try{r=bt(e);}catch(n){if(n instanceof g)return t.err(`Error: ${n.message}`),t.err('Run "evolve-evals help" for usage.'),2;throw n}try{switch(r.command){case "help":return t.out(gt),0;case "run":return await _t(r,t);case "list":return await Ut(r,t);case "get":return await $t(r,t);case "trials":return await Lt(r,t);case "trial":return await Dt(r,t);case "trace":return await Bt(r,t);case "compare":return await jt(r,t);case "cancel":return await Ht(r,t);case "rerun-failed":return await Mt(r,t);case "regrade":return await Vt(r,t);case "regrade-job":return await Kt(r,t);case "export":return await Ft(r,t);case "benchmarks":return await Wt(r,t);case "import":return await Yt(r,t);case "download":return await qt(r,t);case "custom-harnesses":return await zt(r,t);default:return t.err(`Error: unknown command "${r.command}"`),2}}catch(n){return n instanceof g?(t.err(`Error: ${n.message}`),t.err('Run "evolve-evals help" for usage.'),2):(t.err(`Error: ${n.message}`),1)}}var Xt=(()=>{let e=process.argv[1];if(!e)return false;let t=r=>{try{return (typeof document === 'undefined' ? require('u' + 'rl').pathToFileURL(__filename).href : (_documentCurrentScript && _documentCurrentScript.tagName.toUpperCase() === 'SCRIPT' && _documentCurrentScript.src || new URL('cli.cjs', document.baseURI).href))===url.pathToFileURL(r).href}catch{return false}};if(t(e))return true;try{return t(fs.realpathSync(e))}catch{return false}})();Xt&&Qt(process.argv.slice(2)).then(e=>{process.exitCode=e;},e=>{process.stderr.write(`Error: ${e?.message??e}
|
|
92
96
|
`),process.exitCode=1;});exports.CliUsageError=g;exports.USAGE=gt;exports.buildCustomHarnessInput=wt;exports.buildImportInput=kt;exports.buildJobInput=yt;exports.eventLine=Ot;exports.importStatusLine=At;exports.parseArgs=bt;exports.parseHarnessEnv=Rt;exports.parseJobAgent=ht;exports.runCli=Qt;exports.traceEventLine=Et;exports.trialDetailLines=xt;
|
package/dist/hosted/cli.d.cts
CHANGED
|
@@ -12,7 +12,7 @@ import { o as JobAgent, p as JobInput, Z as BenchmarkImportInput, a2 as CustomHa
|
|
|
12
12
|
* runtime/API failure (watch: FAILED or CANCELLED), 2 usage error.
|
|
13
13
|
*/
|
|
14
14
|
|
|
15
|
-
declare const USAGE = "evolve-evals \u2014 Evolve hosted jobs CLI\n\nUsage: evolve-evals <command> [options]\n\nCommands:\n run Create a job (add --watch to follow it)\n list List your jobs (newest first)\n get <id> Show one job\n trials <id> List a job's trials\n trial <id> <trial-id> Show one trial in full detail\n trace <id> <trial-id> Print a trial's trace events (--stream <raw-artifact>, --save <dir> for all)\n compare <id> <id> [...] Compare 2-5 jobs side by side\n cancel <id> Request cancellation of a job\n rerun-failed <id> New job from a terminal job's failed trials\n regrade <id> [trial-id] Re-run the verifier on recorded trials (whole job, or one trial)\n regrade-job <job-id> Show a regrade job's results (rewards, deltas, lineage)\n export <id> Download the research archive (gzipped JSON)\n benchmarks List the benchmark catalog\n benchmarks get <name[@version]> Show one benchmark (versions + tasks + providers)\n import Import a benchmark from a git source or a local directory (--watch to follow)\n import status <id> Show one import job\n download <import-id> Download the original corpus package (owner only)\n custom-harnesses List your registered custom harnesses\n custom-harnesses get <name> Show one custom harness\n custom-harnesses add Register a custom harness (install script or local directory)\n custom-harnesses remove <name> Delete a custom harness\n help Show this help\n\nRun options:\n --benchmark <name[@version]> Benchmark (required; bare name = active version)\n --tasks <k1,k2,...> Task keys (default: every task of the version)\n --agent <harness:model[:version]> Agent; repeatable (at least one required)\n --effort <value> Reasoning effort for EVERY agent (values:\n GET /api/meta limits.job.reasoningEfforts).\n Applied verbatim \u2014 an agent whose harness\n cannot honor it is refused by the server,\n never silently skipped. Per-agent efforts\n need the SDK. Omitted: the server default.\n --runs <n> Runs per task x agent (default 1)\n --concurrency <n> Parallel trials (default 1)\n --max-trial-spend <usd> Model-spend cap for EACH trial (default: the server's, $200)\n --provider <e2b|daytona|modal> e2b | daytona | modal, default e2b\n --watch Stream events until the job finishes\n\nTrial options:\n --status <s1,s2,...> Filter trials by status (e.g. INFRASTRUCTURE_ERROR)\n\nRegrade options (whole-job regrade only):\n --status <s1,s2,...> Only regrade source trials in these statuses\n --task <key> Only regrade source trials of this task\n\nTrace options:\n --cursor <seq> Resume after this trace seq (a trace cursor IS a seq)\n --limit <n> Max events per page\n\nImport options (a git source OR a local directory; --name and --version required):\n --git <url> Git repository URL (with --ref)\n --ref <ref> Git ref: branch, tag, or commit (with --git)\n --dir <path> Local corpus directory (tarred + uploaded)\n --name <benchmark> Catalog benchmark name to create or extend (required)\n --version <v> Version label for the imported version (required)\n --watch Poll until the import is IMPORTED or FAILED\n\nCustom-harness options (\"custom-harnesses add\"; an install script OR a local directory):\n --name <harness> Harness name, later used in --agent (required)\n --install-script <path> Install script file; its contents are uploaded\n --dir <path> Local harness directory (tarred + uploaded)\n --run <command> Run command, executed with sh -c (required)\n --env KEY=VALUE Env injected at run time; repeatable\n\nOther options:\n --limit <n>, --cursor <c> Pagination \u2014 one envelope on every collection\n (list, trials, trace, benchmarks, benchmarks get,\n custom-harnesses, regrade-job)\n --to <dir> Directory to save into, for export and download (default: current dir)\n --format harbor Export the Harbor job-layout bundle\n --json Machine-readable JSON output\n --api-key <key> API key (default: $EVOLVE_API_KEY)\n --base-url <url> API base URL (default: the Evolve dashboard API)";
|
|
15
|
+
declare const USAGE = "evolve-evals \u2014 Evolve hosted jobs CLI\n\nUsage: evolve-evals <command> [options]\n\nCommands:\n run Create a job (add --watch to follow it)\n list List your jobs (newest first)\n get <id> Show one job\n trials <id> List a job's trials\n trial <id> <trial-id> Show one trial in full detail\n trace <id> <trial-id> Print a trial's trace events (--stream <raw-artifact>, --save <dir> for all)\n compare <id> <id> [...] Compare 2-5 jobs side by side\n cancel <id> Request cancellation of a job\n rerun-failed <id> New job from a terminal job's failed trials\n regrade <id> [trial-id] Re-run the verifier on recorded trials (whole job, or one trial)\n regrade-job <job-id> Show a regrade job's results (rewards, deltas, lineage)\n export <id> Download the research archive (gzipped JSON)\n benchmarks List the benchmark catalog\n benchmarks get <name[@version]> Show one benchmark (versions + tasks + providers)\n import Import a benchmark from a git source or a local directory (--watch to follow)\n import status <id> Show one import job\n download <import-id> Download the original corpus package (owner only)\n custom-harnesses List your registered custom harnesses\n custom-harnesses get <name> Show one custom harness\n custom-harnesses add Register a custom harness (install script or local directory)\n custom-harnesses remove <name> Delete a custom harness\n help Show this help\n\nRun options:\n --benchmark <name[@version]> Benchmark (required; bare name = active version)\n --tasks <k1,k2,...> Task keys (default: every task of the version)\n --agent <harness:model[:version]> Agent; repeatable (at least one required)\n --effort <value> Reasoning effort for EVERY agent (values:\n GET /api/meta limits.job.reasoningEfforts).\n Applied verbatim \u2014 an agent whose harness\n cannot honor it is refused by the server,\n never silently skipped. Per-agent efforts\n need the SDK. Omitted: the server default.\n --runs <n> Runs per task x agent (default 1)\n --concurrency <n> Parallel trials (default 1)\n --max-trial-spend <usd> Model-spend cap for EACH trial (default: the server's, $200)\n --provider <e2b|daytona|modal> e2b | daytona | modal, default e2b\n --watch Stream events until the job finishes\n\nTrial options:\n --status <s1,s2,...> Filter trials by status (e.g. INFRASTRUCTURE_ERROR)\n\nRegrade options (whole-job regrade only):\n --status <s1,s2,...> Only regrade source trials in these statuses\n --task <key> Only regrade source trials of this task\n\nTrace options (--stream and --save are exclusive; --cursor/--limit page the events only):\n --cursor <seq> Resume after this trace seq (a trace cursor IS a seq)\n --limit <n> Max events per page\n --stream <artifact> Print ONE raw artifact instead: verifier |\n trace-stdout | trace-stderr | agent-home\n --save <dir> Save everything the trial recorded into <dir>:\n trace-parsed.jsonl, each raw log, agent-home/\n\nImport options (a git source OR a local directory; --name and --version required):\n --git <url> Git repository URL (with --ref)\n --ref <ref> Git ref: branch, tag, or commit (with --git)\n --dir <path> Local corpus directory (tarred + uploaded)\n --name <benchmark> Catalog benchmark name to create or extend (required)\n --version <v> Version label for the imported version (required)\n --watch Poll until the import is IMPORTED or FAILED\n\nCustom-harness options (\"custom-harnesses add\"; an install script OR a local directory):\n --name <harness> Harness name, later used in --agent (required)\n --install-script <path> Install script file; its contents are uploaded\n --dir <path> Local harness directory (tarred + uploaded)\n --run <command> Run command, executed with sh -c (required)\n --env KEY=VALUE Env injected at run time; repeatable\n\nOther options:\n --limit <n>, --cursor <c> Pagination \u2014 one envelope on every collection\n (list, trials, trace, benchmarks, benchmarks get,\n custom-harnesses, regrade-job)\n --to <dir> Directory to save into, for export and download (default: current dir)\n --format harbor Export the Harbor job-layout bundle\n --json Machine-readable JSON output\n --api-key <key> API key (default: $EVOLVE_API_KEY)\n --base-url <url> API base URL (default: the Evolve dashboard API)";
|
|
16
16
|
/** Usage-level error: bad command line, not a runtime failure. Exit code 2. */
|
|
17
17
|
declare class CliUsageError extends Error {
|
|
18
18
|
constructor(message: string);
|
package/dist/hosted/cli.d.ts
CHANGED
|
@@ -12,7 +12,7 @@ import { o as JobAgent, p as JobInput, Z as BenchmarkImportInput, a2 as CustomHa
|
|
|
12
12
|
* runtime/API failure (watch: FAILED or CANCELLED), 2 usage error.
|
|
13
13
|
*/
|
|
14
14
|
|
|
15
|
-
declare const USAGE = "evolve-evals \u2014 Evolve hosted jobs CLI\n\nUsage: evolve-evals <command> [options]\n\nCommands:\n run Create a job (add --watch to follow it)\n list List your jobs (newest first)\n get <id> Show one job\n trials <id> List a job's trials\n trial <id> <trial-id> Show one trial in full detail\n trace <id> <trial-id> Print a trial's trace events (--stream <raw-artifact>, --save <dir> for all)\n compare <id> <id> [...] Compare 2-5 jobs side by side\n cancel <id> Request cancellation of a job\n rerun-failed <id> New job from a terminal job's failed trials\n regrade <id> [trial-id] Re-run the verifier on recorded trials (whole job, or one trial)\n regrade-job <job-id> Show a regrade job's results (rewards, deltas, lineage)\n export <id> Download the research archive (gzipped JSON)\n benchmarks List the benchmark catalog\n benchmarks get <name[@version]> Show one benchmark (versions + tasks + providers)\n import Import a benchmark from a git source or a local directory (--watch to follow)\n import status <id> Show one import job\n download <import-id> Download the original corpus package (owner only)\n custom-harnesses List your registered custom harnesses\n custom-harnesses get <name> Show one custom harness\n custom-harnesses add Register a custom harness (install script or local directory)\n custom-harnesses remove <name> Delete a custom harness\n help Show this help\n\nRun options:\n --benchmark <name[@version]> Benchmark (required; bare name = active version)\n --tasks <k1,k2,...> Task keys (default: every task of the version)\n --agent <harness:model[:version]> Agent; repeatable (at least one required)\n --effort <value> Reasoning effort for EVERY agent (values:\n GET /api/meta limits.job.reasoningEfforts).\n Applied verbatim \u2014 an agent whose harness\n cannot honor it is refused by the server,\n never silently skipped. Per-agent efforts\n need the SDK. Omitted: the server default.\n --runs <n> Runs per task x agent (default 1)\n --concurrency <n> Parallel trials (default 1)\n --max-trial-spend <usd> Model-spend cap for EACH trial (default: the server's, $200)\n --provider <e2b|daytona|modal> e2b | daytona | modal, default e2b\n --watch Stream events until the job finishes\n\nTrial options:\n --status <s1,s2,...> Filter trials by status (e.g. INFRASTRUCTURE_ERROR)\n\nRegrade options (whole-job regrade only):\n --status <s1,s2,...> Only regrade source trials in these statuses\n --task <key> Only regrade source trials of this task\n\nTrace options:\n --cursor <seq> Resume after this trace seq (a trace cursor IS a seq)\n --limit <n> Max events per page\n\nImport options (a git source OR a local directory; --name and --version required):\n --git <url> Git repository URL (with --ref)\n --ref <ref> Git ref: branch, tag, or commit (with --git)\n --dir <path> Local corpus directory (tarred + uploaded)\n --name <benchmark> Catalog benchmark name to create or extend (required)\n --version <v> Version label for the imported version (required)\n --watch Poll until the import is IMPORTED or FAILED\n\nCustom-harness options (\"custom-harnesses add\"; an install script OR a local directory):\n --name <harness> Harness name, later used in --agent (required)\n --install-script <path> Install script file; its contents are uploaded\n --dir <path> Local harness directory (tarred + uploaded)\n --run <command> Run command, executed with sh -c (required)\n --env KEY=VALUE Env injected at run time; repeatable\n\nOther options:\n --limit <n>, --cursor <c> Pagination \u2014 one envelope on every collection\n (list, trials, trace, benchmarks, benchmarks get,\n custom-harnesses, regrade-job)\n --to <dir> Directory to save into, for export and download (default: current dir)\n --format harbor Export the Harbor job-layout bundle\n --json Machine-readable JSON output\n --api-key <key> API key (default: $EVOLVE_API_KEY)\n --base-url <url> API base URL (default: the Evolve dashboard API)";
|
|
15
|
+
declare const USAGE = "evolve-evals \u2014 Evolve hosted jobs CLI\n\nUsage: evolve-evals <command> [options]\n\nCommands:\n run Create a job (add --watch to follow it)\n list List your jobs (newest first)\n get <id> Show one job\n trials <id> List a job's trials\n trial <id> <trial-id> Show one trial in full detail\n trace <id> <trial-id> Print a trial's trace events (--stream <raw-artifact>, --save <dir> for all)\n compare <id> <id> [...] Compare 2-5 jobs side by side\n cancel <id> Request cancellation of a job\n rerun-failed <id> New job from a terminal job's failed trials\n regrade <id> [trial-id] Re-run the verifier on recorded trials (whole job, or one trial)\n regrade-job <job-id> Show a regrade job's results (rewards, deltas, lineage)\n export <id> Download the research archive (gzipped JSON)\n benchmarks List the benchmark catalog\n benchmarks get <name[@version]> Show one benchmark (versions + tasks + providers)\n import Import a benchmark from a git source or a local directory (--watch to follow)\n import status <id> Show one import job\n download <import-id> Download the original corpus package (owner only)\n custom-harnesses List your registered custom harnesses\n custom-harnesses get <name> Show one custom harness\n custom-harnesses add Register a custom harness (install script or local directory)\n custom-harnesses remove <name> Delete a custom harness\n help Show this help\n\nRun options:\n --benchmark <name[@version]> Benchmark (required; bare name = active version)\n --tasks <k1,k2,...> Task keys (default: every task of the version)\n --agent <harness:model[:version]> Agent; repeatable (at least one required)\n --effort <value> Reasoning effort for EVERY agent (values:\n GET /api/meta limits.job.reasoningEfforts).\n Applied verbatim \u2014 an agent whose harness\n cannot honor it is refused by the server,\n never silently skipped. Per-agent efforts\n need the SDK. Omitted: the server default.\n --runs <n> Runs per task x agent (default 1)\n --concurrency <n> Parallel trials (default 1)\n --max-trial-spend <usd> Model-spend cap for EACH trial (default: the server's, $200)\n --provider <e2b|daytona|modal> e2b | daytona | modal, default e2b\n --watch Stream events until the job finishes\n\nTrial options:\n --status <s1,s2,...> Filter trials by status (e.g. INFRASTRUCTURE_ERROR)\n\nRegrade options (whole-job regrade only):\n --status <s1,s2,...> Only regrade source trials in these statuses\n --task <key> Only regrade source trials of this task\n\nTrace options (--stream and --save are exclusive; --cursor/--limit page the events only):\n --cursor <seq> Resume after this trace seq (a trace cursor IS a seq)\n --limit <n> Max events per page\n --stream <artifact> Print ONE raw artifact instead: verifier |\n trace-stdout | trace-stderr | agent-home\n --save <dir> Save everything the trial recorded into <dir>:\n trace-parsed.jsonl, each raw log, agent-home/\n\nImport options (a git source OR a local directory; --name and --version required):\n --git <url> Git repository URL (with --ref)\n --ref <ref> Git ref: branch, tag, or commit (with --git)\n --dir <path> Local corpus directory (tarred + uploaded)\n --name <benchmark> Catalog benchmark name to create or extend (required)\n --version <v> Version label for the imported version (required)\n --watch Poll until the import is IMPORTED or FAILED\n\nCustom-harness options (\"custom-harnesses add\"; an install script OR a local directory):\n --name <harness> Harness name, later used in --agent (required)\n --install-script <path> Install script file; its contents are uploaded\n --dir <path> Local harness directory (tarred + uploaded)\n --run <command> Run command, executed with sh -c (required)\n --env KEY=VALUE Env injected at run time; repeatable\n\nOther options:\n --limit <n>, --cursor <c> Pagination \u2014 one envelope on every collection\n (list, trials, trace, benchmarks, benchmarks get,\n custom-harnesses, regrade-job)\n --to <dir> Directory to save into, for export and download (default: current dir)\n --format harbor Export the Harbor job-layout bundle\n --json Machine-readable JSON output\n --api-key <key> API key (default: $EVOLVE_API_KEY)\n --base-url <url> API base URL (default: the Evolve dashboard API)";
|
|
16
16
|
/** Usage-level error: bad command line, not a runtime failure. Exit code 2. */
|
|
17
17
|
declare class CliUsageError extends Error {
|
|
18
18
|
constructor(message: string);
|
package/dist/hosted/cli.js
CHANGED
|
@@ -50,9 +50,13 @@ Regrade options (whole-job regrade only):
|
|
|
50
50
|
--status <s1,s2,...> Only regrade source trials in these statuses
|
|
51
51
|
--task <key> Only regrade source trials of this task
|
|
52
52
|
|
|
53
|
-
Trace options:
|
|
53
|
+
Trace options (--stream and --save are exclusive; --cursor/--limit page the events only):
|
|
54
54
|
--cursor <seq> Resume after this trace seq (a trace cursor IS a seq)
|
|
55
55
|
--limit <n> Max events per page
|
|
56
|
+
--stream <artifact> Print ONE raw artifact instead: verifier |
|
|
57
|
+
trace-stdout | trace-stderr | agent-home
|
|
58
|
+
--save <dir> Save everything the trial recorded into <dir>:
|
|
59
|
+
trace-parsed.jsonl, each raw log, agent-home/
|
|
56
60
|
|
|
57
61
|
Import options (a git source OR a local directory; --name and --version required):
|
|
58
62
|
--git <url> Git repository URL (with --ref)
|
|
@@ -77,11 +81,11 @@ Other options:
|
|
|
77
81
|
--format harbor Export the Harbor job-layout bundle
|
|
78
82
|
--json Machine-readable JSON output
|
|
79
83
|
--api-key <key> API key (default: $EVOLVE_API_KEY)
|
|
80
|
-
--base-url <url> API base URL (default: the Evolve dashboard API)`,l=class extends Error{constructor(e){super(e),this.name="CliUsageError";}},M={json:"boolean","api-key":"string","base-url":"string"},q={run:{flags:{benchmark:"string",tasks:"string",agent:"repeat",effort:"string",runs:"number",concurrency:"number","max-trial-spend":"number",provider:"string",watch:"boolean"},required:["benchmark","agent"],minPositionals:0,maxPositionals:0},list:{flags:{limit:"number",cursor:"string"},minPositionals:0,maxPositionals:0},get:{flags:{},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},trials:{flags:{status:"string",limit:"number",cursor:"string"},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},trial:{flags:{},minPositionals:2,maxPositionals:2,positionalUsage:"<id> <trial-id>"},trace:{flags:{cursor:"string",limit:"number"},minPositionals:2,maxPositionals:2,positionalUsage:"<id> <trial-id>"},compare:{flags:{},minPositionals:2,maxPositionals:5,positionalUsage:"<id> <id> [...]"},cancel:{flags:{},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},"rerun-failed":{flags:{},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},regrade:{flags:{status:"string",task:"string"},minPositionals:1,maxPositionals:2,positionalUsage:"<id> [trial-id]"},"regrade-job":{flags:{limit:"number",cursor:"string"},minPositionals:1,maxPositionals:1,positionalUsage:"<job-id>"},export:{flags:{to:"string",format:"string"},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},benchmarks:{flags:{limit:"number",cursor:"string"},minPositionals:0,maxPositionals:2},import:{flags:{git:"string",ref:"string",dir:"string",name:"string",version:"string",watch:"boolean"},minPositionals:0,maxPositionals:2},download:{flags:{to:"string"},minPositionals:1,maxPositionals:1,positionalUsage:"<import-id>"},"custom-harnesses":{flags:{name:"string","install-script":"string",dir:"string",run:"string",env:"repeat",limit:"number",cursor:"string"},minPositionals:0,maxPositionals:2},help:{flags:{},minPositionals:0,maxPositionals:0}},P={command:"help",positionals:[],flags:{}};function K(t){if(t.length===0)throw new l("No command given");if(t[0]==="--help"||t[0]==="-h")return P;let e=t[0],s=q[e];if(!s)throw new l(`Unknown command "${e}"`);let r={},n=[];for(let
|
|
84
|
+
--base-url <url> API base URL (default: the Evolve dashboard API)`,l=class extends Error{constructor(e){super(e),this.name="CliUsageError";}},M={json:"boolean","api-key":"string","base-url":"string"},q={run:{flags:{benchmark:"string",tasks:"string",agent:"repeat",effort:"string",runs:"number",concurrency:"number","max-trial-spend":"number",provider:"string",watch:"boolean"},required:["benchmark","agent"],minPositionals:0,maxPositionals:0},list:{flags:{limit:"number",cursor:"string"},minPositionals:0,maxPositionals:0},get:{flags:{},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},trials:{flags:{status:"string",limit:"number",cursor:"string"},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},trial:{flags:{},minPositionals:2,maxPositionals:2,positionalUsage:"<id> <trial-id>"},trace:{flags:{cursor:"string",limit:"number",stream:"string",save:"string"},minPositionals:2,maxPositionals:2,positionalUsage:"<id> <trial-id>"},compare:{flags:{},minPositionals:2,maxPositionals:5,positionalUsage:"<id> <id> [...]"},cancel:{flags:{},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},"rerun-failed":{flags:{},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},regrade:{flags:{status:"string",task:"string"},minPositionals:1,maxPositionals:2,positionalUsage:"<id> [trial-id]"},"regrade-job":{flags:{limit:"number",cursor:"string"},minPositionals:1,maxPositionals:1,positionalUsage:"<job-id>"},export:{flags:{to:"string",format:"string"},minPositionals:1,maxPositionals:1,positionalUsage:"<id>"},benchmarks:{flags:{limit:"number",cursor:"string"},minPositionals:0,maxPositionals:2},import:{flags:{git:"string",ref:"string",dir:"string",name:"string",version:"string",watch:"boolean"},minPositionals:0,maxPositionals:2},download:{flags:{to:"string"},minPositionals:1,maxPositionals:1,positionalUsage:"<import-id>"},"custom-harnesses":{flags:{name:"string","install-script":"string",dir:"string",run:"string",env:"repeat",limit:"number",cursor:"string"},minPositionals:0,maxPositionals:2},help:{flags:{},minPositionals:0,maxPositionals:0}},P={command:"help",positionals:[],flags:{}};function K(t){if(t.length===0)throw new l("No command given");if(t[0]==="--help"||t[0]==="-h")return P;let e=t[0],s=q[e];if(!s)throw new l(`Unknown command "${e}"`);let r={},n=[];for(let i=1;i<t.length;i++){let o=t[i];if(o==="--help"||o==="-h")return P;if(!o.startsWith("--")){n.push(o);continue}let a=o.slice(2),u,c=a.indexOf("=");c!==-1&&(u=a.slice(c+1),a=a.slice(0,c));let h=s.flags[a]??M[a];if(!h)throw new l(`Unknown option --${a} for "${e}"`);if(h==="boolean"){if(u!==void 0)throw new l(`Option --${a} takes no value`);r[a]=true;continue}let p=u;if(p===void 0){let m=t[i+1];if(m===void 0||m.startsWith("--"))throw new l(`Option --${a} requires a value`);p=m,i++;}if(h==="number"){let m=Number(p);if(p.trim()===""||!Number.isFinite(m))throw new l(`Option --${a} expects a number, got "${p}"`);r[a]=m;}else if(h==="repeat"){let m=r[a]??[];m.push(p),r[a]=m;}else r[a]=p;}if(n.length<s.minPositionals)throw new l(`"${e}" requires ${s.positionalUsage??`${s.minPositionals} argument(s)`}`);if(n.length>s.maxPositionals)throw new l(`"${e}" got unexpected argument "${n[s.maxPositionals]}"`);for(let i of s.required??[])if(!(i in r))throw new l(`"${e}" requires --${i}`);return {command:e,positionals:n,flags:r}}function V(t){let e=t.indexOf(":");if(e<=0||e===t.length-1)throw new l(`Invalid --agent "${t}": expected harness:model[:version]`);let s=t.slice(0,e),r=t.slice(e+1),n=r.indexOf(":");if(n===-1)return {harness:s,model:r};let i=r.slice(0,n),o=r.slice(n+1);if(!i||!o)throw new l(`Invalid --agent "${t}": expected harness:model[:version]`);return {harness:s,model:i,harnessVersion:o}}function F(t){let e=t.flags,s;if(e.tasks!==void 0&&(s=String(e.tasks).split(",").map(r=>r.trim()).filter(Boolean),s.length===0))throw new l("--tasks got an empty task list");return {benchmark:e.benchmark,...s!==void 0?{tasks:s}:{},agents:e.agent.map(r=>{let n=V(r);return e.effort!==void 0?{...n,reasoningEffort:e.effort}:n}),...e.runs!==void 0?{runsPerTask:e.runs}:{},...e.concurrency!==void 0?{concurrency:e.concurrency}:{},...e["max-trial-spend"]!==void 0?{maxTrialSpendUsd:e["max-trial-spend"]}:{},...e.provider!==void 0?{sandboxProvider:e.provider}:{}}}function H(t){let e=t.flags,s=typeof e.dir=="string",r=typeof e.git=="string"||typeof e.ref=="string";if(s&&r)throw new l('"import" takes EITHER --dir OR --git/--ref, not both');if(s){for(let n of ["name","version"])if(typeof e[n]!="string")throw new l(`"import" requires --${n}`);return {source:{directory:e.dir},benchmarkName:e.name,version:e.version}}for(let n of ["git","ref","name","version"])if(typeof e[n]!="string"){let i=n==="git"||n==="ref"?" (or --dir for a local corpus directory)":"";throw new l(`"import" requires --${n}${i}`)}return {source:{gitUrl:e.git,ref:e.ref},benchmarkName:e.name,version:e.version}}function B(t){let e={};for(let s of t){let r=s.indexOf("=");if(r<=0)throw new l(`Invalid --env "${s}": expected KEY=VALUE`);e[s.slice(0,r)]=s.slice(r+1);}return e}function G(t,e=s=>readFileSync(s,"utf-8")){let s=t.flags,r=typeof s.dir=="string",n=typeof s["install-script"]=="string";if(r&&n)throw new l('"custom-harnesses add" takes EITHER --dir OR --install-script, not both');if(!r&&!n)throw new l('"custom-harnesses add" requires --install-script (or --dir for a local harness directory)');for(let o of ["name","run"])if(typeof s[o]!="string")throw new l(`"custom-harnesses add" requires --${o}`);let i=B(s.env??[]);return {name:s.name,...r?{directory:s.dir}:{installScript:e(s["install-script"])},runCommand:s.run,...Object.keys(i).length>0?{env:i}:{}}}var _={out:t=>process.stdout.write(t+`
|
|
81
85
|
`),err:t=>process.stderr.write(t+`
|
|
82
|
-
`)};function d(t){if(t.length===0)return [];let e=[];for(let s of t)s.forEach((r,n)=>{e[n]=Math.max(e[n]??0,r.length);});return t.map(s=>s.map((r,n)=>r.padEnd(e[n])).join(" ").trimEnd())}function b(t){return typeof t=="number"?`$${t.toFixed(2)}`:"-"}function
|
|
83
|
-
More: evolve-evals list --cursor ${r.nextCursor}`),0}async function ot(t,e){let r=await J(f(t)).get(t.positionals[0]);if(t.flags.json===true)e.out(JSON.stringify(r));else for(let n of k(r))e.out(n);return 0}async function at(t,e){let s=J(f(t)),r;if(t.flags.status!==void 0&&(r=String(t.flags.status).split(",").map(
|
|
84
|
-
${n.items.length} trial(s) shown`),n.nextCursor&&e.out(`More: evolve-evals trials ${t.positionals[0]} --cursor ${n.nextCursor}`),0}async function ut(t,e){let r=await J(f(t)).trial(t.positionals[0],t.positionals[1]);if(t.flags.json===true)e.out(JSON.stringify(r));else for(let n of z(r))e.out(n);return 0}async function lt(t,e){let s=J(f(t)),r=t.flags.json===true,n=t.flags.stream;if(n!==void 0){if(n==="verifier"||n==="trace-stdout"||n==="trace-stderr"){let a=await s.trialArtifact(t.positionals[0],t.positionals[1],n);return a===null?(e.out(r?JSON.stringify({log:null}):`No ${n} log was stored for this trial.`),0):(e.out(r?JSON.stringify({log:a}):a),0)}if(n==="agent-home"){let a=await s.trialArtifact(t.positionals[0],t.positionals[1],n);if(a===null)return e.out(r?JSON.stringify({files:null}):`No ${n} content was stored for this trial.`),0;if(r)e.out(JSON.stringify({files:a}));else for(let[u,c]of Object.entries(a))e.out(`===== ${u} (${Buffer.byteLength(c,"utf8")} bytes) =====`),e.out(c);return 0}
|
|
86
|
+
`)};function d(t){if(t.length===0)return [];let e=[];for(let s of t)s.forEach((r,n)=>{e[n]=Math.max(e[n]??0,r.length);});return t.map(s=>s.map((r,n)=>r.padEnd(e[n])).join(" ").trimEnd())}function b(t){return typeof t=="number"?`$${t.toFixed(2)}`:"-"}function E(t){let e=`${t.harness}:${t.model}`;return t.harnessVersion?`${e}:${t.harnessVersion}`:e}function I(t,e){return t.length>e?t.slice(0,e-1)+"\u2026":t}function k(t){let e=[["id",t.id],["status",t.status],["benchmark",t.benchmark]];e.push(["agents",t.agents.map(E).join(", ")]),e.push(["size",`${t.counts.agents} agent(s) x ${t.counts.tasks} task(s) = ${t.trials.total} trial(s)`]),e.push(["runs/task",String(t.runsPerTask)]),e.push(["concurrency",String(t.concurrency)]),e.push(["max spend/trial",b(t.maxTrialSpendUsd)]),e.push(["worst case",b(t.worstCaseSpendUsd)]),e.push(["provider",t.sandboxProvider]),e.push(["spent",b(t.spentUsd)]),e.push(["mean reward",t.meanReward!==null?String(t.meanReward):"-"]);let s=Object.entries(t.trials.byStatus).filter(([,r])=>r>0).map(([r,n])=>`${r} ${n}`).join(" \xB7 ");return s&&e.push(["trials",s]),t.sourceJobId&&e.push(["rerun of",t.sourceJobId]),t.idempotentReplay&&e.push(["note","idempotent replay of an existing job"]),t.failure&&e.push(["failure",`${t.failure.code}: ${t.failure.message}`]),e.push(["created",t.createdAt]),e.push(["updated",t.updatedAt]),d(e)}function W(t){return [t.id,t.status,t.benchmark,String(t.trials.total),v(t.meanReward),b(t.spentUsd),t.createdAt]}function Y(t){return [t.taskKey,E(t.agent),String(t.runNumber),t.status,t.reward!==null?String(t.reward):"-",b(t.spentUsd),t.id]}function z(t){let e=[["trial id",t.id],["job",t.jobId],["task",t.taskKey],["agent",E(t.agent)],["run",String(t.runNumber)],["status",t.status],["reward",t.reward!==null?String(t.reward):"-"]];if(t.metrics&&Object.keys(t.metrics).length>0&&e.push(["metrics",Object.entries(t.metrics).map(([s,r])=>`${s}=${r}`).join(" \xB7 ")]),e.push(["spent",b(t.spentUsd)]),(t.status==="RUNNING"||t.status==="SCORING")&&t.liveSpentUsd!==null){let s=t.liveSpendAt?` as of ${t.liveSpendAt}`:"";e.push(["spent (live)",`at least $${t.liveSpentUsd.toFixed(4)}${s}`]);}return t.sandboxProvider&&e.push(["provider",t.sandboxProvider]),t.sandboxId&&e.push(["sandbox",t.sandboxId]),t.verifierMode&&e.push(["verifier",t.verifierMode]),t.verifierSandboxId&&e.push(["verifier sandbox",t.verifierSandboxId]),t.resolvedHarnessVersion&&e.push(["harness version",t.resolvedHarnessVersion]),t.phaseTimingsMs&&Object.keys(t.phaseTimingsMs).length>0&&e.push(["timings",Object.entries(t.phaseTimingsMs).map(([s,r])=>`${s}=${r}ms`).join(" \xB7 ")]),t.failurePhase&&e.push(["failure phase",t.failurePhase]),t.failureDetail&&e.push(["failure detail",t.failureDetail]),t.sessionRef&&e.push(["session",t.sessionRef]),e.push(["created",t.createdAt]),e.push(["updated",t.updatedAt]),d(e)}function Q(t){if(t===null)return "-";let e=Math.round(t*1e3)/1e3;return e>0?`+${e}`:String(e)}function X(t){return [t.taskKey,t.status,v(t.sourceReward),v(t.reward),Q(t.rewardDelta),t.sourceTrialId]}function T(t){let e=[["job id",t.id],["status",t.status],["source job",t.sourceJobId],["provider",t.sandboxProvider],["results",String(t.results.total)]],s=Object.entries(t.results.byStatus).filter(([,n])=>n>0).map(([n,i])=>`${n} ${i}`).join(" \xB7 ");if(s&&e.push(["by status",s]),t.filter&&(t.filter.status?.length||t.filter.taskKey)){let n=[];t.filter.status?.length&&n.push(`status=${t.filter.status.join(",")}`),t.filter.taskKey&&n.push(`task=${t.filter.taskKey}`),e.push(["filter",n.join(" \xB7 ")]);}e.push(["created",t.createdAt]);let r=d(e);if(t.results.items.length>0){r.push("");let n=[["TASK","STATUS","WAS","NOW","\u0394","SOURCE TRIAL ID"]];for(let i of t.results.items)n.push(X(i));r.push(...d(n)),t.results.nextCursor&&r.push("",`More: evolve-evals regrade-job ${t.id} --cursor ${t.results.nextCursor}`);}return r}function C(t){let e=[["name",t.name],["source",t.source],["run command",t.runCommand]],s=Object.keys(t.env??{});return s.length>0&&e.push(["env",s.sort().join(", ")]),e.push(["created",t.createdAt]),e.push(["updated",t.updatedAt]),d(e)}var A=["e2b","daytona","modal"];function Z(t){return A.filter(e=>t?.[e]!==void 0).map(e=>`${e} ${t[e].ok?"ok":"NO"}`).join(" \xB7 ")}function v(t){return t!==null?String(Math.round(t*1e3)/1e3):"-"}function tt(t){let e=I(JSON.stringify(t.data??{}),140);return `#${String(t.seq).padStart(4)} ${t.type.padEnd(26)} ${e}`.trimEnd()}function j(t){let e=t.failures?.length?` (${t.failures.length} task failure${t.failures.length===1?"":"s"})`:"";return `${t.message}${e}`}function O(t){let e=[["id",t.id],["status",t.status]];if(t.benchmarkName!==void 0&&e.push(["benchmark",t.benchmarkName]),t.version!==void 0&&e.push(["version",t.version]),t.taskCount!==void 0&&e.push(["tasks",String(t.taskCount)]),t.failure){e.push(["failure",j(t.failure)]);for(let s of t.failure.failures??[])e.push([` ${s.taskKey}`,s.error]);}return d(e)}function et(t){let e=[];return t.taskCount!==void 0&&e.push(`tasks=${t.taskCount}`),t.failure&&e.push(I(j(t.failure),140)),`status ${t.status.padEnd(12)} ${e.join(" ")}`.trimEnd()}function st(t){let e={...t.data??{}},s=[];typeof e.trialId=="string"&&s.push(e.trialId),typeof e.taskKey=="string"&&s.push(e.taskKey),typeof e.status=="string"&&s.push(e.status),typeof e.reward=="number"&&s.push(`reward=${e.reward}`);let r=new Set(["trialId","taskKey","status","reward"]);for(let[i,o]of Object.entries(e))r.has(i)||s.push(`${i}=${typeof o=="object"?JSON.stringify(o):String(o)}`);let n=I(s.join(" "),140);return `#${String(t.seq).padStart(4)} ${t.type.padEnd(26)} ${n}`.trimEnd()}function f(t){let e={};return typeof t.flags["api-key"]=="string"&&(e.apiKey=t.flags["api-key"]),typeof t.flags["base-url"]=="string"&&(e.baseUrl=t.flags["base-url"]),e}function $(t){return {...t.flags.limit!==void 0?{limit:t.flags.limit}:{},...t.flags.cursor!==void 0?{cursor:String(t.flags.cursor)}:{}}}function rt(t){return t.status==="COMPLETED"?0:t.status==="FAILED"||t.status==="CANCELLED"?1:0}async function nt(t,e){let s=F(t),r=t.flags.json===true,n=t.flags.watch===true,i=J(f(t)),o=await i.run(s);if(!n){if(r)e.out(JSON.stringify(o));else {for(let u of k(o))e.out(u);e.out(""),e.out(`Follow it with: evolve-evals get ${o.id}`);}return 0}r?e.out(JSON.stringify({kind:"job.created",job:o})):e.out(`Job ${o.id} (${o.benchmark}) ${o.status} \u2014 watching\u2026`);let a=await i.watch(o.id,{onEvent:u=>{e.out(r?JSON.stringify({kind:"event",...u}):st(u));}});if(r)e.out(JSON.stringify({kind:"job.final",job:a}));else {e.out("");for(let u of k(a))e.out(u);}return rt(a)}async function it(t,e){let r=await J(f(t)).list({...t.flags.limit!==void 0?{limit:t.flags.limit}:{},...t.flags.cursor!==void 0?{cursor:t.flags.cursor}:{}});if(t.flags.json===true)return e.out(JSON.stringify(r)),0;if(r.items.length===0)return e.out("No jobs."),0;let n=[["ID","STATUS","BENCHMARK","TRIALS","MEAN REWARD","SPENT","CREATED"]];for(let i of r.items)n.push(W(i));for(let i of d(n))e.out(i);return r.nextCursor&&e.out(`
|
|
87
|
+
More: evolve-evals list --cursor ${r.nextCursor}`),0}async function ot(t,e){let r=await J(f(t)).get(t.positionals[0]);if(t.flags.json===true)e.out(JSON.stringify(r));else for(let n of k(r))e.out(n);return 0}async function at(t,e){let s=J(f(t)),r;if(t.flags.status!==void 0&&(r=String(t.flags.status).split(",").map(o=>o.trim()).filter(Boolean),r.length===0))throw new l("--status got an empty status list");let n=await s.trials(t.positionals[0],{...r!==void 0?{status:r}:{},...t.flags.limit!==void 0?{limit:t.flags.limit}:{},...t.flags.cursor!==void 0?{cursor:t.flags.cursor}:{}});if(t.flags.json===true)return e.out(JSON.stringify(n)),0;if(n.items.length===0)return e.out("No trials."),0;let i=[["TASK","AGENT","RUN","STATUS","REWARD","SPENT","TRIAL ID"]];for(let o of n.items)i.push(Y(o));for(let o of d(i))e.out(o);return e.out(`
|
|
88
|
+
${n.items.length} trial(s) shown`),n.nextCursor&&e.out(`More: evolve-evals trials ${t.positionals[0]} --cursor ${n.nextCursor}`),0}async function ut(t,e){let r=await J(f(t)).trial(t.positionals[0],t.positionals[1]);if(t.flags.json===true)e.out(JSON.stringify(r));else for(let n of z(r))e.out(n);return 0}async function lt(t,e){let s=J(f(t)),r=t.flags.json===true,n=t.flags.stream,i=t.flags.save;if(n!==void 0&&i!==void 0)throw new l('"trace" takes EITHER --stream OR --save, not both');if((n!==void 0||i!==void 0)&&(t.flags.cursor!==void 0||t.flags.limit!==void 0))throw new l("--cursor/--limit page the parsed events; they do not apply to --stream/--save");if(n!==void 0){if(n==="verifier"||n==="trace-stdout"||n==="trace-stderr"){let a=await s.trialArtifact(t.positionals[0],t.positionals[1],n);return a===null?(e.out(r?JSON.stringify({log:null}):`No ${n} log was stored for this trial.`),0):(e.out(r?JSON.stringify({log:a}):a),0)}if(n==="agent-home"){let a=await s.trialArtifact(t.positionals[0],t.positionals[1],n);if(a===null)return e.out(r?JSON.stringify({files:null}):`No ${n} content was stored for this trial.`),0;if(r)e.out(JSON.stringify({files:a}));else for(let[u,c]of Object.entries(a))e.out(`===== ${u} (${Buffer.byteLength(c,"utf8")} bytes) =====`),e.out(c);return 0}throw new l('--stream must be "verifier", "trace-stdout", "trace-stderr" or "agent-home"')}if(i!==void 0){let{mkdir:a,writeFile:u}=await import('fs/promises'),{join:c,dirname:h}=await import('path');await a(i,{recursive:true});let p=[];for await(let w of s.trialTraceEvents(t.positionals[0],t.positionals[1]))p.push(JSON.stringify(w));await u(c(i,"trace-parsed.jsonl"),p.join(`
|
|
85
89
|
`)+(p.length?`
|
|
86
|
-
`:"")),e.out(`trace-parsed.jsonl (${p.length} events)`);for(let w of ["verifier","trace-stdout","trace-stderr"]){let y=await s.trialArtifact(t.positionals[0],t.positionals[1],w);y!==null&&(await u(c(
|
|
87
|
-
`),process.exitCode=1;});export{l as CliUsageError,L as USAGE,G as buildCustomHarnessInput,
|
|
90
|
+
`:"")),e.out(`trace-parsed.jsonl (${p.length} events)`);for(let w of ["verifier","trace-stdout","trace-stderr"]){let y=await s.trialArtifact(t.positionals[0],t.positionals[1],w);y!==null&&(await u(c(i,`${w}.log`),y),e.out(`${w}.log (${Buffer.byteLength(y,"utf8")} bytes)`));}let m=await s.trialArtifact(t.positionals[0],t.positionals[1],"agent-home");if(m!==null){for(let[w,y]of Object.entries(m)){let x=c(i,"agent-home",...w.split("/").filter(Boolean));await a(h(x),{recursive:true}),await u(x,y);}e.out(`agent-home/ (${Object.keys(m).length} files)`);}return 0}let o=0;for await(let a of s.trialTraceEvents(t.positionals[0],t.positionals[1],{...t.flags.cursor!==void 0?{cursor:String(t.flags.cursor)}:{},...t.flags.limit!==void 0?{limit:t.flags.limit}:{}}))e.out(r?JSON.stringify(a):tt(a)),o+=1;return !r&&o===0&&e.out("No trace events."),0}function ct(t){let e=[],s=[["ID","BENCHMARK","STATUS","MEAN REWARD","COVERAGE","SPENT"]];for(let r of t.jobs)s.push([r.id,r.benchmark,r.status,v(r.meanReward),`${r.coverage.scored}/${r.coverage.total}`,b(r.spentUsd)]);if(e.push(...d(s)),t.taskMatrix.length>0){e.push("","Task matrix (disagreements first; columns in the order above):");let r=[["TASK","DIFF",...t.jobs.map((i,o)=>`JOB ${o+1}`)]],n=t.jobs.map(i=>i.id);for(let i of t.taskMatrix){let o=new Map(i.cells.map(a=>[a.jobId,a]));r.push([i.taskKey,i.disagreement?"!":"",...n.map(a=>{let u=o.get(a);return u?u.meanReward!==null?`${u.status} ${v(u.meanReward)}`:u.status:"-"})]);}e.push(...d(r));}return e}async function ft(t,e){let r=await J(f(t)).compare(t.positionals);if(t.flags.json===true)e.out(JSON.stringify(r));else for(let n of ct(r))e.out(n);return 0}async function dt(t,e){let r=await J(f(t)).cancel(t.positionals[0]);if(t.flags.json===true)e.out(JSON.stringify(r));else for(let n of k(r))e.out(n);return 0}async function mt(t,e){let r=await J(f(t)).rerunFailed(t.positionals[0]);if(t.flags.json===true)e.out(JSON.stringify(r));else {for(let n of k(r))e.out(n);e.out(""),e.out(`Follow it with: evolve-evals get ${r.id}`);}return 0}async function gt(t,e){let s=J(f(t)),[r,n]=t.positionals,i;if(n!==void 0){if(t.flags.status!==void 0||t.flags.task!==void 0)throw new l("--status/--task apply to a whole-job regrade, not a single trial");i=await s.regradeTrial(r,n);}else {let o={};if(t.flags.status!==void 0){let a=String(t.flags.status).split(",").map(u=>u.trim()).filter(Boolean);if(a.length===0)throw new l("--status got an empty status list");o.status=a;}t.flags.task!==void 0&&(o.taskKey=String(t.flags.task)),i=await s.regrade(r,o);}if(t.flags.json===true)e.out(JSON.stringify(i));else {for(let o of T(i))e.out(o);e.out(""),e.out(`Follow it with: evolve-evals regrade-job ${i.id}`);}return 0}async function pt(t,e){let r=await J(f(t)).getRegrade(t.positionals[0],$(t));if(t.flags.json===true)e.out(JSON.stringify(r));else for(let n of T(r))e.out(n);return 0}async function ht(t,e){let s=t.flags.format;if(s!==void 0&&s!=="harbor")throw new l(`Unknown --format "${s}" (supported: harbor)`);let n=await J(f(t)).export(t.positionals[0],{to:t.flags.to??process.cwd(),...s==="harbor"?{format:"harbor"}:{}});return t.flags.json===true?e.out(JSON.stringify({path:n})):e.out(`Saved ${n}`),0}async function wt(t,e){let r=await H$1(f(t)).downloadPackage(t.positionals[0],{to:t.flags.to??process.cwd()});return t.flags.json===true?e.out(JSON.stringify({path:r})):e.out(`Saved ${r}`),0}function bt(t){let e=d([["name",t.name],["title",t.title??"-"],["description",t.description??"-"],["active version",t.activeVersion?.version??"-"]]);if(t.versions&&t.versions.length>0){e.push("");let s=[["VERSION","STATE","TASKS","CREATED"]];for(let r of t.versions)s.push([r.version,r.state,String(r.taskCount),r.createdAt??"-"]);e.push(...d(s));}if(t.tasks&&t.tasks.items.length>0){e.push("",`Tasks (version ${t.selectedVersion?.version??"?"}):`);let s=[["TASK","AGENT TIMEOUT","VERIFIER TIMEOUT","PROVIDERS"]];for(let n of t.tasks.items)s.push([n.taskKey,`${n.agentTimeoutSec}s`,`${n.verifierTimeoutSec}s`,Z(n.providers)]);e.push(...d(s)),t.tasks.nextCursor&&e.push(`More tasks: evolve-evals benchmarks get ${t.name} --cursor ${t.tasks.nextCursor}`);let r=new Map;for(let n of t.tasks.items)for(let i of A){let o=n.providers?.[i];o&&!o.ok&&!r.has(`${i}:${o.reason}`)&&r.set(`${i}:${o.reason}`,`${i}: ${o.reason}`);}if(r.size>0){e.push("","Provider limitations:");for(let n of r.values())e.push(` ${n}`);}}return e}async function yt(t,e){let s=H$1(f(t)),[r,n]=t.positionals;if(r===void 0){let o=await s.list($(t));if(t.flags.json===true)return e.out(JSON.stringify(o)),0;if(o.items.length===0)return e.out("No benchmarks."),0;let a=[["NAME","ACTIVE","STATE","TASKS","TITLE"]];for(let u of o.items)a.push([u.name,u.activeVersion?.version??"-",u.activeVersion?.state??"-",u.activeVersion?String(u.activeVersion.taskCount):"-",u.title??"-"]);for(let u of d(a))e.out(u);for(let u of N(o.items))e.out(u);return 0}if(r!=="get")throw new l(`Unknown benchmarks subcommand "${r}" (did you mean "benchmarks get ${r}"?)`);if(!n)throw new l('"benchmarks get" requires a <name[@version]> ref');let i=await s.get(n,$(t));if(t.flags.json===true)e.out(JSON.stringify(i));else {for(let o of bt(i))e.out(o);for(let o of N([i]))e.out(o);}return 0}function N(t){let e=[];for(let s of t){if(!s.upstream?.moved)continue;let r=s.activeVersion?`@${s.activeVersion.version}`:"";e.push(`${s.name}${r} \xB7 upstream ${s.upstream.ref} moved \u2014 run: evolve-evals import --benchmark ${s.name} --version <new-version> --git-url <url> --ref ${s.upstream.ref}`);}return e}async function kt(t,e){let[s,r]=t.positionals,n=t.flags.json===true,i=H$1(f(t));if(s==="status"){if(!r)throw new l('"import status" requires an <id>');let c=await i.getImport(r);if(n)e.out(JSON.stringify(c));else for(let h of O(c))e.out(h);return 0}if(s!==void 0)throw new l(`Unknown import subcommand "${s}" (supported: status)`);let o=H(t),a=await i.import(o);if(t.flags.watch!==true){if(n)e.out(JSON.stringify(a));else {for(let c of O(a))e.out(c);e.out(""),e.out(`Follow it with: evolve-evals import status ${a.id}`);}return 0}n?e.out(JSON.stringify({kind:"import.created",benchmarkImport:a})):e.out(`Import ${a.id} (${o.benchmarkName}) ${a.status} \u2014 watching\u2026`);let u=await i.watchImport(a.id,{onStatus:c=>{e.out(n?JSON.stringify({kind:"import.status",benchmarkImport:c}):et(c));}});if(n)e.out(JSON.stringify({kind:"import.final",benchmarkImport:u}));else {e.out("");for(let c of O(u))e.out(c);}return u.status==="FAILED"?1:0}async function vt(t,e){let s=I$1(f(t)),[r,n]=t.positionals,i=t.flags.json===true;if(r===void 0){let o=await s.list($(t));if(i)return e.out(JSON.stringify(o)),0;if(o.items.length===0)return e.out("No custom harnesses."),0;let a=[["NAME","SOURCE","RUN COMMAND","UPDATED"]];for(let u of o.items)a.push([u.name,u.source,I(u.runCommand,60),u.updatedAt]);for(let u of d(a))e.out(u);return 0}if(r==="add"){if(n!==void 0)throw new l(`"custom-harnesses add" got unexpected argument "${n}"`);let o=await s.create(G(t));if(i)e.out(JSON.stringify(o));else {for(let a of C(o))e.out(a);e.out(""),e.out(`Use it with: evolve-evals run --agent ${o.name}:<model> \u2026`);}return 0}if(r==="get"){if(!n)throw new l('"custom-harnesses get" requires a <name>');let o=await s.get(n);if(i)e.out(JSON.stringify(o));else for(let a of C(o))e.out(a);return 0}if(r==="remove"){if(!n)throw new l('"custom-harnesses remove" requires a <name>');return await s.delete(n),i?e.out(JSON.stringify({name:n,deleted:true})):e.out(`Deleted custom harness ${n}`),0}throw new l(`Unknown custom-harnesses subcommand "${r}" (supported: add, get, remove)`)}async function St(t,e=_){let s;try{s=K(t);}catch(r){if(r instanceof l)return e.err(`Error: ${r.message}`),e.err('Run "evolve-evals help" for usage.'),2;throw r}try{switch(s.command){case "help":return e.out(L),0;case "run":return await nt(s,e);case "list":return await it(s,e);case "get":return await ot(s,e);case "trials":return await at(s,e);case "trial":return await ut(s,e);case "trace":return await lt(s,e);case "compare":return await ft(s,e);case "cancel":return await dt(s,e);case "rerun-failed":return await mt(s,e);case "regrade":return await gt(s,e);case "regrade-job":return await pt(s,e);case "export":return await ht(s,e);case "benchmarks":return await yt(s,e);case "import":return await kt(s,e);case "download":return await wt(s,e);case "custom-harnesses":return await vt(s,e);default:return e.err(`Error: unknown command "${s.command}"`),2}}catch(r){return r instanceof l?(e.err(`Error: ${r.message}`),e.err('Run "evolve-evals help" for usage.'),2):(e.err(`Error: ${r.message}`),1)}}var $t=(()=>{let t=process.argv[1];if(!t)return false;let e=s=>{try{return import.meta.url===pathToFileURL(s).href}catch{return false}};if(e(t))return true;try{return e(realpathSync(t))}catch{return false}})();$t&&St(process.argv.slice(2)).then(t=>{process.exitCode=t;},t=>{process.stderr.write(`Error: ${t?.message??t}
|
|
91
|
+
`),process.exitCode=1;});export{l as CliUsageError,L as USAGE,G as buildCustomHarnessInput,H as buildImportInput,F as buildJobInput,st as eventLine,et as importStatusLine,K as parseArgs,B as parseHarnessEnv,V as parseJobAgent,St as runCli,tt as traceEventLine,z as trialDetailLines};
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@evolvingmachines/sdk",
|
|
3
|
-
"version": "0.0.52-project-sable.20260730.
|
|
3
|
+
"version": "0.0.52-project-sable.20260730.c2cb2d2",
|
|
4
4
|
"keywords": [
|
|
5
5
|
"ai",
|
|
6
6
|
"agents",
|
|
@@ -112,9 +112,9 @@
|
|
|
112
112
|
},
|
|
113
113
|
"dependencies": {
|
|
114
114
|
"@agentclientprotocol/sdk": "^0.5.1",
|
|
115
|
-
"@evolvingmachines/e2b": "^0.0.52-project-sable.20260730.
|
|
116
|
-
"@evolvingmachines/daytona": "^0.0.52-project-sable.20260730.
|
|
117
|
-
"@evolvingmachines/modal": "^0.0.52-project-sable.20260730.
|
|
115
|
+
"@evolvingmachines/e2b": "^0.0.52-project-sable.20260730.c2cb2d2",
|
|
116
|
+
"@evolvingmachines/daytona": "^0.0.52-project-sable.20260730.c2cb2d2",
|
|
117
|
+
"@evolvingmachines/modal": "^0.0.52-project-sable.20260730.c2cb2d2",
|
|
118
118
|
"ajv": "^8.17.1",
|
|
119
119
|
"p-map": "^7.0.2",
|
|
120
120
|
"zod": "^3.24.0",
|