@kortyx/cli 0.13.0 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/dist/index.js +251 -28
- package/dist/index.js.map +1 -1
- package/package.json +3 -3
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,19 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [0.14.0](https://github.com/kortyx-io/kortyx/compare/cli-v0.13.0...cli-v0.14.0) (2026-10-07)
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
### Features
|
|
7
|
+
|
|
8
|
+
* **evals:** group suite runs and expose CLI results ([#284](https://github.com/kortyx-io/kortyx/issues/284)) ([146fd71](https://github.com/kortyx-io/kortyx/commit/146fd71a88587e5591dd0511a4c0cb57969f9d07))
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
### Dependencies
|
|
12
|
+
|
|
13
|
+
* The following workspace dependencies were updated
|
|
14
|
+
* dependencies
|
|
15
|
+
* @kortyx/agent bumped to 0.31.0
|
|
16
|
+
|
|
3
17
|
## [0.13.0](https://github.com/kortyx-io/kortyx/compare/cli-v0.12.0...cli-v0.13.0) (2026-10-07)
|
|
4
18
|
|
|
5
19
|
|
package/dist/index.js
CHANGED
|
@@ -1430,7 +1430,9 @@ var parseEvalRunTarget = (input) => {
|
|
|
1430
1430
|
} catch {
|
|
1431
1431
|
throw new StudioReadError("invalid_target", "Invalid eval run URL.");
|
|
1432
1432
|
}
|
|
1433
|
-
const match = locator.pathname.match(
|
|
1433
|
+
const match = locator.pathname.match(
|
|
1434
|
+
/\/evals\/(?:runs|evaluations)\/([^/]+)\/?$/
|
|
1435
|
+
);
|
|
1434
1436
|
if (!match || !["http:", "https:"].includes(locator.protocol) || locator.username || locator.password)
|
|
1435
1437
|
throw new StudioReadError(
|
|
1436
1438
|
"invalid_target",
|
|
@@ -1446,6 +1448,59 @@ var parseEvalRunTarget = (input) => {
|
|
|
1446
1448
|
return { id, url };
|
|
1447
1449
|
};
|
|
1448
1450
|
var StudioEvalClient = class extends StudioApiTransport {
|
|
1451
|
+
evaluations() {
|
|
1452
|
+
return this.requestJson(
|
|
1453
|
+
"GET",
|
|
1454
|
+
"/v1/studio/evals/evaluations",
|
|
1455
|
+
import_evals.StudioEvaluationHistorySchema
|
|
1456
|
+
);
|
|
1457
|
+
}
|
|
1458
|
+
evaluation(id) {
|
|
1459
|
+
this.validateId(id);
|
|
1460
|
+
return this.requestJson(
|
|
1461
|
+
"GET",
|
|
1462
|
+
`/v1/studio/evals/evaluations/${id}`,
|
|
1463
|
+
import_evals.StudioEvaluationDetailSchema
|
|
1464
|
+
);
|
|
1465
|
+
}
|
|
1466
|
+
evaluationResults(id) {
|
|
1467
|
+
this.validateId(id);
|
|
1468
|
+
return this.requestJson(
|
|
1469
|
+
"GET",
|
|
1470
|
+
`/v1/studio/evals/evaluations/${id}/results`,
|
|
1471
|
+
import_evals.StudioEvaluationResultsSchema
|
|
1472
|
+
);
|
|
1473
|
+
}
|
|
1474
|
+
startEvaluation(input) {
|
|
1475
|
+
const parsed = import_evals.StudioEvaluationStartRequestSchema.safeParse(input);
|
|
1476
|
+
if (!parsed.success)
|
|
1477
|
+
throw new StudioReadError(
|
|
1478
|
+
"invalid_eval_request",
|
|
1479
|
+
"Invalid evaluation selection or execution limits."
|
|
1480
|
+
);
|
|
1481
|
+
return this.requestJson(
|
|
1482
|
+
"POST",
|
|
1483
|
+
"/v1/studio/evals/evaluations",
|
|
1484
|
+
import_zod3.z.object({ id: import_zod3.z.uuid() }),
|
|
1485
|
+
{},
|
|
1486
|
+
parsed.data
|
|
1487
|
+
);
|
|
1488
|
+
}
|
|
1489
|
+
cancelEvaluation(id) {
|
|
1490
|
+
this.validateId(id);
|
|
1491
|
+
return this.requestJson(
|
|
1492
|
+
"POST",
|
|
1493
|
+
`/v1/studio/evals/evaluations/${id}/cancel`,
|
|
1494
|
+
import_zod3.z.object({ ok: import_zod3.z.literal(true) })
|
|
1495
|
+
);
|
|
1496
|
+
}
|
|
1497
|
+
validateId(id) {
|
|
1498
|
+
if (!import_zod3.z.uuid().safeParse(id).success)
|
|
1499
|
+
throw new StudioReadError(
|
|
1500
|
+
"invalid_target",
|
|
1501
|
+
"Expected an evaluation UUID."
|
|
1502
|
+
);
|
|
1503
|
+
}
|
|
1449
1504
|
targets() {
|
|
1450
1505
|
return this.requestJson(
|
|
1451
1506
|
"GET",
|
|
@@ -1961,11 +2016,65 @@ function registerStudioEvalCommands(studio, log, request = fetch) {
|
|
|
1961
2016
|
options
|
|
1962
2017
|
);
|
|
1963
2018
|
});
|
|
2019
|
+
const terminal = (status) => status !== "queued" && status !== "running";
|
|
2020
|
+
const resultExitCode = (status) => status === "passed" ? 0 : status === "failed" ? 1 : status === "cancelled" ? 130 : 2;
|
|
2021
|
+
const waitForEvaluation = async (client, id, timeout) => {
|
|
2022
|
+
const deadline = Date.now() + timeout * 1e3;
|
|
2023
|
+
for (; ; ) {
|
|
2024
|
+
const { run: run2 } = await client.evaluation(id);
|
|
2025
|
+
if (terminal(run2.status)) return run2;
|
|
2026
|
+
if (Date.now() >= deadline)
|
|
2027
|
+
throw new StudioReadError(
|
|
2028
|
+
"eval_wait_timeout",
|
|
2029
|
+
`Evaluation ${id} is still running. Inspect it with evals runs get; waiting did not cancel it.`
|
|
2030
|
+
);
|
|
2031
|
+
await new Promise(
|
|
2032
|
+
(resolve6) => setTimeout(resolve6, Math.min(2e3, deadline - Date.now()))
|
|
2033
|
+
);
|
|
2034
|
+
}
|
|
2035
|
+
};
|
|
2036
|
+
const readEvaluation = async (client, id, includeContent) => {
|
|
2037
|
+
const { run: run2 } = await client.evaluationResults(id);
|
|
2038
|
+
return includeContent ? run2 : { ...run2, suites: run2.suites.map(compactDetail) };
|
|
2039
|
+
};
|
|
2040
|
+
const waitOptions = (command) => command.option(
|
|
2041
|
+
"--timeout <seconds>",
|
|
2042
|
+
"Maximum time to wait (1\u20137200 seconds); does not cancel execution.",
|
|
2043
|
+
integer2(7200),
|
|
2044
|
+
1800
|
|
2045
|
+
);
|
|
1964
2046
|
const runs = evals.command("runs").description("Start and inspect saved eval runs.");
|
|
1965
2047
|
selectionOptions(
|
|
1966
|
-
runs.command("start
|
|
1967
|
-
"
|
|
2048
|
+
runs.command("start [suite-id]").description(
|
|
2049
|
+
"Start an evaluation containing all or selected suites; optionally wait for results."
|
|
1968
2050
|
)
|
|
2051
|
+
).option("--all", "Run every suite registered on the selected application.").option(
|
|
2052
|
+
"--suite <id>",
|
|
2053
|
+
"Select a suite; repeat for several suites.",
|
|
2054
|
+
(value, previous) => [...previous, value],
|
|
2055
|
+
[]
|
|
2056
|
+
).option("--name <name>", "Display name for this evaluation run.").option(
|
|
2057
|
+
"--source <source>",
|
|
2058
|
+
"Trigger: manual, deployment, schedule or ci.",
|
|
2059
|
+
(value) => {
|
|
2060
|
+
if (!["manual", "deployment", "schedule", "ci"].includes(value))
|
|
2061
|
+
throw new import_commander3.InvalidArgumentError(
|
|
2062
|
+
"Expected manual, deployment, schedule or ci."
|
|
2063
|
+
);
|
|
2064
|
+
return value;
|
|
2065
|
+
},
|
|
2066
|
+
"manual"
|
|
2067
|
+
).option("--commit <sha>", "Deployed application commit.").option("--deployment-url <url>", "Deployment or CI job URL.").option(
|
|
2068
|
+
"--idempotency-key <key>",
|
|
2069
|
+
"Reuse a matching saved evaluation after a trigger retry."
|
|
2070
|
+
).option(
|
|
2071
|
+
"--wait",
|
|
2072
|
+
"Wait for final results and return a pass/fail exit code."
|
|
2073
|
+
).option(
|
|
2074
|
+
"--timeout <seconds>",
|
|
2075
|
+
"Maximum wait in seconds (1\u20137200); execution continues after timeout.",
|
|
2076
|
+
integer2(7200),
|
|
2077
|
+
1800
|
|
1969
2078
|
).option(
|
|
1970
2079
|
"--case <id>",
|
|
1971
2080
|
"Select a case; repeat for multiple cases.",
|
|
@@ -1986,39 +2095,102 @@ function registerStudioEvalCommands(studio, log, request = fetch) {
|
|
|
1986
2095
|
},
|
|
1987
2096
|
"studio"
|
|
1988
2097
|
).action(async (suiteId, options) => {
|
|
1989
|
-
const
|
|
1990
|
-
|
|
1991
|
-
|
|
2098
|
+
const ids = [...suiteId ? [suiteId] : [], ...options.suite ?? []];
|
|
2099
|
+
if (!options.all && !ids.length || options.all && ids.length || new Set(ids).size !== ids.length)
|
|
2100
|
+
throw new StudioReadError(
|
|
2101
|
+
"invalid_suite_selection",
|
|
2102
|
+
"Choose --all or unique suite IDs (positional or repeated --suite)."
|
|
2103
|
+
);
|
|
2104
|
+
const data = await discover(options);
|
|
2105
|
+
const matches = data.targets.filter(
|
|
2106
|
+
(target2) => target2.manifest && (options.all || ids.every(
|
|
2107
|
+
(id) => target2.manifest.suites.some((suite) => suite.id === id)
|
|
2108
|
+
))
|
|
1992
2109
|
);
|
|
1993
|
-
|
|
2110
|
+
const target = matches[0];
|
|
2111
|
+
if (matches.length !== 1 || !target?.manifest)
|
|
2112
|
+
throw new StudioReadError(
|
|
2113
|
+
"invalid_suite_selection",
|
|
2114
|
+
"Select one available application with --target and --environment; all selected suites must belong to it."
|
|
2115
|
+
);
|
|
2116
|
+
if (!data.canRun)
|
|
1994
2117
|
throw new StudioReadError(
|
|
1995
2118
|
"eval_forbidden",
|
|
1996
2119
|
"This Studio key requires eval:run to execute suites."
|
|
1997
2120
|
);
|
|
2121
|
+
const suites2 = options.all ? target.manifest.suites : ids.map(
|
|
2122
|
+
(id) => target.manifest.suites.find((suite) => suite.id === id)
|
|
2123
|
+
);
|
|
1998
2124
|
const caseIds = options.case ?? [];
|
|
1999
|
-
if (
|
|
2125
|
+
if (caseIds.length && (options.all || suites2.length !== 1))
|
|
2126
|
+
throw new StudioReadError(
|
|
2127
|
+
"invalid_case_selection",
|
|
2128
|
+
"--case requires exactly one selected suite."
|
|
2129
|
+
);
|
|
2130
|
+
if (new Set(caseIds).size !== caseIds.length || caseIds.some(
|
|
2131
|
+
(id) => !suites2[0]?.cases.some((item) => item.id === id)
|
|
2132
|
+
) || suites2.some(
|
|
2133
|
+
(suite) => (caseIds.length || suite.cases.length) * options.repetitions > 100
|
|
2134
|
+
))
|
|
2000
2135
|
throw new StudioReadError(
|
|
2001
2136
|
"invalid_case_selection",
|
|
2002
|
-
"Select unique existing
|
|
2137
|
+
"Select unique existing cases and at most 100 attempts per suite."
|
|
2003
2138
|
);
|
|
2004
|
-
const run2 = await client.
|
|
2139
|
+
const run2 = await data.client.startEvaluation({
|
|
2005
2140
|
targetId: target.id,
|
|
2141
|
+
selection: options.all ? "all" : "selected",
|
|
2142
|
+
suites: suites2.map((suite) => ({
|
|
2143
|
+
suiteId: suite.id,
|
|
2144
|
+
suiteRevision: target.revisions[suite.id] ?? "",
|
|
2145
|
+
...caseIds.length ? { caseIds } : {}
|
|
2146
|
+
})),
|
|
2006
2147
|
judge: options.judge ?? "studio",
|
|
2007
|
-
suiteId,
|
|
2008
|
-
suiteRevision: target.revisions[suiteId] ?? "",
|
|
2009
2148
|
repetitions: options.repetitions,
|
|
2010
2149
|
concurrency: options.concurrency,
|
|
2011
|
-
|
|
2012
|
-
|
|
2013
|
-
|
|
2014
|
-
|
|
2015
|
-
|
|
2016
|
-
id: run2.id,
|
|
2017
|
-
status: "queued",
|
|
2018
|
-
studioUrl: connection.studioUrl ? `${connection.studioUrl}/evals/runs/${run2.id}` : null
|
|
2150
|
+
name: options.name,
|
|
2151
|
+
metadata: {
|
|
2152
|
+
source: options.source ?? "manual",
|
|
2153
|
+
commit: options.commit,
|
|
2154
|
+
deploymentUrl: options.deploymentUrl
|
|
2019
2155
|
},
|
|
2020
|
-
options
|
|
2021
|
-
);
|
|
2156
|
+
idempotencyKey: options.idempotencyKey
|
|
2157
|
+
});
|
|
2158
|
+
const studioUrl = data.connection.studioUrl ? `${data.connection.studioUrl}/evals/evaluations/${run2.id}` : null;
|
|
2159
|
+
if (!options.wait) {
|
|
2160
|
+
print(
|
|
2161
|
+
{
|
|
2162
|
+
connection: data.connection.name,
|
|
2163
|
+
id: run2.id,
|
|
2164
|
+
status: "queued",
|
|
2165
|
+
studioUrl
|
|
2166
|
+
},
|
|
2167
|
+
options
|
|
2168
|
+
);
|
|
2169
|
+
return;
|
|
2170
|
+
}
|
|
2171
|
+
try {
|
|
2172
|
+
const completed = await waitForEvaluation(
|
|
2173
|
+
data.client,
|
|
2174
|
+
run2.id,
|
|
2175
|
+
options.timeout
|
|
2176
|
+
);
|
|
2177
|
+
print(
|
|
2178
|
+
{
|
|
2179
|
+
connection: data.connection.name,
|
|
2180
|
+
studioUrl,
|
|
2181
|
+
run: await readEvaluation(
|
|
2182
|
+
data.client,
|
|
2183
|
+
run2.id,
|
|
2184
|
+
options.includeContent ?? false
|
|
2185
|
+
)
|
|
2186
|
+
},
|
|
2187
|
+
options
|
|
2188
|
+
);
|
|
2189
|
+
process.exitCode = resultExitCode(completed.status);
|
|
2190
|
+
} catch (error) {
|
|
2191
|
+
process.exitCode = 2;
|
|
2192
|
+
throw error;
|
|
2193
|
+
}
|
|
2022
2194
|
});
|
|
2023
2195
|
common(
|
|
2024
2196
|
runs.command("list").description("List the latest 100 saved runs in this project.")
|
|
@@ -2027,12 +2199,12 @@ function registerStudioEvalCommands(studio, log, request = fetch) {
|
|
|
2027
2199
|
"Filter results; defaults to the connection environment."
|
|
2028
2200
|
).action(async (options) => {
|
|
2029
2201
|
const { connection, client } = await clientFor(options);
|
|
2030
|
-
const data = await client.
|
|
2202
|
+
const data = await client.evaluations();
|
|
2031
2203
|
const environment = options.environment ?? connection.environment;
|
|
2032
2204
|
print(
|
|
2033
2205
|
{
|
|
2034
2206
|
connection: connection.name,
|
|
2035
|
-
runs: data.runs.filter((run2) => !environment || run2.environment === environment).map(
|
|
2207
|
+
runs: data.runs.filter((run2) => !environment || run2.environment === environment).map((run2) => run2)
|
|
2036
2208
|
},
|
|
2037
2209
|
options
|
|
2038
2210
|
);
|
|
@@ -2044,15 +2216,59 @@ function registerStudioEvalCommands(studio, log, request = fetch) {
|
|
|
2044
2216
|
).action(async (input, options) => {
|
|
2045
2217
|
const target = parseEvalRunTarget(input);
|
|
2046
2218
|
const { connection, client } = await clientFor(options, target.url);
|
|
2047
|
-
|
|
2219
|
+
let run2;
|
|
2220
|
+
try {
|
|
2221
|
+
run2 = await readEvaluation(
|
|
2222
|
+
client,
|
|
2223
|
+
target.id,
|
|
2224
|
+
options.includeContent ?? false
|
|
2225
|
+
);
|
|
2226
|
+
} catch (error) {
|
|
2227
|
+
if (!(error instanceof StudioReadError) || error.status !== 404)
|
|
2228
|
+
throw error;
|
|
2229
|
+
const legacy = await client.run(target.id);
|
|
2230
|
+
run2 = options.includeContent ? legacy.run : compactDetail(legacy.run);
|
|
2231
|
+
}
|
|
2048
2232
|
print(
|
|
2049
2233
|
{
|
|
2050
2234
|
connection: connection.name,
|
|
2051
|
-
run:
|
|
2235
|
+
run: run2
|
|
2052
2236
|
},
|
|
2053
2237
|
options
|
|
2054
2238
|
);
|
|
2055
2239
|
});
|
|
2240
|
+
waitOptions(
|
|
2241
|
+
common(
|
|
2242
|
+
runs.command("wait <id-or-url>").description(
|
|
2243
|
+
"Wait for a grouped evaluation and print its final suite/case results."
|
|
2244
|
+
)
|
|
2245
|
+
)
|
|
2246
|
+
).action(async (input, options) => {
|
|
2247
|
+
const target = parseEvalRunTarget(input);
|
|
2248
|
+
const { connection, client } = await clientFor(options, target.url);
|
|
2249
|
+
try {
|
|
2250
|
+
const completed = await waitForEvaluation(
|
|
2251
|
+
client,
|
|
2252
|
+
target.id,
|
|
2253
|
+
options.timeout
|
|
2254
|
+
);
|
|
2255
|
+
print(
|
|
2256
|
+
{
|
|
2257
|
+
connection: connection.name,
|
|
2258
|
+
run: await readEvaluation(
|
|
2259
|
+
client,
|
|
2260
|
+
target.id,
|
|
2261
|
+
options.includeContent ?? false
|
|
2262
|
+
)
|
|
2263
|
+
},
|
|
2264
|
+
options
|
|
2265
|
+
);
|
|
2266
|
+
process.exitCode = resultExitCode(completed.status);
|
|
2267
|
+
} catch (error) {
|
|
2268
|
+
process.exitCode = 2;
|
|
2269
|
+
throw error;
|
|
2270
|
+
}
|
|
2271
|
+
});
|
|
2056
2272
|
common(
|
|
2057
2273
|
runs.command("cancel <id-or-url>").description(
|
|
2058
2274
|
"Request cooperative cancellation of a queued or running eval."
|
|
@@ -2060,7 +2276,14 @@ function registerStudioEvalCommands(studio, log, request = fetch) {
|
|
|
2060
2276
|
).action(async (input, options) => {
|
|
2061
2277
|
const target = parseEvalRunTarget(input);
|
|
2062
2278
|
const { connection, client } = await clientFor(options, target.url);
|
|
2063
|
-
|
|
2279
|
+
let data;
|
|
2280
|
+
try {
|
|
2281
|
+
data = await client.cancelEvaluation(target.id);
|
|
2282
|
+
} catch (error) {
|
|
2283
|
+
if (!(error instanceof StudioReadError) || error.status !== 404)
|
|
2284
|
+
throw error;
|
|
2285
|
+
data = await client.cancel(target.id);
|
|
2286
|
+
}
|
|
2064
2287
|
print({ connection: connection.name, id: target.id, ...data }, options);
|
|
2065
2288
|
});
|
|
2066
2289
|
}
|
|
@@ -4093,7 +4316,7 @@ var main = async () => {
|
|
|
4093
4316
|
console.error(
|
|
4094
4317
|
process.argv.includes("--json") ? JSON.stringify(error.toJSON()) : `[${error.code}] ${error.message}`
|
|
4095
4318
|
);
|
|
4096
|
-
process.exitCode
|
|
4319
|
+
process.exitCode ??= 1;
|
|
4097
4320
|
return;
|
|
4098
4321
|
}
|
|
4099
4322
|
const failure = (0, import_errors.serializeFailure)(error);
|