@kortyx/cli 0.12.0 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,42 @@
1
1
  # Changelog
2
2
 
3
+ ## [0.14.0](https://github.com/kortyx-io/kortyx/compare/cli-v0.13.0...cli-v0.14.0) (2026-10-07)
4
+
5
+
6
+ ### Features
7
+
8
+ * **evals:** group suite runs and expose CLI results ([#284](https://github.com/kortyx-io/kortyx/issues/284)) ([146fd71](https://github.com/kortyx-io/kortyx/commit/146fd71a88587e5591dd0511a4c0cb57969f9d07))
9
+
10
+
11
+ ### Dependencies
12
+
13
+ * The following workspace dependencies were updated
14
+ * dependencies
15
+ * @kortyx/agent bumped to 0.31.0
16
+
17
+ ## [0.13.0](https://github.com/kortyx-io/kortyx/compare/cli-v0.12.0...cli-v0.13.0) (2026-10-07)
18
+
19
+
20
+ ### Features
21
+
22
+ * **api:** add typed security and tenant database extensions ([#260](https://github.com/kortyx-io/kortyx/issues/260)) ([47e4928](https://github.com/kortyx-io/kortyx/commit/47e49286f12e4470b3a327e609a65c99678b359f))
23
+ * **evals:** diagnose setup and guide first workflow runs ([#275](https://github.com/kortyx-io/kortyx/issues/275)) ([01dbf80](https://github.com/kortyx-io/kortyx/commit/01dbf803e65eec03187de9eaa316e4294f4f8dc5))
24
+
25
+
26
+ ### Bug Fixes
27
+
28
+ * **security:** remediate production dependency risks ([#267](https://github.com/kortyx-io/kortyx/issues/267)) ([319daa1](https://github.com/kortyx-io/kortyx/commit/319daa1716a8684c332842bc6b6e9bb9474219e7))
29
+ * **studio:** adopt native Drizzle migrations safely ([98238a7](https://github.com/kortyx-io/kortyx/commit/98238a770a900017dd4956f1afc56fddba21d371))
30
+ * **studio:** preserve legacy updater access ([#283](https://github.com/kortyx-io/kortyx/issues/283)) ([fb733d8](https://github.com/kortyx-io/kortyx/commit/fb733d848ac7b6e42851de810aa2f039abb535b2))
31
+
32
+
33
+ ### Dependencies
34
+
35
+ * The following workspace dependencies were updated
36
+ * dependencies
37
+ * @kortyx/agent bumped to 0.30.0
38
+ * @kortyx/telemetry-contracts bumped to 0.14.0
39
+
3
40
  ## [0.12.0](https://github.com/kortyx-io/kortyx/compare/cli-v0.11.4...cli-v0.12.0) (2026-10-03)
4
41
 
5
42
 
package/README.md CHANGED
@@ -447,3 +447,17 @@ Studio draws discovered call/return links before traffic exists. The **Observed
447
447
  `kortyx topology push` discovers shared tool definitions attached via `useTool({tool, input})` and `useReason({tools})` through local imports, custom hooks, and statically bound factories. Discovery does not execute nodes, tool factories or MCP discovery. The configured entry is still imported to obtain the workflow registry.
448
448
 
449
449
  Published node capabilities contain names, descriptions, calling mode, safe input-field summaries and discovery freshness. Dynamic attachments produce an unresolved warning rather than an empty-tools claim. Studio merges real observed tools and shows execution outcomes/durations separately from cached reuse. `--dry-run --json` exposes the discovered attachments and status without publishing.
450
+
451
+ ## Diagnose eval setup
452
+
453
+ ```sh
454
+ kortyx studio evals doctor --connection staging --target catalog --suite catalog-smoke
455
+ ```
456
+
457
+ Checks Studio access, execution permission, target/environment, authenticated
458
+ consumer manifest/suites and advertised judge compatibility. Discovery GET only;
459
+ no workflows, model calls or saved runs are started. Consumer GET wrappers may
460
+ authenticate a test actor. Failures include actionable remedies. Use `--judge app`
461
+ for a code judge or `--json` for a versioned report; failures exit 1. A successful
462
+ check still needs a representative run to verify tool/provider access and saved
463
+ results. See the [first eval guide](https://kortyx.io/docs/studio/first-eval).
package/dist/index.js CHANGED
@@ -257,6 +257,7 @@ services:
257
257
  db-init:
258
258
  image: \${KORTYX_API_IMAGE_REF:-\${KORTYX_API_IMAGE:-ghcr.io/kortyx-io/kortyx-api}:\${KORTYX_STUDIO_IMAGE_TAG:-latest}}
259
259
  pull_policy: \${KORTYX_STUDIO_PULL_POLICY:-always}
260
+ user: "1000:1000"
260
261
  environment:
261
262
  <<: [*api-env, *bootstrap-keys]
262
263
  command: >
@@ -270,6 +271,7 @@ services:
270
271
  api:
271
272
  image: \${KORTYX_API_IMAGE_REF:-\${KORTYX_API_IMAGE:-ghcr.io/kortyx-io/kortyx-api}:\${KORTYX_STUDIO_IMAGE_TAG:-latest}}
272
273
  pull_policy: \${KORTYX_STUDIO_PULL_POLICY:-always}
274
+ user: "1000:1000"
273
275
  environment:
274
276
  <<: *api-env
275
277
  NODE_ENV: production
@@ -329,6 +331,9 @@ services:
329
331
  image: \${KORTYX_API_IMAGE_REF:-\${KORTYX_API_IMAGE:-ghcr.io/kortyx-io/kortyx-api}:\${KORTYX_STUDIO_IMAGE_TAG:-latest}}
330
332
  pull_policy: \${KORTYX_STUDIO_PULL_POLICY:-always}
331
333
  restart: unless-stopped
334
+ # The updater alone needs root for the host Docker socket and ownership-safe
335
+ # writes to the bind-mounted Studio state directory.
336
+ user: "0:0"
332
337
  command: ["node", "apps/api/dist/updater.js", "serve", "\${KORTYX_STUDIO_STATE_DIR}"]
333
338
  volumes:
334
339
  - type: bind
@@ -832,7 +837,7 @@ var createConnectionsCommand = (log = console.log, request = fetch) => {
832
837
  "/v1/studio/context",
833
838
  import_telemetry_contracts2.StudioContextResponseSchema
834
839
  );
835
- if (!context.apiKey.scopes.includes("studio:read"))
840
+ if (!context.apiKey?.scopes.includes("studio:read"))
836
841
  throw new StudioReadError(
837
842
  "missing_scope",
838
843
  "The API key lacks studio:read permission."
@@ -1425,7 +1430,9 @@ var parseEvalRunTarget = (input) => {
1425
1430
  } catch {
1426
1431
  throw new StudioReadError("invalid_target", "Invalid eval run URL.");
1427
1432
  }
1428
- const match = locator.pathname.match(/\/evals\/runs\/([^/]+)\/?$/);
1433
+ const match = locator.pathname.match(
1434
+ /\/evals\/(?:runs|evaluations)\/([^/]+)\/?$/
1435
+ );
1429
1436
  if (!match || !["http:", "https:"].includes(locator.protocol) || locator.username || locator.password)
1430
1437
  throw new StudioReadError(
1431
1438
  "invalid_target",
@@ -1441,6 +1448,59 @@ var parseEvalRunTarget = (input) => {
1441
1448
  return { id, url };
1442
1449
  };
1443
1450
  var StudioEvalClient = class extends StudioApiTransport {
1451
+ evaluations() {
1452
+ return this.requestJson(
1453
+ "GET",
1454
+ "/v1/studio/evals/evaluations",
1455
+ import_evals.StudioEvaluationHistorySchema
1456
+ );
1457
+ }
1458
+ evaluation(id) {
1459
+ this.validateId(id);
1460
+ return this.requestJson(
1461
+ "GET",
1462
+ `/v1/studio/evals/evaluations/${id}`,
1463
+ import_evals.StudioEvaluationDetailSchema
1464
+ );
1465
+ }
1466
+ evaluationResults(id) {
1467
+ this.validateId(id);
1468
+ return this.requestJson(
1469
+ "GET",
1470
+ `/v1/studio/evals/evaluations/${id}/results`,
1471
+ import_evals.StudioEvaluationResultsSchema
1472
+ );
1473
+ }
1474
+ startEvaluation(input) {
1475
+ const parsed = import_evals.StudioEvaluationStartRequestSchema.safeParse(input);
1476
+ if (!parsed.success)
1477
+ throw new StudioReadError(
1478
+ "invalid_eval_request",
1479
+ "Invalid evaluation selection or execution limits."
1480
+ );
1481
+ return this.requestJson(
1482
+ "POST",
1483
+ "/v1/studio/evals/evaluations",
1484
+ import_zod3.z.object({ id: import_zod3.z.uuid() }),
1485
+ {},
1486
+ parsed.data
1487
+ );
1488
+ }
1489
+ cancelEvaluation(id) {
1490
+ this.validateId(id);
1491
+ return this.requestJson(
1492
+ "POST",
1493
+ `/v1/studio/evals/evaluations/${id}/cancel`,
1494
+ import_zod3.z.object({ ok: import_zod3.z.literal(true) })
1495
+ );
1496
+ }
1497
+ validateId(id) {
1498
+ if (!import_zod3.z.uuid().safeParse(id).success)
1499
+ throw new StudioReadError(
1500
+ "invalid_target",
1501
+ "Expected an evaluation UUID."
1502
+ );
1503
+ }
1444
1504
  targets() {
1445
1505
  return this.requestJson(
1446
1506
  "GET",
@@ -1490,6 +1550,162 @@ var StudioEvalClient = class extends StudioApiTransport {
1490
1550
  }
1491
1551
  };
1492
1552
 
1553
+ // src/studio/eval-doctor.ts
1554
+ var advice = {
1555
+ environment_forbidden: {
1556
+ message: "Target environment is not allowed in this Studio project.",
1557
+ remedy: "Allow the target environment in the project, or correct the target's environment label."
1558
+ },
1559
+ environment_unavailable: {
1560
+ message: "Studio could not check the target environment.",
1561
+ remedy: "Inspect Studio API database connectivity and private server logs."
1562
+ },
1563
+ endpoint_not_found: {
1564
+ message: "Application endpoint returned HTTP 404.",
1565
+ remedy: "Ensure the eval route is mounted and enabled in the deployed application; check the target URL and reverse proxy route."
1566
+ },
1567
+ endpoint_unauthorized: {
1568
+ message: "Application endpoint rejected discovery credentials.",
1569
+ remedy: "Match the consumer handler service key to the Studio target. If the app authenticates a test actor during discovery, also check that actor's credentials and permissions."
1570
+ },
1571
+ endpoint_http_error: {
1572
+ message: "Application endpoint returned an unsuccessful HTTP status.",
1573
+ remedy: "Inspect the consumer and proxy logs. Check deployment configuration and any app-owned test-identity initialization."
1574
+ },
1575
+ endpoint_unreachable: {
1576
+ message: "Studio API could not fetch the application manifest.",
1577
+ remedy: "Check API-to-consumer networking, DNS, TLS and timeouts. From Docker Desktop use host.docker.internal for a host app; redirects are not followed."
1578
+ },
1579
+ manifest_invalid: {
1580
+ message: "Application returned an empty, oversized or incompatible manifest.",
1581
+ remedy: "Mount createEvalRouteHandler on the exact target URL and use compatible SDK/Studio releases. Check whether a proxy returned HTML instead of JSON."
1582
+ }
1583
+ };
1584
+ function buildEvalDoctorReport(data, options) {
1585
+ const checks = [
1586
+ {
1587
+ id: "studio_access",
1588
+ status: "passed",
1589
+ message: "Authenticated Studio discovery (studio:read)."
1590
+ },
1591
+ {
1592
+ id: "execution_permission",
1593
+ status: data.canRun ? "passed" : "failed",
1594
+ message: data.canRun ? "Studio key has eval:run." : "Studio key lacks eval:run.",
1595
+ ...!data.canRun ? {
1596
+ remedy: "Grant eval:run to this project key. For local bootstrap rerun with KORTYX_STUDIO_ENABLE_EVALS=1 and the existing stored key."
1597
+ } : {}
1598
+ }
1599
+ ];
1600
+ const targets = data.targets.filter(
1601
+ (target) => (!options.target || target.id === options.target) && (!options.environment || target.environment === options.environment)
1602
+ );
1603
+ checks.push({
1604
+ id: "target_selection",
1605
+ status: targets.length ? "passed" : "failed",
1606
+ message: targets.length ? `${targets.length} matching application target(s).` : "No application target matches this connection and selection.",
1607
+ ...!targets.length ? {
1608
+ remedy: "Register the target with this key's organization/project and environment. Mount KORTYX_EVAL_TARGETS_FILE on the Studio API, restart it, and check --target/--environment and the connection's default environment."
1609
+ } : {}
1610
+ });
1611
+ for (const target of targets) {
1612
+ const prefix = `${target.id} (${target.environment})`;
1613
+ const diagnostic = target.diagnostic;
1614
+ const environmentFailed = diagnostic?.code === "environment_forbidden" || diagnostic?.code === "environment_unavailable";
1615
+ checks.push({
1616
+ id: `${target.id}:environment`,
1617
+ status: environmentFailed ? "failed" : target.manifest || diagnostic ? "passed" : "skipped",
1618
+ message: `${prefix}: ${environmentFailed ? advice[diagnostic.code].message : target.manifest || diagnostic ? "target environment allowed." : "environment check unavailable on this API."}`,
1619
+ ...environmentFailed ? { remedy: advice[diagnostic.code].remedy } : {}
1620
+ });
1621
+ if (!target.manifest) {
1622
+ const guidance = diagnostic && !environmentFailed ? advice[diagnostic.code] : void 0;
1623
+ checks.push({
1624
+ id: `${target.id}:manifest`,
1625
+ status: environmentFailed ? "skipped" : "failed",
1626
+ ...diagnostic ? { code: diagnostic.code } : {},
1627
+ message: `${prefix}: ${environmentFailed ? "discovery skipped because the environment check failed." : guidance?.message ?? "consumer unavailable; this API did not supply a failure category."}${diagnostic?.httpStatus ? ` (HTTP ${diagnostic.httpStatus})` : ""}`,
1628
+ ...!environmentFailed ? {
1629
+ remedy: guidance?.remedy ?? "Check the deployed consumer route, matching service key and API-to-consumer reachability. Upgrade the Studio API for detailed discovery diagnostics."
1630
+ } : {}
1631
+ });
1632
+ checks.push({
1633
+ id: `${target.id}:judge`,
1634
+ status: "skipped",
1635
+ message: `${prefix}: judge compatibility cannot be checked without a manifest.`
1636
+ });
1637
+ continue;
1638
+ }
1639
+ const suites = target.manifest.suites;
1640
+ const selectedSuite = options.suite ? suites.find((suite) => suite.id === options.suite) : void 0;
1641
+ const suitesReady = options.suite ? Boolean(selectedSuite) : suites.length > 0;
1642
+ checks.push({
1643
+ id: `${target.id}:manifest`,
1644
+ status: suitesReady ? "passed" : "failed",
1645
+ message: `${prefix}: authenticated manifest valid; ${suites.length} registered suite(s).${options.suite ? ` Requested suite ${options.suite} ${selectedSuite ? "found" : "missing"}.` : ""}`,
1646
+ ...!suitesReady ? {
1647
+ remedy: "Pass the intended suite to createEvals({ suites: [...] }), deploy the application and refresh discovery. Suites are fetched from the consumer; topology publication does not register them."
1648
+ } : {}
1649
+ });
1650
+ const judgeReady = options.judge === "studio" ? Boolean(data.studioJudge && target.manifest.studioJudging) : Boolean(target.manifest.judge);
1651
+ checks.push({
1652
+ id: `${target.id}:judge`,
1653
+ status: judgeReady ? "passed" : "failed",
1654
+ message: `${prefix}: ${options.judge} judge ${judgeReady ? "configured and advertised" : "unavailable"}.`,
1655
+ ...!judgeReady ? {
1656
+ remedy: options.judge === "app" ? "Provide a code judge to createEvals in the consumer, or explicitly select --judge studio." : !data.studioJudge ? "Configure KORTYX_EVAL_JUDGE_MODEL and KORTYX_EVAL_JUDGE_API_KEY on the Studio API and restart it, or explicitly select an available --judge app." : "Update the consumer SDK to advertise Studio judging support, or select an available --judge app."
1657
+ } : {}
1658
+ });
1659
+ }
1660
+ return {
1661
+ schemaVersion: 1,
1662
+ status: checks.some((check) => check.status === "failed") ? "failed" : "passed",
1663
+ checks
1664
+ };
1665
+ }
1666
+ function evalDoctorFailure(error) {
1667
+ const safe = error instanceof StudioReadError ? error : null;
1668
+ const connectionError = safe && [
1669
+ "invalid_config",
1670
+ "invalid_connection",
1671
+ "not_configured",
1672
+ "unknown_connection",
1673
+ "invalid_key_env",
1674
+ "missing_key",
1675
+ "invalid_key",
1676
+ "invalid_url",
1677
+ "incompatible_studio_api",
1678
+ "schema_mismatch",
1679
+ "connection_failed"
1680
+ ].includes(safe.code);
1681
+ return {
1682
+ schemaVersion: 1,
1683
+ status: "failed",
1684
+ checks: [
1685
+ {
1686
+ id: "studio_access",
1687
+ status: "failed",
1688
+ ...safe ? { code: safe.code } : {},
1689
+ message: connectionError ? safe.message : safe?.status === 401 ? "Studio rejected the project key (HTTP 401)." : safe?.status === 403 ? "Studio discovery requires studio:read (HTTP 403)." : safe?.status === 404 ? "Studio eval discovery endpoint was not found (HTTP 404)." : "Could not resolve the Studio connection or read a compatible discovery response.",
1690
+ remedy: "Check the selected connection, its API URL and project key, network access, and compatible CLI/Studio releases. No consumer or provider credentials belong in this CLI connection."
1691
+ }
1692
+ ]
1693
+ };
1694
+ }
1695
+ function formatEvalDoctorReport(report) {
1696
+ const symbols = { passed: "\u2713", failed: "\u2717", skipped: "\u2013" };
1697
+ return [
1698
+ "Kortyx Evals \xB7 Setup check",
1699
+ ...report.checks.flatMap((check) => [
1700
+ ` ${symbols[check.status]} ${check.message}`,
1701
+ ...check.remedy ? [` \u2192 ${check.remedy}`] : []
1702
+ ]),
1703
+ "",
1704
+ report.status === "passed" ? "Configuration checks passed. Run one representative suite to verify the test identity, tools, model access and saved results." : "Setup checks failed. Resolve the failures and run doctor again.",
1705
+ "No workflow or judge calls were started. Consumer GET discovery may run app-owned authentication logic."
1706
+ ].join("\n").replace(/\p{Cc}/gu, (character) => character === "\n" ? character : "");
1707
+ }
1708
+
1493
1709
  // src/studio/read-output.ts
1494
1710
  var parseStudioTarget = (input, entity) => {
1495
1711
  if (/^[a-z][a-z0-9+.-]*:/i.test(input)) {
@@ -1709,6 +1925,36 @@ function registerStudioEvalCommands(studio, log, request = fetch) {
1709
1925
  "--environment <name>",
1710
1926
  "Filter targets; defaults to the connection environment."
1711
1927
  );
1928
+ selectionOptions(
1929
+ evals.command("doctor").description(
1930
+ "Check deployment wiring without starting workflows or model calls."
1931
+ )
1932
+ ).option("--suite <id>", "Check that a specific suite is registered.").option(
1933
+ "--judge <location>",
1934
+ "Judge to check: studio (default) or app.",
1935
+ (value) => {
1936
+ if (value !== "studio" && value !== "app")
1937
+ throw new import_commander3.InvalidArgumentError("Expected studio or app.");
1938
+ return value;
1939
+ },
1940
+ "studio"
1941
+ ).action(async (options) => {
1942
+ let report;
1943
+ try {
1944
+ const { connection, client } = await clientFor(options);
1945
+ report = buildEvalDoctorReport(await client.targets(), {
1946
+ ...options,
1947
+ environment: options.environment ?? connection.environment,
1948
+ judge: options.judge ?? "studio"
1949
+ });
1950
+ } catch (error) {
1951
+ report = evalDoctorFailure(error);
1952
+ }
1953
+ log(
1954
+ options.json ? JSON.stringify(report) : formatEvalDoctorReport(report)
1955
+ );
1956
+ if (report.status === "failed") process.exitCode = 1;
1957
+ });
1712
1958
  selectionOptions(
1713
1959
  suites.command("list").description("List targets, suites, revisions and case IDs.")
1714
1960
  ).action(async (options) => {
@@ -1770,11 +2016,65 @@ function registerStudioEvalCommands(studio, log, request = fetch) {
1770
2016
  options
1771
2017
  );
1772
2018
  });
2019
+ const terminal = (status) => status !== "queued" && status !== "running";
2020
+ const resultExitCode = (status) => status === "passed" ? 0 : status === "failed" ? 1 : status === "cancelled" ? 130 : 2;
2021
+ const waitForEvaluation = async (client, id, timeout) => {
2022
+ const deadline = Date.now() + timeout * 1e3;
2023
+ for (; ; ) {
2024
+ const { run: run2 } = await client.evaluation(id);
2025
+ if (terminal(run2.status)) return run2;
2026
+ if (Date.now() >= deadline)
2027
+ throw new StudioReadError(
2028
+ "eval_wait_timeout",
2029
+ `Evaluation ${id} is still running. Inspect it with evals runs get; waiting did not cancel it.`
2030
+ );
2031
+ await new Promise(
2032
+ (resolve6) => setTimeout(resolve6, Math.min(2e3, deadline - Date.now()))
2033
+ );
2034
+ }
2035
+ };
2036
+ const readEvaluation = async (client, id, includeContent) => {
2037
+ const { run: run2 } = await client.evaluationResults(id);
2038
+ return includeContent ? run2 : { ...run2, suites: run2.suites.map(compactDetail) };
2039
+ };
2040
+ const waitOptions = (command) => command.option(
2041
+ "--timeout <seconds>",
2042
+ "Maximum time to wait (1\u20137200 seconds); does not cancel execution.",
2043
+ integer2(7200),
2044
+ 1800
2045
+ );
1773
2046
  const runs = evals.command("runs").description("Start and inspect saved eval runs.");
1774
2047
  selectionOptions(
1775
- runs.command("start <suite-id>").description(
1776
- "Enqueue once and return immediately; does not wait for grades."
2048
+ runs.command("start [suite-id]").description(
2049
+ "Start an evaluation containing all or selected suites; optionally wait for results."
1777
2050
  )
2051
+ ).option("--all", "Run every suite registered on the selected application.").option(
2052
+ "--suite <id>",
2053
+ "Select a suite; repeat for several suites.",
2054
+ (value, previous) => [...previous, value],
2055
+ []
2056
+ ).option("--name <name>", "Display name for this evaluation run.").option(
2057
+ "--source <source>",
2058
+ "Trigger: manual, deployment, schedule or ci.",
2059
+ (value) => {
2060
+ if (!["manual", "deployment", "schedule", "ci"].includes(value))
2061
+ throw new import_commander3.InvalidArgumentError(
2062
+ "Expected manual, deployment, schedule or ci."
2063
+ );
2064
+ return value;
2065
+ },
2066
+ "manual"
2067
+ ).option("--commit <sha>", "Deployed application commit.").option("--deployment-url <url>", "Deployment or CI job URL.").option(
2068
+ "--idempotency-key <key>",
2069
+ "Reuse a matching saved evaluation after a trigger retry."
2070
+ ).option(
2071
+ "--wait",
2072
+ "Wait for final results and return a pass/fail exit code."
2073
+ ).option(
2074
+ "--timeout <seconds>",
2075
+ "Maximum wait in seconds (1\u20137200); execution continues after timeout.",
2076
+ integer2(7200),
2077
+ 1800
1778
2078
  ).option(
1779
2079
  "--case <id>",
1780
2080
  "Select a case; repeat for multiple cases.",
@@ -1795,39 +2095,102 @@ function registerStudioEvalCommands(studio, log, request = fetch) {
1795
2095
  },
1796
2096
  "studio"
1797
2097
  ).action(async (suiteId, options) => {
1798
- const { connection, client, target, suite, canRun } = await select(
1799
- suiteId,
1800
- options
2098
+ const ids = [...suiteId ? [suiteId] : [], ...options.suite ?? []];
2099
+ if (!options.all && !ids.length || options.all && ids.length || new Set(ids).size !== ids.length)
2100
+ throw new StudioReadError(
2101
+ "invalid_suite_selection",
2102
+ "Choose --all or unique suite IDs (positional or repeated --suite)."
2103
+ );
2104
+ const data = await discover(options);
2105
+ const matches = data.targets.filter(
2106
+ (target2) => target2.manifest && (options.all || ids.every(
2107
+ (id) => target2.manifest.suites.some((suite) => suite.id === id)
2108
+ ))
1801
2109
  );
1802
- if (!canRun)
2110
+ const target = matches[0];
2111
+ if (matches.length !== 1 || !target?.manifest)
2112
+ throw new StudioReadError(
2113
+ "invalid_suite_selection",
2114
+ "Select one available application with --target and --environment; all selected suites must belong to it."
2115
+ );
2116
+ if (!data.canRun)
1803
2117
  throw new StudioReadError(
1804
2118
  "eval_forbidden",
1805
2119
  "This Studio key requires eval:run to execute suites."
1806
2120
  );
2121
+ const suites2 = options.all ? target.manifest.suites : ids.map(
2122
+ (id) => target.manifest.suites.find((suite) => suite.id === id)
2123
+ );
1807
2124
  const caseIds = options.case ?? [];
1808
- if (new Set(caseIds).size !== caseIds.length || caseIds.some((id) => !suite.cases.some((item) => item.id === id)) || (caseIds.length || suite.cases.length) * options.repetitions > 100)
2125
+ if (caseIds.length && (options.all || suites2.length !== 1))
1809
2126
  throw new StudioReadError(
1810
2127
  "invalid_case_selection",
1811
- "Select unique existing case IDs and at most 100 total attempts."
2128
+ "--case requires exactly one selected suite."
1812
2129
  );
1813
- const run2 = await client.start({
2130
+ if (new Set(caseIds).size !== caseIds.length || caseIds.some(
2131
+ (id) => !suites2[0]?.cases.some((item) => item.id === id)
2132
+ ) || suites2.some(
2133
+ (suite) => (caseIds.length || suite.cases.length) * options.repetitions > 100
2134
+ ))
2135
+ throw new StudioReadError(
2136
+ "invalid_case_selection",
2137
+ "Select unique existing cases and at most 100 attempts per suite."
2138
+ );
2139
+ const run2 = await data.client.startEvaluation({
1814
2140
  targetId: target.id,
2141
+ selection: options.all ? "all" : "selected",
2142
+ suites: suites2.map((suite) => ({
2143
+ suiteId: suite.id,
2144
+ suiteRevision: target.revisions[suite.id] ?? "",
2145
+ ...caseIds.length ? { caseIds } : {}
2146
+ })),
1815
2147
  judge: options.judge ?? "studio",
1816
- suiteId,
1817
- suiteRevision: target.revisions[suiteId] ?? "",
1818
2148
  repetitions: options.repetitions,
1819
2149
  concurrency: options.concurrency,
1820
- ...caseIds.length ? { caseIds } : {}
1821
- });
1822
- print(
1823
- {
1824
- connection: connection.name,
1825
- id: run2.id,
1826
- status: "queued",
1827
- studioUrl: connection.studioUrl ? `${connection.studioUrl}/evals/runs/${run2.id}` : null
2150
+ name: options.name,
2151
+ metadata: {
2152
+ source: options.source ?? "manual",
2153
+ commit: options.commit,
2154
+ deploymentUrl: options.deploymentUrl
1828
2155
  },
1829
- options
1830
- );
2156
+ idempotencyKey: options.idempotencyKey
2157
+ });
2158
+ const studioUrl = data.connection.studioUrl ? `${data.connection.studioUrl}/evals/evaluations/${run2.id}` : null;
2159
+ if (!options.wait) {
2160
+ print(
2161
+ {
2162
+ connection: data.connection.name,
2163
+ id: run2.id,
2164
+ status: "queued",
2165
+ studioUrl
2166
+ },
2167
+ options
2168
+ );
2169
+ return;
2170
+ }
2171
+ try {
2172
+ const completed = await waitForEvaluation(
2173
+ data.client,
2174
+ run2.id,
2175
+ options.timeout
2176
+ );
2177
+ print(
2178
+ {
2179
+ connection: data.connection.name,
2180
+ studioUrl,
2181
+ run: await readEvaluation(
2182
+ data.client,
2183
+ run2.id,
2184
+ options.includeContent ?? false
2185
+ )
2186
+ },
2187
+ options
2188
+ );
2189
+ process.exitCode = resultExitCode(completed.status);
2190
+ } catch (error) {
2191
+ process.exitCode = 2;
2192
+ throw error;
2193
+ }
1831
2194
  });
1832
2195
  common(
1833
2196
  runs.command("list").description("List the latest 100 saved runs in this project.")
@@ -1836,12 +2199,12 @@ function registerStudioEvalCommands(studio, log, request = fetch) {
1836
2199
  "Filter results; defaults to the connection environment."
1837
2200
  ).action(async (options) => {
1838
2201
  const { connection, client } = await clientFor(options);
1839
- const data = await client.runs();
2202
+ const data = await client.evaluations();
1840
2203
  const environment = options.environment ?? connection.environment;
1841
2204
  print(
1842
2205
  {
1843
2206
  connection: connection.name,
1844
- runs: data.runs.filter((run2) => !environment || run2.environment === environment).map(summary)
2207
+ runs: data.runs.filter((run2) => !environment || run2.environment === environment).map((run2) => run2)
1845
2208
  },
1846
2209
  options
1847
2210
  );
@@ -1853,15 +2216,59 @@ function registerStudioEvalCommands(studio, log, request = fetch) {
1853
2216
  ).action(async (input, options) => {
1854
2217
  const target = parseEvalRunTarget(input);
1855
2218
  const { connection, client } = await clientFor(options, target.url);
1856
- const { run: run2 } = await client.run(target.id);
2219
+ let run2;
2220
+ try {
2221
+ run2 = await readEvaluation(
2222
+ client,
2223
+ target.id,
2224
+ options.includeContent ?? false
2225
+ );
2226
+ } catch (error) {
2227
+ if (!(error instanceof StudioReadError) || error.status !== 404)
2228
+ throw error;
2229
+ const legacy = await client.run(target.id);
2230
+ run2 = options.includeContent ? legacy.run : compactDetail(legacy.run);
2231
+ }
1857
2232
  print(
1858
2233
  {
1859
2234
  connection: connection.name,
1860
- run: options.includeContent ? run2 : compactDetail(run2)
2235
+ run: run2
1861
2236
  },
1862
2237
  options
1863
2238
  );
1864
2239
  });
2240
+ waitOptions(
2241
+ common(
2242
+ runs.command("wait <id-or-url>").description(
2243
+ "Wait for a grouped evaluation and print its final suite/case results."
2244
+ )
2245
+ )
2246
+ ).action(async (input, options) => {
2247
+ const target = parseEvalRunTarget(input);
2248
+ const { connection, client } = await clientFor(options, target.url);
2249
+ try {
2250
+ const completed = await waitForEvaluation(
2251
+ client,
2252
+ target.id,
2253
+ options.timeout
2254
+ );
2255
+ print(
2256
+ {
2257
+ connection: connection.name,
2258
+ run: await readEvaluation(
2259
+ client,
2260
+ target.id,
2261
+ options.includeContent ?? false
2262
+ )
2263
+ },
2264
+ options
2265
+ );
2266
+ process.exitCode = resultExitCode(completed.status);
2267
+ } catch (error) {
2268
+ process.exitCode = 2;
2269
+ throw error;
2270
+ }
2271
+ });
1865
2272
  common(
1866
2273
  runs.command("cancel <id-or-url>").description(
1867
2274
  "Request cooperative cancellation of a queued or running eval."
@@ -1869,7 +2276,14 @@ function registerStudioEvalCommands(studio, log, request = fetch) {
1869
2276
  ).action(async (input, options) => {
1870
2277
  const target = parseEvalRunTarget(input);
1871
2278
  const { connection, client } = await clientFor(options, target.url);
1872
- const data = await client.cancel(target.id);
2279
+ let data;
2280
+ try {
2281
+ data = await client.cancelEvaluation(target.id);
2282
+ } catch (error) {
2283
+ if (!(error instanceof StudioReadError) || error.status !== 404)
2284
+ throw error;
2285
+ data = await client.cancel(target.id);
2286
+ }
1873
2287
  print({ connection: connection.name, id: target.id, ...data }, options);
1874
2288
  });
1875
2289
  }
@@ -3902,7 +4316,7 @@ var main = async () => {
3902
4316
  console.error(
3903
4317
  process.argv.includes("--json") ? JSON.stringify(error.toJSON()) : `[${error.code}] ${error.message}`
3904
4318
  );
3905
- process.exitCode = 1;
4319
+ process.exitCode ??= 1;
3906
4320
  return;
3907
4321
  }
3908
4322
  const failure = (0, import_errors.serializeFailure)(error);