@cursor/july 0.1.5 → 0.1.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (177) hide show
  1. package/dist/ab.d.ts +8 -95
  2. package/dist/ab.d.ts.map +1 -1
  3. package/dist/ab.js +9 -150
  4. package/dist/bin/agent-serve.js +41 -8
  5. package/dist/channels/slack/post-update-delivery.d.ts +85 -0
  6. package/dist/channels/slack/post-update-delivery.d.ts.map +1 -0
  7. package/dist/docs/404.html +2 -2
  8. package/dist/docs/ab.html +4 -4
  9. package/dist/docs/assets/{app.DabPG-io.js → app.COTN7wgo.js} +1 -1
  10. package/dist/docs/assets/chunks/@localSearchIndexroot.B7UcKvIn.js +1 -0
  11. package/dist/docs/assets/chunks/{VPLocalSearchBox.jmyr0bU0.js → VPLocalSearchBox.BW3TBdT0.js} +1 -1
  12. package/dist/docs/assets/chunks/{theme.DysN9-VN.js → theme.BEJW0vE7.js} +2 -2
  13. package/dist/docs/assets/deployment.md.BtfEsc9S.js +55 -0
  14. package/dist/docs/assets/deployment.md.BtfEsc9S.lean.js +1 -0
  15. package/dist/docs/assets/example-agents_approval-buddy.md.8R5phXb5.js +10 -0
  16. package/dist/docs/assets/example-agents_approval-buddy.md.8R5phXb5.lean.js +1 -0
  17. package/dist/docs/assets/example-agents_benny.md.B0gjhI-p.js +7 -0
  18. package/dist/docs/assets/example-agents_benny.md.B0gjhI-p.lean.js +1 -0
  19. package/dist/docs/assets/example-agents_bugbot.md.DelIdhxB.js +11 -0
  20. package/dist/docs/assets/example-agents_bugbot.md.DelIdhxB.lean.js +1 -0
  21. package/dist/docs/assets/example-agents_codebase-wiki.md.DC6sgwn0.js +8 -0
  22. package/dist/docs/assets/example-agents_codebase-wiki.md.DC6sgwn0.lean.js +1 -0
  23. package/dist/docs/assets/example-agents_codeowners-review.md.Ku_tG2RY.js +8 -0
  24. package/dist/docs/assets/example-agents_codeowners-review.md.Ku_tG2RY.lean.js +1 -0
  25. package/dist/docs/assets/example-agents_concierge.md.4rQTSMXt.js +23 -0
  26. package/dist/docs/assets/example-agents_concierge.md.4rQTSMXt.lean.js +1 -0
  27. package/dist/docs/assets/example-agents_fsd.md.CzgUrDfi.js +15 -0
  28. package/dist/docs/assets/example-agents_fsd.md.CzgUrDfi.lean.js +1 -0
  29. package/dist/docs/assets/example-agents_index.md.CRqJlnIf.js +2 -0
  30. package/dist/docs/assets/example-agents_index.md.CRqJlnIf.lean.js +1 -0
  31. package/dist/docs/assets/example-agents_knowledge-base.md.BPJiVueF.js +11 -0
  32. package/dist/docs/assets/example-agents_knowledge-base.md.BPJiVueF.lean.js +1 -0
  33. package/dist/docs/assets/example-agents_security-reviewer.md.D2rtwDTO.js +19 -0
  34. package/dist/docs/assets/example-agents_security-reviewer.md.D2rtwDTO.lean.js +1 -0
  35. package/dist/docs/assets/example-agents_slack-agent.md.buLbgvBf.js +5 -0
  36. package/dist/docs/assets/example-agents_slack-agent.md.buLbgvBf.lean.js +1 -0
  37. package/dist/docs/assets/example-agents_weather-agent.md.C9Qv-W0o.js +24 -0
  38. package/dist/docs/assets/example-agents_weather-agent.md.C9Qv-W0o.lean.js +1 -0
  39. package/dist/docs/assets/index.md.COiu-1jL.js +20 -0
  40. package/dist/docs/assets/{index.md.Cylk70gg.lean.js → index.md.COiu-1jL.lean.js} +1 -1
  41. package/dist/docs/assets/reference_cli.md.D189RBCH.js +60 -0
  42. package/dist/docs/assets/reference_cli.md.D189RBCH.lean.js +1 -0
  43. package/dist/docs/building-with-agents.html +4 -4
  44. package/dist/docs/concepts.html +4 -4
  45. package/dist/docs/deployment.html +58 -17
  46. package/dist/docs/evals.html +4 -4
  47. package/dist/docs/example-agents/approval-buddy.html +34 -0
  48. package/dist/docs/example-agents/benny.html +31 -0
  49. package/dist/docs/example-agents/bugbot.html +35 -0
  50. package/dist/docs/example-agents/codebase-wiki.html +32 -0
  51. package/dist/docs/example-agents/codeowners-review.html +32 -0
  52. package/dist/docs/example-agents/concierge.html +47 -0
  53. package/dist/docs/example-agents/fsd.html +39 -0
  54. package/dist/docs/example-agents/index.html +26 -0
  55. package/dist/docs/example-agents/knowledge-base.html +35 -0
  56. package/dist/docs/example-agents/security-reviewer.html +43 -0
  57. package/dist/docs/example-agents/slack-agent.html +29 -0
  58. package/dist/docs/example-agents/weather-agent.html +48 -0
  59. package/dist/docs/guides/agent-to-agent.html +4 -4
  60. package/dist/docs/guides/cloud-runtime.html +5 -5
  61. package/dist/docs/guides/github.html +4 -4
  62. package/dist/docs/guides/human-in-the-loop.html +4 -4
  63. package/dist/docs/guides/slack.html +4 -4
  64. package/dist/docs/guides/webhooks.html +4 -4
  65. package/dist/docs/hashmap.json +1 -1
  66. package/dist/docs/hillclimbing.html +4 -4
  67. package/dist/docs/index.html +7 -7
  68. package/dist/docs/quickstart.html +4 -4
  69. package/dist/docs/reference/agent-config.html +4 -4
  70. package/dist/docs/reference/channels.html +4 -4
  71. package/dist/docs/reference/cli.html +52 -30
  72. package/dist/docs/reference/connections.html +4 -4
  73. package/dist/docs/reference/hooks.html +4 -4
  74. package/dist/docs/reference/http-api.html +4 -4
  75. package/dist/docs/reference/instructions.html +4 -4
  76. package/dist/docs/reference/playground.html +4 -4
  77. package/dist/docs/reference/project-layout.html +4 -4
  78. package/dist/docs/reference/schedules.html +4 -4
  79. package/dist/docs/reference/sessions.html +4 -4
  80. package/dist/docs/reference/skills.html +4 -4
  81. package/dist/docs/reference/subagents.html +4 -4
  82. package/dist/docs/reference/tools.html +4 -4
  83. package/dist/docs/scaffolding-agents.html +4 -4
  84. package/dist/docs/storage.html +4 -4
  85. package/dist/docs/troubleshooting.html +4 -4
  86. package/dist/evals.d.ts +5 -62
  87. package/dist/evals.d.ts.map +1 -1
  88. package/dist/evals.js +3 -66
  89. package/dist/index.d.ts +1 -1
  90. package/dist/index.d.ts.map +1 -1
  91. package/dist/internal/ab-collector.d.ts +7 -5
  92. package/dist/internal/ab-collector.d.ts.map +1 -1
  93. package/dist/internal/ab-collector.js +3 -14
  94. package/dist/internal/ab-snapshot.d.ts +2 -4
  95. package/dist/internal/ab-snapshot.d.ts.map +1 -1
  96. package/dist/internal/cli-ax.d.ts +33 -5
  97. package/dist/internal/cli-ax.d.ts.map +1 -1
  98. package/dist/internal/cli-ax.js +428 -87
  99. package/dist/internal/cli-deploy.js +1 -1
  100. package/dist/internal/discovery.js +3 -3
  101. package/dist/internal/eval-run-store.d.ts +35 -30
  102. package/dist/internal/eval-run-store.d.ts.map +1 -1
  103. package/dist/internal/eval-run-store.js +88 -100
  104. package/dist/internal/evals-client.d.ts +96 -0
  105. package/dist/internal/evals-client.d.ts.map +1 -0
  106. package/dist/internal/evals-client.js +262 -0
  107. package/dist/internal/init-project.d.ts.map +1 -1
  108. package/dist/internal/init-project.js +1 -0
  109. package/dist/internal/persistence-coordinator.d.ts +127 -0
  110. package/dist/internal/persistence-coordinator.d.ts.map +1 -0
  111. package/dist/internal/playground-proxy.d.ts +5 -5
  112. package/dist/internal/playground-proxy.js +3 -3
  113. package/dist/internal/resolve-prod-target.d.ts +30 -0
  114. package/dist/internal/resolve-prod-target.d.ts.map +1 -1
  115. package/dist/internal/resolve-prod-target.js +74 -2
  116. package/dist/internal/server.d.ts.map +1 -1
  117. package/dist/internal/server.js +16 -5
  118. package/dist/internal/session-engine.d.ts +1 -2
  119. package/dist/internal/session-engine.d.ts.map +1 -1
  120. package/dist/internal/session-engine.js +14 -31
  121. package/dist/internal/storage-coordinator.d.ts +16 -15
  122. package/dist/internal/storage-coordinator.d.ts.map +1 -1
  123. package/dist/internal/storage-coordinator.js +73 -80
  124. package/dist/persistence.d.ts +184 -0
  125. package/dist/persistence.d.ts.map +1 -0
  126. package/dist/playground/assets/cursor-icons-16-CQ50JpfO.woff2 +0 -0
  127. package/dist/playground/assets/index-72vCOBWO.js +86 -0
  128. package/dist/playground/assets/index-BjnMwYoR.css +1 -0
  129. package/dist/playground/index.html +2 -2
  130. package/dist/storage.d.ts +51 -10
  131. package/dist/storage.d.ts.map +1 -1
  132. package/dist/storage.js +27 -10
  133. package/docs/README.md +34 -5
  134. package/docs/deployment.md +352 -149
  135. package/docs/example-agents/approval-buddy.md +270 -0
  136. package/docs/example-agents/benny.md +186 -0
  137. package/docs/example-agents/bugbot.md +231 -0
  138. package/docs/example-agents/codebase-wiki.md +174 -0
  139. package/docs/example-agents/codeowners-review.md +195 -0
  140. package/docs/example-agents/concierge.md +205 -0
  141. package/docs/example-agents/fsd.md +330 -0
  142. package/docs/example-agents/index.md +102 -0
  143. package/docs/example-agents/knowledge-base.md +171 -0
  144. package/docs/example-agents/security-reviewer.md +296 -0
  145. package/docs/example-agents/slack-agent.md +146 -0
  146. package/docs/example-agents/weather-agent.md +302 -0
  147. package/docs/reference/cli.md +546 -147
  148. package/package.json +1 -1
  149. package/src/ab.ts +9 -261
  150. package/src/bin/agent-serve.ts +46 -7
  151. package/src/evals.ts +5 -119
  152. package/src/index.ts +2 -0
  153. package/src/internal/ab-collector.ts +12 -22
  154. package/src/internal/ab-snapshot.ts +2 -4
  155. package/src/internal/cli-ax.ts +551 -104
  156. package/src/internal/cli-deploy.ts +1 -1
  157. package/src/internal/discovery.ts +2 -2
  158. package/src/internal/eval-run-store.ts +91 -100
  159. package/src/internal/evals-client.ts +431 -0
  160. package/src/internal/init-project.ts +1 -0
  161. package/src/internal/playground-proxy.ts +5 -5
  162. package/src/internal/resolve-prod-target.ts +101 -3
  163. package/src/internal/server.ts +17 -3
  164. package/src/internal/session-engine.ts +9 -29
  165. package/src/internal/storage-coordinator.ts +109 -101
  166. package/src/storage.ts +79 -14
  167. package/dist/docs/assets/chunks/@localSearchIndexroot.QwK5BtEH.js +0 -1
  168. package/dist/docs/assets/deployment.md.DTKwE15Z.js +0 -14
  169. package/dist/docs/assets/deployment.md.DTKwE15Z.lean.js +0 -1
  170. package/dist/docs/assets/index.md.Cylk70gg.js +0 -20
  171. package/dist/docs/assets/reference_cli.md.Bv6pOxcF.js +0 -38
  172. package/dist/docs/assets/reference_cli.md.Bv6pOxcF.lean.js +0 -1
  173. package/dist/internal/json-dir-store.js +0 -100
  174. package/dist/playground/assets/cursor-icons-outline-BxTT_FVJ.woff2 +0 -0
  175. package/dist/playground/assets/index-BEauYlII.css +0 -1
  176. package/dist/playground/assets/index-BtM0wEGg.js +0 -319
  177. package/src/internal/json-dir-store.ts +0 -109
@@ -10,6 +10,7 @@
10
10
  import { mkdir, mkdtemp, writeFile } from "node:fs/promises";
11
11
  import { tmpdir } from "node:os";
12
12
  import { join, resolve } from "node:path";
13
+ import type { EvalRunSnapshot } from "../evals.js";
13
14
  import { serve } from "../index.js";
14
15
  import type { AgentServeHandle } from "../types.js";
15
16
  import { lookupSessionContinuation, runChat } from "./chat-client.js";
@@ -23,6 +24,14 @@ import {
23
24
  filterDiscoveredEvals,
24
25
  runDiscoveredEvals,
25
26
  } from "./eval-runner.js";
27
+ import {
28
+ getRemoteEvalRun,
29
+ listRemoteEvalRuns,
30
+ listRemoteEvals,
31
+ RemoteEvalError,
32
+ runRemoteEvals,
33
+ startRemoteEvalRun,
34
+ } from "./evals-client.js";
26
35
  import {
27
36
  formatInitNextSteps,
28
37
  formatInitScaffoldSummary,
@@ -35,7 +44,12 @@ import { offerInstallCursorSkills } from "./install-cursor-skills.js";
35
44
  import { runLogsTail, sanitizeCustomerError } from "./logs-client.js";
36
45
  import { openBrowser } from "./open-browser.js";
37
46
  import { startPlaygroundProxy } from "./playground-proxy.js";
38
- import { resolveProdSlug, resolveProdTarget } from "./resolve-prod-target.js";
47
+ import {
48
+ ProdTargetError,
49
+ resolveProdAliasPlaygroundUrl,
50
+ resolveProdSlug,
51
+ resolveProdTarget,
52
+ } from "./resolve-prod-target.js";
39
53
  import { runSession } from "./run-client.js";
40
54
  import {
41
55
  fetchSessionEvents,
@@ -90,10 +104,18 @@ export interface AxCliOptions {
90
104
  url?: string;
91
105
  /**
92
106
  * Route chat/resume/run/call/eval/logs/sessions at the team's hosted
93
- * deployment
94
- * instead of a local ephemeral server. Mutually exclusive with `--url`.
107
+ * deployment instead of a local ephemeral server. Mutually exclusive with
108
+ * `--url`. For `eval`, this runs a server-side batch via
109
+ * `/v1/dev/evals/runs` (persisted when `defineStorage` has an `evals` table);
110
+ * without `--prod`, eval defaults to an ephemeral local harness.
95
111
  */
96
112
  prod?: boolean;
113
+ /**
114
+ * `eval` (`--prod` / `--url` only): start the batch via
115
+ * `POST /v1/dev/evals/runs` and return the `runId` without polling.
116
+ * Mutually exclusive with `--list`.
117
+ */
118
+ noWait?: boolean;
97
119
  /** Cursor team id for `--prod` (defaults to the account's team). */
98
120
  team?: string;
99
121
  /** Cursor API key override for `--prod` management calls. */
@@ -654,95 +676,552 @@ export async function cmdTrajectory(options: AxCliOptions): Promise<number> {
654
676
  return trajectory.ok ? 0 : 1;
655
677
  }
656
678
 
679
+ /**
680
+ * Which surface `eval` should target. Local (ephemeral) is the default;
681
+ * `--prod` uses the hosted deployment, `--url` a running server.
682
+ * Exported for tests.
683
+ */
684
+ export function resolveEvalTargetMode(
685
+ options: Pick<AxCliOptions, "prod" | "url">
686
+ ): "local" | "url" | "deployment" | { error: string } {
687
+ const flags = [
688
+ options.prod === true ? "--prod" : undefined,
689
+ options.url !== undefined ? "--url" : undefined,
690
+ ].filter((flag) => flag !== undefined);
691
+ if (flags.length > 1) {
692
+ return { error: `${flags.join(" and ")} are mutually exclusive` };
693
+ }
694
+ if (options.url !== undefined) {
695
+ return "url";
696
+ }
697
+ if (options.prod === true) {
698
+ return "deployment";
699
+ }
700
+ return "local";
701
+ }
702
+
703
+ /** Resolve the eval target; prints a CLI-ready error and returns an exit code on failure. */
704
+ async function resolveEvalTarget(
705
+ options: AxCliOptions
706
+ ): Promise<ResolvedAgentTarget | number> {
707
+ const mode = resolveEvalTargetMode(options);
708
+ if (typeof mode === "object") {
709
+ process.stderr.write(`${mode.error}\n`);
710
+ return 2;
711
+ }
712
+ if (mode === "url") {
713
+ return { agentUrl: options.url!.replace(/\/$/, "") };
714
+ }
715
+ if (mode === "local") {
716
+ return startEphemeral(options);
717
+ }
718
+ try {
719
+ const prod = await resolveProdTarget({
720
+ dir: options.dir,
721
+ cwd: options.cwd,
722
+ slug: options.slug,
723
+ team: options.team,
724
+ apiKey: options.apiKey,
725
+ fetchImpl: options.fetchImpl,
726
+ });
727
+ return { agentUrl: prod.agentUrl, headers: prod.headers };
728
+ } catch (error) {
729
+ if (error instanceof ProdTargetError && error.notFound) {
730
+ // Keep the API error text: a 404 can also mean a wrong API host or a
731
+ // disabled feature gate, and the message distinguishes those from a
732
+ // genuinely missing deployment.
733
+ process.stderr.write(
734
+ `${sanitizeCustomerError(error.message)}\n` +
735
+ ` deploy it first: ${CLI_COMMAND_NAME} deploy --slug ${error.slug}\n` +
736
+ ` or omit --prod: ${CLI_COMMAND_NAME} eval --dir ${options.dir}\n`
737
+ );
738
+ return 1;
739
+ }
740
+ process.stderr.write(
741
+ `${sanitizeCustomerError(
742
+ error instanceof Error ? error.message : String(error)
743
+ )}\n` +
744
+ `Pass --prod for the hosted deployment, or omit it to run on an ephemeral local server.\n`
745
+ );
746
+ return 1;
747
+ }
748
+ }
749
+
657
750
  export async function cmdEval(options: AxCliOptions): Promise<number> {
658
751
  const projectRoot = resolve(options.cwd ?? process.cwd(), options.dir);
752
+ const mode = resolveEvalTargetMode(options);
753
+ if (typeof mode === "object") {
754
+ process.stderr.write(`${mode.error}\n`);
755
+ return 2;
756
+ }
757
+ if (options.noWait === true) {
758
+ if (options.list === true) {
759
+ process.stderr.write("--no-wait and --list are mutually exclusive\n");
760
+ return 2;
761
+ }
762
+ if (mode === "local") {
763
+ process.stderr.write(
764
+ "--no-wait requires --prod or --url (not the default local target)\n"
765
+ );
766
+ return 2;
767
+ }
768
+ }
769
+
770
+ const filterIds =
771
+ options.evalIds === undefined || options.evalIds.length === 0
772
+ ? undefined
773
+ : options.evalIds;
774
+ const tags =
775
+ options.tags === undefined || options.tags.length === 0
776
+ ? undefined
777
+ : options.tags;
659
778
 
660
779
  if (options.list) {
661
- const { evals } = await discoverEvals(projectRoot);
662
- const listed = filterDiscoveredEvals(evals, {
663
- filterIds: options.evalIds,
664
- tags: options.tags,
665
- });
666
- if (options.json) {
667
- console.log(
668
- JSON.stringify(
669
- listed.map((e) => ({
670
- id: e.id,
671
- fileId: e.fileId,
672
- path: e.path,
673
- description: e.definition.description,
674
- tags: e.definition.tags ?? [],
675
- })),
676
- null,
677
- 2
678
- )
780
+ // List the same surface a run will execute: local filesystem by default,
781
+ // GET /v1/dev/evals on --prod / --url.
782
+ if (mode === "local") {
783
+ return listLocalEvals(projectRoot, options, filterIds, tags);
784
+ }
785
+ const target = await resolveEvalTarget(options);
786
+ if (typeof target === "number") {
787
+ return target;
788
+ }
789
+ try {
790
+ const listed = await listRemoteEvals({
791
+ baseUrl: target.agentUrl,
792
+ auth: {
793
+ bearerToken: options.bearerToken,
794
+ headers: target.headers,
795
+ },
796
+ fetchImpl: options.fetchImpl,
797
+ filterIds,
798
+ tags,
799
+ });
800
+ return printListedEvals(
801
+ listed.map((e) => ({
802
+ id: e.id,
803
+ fileId: e.fileId,
804
+ description: e.description,
805
+ tags: e.tags ?? [],
806
+ })),
807
+ options,
808
+ `No evals on ${target.agentUrl}`
679
809
  );
680
- } else if (listed.length === 0) {
681
- process.stdout.write(`No evals under ${projectRoot}/evals\n`);
682
- } else {
683
- for (const e of listed) {
684
- const tags =
685
- e.definition.tags && e.definition.tags.length > 0
686
- ? ` [${e.definition.tags.join(", ")}]`
687
- : "";
688
- const desc =
689
- e.definition.description !== undefined
690
- ? ` — ${e.definition.description}`
691
- : "";
692
- process.stdout.write(`${e.id}${tags}${desc}\n`);
810
+ } catch (error) {
811
+ process.stderr.write(
812
+ `${sanitizeCustomerError(
813
+ error instanceof Error ? error.message : String(error)
814
+ )}\n`
815
+ );
816
+ return 1;
817
+ } finally {
818
+ if (target.close !== undefined) {
819
+ await target.close();
693
820
  }
694
821
  }
695
- return 0;
696
822
  }
697
823
 
698
- const target = await resolveTarget(options);
824
+ const target = await resolveEvalTarget(options);
825
+ if (typeof target === "number") {
826
+ return target;
827
+ }
699
828
  const stream = shouldStreamProgress(options);
700
- const progress = stream ? createStreamProgress() : undefined;
829
+
701
830
  try {
702
831
  if (options.verbose || stream) {
703
832
  process.stderr.write(`eval target: ${target.agentUrl}\n`);
704
833
  }
705
- const results = await runDiscoveredEvals({
706
- projectRoot,
834
+
835
+ // Default local: ephemeral in-process harness against local eval sources.
836
+ // --prod / --url: run on the serve process via /v1/dev/evals/runs so the
837
+ // batch lands in playground history and defineStorage evals (when set).
838
+ if (mode === "local") {
839
+ const progress = stream ? createStreamProgress() : undefined;
840
+ try {
841
+ const results = await runDiscoveredEvals({
842
+ projectRoot,
843
+ baseUrl: target.agentUrl,
844
+ filterIds,
845
+ tags,
846
+ timeoutMs: options.timeoutMs,
847
+ bearerToken: options.bearerToken,
848
+ headers: target.headers,
849
+ verbose: Boolean(options.verbose) || stream,
850
+ onEvent:
851
+ progress === undefined ? undefined : (e) => progress.onEvent(e),
852
+ });
853
+ if (results.length === 0) {
854
+ process.stderr.write("No matching evals found.\n");
855
+ return 2;
856
+ }
857
+ return printEvalResults(results, options);
858
+ } finally {
859
+ progress?.close();
860
+ }
861
+ }
862
+
863
+ try {
864
+ if (options.noWait === true) {
865
+ const started = await startRemoteEvalRun({
866
+ baseUrl: target.agentUrl,
867
+ filterIds,
868
+ tags,
869
+ timeoutMs: options.timeoutMs,
870
+ verbose: Boolean(options.verbose),
871
+ auth: {
872
+ bearerToken: options.bearerToken,
873
+ headers: target.headers,
874
+ },
875
+ fetchImpl: options.fetchImpl,
876
+ });
877
+ return printEvalAccepted(started, options, target.agentUrl);
878
+ }
879
+
880
+ const progress = createRemoteEvalProgress(
881
+ options.verbose === true || stream
882
+ );
883
+ const { snapshot, results } = await runRemoteEvals({
884
+ baseUrl: target.agentUrl,
885
+ filterIds,
886
+ tags,
887
+ timeoutMs: options.timeoutMs,
888
+ verbose: Boolean(options.verbose) || stream,
889
+ bearerToken: options.bearerToken,
890
+ headers: target.headers,
891
+ fetchImpl: options.fetchImpl,
892
+ onSnapshot: progress === undefined ? undefined : progress.onSnapshot,
893
+ });
894
+ if (results.length === 0 && snapshot.status !== "failed") {
895
+ process.stderr.write("No matching evals found.\n");
896
+ return 2;
897
+ }
898
+ return printEvalResults(results, options, {
899
+ runId: snapshot.runId,
900
+ status: snapshot.status,
901
+ error: snapshot.error,
902
+ });
903
+ } catch (error) {
904
+ if (
905
+ error instanceof RemoteEvalError &&
906
+ error.code === "no_matching_evals"
907
+ ) {
908
+ process.stderr.write(`${error.message}\n`);
909
+ return 2;
910
+ }
911
+ process.stderr.write(
912
+ `${sanitizeCustomerError(
913
+ error instanceof Error ? error.message : String(error)
914
+ )}\n`
915
+ );
916
+ return 1;
917
+ }
918
+ } finally {
919
+ if (target.close !== undefined) {
920
+ await target.close();
921
+ }
922
+ }
923
+ }
924
+
925
+ /**
926
+ * Flags to re-target the same surface from an `eval --no-wait` hint.
927
+ * Exported for tests.
928
+ */
929
+ export function formatEvalStatusFlags(
930
+ options: Pick<AxCliOptions, "prod" | "url" | "slug" | "team">
931
+ ): string {
932
+ const parts: string[] = [];
933
+ if (options.url !== undefined) {
934
+ parts.push(`--url ${options.url.replace(/\/$/, "")}`);
935
+ } else {
936
+ parts.push("--prod");
937
+ if (options.slug !== undefined && options.slug !== "") {
938
+ parts.push(`--slug ${options.slug}`);
939
+ }
940
+ }
941
+ if (options.team !== undefined && options.team !== "") {
942
+ parts.push(`--team ${options.team}`);
943
+ }
944
+ return parts.length === 0 ? "" : ` ${parts.join(" ")}`;
945
+ }
946
+
947
+ /** `--no-wait`: print the accepted batch id (and how to poll) then exit. */
948
+ function printEvalAccepted(
949
+ started: EvalRunSnapshot,
950
+ options: AxCliOptions,
951
+ agentUrl: string
952
+ ): number {
953
+ const pollPath = `${agentUrl.replace(/\/$/, "")}/v1/dev/evals/runs/${started.runId}`;
954
+ const statusFlags = formatEvalStatusFlags(options);
955
+ if (options.json) {
956
+ console.log(
957
+ JSON.stringify(
958
+ {
959
+ ok: true,
960
+ accepted: true,
961
+ runId: started.runId,
962
+ status: started.status,
963
+ summary: started.summary,
964
+ pollUrl: pollPath,
965
+ durableRuns: started.config.durableRuns === true,
966
+ },
967
+ null,
968
+ 2
969
+ )
970
+ );
971
+ } else {
972
+ const durable = started.config.durableRuns === true ? " (persisted)" : "";
973
+ process.stdout.write(
974
+ `Accepted eval run ${started.runId}${durable} — ${started.summary.total} case(s)\n`
975
+ );
976
+ process.stdout.write(
977
+ `Status: ${CLI_COMMAND_NAME} eval status ${started.runId}${statusFlags}\n`
978
+ );
979
+ process.stdout.write(`Poll: ${pollPath}\n`);
980
+ }
981
+ return 0;
982
+ }
983
+
984
+ /**
985
+ * `eval status [runId]` — fetch one batch (or list recent) on the deployment /
986
+ * `--url` target. Exit `0` when the batch is completed with all cases passing,
987
+ * `1` when failed / any case failed / unknown, `2` for usage errors, `3` when
988
+ * the batch is still running (for poll loops after `--no-wait`).
989
+ */
990
+ export async function cmdEvalStatus(
991
+ runId: string | undefined,
992
+ options: AxCliOptions
993
+ ): Promise<number> {
994
+ const mode = resolveEvalTargetMode(options);
995
+ if (typeof mode === "object") {
996
+ process.stderr.write(`${mode.error}\n`);
997
+ return 2;
998
+ }
999
+ if (mode === "local") {
1000
+ process.stderr.write(
1001
+ "eval status requires --prod or --url (not the default local target)\n"
1002
+ );
1003
+ return 2;
1004
+ }
1005
+ if (options.list === true || options.noWait === true) {
1006
+ process.stderr.write("eval status does not accept --list or --no-wait\n");
1007
+ return 2;
1008
+ }
1009
+
1010
+ const target = await resolveEvalTarget(options);
1011
+ if (typeof target === "number") {
1012
+ return target;
1013
+ }
1014
+ const auth = {
1015
+ bearerToken: options.bearerToken,
1016
+ headers: target.headers,
1017
+ };
1018
+
1019
+ try {
1020
+ if (runId === undefined || runId === "") {
1021
+ const { runs, activeRunId } = await listRemoteEvalRuns({
1022
+ baseUrl: target.agentUrl,
1023
+ auth,
1024
+ fetchImpl: options.fetchImpl,
1025
+ });
1026
+ if (options.json) {
1027
+ console.log(JSON.stringify({ runs, activeRunId }, null, 2));
1028
+ } else if (runs.length === 0) {
1029
+ process.stdout.write("No eval runs on this agent.\n");
1030
+ } else {
1031
+ if (activeRunId !== undefined) {
1032
+ process.stdout.write(`active: ${activeRunId}\n`);
1033
+ }
1034
+ for (const run of runs) {
1035
+ process.stdout.write(`${formatEvalRunSummaryLine(run)}\n`);
1036
+ }
1037
+ }
1038
+ return 0;
1039
+ }
1040
+
1041
+ const snap = await getRemoteEvalRun({
707
1042
  baseUrl: target.agentUrl,
708
- filterIds:
709
- options.evalIds === undefined || options.evalIds.length === 0
710
- ? undefined
711
- : options.evalIds,
712
- tags:
713
- options.tags === undefined || options.tags.length === 0
714
- ? undefined
715
- : options.tags,
716
- timeoutMs: options.timeoutMs,
717
- bearerToken: options.bearerToken,
718
- headers: target.headers,
719
- verbose: Boolean(options.verbose) || stream,
720
- onEvent: progress === undefined ? undefined : (e) => progress.onEvent(e),
1043
+ runId,
1044
+ auth,
1045
+ fetchImpl: options.fetchImpl,
721
1046
  });
722
-
723
- if (results.length === 0) {
724
- process.stderr.write("No matching evals found.\n");
725
- return 2;
1047
+ if (options.json) {
1048
+ console.log(JSON.stringify(snap, null, 2));
1049
+ } else {
1050
+ process.stdout.write(`${formatEvalRunSummaryLine(snap)}\n`);
1051
+ if (snap.error !== undefined && snap.error !== "") {
1052
+ process.stdout.write(`error: ${snap.error}\n`);
1053
+ }
1054
+ for (const c of snap.cases) {
1055
+ const mark =
1056
+ c.status !== "done" ? c.status : c.ok === true ? "PASS" : "FAIL";
1057
+ const dur =
1058
+ c.durationMs === undefined
1059
+ ? ""
1060
+ : ` (${(c.durationMs / 1000).toFixed(1)}s)`;
1061
+ process.stdout.write(` ${mark} ${c.id}${dur}\n`);
1062
+ if (c.error !== undefined && c.error !== "") {
1063
+ process.stdout.write(` error: ${c.error}\n`);
1064
+ }
1065
+ }
1066
+ }
1067
+ if (snap.status === "running") {
1068
+ return 3;
726
1069
  }
727
- return printEvalResults(results, options);
1070
+ const failed =
1071
+ snap.status === "failed" || snap.cases.some((c) => c.ok !== true);
1072
+ return failed ? 1 : 0;
1073
+ } catch (error) {
1074
+ process.stderr.write(
1075
+ `${sanitizeCustomerError(
1076
+ error instanceof Error ? error.message : String(error)
1077
+ )}\n`
1078
+ );
1079
+ return 1;
728
1080
  } finally {
729
- progress?.close();
730
1081
  if (target.close !== undefined) {
731
1082
  await target.close();
732
1083
  }
733
1084
  }
734
1085
  }
735
1086
 
1087
+ function formatEvalRunSummaryLine(run: EvalRunSnapshot): string {
1088
+ const { passed, failed, total, done } = run.summary;
1089
+ const durable = run.config.durableRuns === true ? " persisted" : "";
1090
+ return `${run.runId} ${run.status} ${done}/${total} done (${passed} passed, ${failed} failed)${durable}`;
1091
+ }
1092
+
1093
+ async function listLocalEvals(
1094
+ projectRoot: string,
1095
+ options: AxCliOptions,
1096
+ filterIds: string[] | undefined,
1097
+ tags: string[] | undefined
1098
+ ): Promise<number> {
1099
+ const { evals } = await discoverEvals(projectRoot);
1100
+ const listed = filterDiscoveredEvals(evals, { filterIds, tags });
1101
+ return printListedEvals(
1102
+ listed.map((e) => ({
1103
+ id: e.id,
1104
+ fileId: e.fileId,
1105
+ path: e.path,
1106
+ description: e.definition.description,
1107
+ tags: e.definition.tags ?? [],
1108
+ })),
1109
+ options,
1110
+ `No evals under ${projectRoot}/evals`
1111
+ );
1112
+ }
1113
+
1114
+ function printListedEvals(
1115
+ listed: Array<{
1116
+ id: string;
1117
+ fileId: string;
1118
+ path?: string;
1119
+ description?: string;
1120
+ tags: string[];
1121
+ }>,
1122
+ options: AxCliOptions,
1123
+ emptyMessage: string
1124
+ ): number {
1125
+ if (options.json) {
1126
+ console.log(JSON.stringify(listed, null, 2));
1127
+ } else if (listed.length === 0) {
1128
+ process.stdout.write(`${emptyMessage}\n`);
1129
+ } else {
1130
+ for (const e of listed) {
1131
+ const tagSuffix = e.tags.length > 0 ? ` [${e.tags.join(", ")}]` : "";
1132
+ const desc = e.description !== undefined ? ` — ${e.description}` : "";
1133
+ process.stdout.write(`${e.id}${tagSuffix}${desc}\n`);
1134
+ }
1135
+ }
1136
+ return 0;
1137
+ }
1138
+
1139
+ /**
1140
+ * Remote batches don't stream turn events to the CLI; print run/case progress
1141
+ * (and case logs once they appear on the snapshot) instead.
1142
+ */
1143
+ function createRemoteEvalProgress(
1144
+ enabled: boolean
1145
+ ): { onSnapshot: (snap: EvalRunSnapshot) => void } | undefined {
1146
+ if (!enabled) {
1147
+ return undefined;
1148
+ }
1149
+ let announced = false;
1150
+ let lastSummaryKey = "";
1151
+ const seenCaseKeys = new Set<string>();
1152
+ return {
1153
+ onSnapshot(snap) {
1154
+ if (!announced) {
1155
+ announced = true;
1156
+ const durable = snap.config.durableRuns === true ? " (persisted)" : "";
1157
+ process.stderr.write(`eval run: ${snap.runId}${durable}\n`);
1158
+ }
1159
+ const summaryKey = `${snap.status}:${snap.summary.done}:${snap.summary.passed}:${snap.summary.failed}`;
1160
+ if (summaryKey !== lastSummaryKey) {
1161
+ lastSummaryKey = summaryKey;
1162
+ const { done, total, passed, failed } = snap.summary;
1163
+ process.stderr.write(
1164
+ `eval progress: ${done}/${total} done (${passed} passed, ${failed} failed) [${snap.status}]\n`
1165
+ );
1166
+ }
1167
+ for (const c of snap.cases) {
1168
+ if (c.status !== "done") {
1169
+ continue;
1170
+ }
1171
+ const caseKey = `${c.id}:${c.ok === true ? "pass" : "fail"}:${c.durationMs ?? 0}`;
1172
+ if (seenCaseKeys.has(caseKey)) {
1173
+ continue;
1174
+ }
1175
+ seenCaseKeys.add(caseKey);
1176
+ process.stderr.write(
1177
+ ` eval ${c.id}: ${c.ok === true ? "PASS" : "FAIL"}${
1178
+ c.error !== undefined ? ` — ${c.error}` : ""
1179
+ }\n`
1180
+ );
1181
+ for (const line of c.logs ?? []) {
1182
+ process.stderr.write(` ${line}\n`);
1183
+ }
1184
+ }
1185
+ },
1186
+ };
1187
+ }
1188
+
736
1189
  function printEvalResults(
737
1190
  results: EvalRunResult[],
738
- options: AxCliOptions
1191
+ options: AxCliOptions,
1192
+ batch?: {
1193
+ runId?: string;
1194
+ status?: EvalRunSnapshot["status"];
1195
+ error?: string;
1196
+ }
739
1197
  ): number {
740
1198
  const passed = results.filter((r) => r.ok).length;
741
1199
  const failed = results.length - passed;
1200
+ const batchFailed = batch?.status === "failed";
1201
+ const ok = failed === 0 && !batchFailed;
1202
+
1203
+ if (batch?.error !== undefined && batch.error !== "" && !options.json) {
1204
+ // Always surface harness-level failures (not only --verbose).
1205
+ process.stderr.write(`eval run error: ${batch.error}\n`);
1206
+ }
742
1207
 
743
1208
  if (options.json) {
744
1209
  console.log(
745
- JSON.stringify({ ok: failed === 0, passed, failed, results }, null, 2)
1210
+ JSON.stringify(
1211
+ {
1212
+ ok,
1213
+ passed,
1214
+ failed,
1215
+ results,
1216
+ ...(batch?.runId === undefined ? {} : { runId: batch.runId }),
1217
+ ...(batch?.status === undefined ? {} : { status: batch.status }),
1218
+ ...(batch?.error === undefined || batch.error === ""
1219
+ ? {}
1220
+ : { error: batch.error }),
1221
+ },
1222
+ null,
1223
+ 2
1224
+ )
746
1225
  );
747
1226
  } else {
748
1227
  for (const r of results) {
@@ -773,7 +1252,7 @@ function printEvalResults(
773
1252
  `\n${passed} passed, ${failed} failed, ${results.length} total\n`
774
1253
  );
775
1254
  }
776
- return failed === 0 ? 0 : 1;
1255
+ return ok ? 0 : 1;
777
1256
  }
778
1257
 
779
1258
  /**
@@ -991,9 +1470,9 @@ export async function cmdSession(
991
1470
  /**
992
1471
  * `playground` — open the web UI against local / `--url` / `--prod`.
993
1472
  *
994
- * Hosted engines need short-lived access headers a browser cannot set, so
995
- * `--prod` (and `--bearer-token`) run a loopback reverse proxy that injects
996
- * them. Ctrl-C stops the proxy.
1473
+ * `--prod` opens the stable hosted alias playground URL; the SPA collects
1474
+ * the alias token. `--bearer-token` (and other injected headers) still run a
1475
+ * loopback reverse proxy — Ctrl-C stops that proxy.
997
1476
  */
998
1477
  export async function cmdPlayground(options: AxCliOptions): Promise<number> {
999
1478
  if (options.prod === true && options.url !== undefined) {
@@ -1006,17 +1485,18 @@ export async function cmdPlayground(options: AxCliOptions): Promise<number> {
1006
1485
  ? options.sessionId.trim()
1007
1486
  : undefined;
1008
1487
 
1009
- // Hosted: mint engineAccess and refresh it for the life of the proxy.
1488
+ // Hosted: open the alias playground; SPA owns alias-token sign-in.
1010
1489
  if (options.prod === true) {
1011
- let initial: Awaited<ReturnType<typeof resolveProdTarget>>;
1490
+ let resolved: Awaited<ReturnType<typeof resolveProdAliasPlaygroundUrl>>;
1012
1491
  try {
1013
- initial = await resolveProdTarget({
1492
+ resolved = await resolveProdAliasPlaygroundUrl({
1014
1493
  dir: options.dir,
1015
1494
  cwd: options.cwd,
1016
1495
  slug: options.slug,
1017
1496
  team: options.team,
1018
1497
  apiKey: options.apiKey,
1019
1498
  fetchImpl: options.fetchImpl,
1499
+ sessionId,
1020
1500
  });
1021
1501
  } catch (error) {
1022
1502
  process.stderr.write(
@@ -1027,44 +1507,11 @@ export async function cmdPlayground(options: AxCliOptions): Promise<number> {
1027
1507
  return 1;
1028
1508
  }
1029
1509
 
1030
- let proxy: Awaited<ReturnType<typeof startPlaygroundProxy>>;
1031
- try {
1032
- proxy = await startPlaygroundProxy({
1033
- upstreamUrl: initial.agentUrl,
1034
- headers: initial.headers,
1035
- expiresAt: initial.expiresAt,
1036
- refreshUpstream: async () => {
1037
- const next = await resolveProdTarget({
1038
- dir: options.dir,
1039
- cwd: options.cwd,
1040
- slug: options.slug ?? initial.slug,
1041
- team: options.team,
1042
- apiKey: options.apiKey,
1043
- fetchImpl: options.fetchImpl,
1044
- err: () => {},
1045
- });
1046
- return {
1047
- upstreamUrl: next.agentUrl,
1048
- headers: next.headers,
1049
- expiresAt: next.expiresAt,
1050
- };
1051
- },
1052
- });
1053
- } catch (error) {
1054
- process.stderr.write(
1055
- `${sanitizeCustomerError(
1056
- error instanceof Error ? error.message : String(error)
1057
- )}\n`
1058
- );
1059
- return 1;
1510
+ process.stdout.write(`${resolved.playgroundUrl}\n`);
1511
+ if (options.print !== true) {
1512
+ openBrowser(resolved.playgroundUrl);
1060
1513
  }
1061
-
1062
- return await runPlaygroundProxySession({
1063
- proxy,
1064
- sessionId,
1065
- print: options.print === true,
1066
- signal: options.signal,
1067
- });
1514
+ return 0;
1068
1515
  }
1069
1516
 
1070
1517
  const target = await resolveSessionTarget(options);