@moda-ai/cli 1.42.0 → 1.43.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5586,6 +5586,11 @@ var REPLAY_MODEL_ALLOWLIST = [
5586
5586
  var DEFAULT_POLL_INTERVAL_MS = 15000;
5587
5587
  var DEFAULT_WAIT_TIMEOUT_MS = 2 * 60 * 60 * 1000;
5588
5588
  var TERMINAL_RUN_STATUSES = new Set(["completed", "skipped", "error"]);
5589
+ var FALLBACK_SUCCESS_CRITERIA = [
5590
+ "The agent understands the user request",
5591
+ "The agent uses tools appropriately when needed",
5592
+ "The agent provides a helpful final answer"
5593
+ ];
5589
5594
  async function runPromptAb(flags, profileOptions, context) {
5590
5595
  validateConfig();
5591
5596
  const tenantId = flags["tenant-id"] || resolveApiTenantId(profileOptions);
@@ -5626,14 +5631,20 @@ async function runPromptAb(flags, profileOptions, context) {
5626
5631
  prod: resolvePromptArmSpec(baselineSource, "baseline"),
5627
5632
  proposed: resolvePromptArmSpec(candidateSource, "candidate")
5628
5633
  };
5629
- const replaySetId = await ensureReplaySet(plan, flags, tenantId, profileOptions);
5634
+ const { setId: replaySetId, caseSources, warnings } = await ensureReplaySet(plan, flags, tenantId, profileOptions);
5635
+ if (context.outputMode === "human") {
5636
+ for (const warning of warnings)
5637
+ process.stderr.write(`Warning: ${warning}
5638
+ `);
5639
+ }
5640
+ const writeOptions = warnings.length ? { warnings } : undefined;
5630
5641
  const enqueue = await callControlAPI(`/tenants/${encodeURIComponent(tenantId)}/replay-sets/${encodeURIComponent(replaySetId)}/run`, {
5631
5642
  method: "POST",
5632
5643
  body: JSON.stringify({
5633
5644
  promptArms,
5634
5645
  seedsPerCase,
5635
5646
  assistantModel: assistantModel || undefined,
5636
- promotePrimary: flags["no-promote"] !== "true",
5647
+ promotePrimary: flags.promote === "true" && flags["no-promote"] !== "true",
5637
5648
  ...plan.kind === "existing" ? replayCasePin(plan.caseIds) : {}
5638
5649
  })
5639
5650
  }, profileOptions, { timeoutMs: 120000, retries: 0 });
@@ -5643,11 +5654,12 @@ async function runPromptAb(flags, profileOptions, context) {
5643
5654
  if (noWait) {
5644
5655
  context.output.writeData({
5645
5656
  replaySetId,
5657
+ ...caseSources ? { caseSources } : {},
5646
5658
  runId: enqueue.runId,
5647
5659
  status: enqueue.status,
5648
5660
  message: enqueue.message ?? "Replay comparison queued",
5649
5661
  plannedPlayouts
5650
- });
5662
+ }, writeOptions);
5651
5663
  return;
5652
5664
  }
5653
5665
  const deadline = Date.now() + timeoutMs;
@@ -5670,6 +5682,7 @@ async function runPromptAb(flags, profileOptions, context) {
5670
5682
  }
5671
5683
  const result = {
5672
5684
  replaySetId,
5685
+ ...caseSources ? { caseSources } : {},
5673
5686
  plannedPlayouts,
5674
5687
  runId: latest.run.runId,
5675
5688
  status: latest.run.status,
@@ -5680,7 +5693,7 @@ async function runPromptAb(flags, profileOptions, context) {
5680
5693
  if (context.outputMode === "human") {
5681
5694
  printHumanVerdict(result);
5682
5695
  }
5683
- context.output.writeData(result);
5696
+ context.output.writeData(result, writeOptions);
5684
5697
  }
5685
5698
  function warnIfQueued(enqueue) {
5686
5699
  const { runsAhead, maxActiveRuns } = enqueue;
@@ -5793,7 +5806,7 @@ async function planReplaySet(flags, tenantId, profileOptions) {
5793
5806
  const snapshot = await fetchReplaySetCases(tenantId, existingSetId, profileOptions);
5794
5807
  return { kind: "existing", setId: existingSetId, cases: snapshot?.ids.length ?? null, caseIds: snapshot?.ids };
5795
5808
  }
5796
- const conversations = parseCsv(flags.traces ?? flags.conversations ?? flags["conversation-ids"]);
5809
+ const conversations = [...new Set(parseCsv(flags.traces ?? flags.conversations ?? flags["conversation-ids"]))];
5797
5810
  if (conversations.length > MAX_REPLAY_TRACES) {
5798
5811
  throw new CliInputError(`--traces has ${conversations.length} ids; the limit is ${MAX_REPLAY_TRACES} per replay set, so nothing was sent.`, "Split the traces across several runs, or use --auto-generate.");
5799
5812
  }
@@ -5861,9 +5874,9 @@ function rerunWithYes(command, positionals, flags) {
5861
5874
  }
5862
5875
  async function ensureReplaySet(plan, flags, tenantId, profileOptions) {
5863
5876
  if (plan.kind === "existing")
5864
- return plan.setId;
5877
+ return { setId: plan.setId, warnings: [] };
5865
5878
  if (plan.kind === "traces") {
5866
- return createSetFromConversations(plan.traces, flags, tenantId, profileOptions);
5879
+ return createSetFromTraces(plan.traces, flags, tenantId, profileOptions);
5867
5880
  }
5868
5881
  const caseCount = plan.cases;
5869
5882
  const lookbackDays = parsePositiveInt(flags["lookback-days"], 30, 365, "--lookback-days");
@@ -5875,7 +5888,7 @@ async function ensureReplaySet(plan, flags, tenantId, profileOptions) {
5875
5888
  if (!generated?.id) {
5876
5889
  throw new Error("Auto-generate did not return a replay set id");
5877
5890
  }
5878
- return generated.id;
5891
+ return { setId: generated.id, warnings: [] };
5879
5892
  }
5880
5893
  function parseTraceRef(ref) {
5881
5894
  const match = /^(.+)@(\d+)$/.exec(ref.trim());
@@ -5883,38 +5896,75 @@ function parseTraceRef(ref) {
5883
5896
  return { conversationId: ref.trim() };
5884
5897
  return { conversationId: match[1], startMsgIndex: Number(match[2]) };
5885
5898
  }
5886
- async function createSetFromConversations(conversationIds, flags, tenantId, profileOptions) {
5887
- const name = (flags.name ?? flags["set-name"] ?? `Prompt A/B ${conversationIds.length} trace(s)`).trim();
5888
- const created = await callControlAPI(`/tenants/${encodeURIComponent(tenantId)}/replay-sets`, {
5889
- method: "POST",
5890
- body: JSON.stringify({ name, description: "Created by moda prompts ab" })
5891
- }, profileOptions);
5892
- let position = 0;
5893
- for (const ref of conversationIds) {
5894
- const { conversationId, startMsgIndex } = parseTraceRef(ref);
5899
+ async function createSetFromTraces(conversationIds, flags, tenantId, profileOptions) {
5900
+ const refs = [...new Set(conversationIds)].map(parseTraceRef);
5901
+ const traces = [...new Set(refs.map((ref) => ref.conversationId))];
5902
+ const startByConversation = new Map;
5903
+ for (const ref of refs) {
5904
+ if (ref.startMsgIndex !== undefined)
5905
+ startByConversation.set(ref.conversationId, ref.startMsgIndex);
5906
+ }
5907
+ const name = (flags.name ?? flags["set-name"] ?? `Prompt A/B ${traces.length} trace(s)`).trim();
5908
+ const base = `/tenants/${encodeURIComponent(tenantId)}/replay-sets`;
5909
+ let set = null;
5910
+ let unavailable = null;
5911
+ try {
5912
+ set = await callControlAPI(`${base}/auto-generate`, { method: "POST", body: JSON.stringify({ name, conversationIds: traces }) }, profileOptions, { timeoutMs: 300000, retries: 0 });
5913
+ if (!set?.id)
5914
+ throw new Error("Auto-generate did not return a replay set id");
5915
+ } catch (error) {
5916
+ const status = error?.statusCode;
5917
+ if (status === 401 || status === 403)
5918
+ throw error;
5919
+ unavailable = error instanceof Error ? error.message : String(error);
5920
+ set = null;
5921
+ }
5922
+ const distilledIds = new Set((set?.cases ?? []).map((c) => String(c?.sourceConversationId ?? "")).filter(Boolean));
5923
+ const distilled = traces.filter((id) => distilledIds.has(id));
5924
+ const fallback = traces.filter((id) => !distilledIds.has(id));
5925
+ if (!set) {
5926
+ set = await callControlAPI(base, { method: "POST", body: JSON.stringify({ name, description: "Created by moda prompts ab" }) }, profileOptions);
5927
+ }
5928
+ const requestedIndex = new Map(traces.map((id, index) => [id, index]));
5929
+ for (const existing of set.cases ?? []) {
5930
+ const conversationId = String(existing?.sourceConversationId ?? "");
5931
+ const index = requestedIndex.get(conversationId);
5932
+ const startMsgIndex = startByConversation.get(conversationId);
5933
+ const update = {};
5934
+ if (fallback.length && index !== undefined && existing.position !== index)
5935
+ update.position = index;
5936
+ if (startMsgIndex !== undefined)
5937
+ update.sourceStartMsgIndex = startMsgIndex;
5938
+ if (!Object.keys(update).length)
5939
+ continue;
5940
+ await callControlAPI(`${base}/${encodeURIComponent(set.id)}/cases/${encodeURIComponent(existing.id)}`, { method: "PUT", body: JSON.stringify(update) }, profileOptions);
5941
+ }
5942
+ for (const conversationId of fallback) {
5895
5943
  const scenario = await loadScenarioFromConversation(conversationId);
5896
- await callControlAPI(`/tenants/${encodeURIComponent(tenantId)}/replay-sets/${encodeURIComponent(created.id)}/cases`, {
5944
+ await callControlAPI(`${base}/${encodeURIComponent(set.id)}/cases`, {
5897
5945
  method: "POST",
5898
5946
  body: JSON.stringify({
5899
5947
  title: conversationId,
5900
5948
  scenario,
5901
5949
  sourceConversationId: conversationId,
5902
- ...startMsgIndex !== undefined ? { sourceStartMsgIndex: startMsgIndex } : {},
5903
- successCriteria: [
5904
- "The agent understands the user request",
5905
- "The agent uses tools appropriately when needed",
5906
- "The agent provides a helpful final answer"
5907
- ],
5908
- position: position++
5950
+ ...startByConversation.has(conversationId) ? { sourceStartMsgIndex: startByConversation.get(conversationId) } : {},
5951
+ successCriteria: FALLBACK_SUCCESS_CRITERIA,
5952
+ position: requestedIndex.get(conversationId)
5909
5953
  })
5910
5954
  }, profileOptions);
5911
5955
  }
5912
- return created.id;
5956
+ const warnings = [];
5957
+ if (fallback.length) {
5958
+ const why = unavailable ? `Moda could not distill these traces (${unavailable.slice(0, 200)})` : "Moda could not distill a scenario for these traces";
5959
+ warnings.push(`${why}, so ${fallback.length} of ${traces.length} case(s) use only the opening user message and three generic ` + `success criteria, which long traces pass easily: ${fallback.slice(0, 10).join(", ")}${fallback.length > 10 ? ", …" : ""}. ` + "Edit their scenario/criteria in the dashboard, or pass --set-id with a curated set.");
5960
+ }
5961
+ return { setId: set.id, caseSources: { distilled, fallback }, warnings };
5913
5962
  }
5914
5963
  async function loadScenarioFromConversation(conversationId) {
5915
5964
  try {
5916
- const data = await callDataAPI(`/conversations/${encodeURIComponent(conversationId)}/context?msg_index=0&window=0`);
5917
- const firstUser = (data.messages ?? []).find((message) => message.role === "user" && String(message.content ?? "").trim());
5965
+ const data = await callDataAPI(`/conversations/${encodeURIComponent(conversationId)}/context?msg_index=0&window=5`);
5966
+ const messages = data.context?.messages ?? data.messages ?? [];
5967
+ const firstUser = messages.find((message) => message.role === "user" && String(message.content ?? "").trim());
5918
5968
  const opening = String(firstUser?.content ?? "").trim();
5919
5969
  if (opening) {
5920
5970
  return `The user says: "${opening}"
@@ -6446,6 +6496,7 @@ var PROMPT_WRITE_SUBCOMMAND_FLAGS = {
6446
6496
  "lookback-days",
6447
6497
  "no-promote",
6448
6498
  "no-wait",
6499
+ "promote",
6449
6500
  "poll-interval",
6450
6501
  "replay-set-id",
6451
6502
  "seeds-per-case",
package/dist/cli.js CHANGED
@@ -8,7 +8,7 @@ import {
8
8
  runPromptsCommand,
9
9
  runSkillsCommand,
10
10
  runStatusCommand
11
- } from "./cli-pr6s56fc.js";
11
+ } from "./cli-ecrx32mm.js";
12
12
  import {
13
13
  ApiError,
14
14
  HARNESS_REPORT_APPROVAL_PATH,
@@ -151,7 +151,7 @@ var WorldStateSchema = z.object({
151
151
  var ContextSchema = z.object({
152
152
  conversation_id: z.string(),
153
153
  msg_index: z.number().min(0).optional(),
154
- window: z.number().min(1).max(5).default(2).optional(),
154
+ window: z.number().int().min(0).optional(),
155
155
  all: boolFlag(),
156
156
  from: z.number().int().min(0).optional(),
157
157
  max_messages: z.number().int().min(1).max(5000).optional()
@@ -8194,6 +8194,8 @@ function pageMessages(page) {
8194
8194
  const messages = asRecord(page.context)?.messages ?? page.messages;
8195
8195
  return (Array.isArray(messages) ? messages : []).map((m) => asRecord(m) ?? {});
8196
8196
  }
8197
+ var CONTEXT_MAX_WINDOW = 5;
8198
+ var CONTEXT_ALL_MAX_MESSAGES = 5000;
8197
8199
  async function readFullTranscript(conversationId, opts = {}) {
8198
8200
  const start = opts.from ?? 0;
8199
8201
  const requested = opts.maxMessages ?? FULL_TRANSCRIPT_DEFAULT_MAX_MESSAGES;
@@ -8579,13 +8581,26 @@ async function runCommand(command, positional, flags, positionals = positional ?
8579
8581
  const query2 = new URLSearchParams;
8580
8582
  if (params.msg_index !== undefined)
8581
8583
  query2.set("msg_index", params.msg_index.toString());
8582
- if (params.window)
8583
- query2.set("window", params.window.toString());
8584
+ const window = params.window === undefined ? undefined : Math.min(params.window, CONTEXT_MAX_WINDOW);
8585
+ if (window !== undefined)
8586
+ query2.set("window", window.toString());
8584
8587
  const queryString = query2.toString() ? `?${query2.toString()}` : "";
8585
8588
  const data = await callDataAPI(`/conversations/${params.conversation_id}/context${queryString}`);
8586
8589
  const ctxRecord = asRecord(data) ?? {};
8587
8590
  const warnings = notFoundWarning("trace", params.conversation_id, asNumber(ctxRecord.total_messages) === 0);
8588
- context.output.writeData(data, warnings.length > 0 ? { warnings } : undefined);
8591
+ const nextCommands = [];
8592
+ if (params.window !== undefined && params.window > CONTEXT_MAX_WINDOW) {
8593
+ const center = asNumber(asRecord(ctxRecord.context)?.center_index) ?? params.msg_index ?? 0;
8594
+ const span = Math.min(params.window * 2 + 1, CONTEXT_ALL_MAX_MESSAGES);
8595
+ const from = Math.max(0, center - Math.floor((span - 1) / 2));
8596
+ const wide = `moda context ${shellArg(params.conversation_id)} --all --from=${from} --max-messages=${span}`;
8597
+ warnings.push(`--window=${params.window} is above the maximum of ${CONTEXT_MAX_WINDOW}, so this shows ±${CONTEXT_MAX_WINDOW} messages. For a wider slice run: ${wide}`);
8598
+ nextCommands.push({ command: wide, purpose: `Read ±${params.window} messages around the anchor.`, mutability: "read", requires_approval: false });
8599
+ }
8600
+ context.output.writeData(data, {
8601
+ ...warnings.length > 0 ? { warnings } : {},
8602
+ ...nextCommands.length > 0 ? { nextCommands } : {}
8603
+ });
8589
8604
  break;
8590
8605
  }
8591
8606
  case "audit":
@@ -9360,7 +9375,7 @@ var commandRegistry = createCommandRegistry([
9360
9375
  resetApiRequestCountBeforeRun: true,
9361
9376
  telemetry: "result",
9362
9377
  handler: async (context) => {
9363
- const { runInit } = await import("./index-ah6yf5x2.js");
9378
+ const { runInit } = await import("./index-r1egcg99.js");
9364
9379
  if (context.outputMode === "agent-stream") {
9365
9380
  context.output.writeEvent({
9366
9381
  event: "started",
@@ -9828,6 +9843,7 @@ var commandRegistry = createCommandRegistry([
9828
9843
  description: "Get windowed trace context (messages around one turn), or the whole trace with --all",
9829
9844
  examples: [
9830
9845
  "moda context <conversation_id> --window=3",
9846
+ "moda context <conversation_id> --msg-index=40 --window=0",
9831
9847
  "moda context <conversation_id> --all",
9832
9848
  "moda context <conversation_id> --all --from=500 --max-messages=500"
9833
9849
  ],
@@ -10125,11 +10141,12 @@ var commandRegistry = createCommandRegistry([
10125
10141
  flags: [
10126
10142
  { name: "--baseline=<path>", description: "Prompt file to treat as the control." },
10127
10143
  { name: "--candidate=<path>", description: "Prompt file to treat as the variant." },
10128
- { name: "--traces=<ids>", description: "Comma-separated trace ids (conversation_id values) to replay, at most 100. Legacy alias: --conversations=<ids>." },
10144
+ { name: "--traces=<ids>", description: "Comma-separated trace ids (conversation_id values) to replay, at most 100. Each case uses Moda's distilled scenario and success criteria for the trace; a trace that cannot be distilled falls back to its opening user message with generic criteria, and the run warns. Legacy alias: --conversations=<ids>." },
10129
10145
  { name: "--cases=<n>", description: "Cases to auto-generate (default 5, max 100)." },
10130
10146
  { name: "--seeds=<n>", description: "Repeats per case per arm (default 3, max 10)." },
10131
10147
  { name: "--model=<slug>", description: "Assistant model for both arms; must be in the replay allowlist unless --allow-any-model." },
10132
- { name: "--yes", description: "Confirm a run above 200 planned playouts (cases x seeds x 2 arms), or one whose case count is unknown." }
10148
+ { name: "--yes", description: "Confirm a run above 200 planned playouts (cases x seeds x 2 arms), or one whose case count is unknown." },
10149
+ { name: "--promote", description: "Make this run's replay set the tenant's primary replay set. Off by default, so an ad-hoc A/B never replaces it." }
10133
10150
  ],
10134
10151
  examples: [
10135
10152
  "moda prompts ab --baseline=prompts/agent.prompt.md --candidate=prompts/agent-v2.prompt.md --traces=conv_1,conv_2"
@@ -3,7 +3,7 @@ import {
3
3
  initPrompts,
4
4
  runPromptSync,
5
5
  runSkillSync
6
- } from "./cli-pr6s56fc.js";
6
+ } from "./cli-ecrx32mm.js";
7
7
  import {
8
8
  codingAgentDisplayName,
9
9
  describeCodingAgentEvent,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@moda-ai/cli",
3
- "version": "1.42.0",
3
+ "version": "1.43.0",
4
4
  "description": "CLI for Moda - AI agent analytics and observability",
5
5
  "type": "module",
6
6
  "bin": {
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "schema_version": "moda.skill_index.v1",
3
- "bundled_at": "2026-10-09T02:07:16.635Z",
4
- "cli_version": "1.42.0",
3
+ "bundled_at": "2026-10-09T02:13:14.110Z",
4
+ "cli_version": "1.43.0",
5
5
  "skills": [
6
6
  {
7
7
  "id": "integration-cloudflare-think",
@@ -442,7 +442,9 @@ fail instead of being clamped.
442
442
 
443
443
  `--window=N` on `context`, `frustrations --include-window` and
444
444
  `tool-failure-detail --include-window` is a message half-width (1–5), not a
445
- time window.
445
+ time window. On `context`, `--window=0` returns only the anchor message (plus
446
+ `total_messages`), and a value above 5 is clamped to 5 with a warning that
447
+ prints the `--all --from --max-messages` command for a wider slice.
446
448
 
447
449
  ### Schema introspection
448
450
 
@@ -631,6 +633,7 @@ Filters: `--search`, `--cluster-id`, `--user-id`, `--time-range`
631
633
  moda context <conversation_id> # default window around middle
632
634
  moda context <conversation_id> --msg-index=5 # center on message 5
633
635
  moda context <conversation_id> --window=3 # 3 messages each side (max 5)
636
+ moda context <conversation_id> --msg-index=40 --window=0 # just message 40 + total_messages
634
637
  moda context <conversation_id> --all # whole trace in order, first 500 messages
635
638
  moda context <conversation_id> --all --from=500 --max-messages=500 # continue (max 5000)
636
639
  ```
@@ -767,6 +770,13 @@ earlier turns as history);
767
770
  `anthropic/claude-haiku-4-5`, `google/gemini-2.5-flash`,
768
771
  `google/gemini-2.5-pro` unless you pass `--allow-any-model`.
769
772
 
773
+ `prompts ab --traces=<ids>` builds one case per trace from Moda's distilled
774
+ scenario and success criteria for that trace. A trace Moda cannot distill
775
+ falls back to its opening user message with three generic criteria (which long
776
+ traces pass easily); the run warns and lists it under `caseSources.fallback`.
777
+ `prompts ab` never makes its replay set the tenant's primary set unless you
778
+ pass `--promote`.
779
+
770
780
  Runtime code should render synced prompts through the SDK:
771
781
 
772
782
  ```typescript