@moda-ai/cli 1.39.0 → 1.41.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5595,7 +5595,7 @@ async function runPromptAb(flags, profileOptions, context) {
5595
5595
  const baselineSource = flags.baseline || flags["baseline-file"] || flags["baseline-key"];
5596
5596
  const candidateSource = flags.candidate || flags["candidate-file"] || flags["candidate-key"];
5597
5597
  if (!baselineSource || !candidateSource) {
5598
- throw new Error("Usage: moda prompts ab --baseline=<file|key> --candidate=<file|key> [--auto-generate [--cases=1-100]|--set-id=ID|--traces=id1,id2 (max 100)] [--seeds=1-10] [--model=<slug>] [--yes] (legacy alias: --conversations=)");
5598
+ throw new Error("Usage: moda prompts ab --baseline=<file|key> --candidate=<file|key> [--auto-generate [--cases=1-100]|--set-id=ID|--traces=id1,id2[@msg_index] (max 100)] [--seeds=1-10] [--model=<slug>] [--yes] (legacy alias: --conversations=)");
5599
5599
  }
5600
5600
  assertArmSourceShape(baselineSource, "baseline");
5601
5601
  assertArmSourceShape(candidateSource, "candidate");
@@ -5637,6 +5637,8 @@ async function runPromptAb(flags, profileOptions, context) {
5637
5637
  ...plan.kind === "existing" ? replayCasePin(plan.caseIds) : {}
5638
5638
  })
5639
5639
  }, profileOptions, { timeoutMs: 120000, retries: 0 });
5640
+ if (context.outputMode === "human")
5641
+ warnIfQueued(enqueue);
5640
5642
  const noWait = flags.wait === "false" || flags["no-wait"] === "true";
5641
5643
  if (noWait) {
5642
5644
  context.output.writeData({
@@ -5658,7 +5660,7 @@ async function runPromptAb(flags, profileOptions, context) {
5658
5660
  break;
5659
5661
  }
5660
5662
  if (context.outputMode === "human") {
5661
- process.stderr.write(`Waiting for replay run (${status || "pending"})...
5663
+ process.stderr.write(`Waiting for replay run (${waitLabel(latest.run)})...
5662
5664
  `);
5663
5665
  }
5664
5666
  await sleep(pollIntervalMs);
@@ -5680,6 +5682,23 @@ async function runPromptAb(flags, profileOptions, context) {
5680
5682
  }
5681
5683
  context.output.writeData(result);
5682
5684
  }
5685
+ function warnIfQueued(enqueue) {
5686
+ const { runsAhead, maxActiveRuns } = enqueue;
5687
+ if (typeof runsAhead !== "number" || typeof maxActiveRuns !== "number")
5688
+ return;
5689
+ if (runsAhead < maxActiveRuns)
5690
+ return;
5691
+ process.stderr.write(`Note: ${runsAhead} replay runs for this tenant are already running or queued ` + `(${maxActiveRuns} at a time); run ${enqueue.runId} waits in the queue until a slot frees.
5692
+ `);
5693
+ }
5694
+ function waitLabel(run) {
5695
+ if (!run?.status)
5696
+ return "pending";
5697
+ if (run.status === "accepted") {
5698
+ return typeof run.runsAhead === "number" ? `queued, ${run.runsAhead} run${run.runsAhead === 1 ? "" : "s"} ahead` : "queued";
5699
+ }
5700
+ return run.status;
5701
+ }
5683
5702
  var FILE_ARM_RE = /[\\/]|\.(?:md|txt|json|ya?ml|jinja2?|j2|hbs|tmpl|prompt|[cm]?[jt]sx?|py)$/i;
5684
5703
  var PROMPT_KEY_RE = /^[^\s"'`\x00-\x1f\x7f]+$/;
5685
5704
  function assertArmSourceShape(source, label) {
@@ -5858,6 +5877,12 @@ async function ensureReplaySet(plan, flags, tenantId, profileOptions) {
5858
5877
  }
5859
5878
  return generated.id;
5860
5879
  }
5880
+ function parseTraceRef(ref) {
5881
+ const match = /^(.+)@(\d+)$/.exec(ref.trim());
5882
+ if (!match)
5883
+ return { conversationId: ref.trim() };
5884
+ return { conversationId: match[1], startMsgIndex: Number(match[2]) };
5885
+ }
5861
5886
  async function createSetFromConversations(conversationIds, flags, tenantId, profileOptions) {
5862
5887
  const name = (flags.name ?? flags["set-name"] ?? `Prompt A/B ${conversationIds.length} trace(s)`).trim();
5863
5888
  const created = await callControlAPI(`/tenants/${encodeURIComponent(tenantId)}/replay-sets`, {
@@ -5865,7 +5890,8 @@ async function createSetFromConversations(conversationIds, flags, tenantId, prof
5865
5890
  body: JSON.stringify({ name, description: "Created by moda prompts ab" })
5866
5891
  }, profileOptions);
5867
5892
  let position = 0;
5868
- for (const conversationId of conversationIds) {
5893
+ for (const ref of conversationIds) {
5894
+ const { conversationId, startMsgIndex } = parseTraceRef(ref);
5869
5895
  const scenario = await loadScenarioFromConversation(conversationId);
5870
5896
  await callControlAPI(`/tenants/${encodeURIComponent(tenantId)}/replay-sets/${encodeURIComponent(created.id)}/cases`, {
5871
5897
  method: "POST",
@@ -5873,6 +5899,7 @@ async function createSetFromConversations(conversationIds, flags, tenantId, prof
5873
5899
  title: conversationId,
5874
5900
  scenario,
5875
5901
  sourceConversationId: conversationId,
5902
+ ...startMsgIndex !== undefined ? { sourceStartMsgIndex: startMsgIndex } : {},
5876
5903
  successCriteria: [
5877
5904
  "The agent understands the user request",
5878
5905
  "The agent uses tools appropriately when needed",
@@ -5975,6 +6002,8 @@ async function enqueueAndPollComparison(args, profileOptions) {
5975
6002
  ...replayCasePin(args.caseIds)
5976
6003
  })
5977
6004
  }, profileOptions, { timeoutMs: 120000, retries: 0 });
6005
+ if (args.onWaiting)
6006
+ warnIfQueued(enqueue);
5978
6007
  const timeoutMs = args.timeoutMs ?? DEFAULT_WAIT_TIMEOUT_MS;
5979
6008
  const pollIntervalMs = args.pollIntervalMs ?? DEFAULT_POLL_INTERVAL_MS;
5980
6009
  const deadline = Date.now() + timeoutMs;
@@ -5986,7 +6015,7 @@ async function enqueueAndPollComparison(args, profileOptions) {
5986
6015
  if (TERMINAL_RUN_STATUSES.has(status) && runId === enqueue.runId) {
5987
6016
  return latest;
5988
6017
  }
5989
- args.onWaiting?.(status || "pending");
6018
+ args.onWaiting?.(waitLabel(latest.run));
5990
6019
  await sleep(pollIntervalMs);
5991
6020
  }
5992
6021
  if (!latest?.run || latest.run.runId !== enqueue.runId || !TERMINAL_RUN_STATUSES.has(latest.run.status)) {
package/dist/cli.js CHANGED
@@ -8,7 +8,7 @@ import {
8
8
  runPromptsCommand,
9
9
  runSkillsCommand,
10
10
  runStatusCommand
11
- } from "./cli-a7ypptsf.js";
11
+ } from "./cli-pr6s56fc.js";
12
12
  import {
13
13
  ApiError,
14
14
  HARNESS_REPORT_APPROVAL_PATH,
@@ -7296,6 +7296,7 @@ Examples:
7296
7296
  moda prompts sync
7297
7297
  moda prompts promote support.triage --label=prod --version=pver_abc123
7298
7298
  moda prompts ab --baseline=prompts/agent.prompt.md --candidate=prompts/agent-v2.prompt.md --traces=conv_1,conv_2
7299
+ moda prompts ab --baseline=prompts/agent.prompt.md --candidate=prompts/agent-v2.prompt.md --traces=conv_1@412 # replay from msg 412
7299
7300
  moda skills pull
7300
7301
  moda fixes
7301
7302
  moda fix start <problem_id> --wait
@@ -9168,7 +9169,7 @@ var commandRegistry = createCommandRegistry([
9168
9169
  resetApiRequestCountBeforeRun: true,
9169
9170
  telemetry: "result",
9170
9171
  handler: async (context) => {
9171
- const { runInit } = await import("./index-qgc37efj.js");
9172
+ const { runInit } = await import("./index-ah6yf5x2.js");
9172
9173
  if (context.outputMode === "agent-stream") {
9173
9174
  context.output.writeEvent({
9174
9175
  event: "started",
@@ -3,7 +3,7 @@ import {
3
3
  initPrompts,
4
4
  runPromptSync,
5
5
  runSkillSync
6
- } from "./cli-a7ypptsf.js";
6
+ } from "./cli-pr6s56fc.js";
7
7
  import {
8
8
  codingAgentDisplayName,
9
9
  describeCodingAgentEvent,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@moda-ai/cli",
3
- "version": "1.39.0",
3
+ "version": "1.41.0",
4
4
  "description": "CLI for Moda - AI agent analytics and observability",
5
5
  "type": "module",
6
6
  "bin": {
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "schema_version": "moda.skill_index.v1",
3
- "bundled_at": "2026-10-05T22:35:48.948Z",
4
- "cli_version": "1.39.0",
3
+ "bundled_at": "2026-10-09T01:07:28.008Z",
4
+ "cli_version": "1.41.0",
5
5
  "skills": [
6
6
  {
7
7
  "id": "integration-cloudflare-think",
@@ -759,7 +759,9 @@ anything they compute planned playouts (cases × seeds × 2 arms) and print it
759
759
  (`plannedPlayouts`). Above 200, or when an existing set's case count can't be
760
760
  read, they fail (exit 1) with the number and the exact command to re-run with `--yes`.
761
761
  Only add `--yes` when the user has agreed to that spend. Limits: `--cases`
762
- default 5, max 100; `--seeds` default 3, max 10; `--traces` at most 100;
762
+ default 5, max 100; `--seeds` default 3, max 10; `--traces` at most 100
763
+ (`--traces=conv_id@412` replays that trace from msg_index 412, with the
764
+ earlier turns as history);
763
765
  `--model` / `--assistant-model` must be one of `openai/gpt-5.6-luna`,
764
766
  `openai/gpt-4o-mini`, `openai/gpt-4o`, `anthropic/claude-sonnet-4-5`,
765
767
  `anthropic/claude-haiku-4-5`, `google/gemini-2.5-flash`,