@moda-ai/cli 1.39.0 → 1.41.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -5595,7 +5595,7 @@ async function runPromptAb(flags, profileOptions, context) {
|
|
|
5595
5595
|
const baselineSource = flags.baseline || flags["baseline-file"] || flags["baseline-key"];
|
|
5596
5596
|
const candidateSource = flags.candidate || flags["candidate-file"] || flags["candidate-key"];
|
|
5597
5597
|
if (!baselineSource || !candidateSource) {
|
|
5598
|
-
throw new Error("Usage: moda prompts ab --baseline=<file|key> --candidate=<file|key> [--auto-generate [--cases=1-100]|--set-id=ID|--traces=id1,id2 (max 100)] [--seeds=1-10] [--model=<slug>] [--yes] (legacy alias: --conversations=)");
|
|
5598
|
+
throw new Error("Usage: moda prompts ab --baseline=<file|key> --candidate=<file|key> [--auto-generate [--cases=1-100]|--set-id=ID|--traces=id1,id2[@msg_index] (max 100)] [--seeds=1-10] [--model=<slug>] [--yes] (legacy alias: --conversations=)");
|
|
5599
5599
|
}
|
|
5600
5600
|
assertArmSourceShape(baselineSource, "baseline");
|
|
5601
5601
|
assertArmSourceShape(candidateSource, "candidate");
|
|
@@ -5637,6 +5637,8 @@ async function runPromptAb(flags, profileOptions, context) {
|
|
|
5637
5637
|
...plan.kind === "existing" ? replayCasePin(plan.caseIds) : {}
|
|
5638
5638
|
})
|
|
5639
5639
|
}, profileOptions, { timeoutMs: 120000, retries: 0 });
|
|
5640
|
+
if (context.outputMode === "human")
|
|
5641
|
+
warnIfQueued(enqueue);
|
|
5640
5642
|
const noWait = flags.wait === "false" || flags["no-wait"] === "true";
|
|
5641
5643
|
if (noWait) {
|
|
5642
5644
|
context.output.writeData({
|
|
@@ -5658,7 +5660,7 @@ async function runPromptAb(flags, profileOptions, context) {
|
|
|
5658
5660
|
break;
|
|
5659
5661
|
}
|
|
5660
5662
|
if (context.outputMode === "human") {
|
|
5661
|
-
process.stderr.write(`Waiting for replay run (${
|
|
5663
|
+
process.stderr.write(`Waiting for replay run (${waitLabel(latest.run)})...
|
|
5662
5664
|
`);
|
|
5663
5665
|
}
|
|
5664
5666
|
await sleep(pollIntervalMs);
|
|
@@ -5680,6 +5682,23 @@ async function runPromptAb(flags, profileOptions, context) {
|
|
|
5680
5682
|
}
|
|
5681
5683
|
context.output.writeData(result);
|
|
5682
5684
|
}
|
|
5685
|
+
function warnIfQueued(enqueue) {
|
|
5686
|
+
const { runsAhead, maxActiveRuns } = enqueue;
|
|
5687
|
+
if (typeof runsAhead !== "number" || typeof maxActiveRuns !== "number")
|
|
5688
|
+
return;
|
|
5689
|
+
if (runsAhead < maxActiveRuns)
|
|
5690
|
+
return;
|
|
5691
|
+
process.stderr.write(`Note: ${runsAhead} replay runs for this tenant are already running or queued ` + `(${maxActiveRuns} at a time); run ${enqueue.runId} waits in the queue until a slot frees.
|
|
5692
|
+
`);
|
|
5693
|
+
}
|
|
5694
|
+
function waitLabel(run) {
|
|
5695
|
+
if (!run?.status)
|
|
5696
|
+
return "pending";
|
|
5697
|
+
if (run.status === "accepted") {
|
|
5698
|
+
return typeof run.runsAhead === "number" ? `queued, ${run.runsAhead} run${run.runsAhead === 1 ? "" : "s"} ahead` : "queued";
|
|
5699
|
+
}
|
|
5700
|
+
return run.status;
|
|
5701
|
+
}
|
|
5683
5702
|
var FILE_ARM_RE = /[\\/]|\.(?:md|txt|json|ya?ml|jinja2?|j2|hbs|tmpl|prompt|[cm]?[jt]sx?|py)$/i;
|
|
5684
5703
|
var PROMPT_KEY_RE = /^[^\s"'`\x00-\x1f\x7f]+$/;
|
|
5685
5704
|
function assertArmSourceShape(source, label) {
|
|
@@ -5858,6 +5877,12 @@ async function ensureReplaySet(plan, flags, tenantId, profileOptions) {
|
|
|
5858
5877
|
}
|
|
5859
5878
|
return generated.id;
|
|
5860
5879
|
}
|
|
5880
|
+
function parseTraceRef(ref) {
|
|
5881
|
+
const match = /^(.+)@(\d+)$/.exec(ref.trim());
|
|
5882
|
+
if (!match)
|
|
5883
|
+
return { conversationId: ref.trim() };
|
|
5884
|
+
return { conversationId: match[1], startMsgIndex: Number(match[2]) };
|
|
5885
|
+
}
|
|
5861
5886
|
async function createSetFromConversations(conversationIds, flags, tenantId, profileOptions) {
|
|
5862
5887
|
const name = (flags.name ?? flags["set-name"] ?? `Prompt A/B ${conversationIds.length} trace(s)`).trim();
|
|
5863
5888
|
const created = await callControlAPI(`/tenants/${encodeURIComponent(tenantId)}/replay-sets`, {
|
|
@@ -5865,7 +5890,8 @@ async function createSetFromConversations(conversationIds, flags, tenantId, prof
|
|
|
5865
5890
|
body: JSON.stringify({ name, description: "Created by moda prompts ab" })
|
|
5866
5891
|
}, profileOptions);
|
|
5867
5892
|
let position = 0;
|
|
5868
|
-
for (const
|
|
5893
|
+
for (const ref of conversationIds) {
|
|
5894
|
+
const { conversationId, startMsgIndex } = parseTraceRef(ref);
|
|
5869
5895
|
const scenario = await loadScenarioFromConversation(conversationId);
|
|
5870
5896
|
await callControlAPI(`/tenants/${encodeURIComponent(tenantId)}/replay-sets/${encodeURIComponent(created.id)}/cases`, {
|
|
5871
5897
|
method: "POST",
|
|
@@ -5873,6 +5899,7 @@ async function createSetFromConversations(conversationIds, flags, tenantId, prof
|
|
|
5873
5899
|
title: conversationId,
|
|
5874
5900
|
scenario,
|
|
5875
5901
|
sourceConversationId: conversationId,
|
|
5902
|
+
...startMsgIndex !== undefined ? { sourceStartMsgIndex: startMsgIndex } : {},
|
|
5876
5903
|
successCriteria: [
|
|
5877
5904
|
"The agent understands the user request",
|
|
5878
5905
|
"The agent uses tools appropriately when needed",
|
|
@@ -5975,6 +6002,8 @@ async function enqueueAndPollComparison(args, profileOptions) {
|
|
|
5975
6002
|
...replayCasePin(args.caseIds)
|
|
5976
6003
|
})
|
|
5977
6004
|
}, profileOptions, { timeoutMs: 120000, retries: 0 });
|
|
6005
|
+
if (args.onWaiting)
|
|
6006
|
+
warnIfQueued(enqueue);
|
|
5978
6007
|
const timeoutMs = args.timeoutMs ?? DEFAULT_WAIT_TIMEOUT_MS;
|
|
5979
6008
|
const pollIntervalMs = args.pollIntervalMs ?? DEFAULT_POLL_INTERVAL_MS;
|
|
5980
6009
|
const deadline = Date.now() + timeoutMs;
|
|
@@ -5986,7 +6015,7 @@ async function enqueueAndPollComparison(args, profileOptions) {
|
|
|
5986
6015
|
if (TERMINAL_RUN_STATUSES.has(status) && runId === enqueue.runId) {
|
|
5987
6016
|
return latest;
|
|
5988
6017
|
}
|
|
5989
|
-
args.onWaiting?.(
|
|
6018
|
+
args.onWaiting?.(waitLabel(latest.run));
|
|
5990
6019
|
await sleep(pollIntervalMs);
|
|
5991
6020
|
}
|
|
5992
6021
|
if (!latest?.run || latest.run.runId !== enqueue.runId || !TERMINAL_RUN_STATUSES.has(latest.run.status)) {
|
package/dist/cli.js
CHANGED
|
@@ -8,7 +8,7 @@ import {
|
|
|
8
8
|
runPromptsCommand,
|
|
9
9
|
runSkillsCommand,
|
|
10
10
|
runStatusCommand
|
|
11
|
-
} from "./cli-
|
|
11
|
+
} from "./cli-pr6s56fc.js";
|
|
12
12
|
import {
|
|
13
13
|
ApiError,
|
|
14
14
|
HARNESS_REPORT_APPROVAL_PATH,
|
|
@@ -7296,6 +7296,7 @@ Examples:
|
|
|
7296
7296
|
moda prompts sync
|
|
7297
7297
|
moda prompts promote support.triage --label=prod --version=pver_abc123
|
|
7298
7298
|
moda prompts ab --baseline=prompts/agent.prompt.md --candidate=prompts/agent-v2.prompt.md --traces=conv_1,conv_2
|
|
7299
|
+
moda prompts ab --baseline=prompts/agent.prompt.md --candidate=prompts/agent-v2.prompt.md --traces=conv_1@412 # replay from msg 412
|
|
7299
7300
|
moda skills pull
|
|
7300
7301
|
moda fixes
|
|
7301
7302
|
moda fix start <problem_id> --wait
|
|
@@ -9168,7 +9169,7 @@ var commandRegistry = createCommandRegistry([
|
|
|
9168
9169
|
resetApiRequestCountBeforeRun: true,
|
|
9169
9170
|
telemetry: "result",
|
|
9170
9171
|
handler: async (context) => {
|
|
9171
|
-
const { runInit } = await import("./index-
|
|
9172
|
+
const { runInit } = await import("./index-ah6yf5x2.js");
|
|
9172
9173
|
if (context.outputMode === "agent-stream") {
|
|
9173
9174
|
context.output.writeEvent({
|
|
9174
9175
|
event: "started",
|
package/package.json
CHANGED
package/skills/moda-cli/SKILL.md
CHANGED
|
@@ -759,7 +759,9 @@ anything they compute planned playouts (cases × seeds × 2 arms) and print it
|
|
|
759
759
|
(`plannedPlayouts`). Above 200, or when an existing set's case count can't be
|
|
760
760
|
read, they fail (exit 1) with the number and the exact command to re-run with `--yes`.
|
|
761
761
|
Only add `--yes` when the user has agreed to that spend. Limits: `--cases`
|
|
762
|
-
default 5, max 100; `--seeds` default 3, max 10; `--traces` at most 100
|
|
762
|
+
default 5, max 100; `--seeds` default 3, max 10; `--traces` at most 100
|
|
763
|
+
(`--traces=conv_id@412` replays that trace from msg_index 412, with the
|
|
764
|
+
earlier turns as history);
|
|
763
765
|
`--model` / `--assistant-model` must be one of `openai/gpt-5.6-luna`,
|
|
764
766
|
`openai/gpt-4o-mini`, `openai/gpt-4o`, `anthropic/claude-sonnet-4-5`,
|
|
765
767
|
`anthropic/claude-haiku-4-5`, `google/gemini-2.5-flash`,
|