@moda-ai/cli 1.40.0 → 1.41.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -5595,7 +5595,7 @@ async function runPromptAb(flags, profileOptions, context) {
|
|
|
5595
5595
|
const baselineSource = flags.baseline || flags["baseline-file"] || flags["baseline-key"];
|
|
5596
5596
|
const candidateSource = flags.candidate || flags["candidate-file"] || flags["candidate-key"];
|
|
5597
5597
|
if (!baselineSource || !candidateSource) {
|
|
5598
|
-
throw new Error("Usage: moda prompts ab --baseline=<file|key> --candidate=<file|key> [--auto-generate [--cases=1-100]|--set-id=ID|--traces=id1,id2 (max 100)] [--seeds=1-10] [--model=<slug>] [--yes] (legacy alias: --conversations=)");
|
|
5598
|
+
throw new Error("Usage: moda prompts ab --baseline=<file|key> --candidate=<file|key> [--auto-generate [--cases=1-100]|--set-id=ID|--traces=id1,id2[@msg_index] (max 100)] [--seeds=1-10] [--model=<slug>] [--yes] (legacy alias: --conversations=)");
|
|
5599
5599
|
}
|
|
5600
5600
|
assertArmSourceShape(baselineSource, "baseline");
|
|
5601
5601
|
assertArmSourceShape(candidateSource, "candidate");
|
|
@@ -5877,6 +5877,12 @@ async function ensureReplaySet(plan, flags, tenantId, profileOptions) {
|
|
|
5877
5877
|
}
|
|
5878
5878
|
return generated.id;
|
|
5879
5879
|
}
|
|
5880
|
+
function parseTraceRef(ref) {
|
|
5881
|
+
const match = /^(.+)@(\d+)$/.exec(ref.trim());
|
|
5882
|
+
if (!match)
|
|
5883
|
+
return { conversationId: ref.trim() };
|
|
5884
|
+
return { conversationId: match[1], startMsgIndex: Number(match[2]) };
|
|
5885
|
+
}
|
|
5880
5886
|
async function createSetFromConversations(conversationIds, flags, tenantId, profileOptions) {
|
|
5881
5887
|
const name = (flags.name ?? flags["set-name"] ?? `Prompt A/B ${conversationIds.length} trace(s)`).trim();
|
|
5882
5888
|
const created = await callControlAPI(`/tenants/${encodeURIComponent(tenantId)}/replay-sets`, {
|
|
@@ -5884,7 +5890,8 @@ async function createSetFromConversations(conversationIds, flags, tenantId, prof
|
|
|
5884
5890
|
body: JSON.stringify({ name, description: "Created by moda prompts ab" })
|
|
5885
5891
|
}, profileOptions);
|
|
5886
5892
|
let position = 0;
|
|
5887
|
-
for (const
|
|
5893
|
+
for (const ref of conversationIds) {
|
|
5894
|
+
const { conversationId, startMsgIndex } = parseTraceRef(ref);
|
|
5888
5895
|
const scenario = await loadScenarioFromConversation(conversationId);
|
|
5889
5896
|
await callControlAPI(`/tenants/${encodeURIComponent(tenantId)}/replay-sets/${encodeURIComponent(created.id)}/cases`, {
|
|
5890
5897
|
method: "POST",
|
|
@@ -5892,6 +5899,7 @@ async function createSetFromConversations(conversationIds, flags, tenantId, prof
|
|
|
5892
5899
|
title: conversationId,
|
|
5893
5900
|
scenario,
|
|
5894
5901
|
sourceConversationId: conversationId,
|
|
5902
|
+
...startMsgIndex !== undefined ? { sourceStartMsgIndex: startMsgIndex } : {},
|
|
5895
5903
|
successCriteria: [
|
|
5896
5904
|
"The agent understands the user request",
|
|
5897
5905
|
"The agent uses tools appropriately when needed",
|
package/dist/cli.js
CHANGED
|
@@ -8,7 +8,7 @@ import {
|
|
|
8
8
|
runPromptsCommand,
|
|
9
9
|
runSkillsCommand,
|
|
10
10
|
runStatusCommand
|
|
11
|
-
} from "./cli-
|
|
11
|
+
} from "./cli-pr6s56fc.js";
|
|
12
12
|
import {
|
|
13
13
|
ApiError,
|
|
14
14
|
HARNESS_REPORT_APPROVAL_PATH,
|
|
@@ -1799,7 +1799,8 @@ function gateFindings(fix) {
|
|
|
1799
1799
|
function formatGateScoreboard(gate) {
|
|
1800
1800
|
const holdout = gate.holdout ?? {};
|
|
1801
1801
|
const repair = gate.repair ?? {};
|
|
1802
|
-
|
|
1802
|
+
const guard = gate.byScope?.regression_guard;
|
|
1803
|
+
return `${gate.byScope ? "in-problem holdout" : "holdout"}: candidate ${holdout.candidate ?? 0}/${holdout.n ?? 0} vs baseline ${holdout.baseline ?? 0}/${holdout.n ?? 0} ` + `(wins ${holdout.wins ?? 0}, losses ${holdout.losses ?? 0}); ` + (guard ? `regression guards: candidate ${guard.candidate ?? 0}/${guard.n ?? 0} vs baseline ${guard.baseline ?? 0}/${guard.n ?? 0} ` + `(losses ${guard.losses ?? 0}, noise floor ${guard.noiseFloor ?? 0}); ` : "") + `training cases (repair): candidate ${repair.candidate ?? 0}/${repair.n ?? 0} vs baseline ${repair.baseline ?? 0}/${repair.n ?? 0}; ` + `noise floor ${gate.noiseFloor ?? 0}; abstained ${gate.abstained ?? 0}`;
|
|
1803
1804
|
}
|
|
1804
1805
|
function printHumanGateVerdict(fix, gateRunId, gateResult, cases) {
|
|
1805
1806
|
const w = (line) => process.stderr.write(`${line}
|
|
@@ -7296,6 +7297,7 @@ Examples:
|
|
|
7296
7297
|
moda prompts sync
|
|
7297
7298
|
moda prompts promote support.triage --label=prod --version=pver_abc123
|
|
7298
7299
|
moda prompts ab --baseline=prompts/agent.prompt.md --candidate=prompts/agent-v2.prompt.md --traces=conv_1,conv_2
|
|
7300
|
+
moda prompts ab --baseline=prompts/agent.prompt.md --candidate=prompts/agent-v2.prompt.md --traces=conv_1@412 # replay from msg 412
|
|
7299
7301
|
moda skills pull
|
|
7300
7302
|
moda fixes
|
|
7301
7303
|
moda fix start <problem_id> --wait
|
|
@@ -9168,7 +9170,7 @@ var commandRegistry = createCommandRegistry([
|
|
|
9168
9170
|
resetApiRequestCountBeforeRun: true,
|
|
9169
9171
|
telemetry: "result",
|
|
9170
9172
|
handler: async (context) => {
|
|
9171
|
-
const { runInit } = await import("./index-
|
|
9173
|
+
const { runInit } = await import("./index-ah6yf5x2.js");
|
|
9172
9174
|
if (context.outputMode === "agent-stream") {
|
|
9173
9175
|
context.output.writeEvent({
|
|
9174
9176
|
event: "started",
|
package/package.json
CHANGED
package/skills/moda-cli/SKILL.md
CHANGED
|
@@ -759,7 +759,9 @@ anything they compute planned playouts (cases × seeds × 2 arms) and print it
|
|
|
759
759
|
(`plannedPlayouts`). Above 200, or when an existing set's case count can't be
|
|
760
760
|
read, they fail (exit 1) with the number and the exact command to re-run with `--yes`.
|
|
761
761
|
Only add `--yes` when the user has agreed to that spend. Limits: `--cases`
|
|
762
|
-
default 5, max 100; `--seeds` default 3, max 10; `--traces` at most 100
|
|
762
|
+
default 5, max 100; `--seeds` default 3, max 10; `--traces` at most 100
|
|
763
|
+
(`--traces=conv_id@412` replays that trace from msg_index 412, with the
|
|
764
|
+
earlier turns as history);
|
|
763
765
|
`--model` / `--assistant-model` must be one of `openai/gpt-5.6-luna`,
|
|
764
766
|
`openai/gpt-4o-mini`, `openai/gpt-4o`, `anthropic/claude-sonnet-4-5`,
|
|
765
767
|
`anthropic/claude-haiku-4-5`, `google/gemini-2.5-flash`,
|