@moda-ai/cli 1.42.0 → 1.43.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -5586,6 +5586,11 @@ var REPLAY_MODEL_ALLOWLIST = [
|
|
|
5586
5586
|
var DEFAULT_POLL_INTERVAL_MS = 15000;
|
|
5587
5587
|
var DEFAULT_WAIT_TIMEOUT_MS = 2 * 60 * 60 * 1000;
|
|
5588
5588
|
var TERMINAL_RUN_STATUSES = new Set(["completed", "skipped", "error"]);
|
|
5589
|
+
var FALLBACK_SUCCESS_CRITERIA = [
|
|
5590
|
+
"The agent understands the user request",
|
|
5591
|
+
"The agent uses tools appropriately when needed",
|
|
5592
|
+
"The agent provides a helpful final answer"
|
|
5593
|
+
];
|
|
5589
5594
|
async function runPromptAb(flags, profileOptions, context) {
|
|
5590
5595
|
validateConfig();
|
|
5591
5596
|
const tenantId = flags["tenant-id"] || resolveApiTenantId(profileOptions);
|
|
@@ -5626,14 +5631,20 @@ async function runPromptAb(flags, profileOptions, context) {
|
|
|
5626
5631
|
prod: resolvePromptArmSpec(baselineSource, "baseline"),
|
|
5627
5632
|
proposed: resolvePromptArmSpec(candidateSource, "candidate")
|
|
5628
5633
|
};
|
|
5629
|
-
const replaySetId = await ensureReplaySet(plan, flags, tenantId, profileOptions);
|
|
5634
|
+
const { setId: replaySetId, caseSources, warnings } = await ensureReplaySet(plan, flags, tenantId, profileOptions);
|
|
5635
|
+
if (context.outputMode === "human") {
|
|
5636
|
+
for (const warning of warnings)
|
|
5637
|
+
process.stderr.write(`Warning: ${warning}
|
|
5638
|
+
`);
|
|
5639
|
+
}
|
|
5640
|
+
const writeOptions = warnings.length ? { warnings } : undefined;
|
|
5630
5641
|
const enqueue = await callControlAPI(`/tenants/${encodeURIComponent(tenantId)}/replay-sets/${encodeURIComponent(replaySetId)}/run`, {
|
|
5631
5642
|
method: "POST",
|
|
5632
5643
|
body: JSON.stringify({
|
|
5633
5644
|
promptArms,
|
|
5634
5645
|
seedsPerCase,
|
|
5635
5646
|
assistantModel: assistantModel || undefined,
|
|
5636
|
-
promotePrimary: flags["no-promote"] !== "true",
|
|
5647
|
+
promotePrimary: flags.promote === "true" && flags["no-promote"] !== "true",
|
|
5637
5648
|
...plan.kind === "existing" ? replayCasePin(plan.caseIds) : {}
|
|
5638
5649
|
})
|
|
5639
5650
|
}, profileOptions, { timeoutMs: 120000, retries: 0 });
|
|
@@ -5643,11 +5654,12 @@ async function runPromptAb(flags, profileOptions, context) {
|
|
|
5643
5654
|
if (noWait) {
|
|
5644
5655
|
context.output.writeData({
|
|
5645
5656
|
replaySetId,
|
|
5657
|
+
...caseSources ? { caseSources } : {},
|
|
5646
5658
|
runId: enqueue.runId,
|
|
5647
5659
|
status: enqueue.status,
|
|
5648
5660
|
message: enqueue.message ?? "Replay comparison queued",
|
|
5649
5661
|
plannedPlayouts
|
|
5650
|
-
});
|
|
5662
|
+
}, writeOptions);
|
|
5651
5663
|
return;
|
|
5652
5664
|
}
|
|
5653
5665
|
const deadline = Date.now() + timeoutMs;
|
|
@@ -5670,6 +5682,7 @@ async function runPromptAb(flags, profileOptions, context) {
|
|
|
5670
5682
|
}
|
|
5671
5683
|
const result = {
|
|
5672
5684
|
replaySetId,
|
|
5685
|
+
...caseSources ? { caseSources } : {},
|
|
5673
5686
|
plannedPlayouts,
|
|
5674
5687
|
runId: latest.run.runId,
|
|
5675
5688
|
status: latest.run.status,
|
|
@@ -5680,7 +5693,7 @@ async function runPromptAb(flags, profileOptions, context) {
|
|
|
5680
5693
|
if (context.outputMode === "human") {
|
|
5681
5694
|
printHumanVerdict(result);
|
|
5682
5695
|
}
|
|
5683
|
-
context.output.writeData(result);
|
|
5696
|
+
context.output.writeData(result, writeOptions);
|
|
5684
5697
|
}
|
|
5685
5698
|
function warnIfQueued(enqueue) {
|
|
5686
5699
|
const { runsAhead, maxActiveRuns } = enqueue;
|
|
@@ -5793,7 +5806,7 @@ async function planReplaySet(flags, tenantId, profileOptions) {
|
|
|
5793
5806
|
const snapshot = await fetchReplaySetCases(tenantId, existingSetId, profileOptions);
|
|
5794
5807
|
return { kind: "existing", setId: existingSetId, cases: snapshot?.ids.length ?? null, caseIds: snapshot?.ids };
|
|
5795
5808
|
}
|
|
5796
|
-
const conversations = parseCsv(flags.traces ?? flags.conversations ?? flags["conversation-ids"]);
|
|
5809
|
+
const conversations = [...new Set(parseCsv(flags.traces ?? flags.conversations ?? flags["conversation-ids"]))];
|
|
5797
5810
|
if (conversations.length > MAX_REPLAY_TRACES) {
|
|
5798
5811
|
throw new CliInputError(`--traces has ${conversations.length} ids; the limit is ${MAX_REPLAY_TRACES} per replay set, so nothing was sent.`, "Split the traces across several runs, or use --auto-generate.");
|
|
5799
5812
|
}
|
|
@@ -5861,9 +5874,9 @@ function rerunWithYes(command, positionals, flags) {
|
|
|
5861
5874
|
}
|
|
5862
5875
|
async function ensureReplaySet(plan, flags, tenantId, profileOptions) {
|
|
5863
5876
|
if (plan.kind === "existing")
|
|
5864
|
-
return plan.setId;
|
|
5877
|
+
return { setId: plan.setId, warnings: [] };
|
|
5865
5878
|
if (plan.kind === "traces") {
|
|
5866
|
-
return
|
|
5879
|
+
return createSetFromTraces(plan.traces, flags, tenantId, profileOptions);
|
|
5867
5880
|
}
|
|
5868
5881
|
const caseCount = plan.cases;
|
|
5869
5882
|
const lookbackDays = parsePositiveInt(flags["lookback-days"], 30, 365, "--lookback-days");
|
|
@@ -5875,7 +5888,7 @@ async function ensureReplaySet(plan, flags, tenantId, profileOptions) {
|
|
|
5875
5888
|
if (!generated?.id) {
|
|
5876
5889
|
throw new Error("Auto-generate did not return a replay set id");
|
|
5877
5890
|
}
|
|
5878
|
-
return generated.id;
|
|
5891
|
+
return { setId: generated.id, warnings: [] };
|
|
5879
5892
|
}
|
|
5880
5893
|
function parseTraceRef(ref) {
|
|
5881
5894
|
const match = /^(.+)@(\d+)$/.exec(ref.trim());
|
|
@@ -5883,38 +5896,75 @@ function parseTraceRef(ref) {
|
|
|
5883
5896
|
return { conversationId: ref.trim() };
|
|
5884
5897
|
return { conversationId: match[1], startMsgIndex: Number(match[2]) };
|
|
5885
5898
|
}
|
|
5886
|
-
async function
|
|
5887
|
-
const
|
|
5888
|
-
const
|
|
5889
|
-
|
|
5890
|
-
|
|
5891
|
-
|
|
5892
|
-
|
|
5893
|
-
|
|
5894
|
-
|
|
5899
|
+
async function createSetFromTraces(conversationIds, flags, tenantId, profileOptions) {
|
|
5900
|
+
const refs = [...new Set(conversationIds)].map(parseTraceRef);
|
|
5901
|
+
const traces = [...new Set(refs.map((ref) => ref.conversationId))];
|
|
5902
|
+
const startByConversation = new Map;
|
|
5903
|
+
for (const ref of refs) {
|
|
5904
|
+
if (ref.startMsgIndex !== undefined)
|
|
5905
|
+
startByConversation.set(ref.conversationId, ref.startMsgIndex);
|
|
5906
|
+
}
|
|
5907
|
+
const name = (flags.name ?? flags["set-name"] ?? `Prompt A/B ${traces.length} trace(s)`).trim();
|
|
5908
|
+
const base = `/tenants/${encodeURIComponent(tenantId)}/replay-sets`;
|
|
5909
|
+
let set = null;
|
|
5910
|
+
let unavailable = null;
|
|
5911
|
+
try {
|
|
5912
|
+
set = await callControlAPI(`${base}/auto-generate`, { method: "POST", body: JSON.stringify({ name, conversationIds: traces }) }, profileOptions, { timeoutMs: 300000, retries: 0 });
|
|
5913
|
+
if (!set?.id)
|
|
5914
|
+
throw new Error("Auto-generate did not return a replay set id");
|
|
5915
|
+
} catch (error) {
|
|
5916
|
+
const status = error?.statusCode;
|
|
5917
|
+
if (status === 401 || status === 403)
|
|
5918
|
+
throw error;
|
|
5919
|
+
unavailable = error instanceof Error ? error.message : String(error);
|
|
5920
|
+
set = null;
|
|
5921
|
+
}
|
|
5922
|
+
const distilledIds = new Set((set?.cases ?? []).map((c) => String(c?.sourceConversationId ?? "")).filter(Boolean));
|
|
5923
|
+
const distilled = traces.filter((id) => distilledIds.has(id));
|
|
5924
|
+
const fallback = traces.filter((id) => !distilledIds.has(id));
|
|
5925
|
+
if (!set) {
|
|
5926
|
+
set = await callControlAPI(base, { method: "POST", body: JSON.stringify({ name, description: "Created by moda prompts ab" }) }, profileOptions);
|
|
5927
|
+
}
|
|
5928
|
+
const requestedIndex = new Map(traces.map((id, index) => [id, index]));
|
|
5929
|
+
for (const existing of set.cases ?? []) {
|
|
5930
|
+
const conversationId = String(existing?.sourceConversationId ?? "");
|
|
5931
|
+
const index = requestedIndex.get(conversationId);
|
|
5932
|
+
const startMsgIndex = startByConversation.get(conversationId);
|
|
5933
|
+
const update = {};
|
|
5934
|
+
if (fallback.length && index !== undefined && existing.position !== index)
|
|
5935
|
+
update.position = index;
|
|
5936
|
+
if (startMsgIndex !== undefined)
|
|
5937
|
+
update.sourceStartMsgIndex = startMsgIndex;
|
|
5938
|
+
if (!Object.keys(update).length)
|
|
5939
|
+
continue;
|
|
5940
|
+
await callControlAPI(`${base}/${encodeURIComponent(set.id)}/cases/${encodeURIComponent(existing.id)}`, { method: "PUT", body: JSON.stringify(update) }, profileOptions);
|
|
5941
|
+
}
|
|
5942
|
+
for (const conversationId of fallback) {
|
|
5895
5943
|
const scenario = await loadScenarioFromConversation(conversationId);
|
|
5896
|
-
await callControlAPI(
|
|
5944
|
+
await callControlAPI(`${base}/${encodeURIComponent(set.id)}/cases`, {
|
|
5897
5945
|
method: "POST",
|
|
5898
5946
|
body: JSON.stringify({
|
|
5899
5947
|
title: conversationId,
|
|
5900
5948
|
scenario,
|
|
5901
5949
|
sourceConversationId: conversationId,
|
|
5902
|
-
...
|
|
5903
|
-
successCriteria:
|
|
5904
|
-
|
|
5905
|
-
"The agent uses tools appropriately when needed",
|
|
5906
|
-
"The agent provides a helpful final answer"
|
|
5907
|
-
],
|
|
5908
|
-
position: position++
|
|
5950
|
+
...startByConversation.has(conversationId) ? { sourceStartMsgIndex: startByConversation.get(conversationId) } : {},
|
|
5951
|
+
successCriteria: FALLBACK_SUCCESS_CRITERIA,
|
|
5952
|
+
position: requestedIndex.get(conversationId)
|
|
5909
5953
|
})
|
|
5910
5954
|
}, profileOptions);
|
|
5911
5955
|
}
|
|
5912
|
-
|
|
5956
|
+
const warnings = [];
|
|
5957
|
+
if (fallback.length) {
|
|
5958
|
+
const why = unavailable ? `Moda could not distill these traces (${unavailable.slice(0, 200)})` : "Moda could not distill a scenario for these traces";
|
|
5959
|
+
warnings.push(`${why}, so ${fallback.length} of ${traces.length} case(s) use only the opening user message and three generic ` + `success criteria, which long traces pass easily: ${fallback.slice(0, 10).join(", ")}${fallback.length > 10 ? ", …" : ""}. ` + "Edit their scenario/criteria in the dashboard, or pass --set-id with a curated set.");
|
|
5960
|
+
}
|
|
5961
|
+
return { setId: set.id, caseSources: { distilled, fallback }, warnings };
|
|
5913
5962
|
}
|
|
5914
5963
|
async function loadScenarioFromConversation(conversationId) {
|
|
5915
5964
|
try {
|
|
5916
|
-
const data = await callDataAPI(`/conversations/${encodeURIComponent(conversationId)}/context?msg_index=0&window=
|
|
5917
|
-
const
|
|
5965
|
+
const data = await callDataAPI(`/conversations/${encodeURIComponent(conversationId)}/context?msg_index=0&window=5`);
|
|
5966
|
+
const messages = data.context?.messages ?? data.messages ?? [];
|
|
5967
|
+
const firstUser = messages.find((message) => message.role === "user" && String(message.content ?? "").trim());
|
|
5918
5968
|
const opening = String(firstUser?.content ?? "").trim();
|
|
5919
5969
|
if (opening) {
|
|
5920
5970
|
return `The user says: "${opening}"
|
|
@@ -6446,6 +6496,7 @@ var PROMPT_WRITE_SUBCOMMAND_FLAGS = {
|
|
|
6446
6496
|
"lookback-days",
|
|
6447
6497
|
"no-promote",
|
|
6448
6498
|
"no-wait",
|
|
6499
|
+
"promote",
|
|
6449
6500
|
"poll-interval",
|
|
6450
6501
|
"replay-set-id",
|
|
6451
6502
|
"seeds-per-case",
|
package/dist/cli.js
CHANGED
|
@@ -8,7 +8,7 @@ import {
|
|
|
8
8
|
runPromptsCommand,
|
|
9
9
|
runSkillsCommand,
|
|
10
10
|
runStatusCommand
|
|
11
|
-
} from "./cli-
|
|
11
|
+
} from "./cli-ecrx32mm.js";
|
|
12
12
|
import {
|
|
13
13
|
ApiError,
|
|
14
14
|
HARNESS_REPORT_APPROVAL_PATH,
|
|
@@ -151,7 +151,7 @@ var WorldStateSchema = z.object({
|
|
|
151
151
|
var ContextSchema = z.object({
|
|
152
152
|
conversation_id: z.string(),
|
|
153
153
|
msg_index: z.number().min(0).optional(),
|
|
154
|
-
window: z.number().
|
|
154
|
+
window: z.number().int().min(0).optional(),
|
|
155
155
|
all: boolFlag(),
|
|
156
156
|
from: z.number().int().min(0).optional(),
|
|
157
157
|
max_messages: z.number().int().min(1).max(5000).optional()
|
|
@@ -8194,6 +8194,8 @@ function pageMessages(page) {
|
|
|
8194
8194
|
const messages = asRecord(page.context)?.messages ?? page.messages;
|
|
8195
8195
|
return (Array.isArray(messages) ? messages : []).map((m) => asRecord(m) ?? {});
|
|
8196
8196
|
}
|
|
8197
|
+
var CONTEXT_MAX_WINDOW = 5;
|
|
8198
|
+
var CONTEXT_ALL_MAX_MESSAGES = 5000;
|
|
8197
8199
|
async function readFullTranscript(conversationId, opts = {}) {
|
|
8198
8200
|
const start = opts.from ?? 0;
|
|
8199
8201
|
const requested = opts.maxMessages ?? FULL_TRANSCRIPT_DEFAULT_MAX_MESSAGES;
|
|
@@ -8579,13 +8581,26 @@ async function runCommand(command, positional, flags, positionals = positional ?
|
|
|
8579
8581
|
const query2 = new URLSearchParams;
|
|
8580
8582
|
if (params.msg_index !== undefined)
|
|
8581
8583
|
query2.set("msg_index", params.msg_index.toString());
|
|
8582
|
-
|
|
8583
|
-
|
|
8584
|
+
const window = params.window === undefined ? undefined : Math.min(params.window, CONTEXT_MAX_WINDOW);
|
|
8585
|
+
if (window !== undefined)
|
|
8586
|
+
query2.set("window", window.toString());
|
|
8584
8587
|
const queryString = query2.toString() ? `?${query2.toString()}` : "";
|
|
8585
8588
|
const data = await callDataAPI(`/conversations/${params.conversation_id}/context${queryString}`);
|
|
8586
8589
|
const ctxRecord = asRecord(data) ?? {};
|
|
8587
8590
|
const warnings = notFoundWarning("trace", params.conversation_id, asNumber(ctxRecord.total_messages) === 0);
|
|
8588
|
-
|
|
8591
|
+
const nextCommands = [];
|
|
8592
|
+
if (params.window !== undefined && params.window > CONTEXT_MAX_WINDOW) {
|
|
8593
|
+
const center = asNumber(asRecord(ctxRecord.context)?.center_index) ?? params.msg_index ?? 0;
|
|
8594
|
+
const span = Math.min(params.window * 2 + 1, CONTEXT_ALL_MAX_MESSAGES);
|
|
8595
|
+
const from = Math.max(0, center - Math.floor((span - 1) / 2));
|
|
8596
|
+
const wide = `moda context ${shellArg(params.conversation_id)} --all --from=${from} --max-messages=${span}`;
|
|
8597
|
+
warnings.push(`--window=${params.window} is above the maximum of ${CONTEXT_MAX_WINDOW}, so this shows ±${CONTEXT_MAX_WINDOW} messages. For a wider slice run: ${wide}`);
|
|
8598
|
+
nextCommands.push({ command: wide, purpose: `Read ±${params.window} messages around the anchor.`, mutability: "read", requires_approval: false });
|
|
8599
|
+
}
|
|
8600
|
+
context.output.writeData(data, {
|
|
8601
|
+
...warnings.length > 0 ? { warnings } : {},
|
|
8602
|
+
...nextCommands.length > 0 ? { nextCommands } : {}
|
|
8603
|
+
});
|
|
8589
8604
|
break;
|
|
8590
8605
|
}
|
|
8591
8606
|
case "audit":
|
|
@@ -9360,7 +9375,7 @@ var commandRegistry = createCommandRegistry([
|
|
|
9360
9375
|
resetApiRequestCountBeforeRun: true,
|
|
9361
9376
|
telemetry: "result",
|
|
9362
9377
|
handler: async (context) => {
|
|
9363
|
-
const { runInit } = await import("./index-
|
|
9378
|
+
const { runInit } = await import("./index-r1egcg99.js");
|
|
9364
9379
|
if (context.outputMode === "agent-stream") {
|
|
9365
9380
|
context.output.writeEvent({
|
|
9366
9381
|
event: "started",
|
|
@@ -9828,6 +9843,7 @@ var commandRegistry = createCommandRegistry([
|
|
|
9828
9843
|
description: "Get windowed trace context (messages around one turn), or the whole trace with --all",
|
|
9829
9844
|
examples: [
|
|
9830
9845
|
"moda context <conversation_id> --window=3",
|
|
9846
|
+
"moda context <conversation_id> --msg-index=40 --window=0",
|
|
9831
9847
|
"moda context <conversation_id> --all",
|
|
9832
9848
|
"moda context <conversation_id> --all --from=500 --max-messages=500"
|
|
9833
9849
|
],
|
|
@@ -10125,11 +10141,12 @@ var commandRegistry = createCommandRegistry([
|
|
|
10125
10141
|
flags: [
|
|
10126
10142
|
{ name: "--baseline=<path>", description: "Prompt file to treat as the control." },
|
|
10127
10143
|
{ name: "--candidate=<path>", description: "Prompt file to treat as the variant." },
|
|
10128
|
-
{ name: "--traces=<ids>", description: "Comma-separated trace ids (conversation_id values) to replay, at most 100. Legacy alias: --conversations=<ids>." },
|
|
10144
|
+
{ name: "--traces=<ids>", description: "Comma-separated trace ids (conversation_id values) to replay, at most 100. Each case uses Moda's distilled scenario and success criteria for the trace; a trace that cannot be distilled falls back to its opening user message with generic criteria, and the run warns. Legacy alias: --conversations=<ids>." },
|
|
10129
10145
|
{ name: "--cases=<n>", description: "Cases to auto-generate (default 5, max 100)." },
|
|
10130
10146
|
{ name: "--seeds=<n>", description: "Repeats per case per arm (default 3, max 10)." },
|
|
10131
10147
|
{ name: "--model=<slug>", description: "Assistant model for both arms; must be in the replay allowlist unless --allow-any-model." },
|
|
10132
|
-
{ name: "--yes", description: "Confirm a run above 200 planned playouts (cases x seeds x 2 arms), or one whose case count is unknown." }
|
|
10148
|
+
{ name: "--yes", description: "Confirm a run above 200 planned playouts (cases x seeds x 2 arms), or one whose case count is unknown." },
|
|
10149
|
+
{ name: "--promote", description: "Make this run's replay set the tenant's primary replay set. Off by default, so an ad-hoc A/B never replaces it." }
|
|
10133
10150
|
],
|
|
10134
10151
|
examples: [
|
|
10135
10152
|
"moda prompts ab --baseline=prompts/agent.prompt.md --candidate=prompts/agent-v2.prompt.md --traces=conv_1,conv_2"
|
package/package.json
CHANGED
package/skills/moda-cli/SKILL.md
CHANGED
|
@@ -442,7 +442,9 @@ fail instead of being clamped.
|
|
|
442
442
|
|
|
443
443
|
`--window=N` on `context`, `frustrations --include-window` and
|
|
444
444
|
`tool-failure-detail --include-window` is a message half-width (1–5), not a
|
|
445
|
-
time window.
|
|
445
|
+
time window. On `context`, `--window=0` returns only the anchor message (plus
|
|
446
|
+
`total_messages`), and a value above 5 is clamped to 5 with a warning that
|
|
447
|
+
prints the `--all --from --max-messages` command for a wider slice.
|
|
446
448
|
|
|
447
449
|
### Schema introspection
|
|
448
450
|
|
|
@@ -631,6 +633,7 @@ Filters: `--search`, `--cluster-id`, `--user-id`, `--time-range`
|
|
|
631
633
|
moda context <conversation_id> # default window around middle
|
|
632
634
|
moda context <conversation_id> --msg-index=5 # center on message 5
|
|
633
635
|
moda context <conversation_id> --window=3 # 3 messages each side (max 5)
|
|
636
|
+
moda context <conversation_id> --msg-index=40 --window=0 # just message 40 + total_messages
|
|
634
637
|
moda context <conversation_id> --all # whole trace in order, first 500 messages
|
|
635
638
|
moda context <conversation_id> --all --from=500 --max-messages=500 # continue (max 5000)
|
|
636
639
|
```
|
|
@@ -767,6 +770,13 @@ earlier turns as history);
|
|
|
767
770
|
`anthropic/claude-haiku-4-5`, `google/gemini-2.5-flash`,
|
|
768
771
|
`google/gemini-2.5-pro` unless you pass `--allow-any-model`.
|
|
769
772
|
|
|
773
|
+
`prompts ab --traces=<ids>` builds one case per trace from Moda's distilled
|
|
774
|
+
scenario and success criteria for that trace. A trace Moda cannot distill
|
|
775
|
+
falls back to its opening user message with three generic criteria (which long
|
|
776
|
+
traces pass easily); the run warns and lists it under `caseSources.fallback`.
|
|
777
|
+
`prompts ab` never makes its replay set the tenant's primary set unless you
|
|
778
|
+
pass `--promote`.
|
|
779
|
+
|
|
770
780
|
Runtime code should render synced prompts through the SDK:
|
|
771
781
|
|
|
772
782
|
```typescript
|