@moda-ai/cli 1.38.0 → 1.40.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -0
- package/dist/{cli-kegd748v.js → cli-dyeywy9d.js} +1 -1
- package/dist/{cli-q79sq80a.js → cli-rp912ypf.js} +163 -21
- package/dist/{cli-3kcy61fr.js → cli-w9an3j54.js} +538 -61
- package/dist/{cli-07tv75te.js → cli-xsmsjny3.js} +696 -139
- package/dist/{cli-k38j2fpq.js → cli-yy8fwg1a.js} +2 -2
- package/dist/cli.js +2522 -263
- package/dist/{harness-n2xgzz31.js → harness-191xt7bq.js} +2 -2
- package/dist/{harness-github-actions-qz5d22rz.js → harness-github-actions-1qt4sgxd.js} +2 -2
- package/dist/{index-hwfe7gs8.js → index-1pb2ftqr.js} +5 -5
- package/dist/prompt-source-mcp.js +2 -2
- package/dist/{provision-64c2n9aw.js → provision-gabtks29.js} +3 -3
- package/package.json +1 -1
- package/skills/integration/index.json +2 -2
- package/skills/moda-cli/SKILL.md +239 -20
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
import {
|
|
2
2
|
$parseDocument,
|
|
3
3
|
$stringify,
|
|
4
|
+
assertFlags,
|
|
5
|
+
assertPositionals,
|
|
4
6
|
callControlAPI,
|
|
5
7
|
callDataAPI,
|
|
6
8
|
inventoryFiles,
|
|
@@ -11,7 +13,7 @@ import {
|
|
|
11
13
|
resolvePromptSource,
|
|
12
14
|
safePromptEvidencePath,
|
|
13
15
|
scanHarness
|
|
14
|
-
} from "./cli-
|
|
16
|
+
} from "./cli-xsmsjny3.js";
|
|
15
17
|
import {
|
|
16
18
|
SKILL_SOURCE_HARNESS,
|
|
17
19
|
SKILL_SOURCE_INTEGRATION_BUNDLED,
|
|
@@ -24,7 +26,7 @@ import {
|
|
|
24
26
|
import {
|
|
25
27
|
isAuthSessionValid,
|
|
26
28
|
loadAuthSession
|
|
27
|
-
} from "./cli-
|
|
29
|
+
} from "./cli-dyeywy9d.js";
|
|
28
30
|
import {
|
|
29
31
|
CliInputError,
|
|
30
32
|
createCommandContext,
|
|
@@ -34,13 +36,14 @@ import {
|
|
|
34
36
|
isLoopbackBaseUrl,
|
|
35
37
|
loadProfileConfig,
|
|
36
38
|
resolveApiKey,
|
|
39
|
+
resolveApiTenantId,
|
|
37
40
|
resolveIngestUrl,
|
|
38
41
|
resolveModaBaseUrl,
|
|
39
42
|
resolveSecretReference,
|
|
40
43
|
resolveTenantId,
|
|
41
44
|
secretFilePathFromRef,
|
|
42
45
|
validateConfig
|
|
43
|
-
} from "./cli-
|
|
46
|
+
} from "./cli-rp912ypf.js";
|
|
44
47
|
import {
|
|
45
48
|
__commonJS,
|
|
46
49
|
__require,
|
|
@@ -5564,54 +5567,89 @@ import {
|
|
|
5564
5567
|
import { dirname as dirname2, isAbsolute as isAbsolute3, join, relative, resolve as resolve4 } from "node:path";
|
|
5565
5568
|
|
|
5566
5569
|
// src/prompts-ab.ts
|
|
5567
|
-
import { existsSync } from "node:fs";
|
|
5570
|
+
import { accessSync, constants as fsConstants, existsSync, statSync } from "node:fs";
|
|
5568
5571
|
import { isAbsolute, resolve } from "node:path";
|
|
5569
5572
|
var DEFAULT_SEEDS_PER_CASE = 3;
|
|
5573
|
+
var DEFAULT_AUTO_GENERATE_CASES = 5;
|
|
5574
|
+
var MAX_AUTO_GENERATE_CASES = 100;
|
|
5575
|
+
var MAX_REPLAY_TRACES = 100;
|
|
5576
|
+
var PLAYOUT_CONFIRM_THRESHOLD = 200;
|
|
5577
|
+
var REPLAY_MODEL_ALLOWLIST = [
|
|
5578
|
+
"openai/gpt-5.6-luna",
|
|
5579
|
+
"openai/gpt-4o-mini",
|
|
5580
|
+
"openai/gpt-4o",
|
|
5581
|
+
"anthropic/claude-sonnet-4-5",
|
|
5582
|
+
"anthropic/claude-haiku-4-5",
|
|
5583
|
+
"google/gemini-2.5-flash",
|
|
5584
|
+
"google/gemini-2.5-pro"
|
|
5585
|
+
];
|
|
5570
5586
|
var DEFAULT_POLL_INTERVAL_MS = 15000;
|
|
5571
5587
|
var DEFAULT_WAIT_TIMEOUT_MS = 2 * 60 * 60 * 1000;
|
|
5572
5588
|
var TERMINAL_RUN_STATUSES = new Set(["completed", "skipped", "error"]);
|
|
5573
5589
|
async function runPromptAb(flags, profileOptions, context) {
|
|
5574
5590
|
validateConfig();
|
|
5575
|
-
const tenantId = flags["tenant-id"] ||
|
|
5591
|
+
const tenantId = flags["tenant-id"] || resolveApiTenantId(profileOptions);
|
|
5576
5592
|
if (!tenantId) {
|
|
5577
5593
|
throw new Error("Missing tenant id. Run `moda init` or pass --tenant-id=<id>.");
|
|
5578
5594
|
}
|
|
5579
5595
|
const baselineSource = flags.baseline || flags["baseline-file"] || flags["baseline-key"];
|
|
5580
5596
|
const candidateSource = flags.candidate || flags["candidate-file"] || flags["candidate-key"];
|
|
5581
5597
|
if (!baselineSource || !candidateSource) {
|
|
5582
|
-
throw new Error("Usage: moda prompts ab --baseline=<file|key> --candidate=<file|key> [--auto-generate|--set-id=ID|--traces=id1,id2] (legacy alias: --conversations=)");
|
|
5598
|
+
throw new Error("Usage: moda prompts ab --baseline=<file|key> --candidate=<file|key> [--auto-generate [--cases=1-100]|--set-id=ID|--traces=id1,id2 (max 100)] [--seeds=1-10] [--model=<slug>] [--yes] (legacy alias: --conversations=)");
|
|
5599
|
+
}
|
|
5600
|
+
assertArmSourceShape(baselineSource, "baseline");
|
|
5601
|
+
assertArmSourceShape(candidateSource, "candidate");
|
|
5602
|
+
const seedsPerCase = parsePositiveInt(flags.seeds ?? flags["seeds-per-case"], DEFAULT_SEEDS_PER_CASE, 10, flags.seeds !== undefined ? "--seeds" : "--seeds-per-case");
|
|
5603
|
+
parsePositiveInt(flags["lookback-days"], 30, 365, "--lookback-days");
|
|
5604
|
+
const timeoutMs = parsePositiveInt(flags.timeout, DEFAULT_WAIT_TIMEOUT_MS, 24 * 60 * 60 * 1000, "--timeout");
|
|
5605
|
+
const pollIntervalMs = parsePositiveInt(flags["poll-interval"], DEFAULT_POLL_INTERVAL_MS, 120000, "--poll-interval");
|
|
5606
|
+
const assistantModel = assertAllowedModel(flags.model ?? flags["assistant-model"], flags, "--model");
|
|
5607
|
+
const plan = await planReplaySet(flags, tenantId, profileOptions);
|
|
5608
|
+
const plannedPlayouts = plan.cases === null ? null : plan.cases * seedsPerCase * 2;
|
|
5609
|
+
assertPlayoutBudget(plannedPlayouts, { seedsPerCase, cases: plan.cases }, context.yes, rerunWithYes("prompts ab", [], flags));
|
|
5610
|
+
if (context.outputMode === "human") {
|
|
5611
|
+
process.stderr.write(`${describePlayouts(plannedPlayouts, plan.cases, seedsPerCase)}
|
|
5612
|
+
`);
|
|
5583
5613
|
}
|
|
5584
5614
|
if (flags.sync === "true") {
|
|
5615
|
+
for (const [source, label] of [[baselineSource, "baseline"], [candidateSource, "candidate"]]) {
|
|
5616
|
+
try {
|
|
5617
|
+
resolvePromptArmSpec(source, label);
|
|
5618
|
+
} catch (error) {
|
|
5619
|
+
if (!(error instanceof UnresolvedPromptError))
|
|
5620
|
+
throw error;
|
|
5621
|
+
}
|
|
5622
|
+
}
|
|
5585
5623
|
await runPromptSync({ ...flags, watch: "false" }, profileOptions);
|
|
5586
5624
|
}
|
|
5587
5625
|
const promptArms = {
|
|
5588
5626
|
prod: resolvePromptArmSpec(baselineSource, "baseline"),
|
|
5589
5627
|
proposed: resolvePromptArmSpec(candidateSource, "candidate")
|
|
5590
5628
|
};
|
|
5591
|
-
const replaySetId = await ensureReplaySet(flags, tenantId, profileOptions);
|
|
5592
|
-
const seedsPerCase = parsePositiveInt(flags.seeds ?? flags["seeds-per-case"], DEFAULT_SEEDS_PER_CASE, 10);
|
|
5593
|
-
const assistantModel = (flags.model ?? flags["assistant-model"] ?? "").trim();
|
|
5629
|
+
const replaySetId = await ensureReplaySet(plan, flags, tenantId, profileOptions);
|
|
5594
5630
|
const enqueue = await callControlAPI(`/tenants/${encodeURIComponent(tenantId)}/replay-sets/${encodeURIComponent(replaySetId)}/run`, {
|
|
5595
5631
|
method: "POST",
|
|
5596
5632
|
body: JSON.stringify({
|
|
5597
5633
|
promptArms,
|
|
5598
5634
|
seedsPerCase,
|
|
5599
5635
|
assistantModel: assistantModel || undefined,
|
|
5600
|
-
promotePrimary: flags["no-promote"] !== "true"
|
|
5636
|
+
promotePrimary: flags["no-promote"] !== "true",
|
|
5637
|
+
...plan.kind === "existing" ? replayCasePin(plan.caseIds) : {}
|
|
5601
5638
|
})
|
|
5602
|
-
}, profileOptions, { timeoutMs: 120000, retries:
|
|
5639
|
+
}, profileOptions, { timeoutMs: 120000, retries: 0 });
|
|
5640
|
+
if (context.outputMode === "human")
|
|
5641
|
+
warnIfQueued(enqueue);
|
|
5603
5642
|
const noWait = flags.wait === "false" || flags["no-wait"] === "true";
|
|
5604
5643
|
if (noWait) {
|
|
5605
5644
|
context.output.writeData({
|
|
5606
5645
|
replaySetId,
|
|
5607
5646
|
runId: enqueue.runId,
|
|
5608
5647
|
status: enqueue.status,
|
|
5609
|
-
message: enqueue.message ?? "Replay comparison queued"
|
|
5648
|
+
message: enqueue.message ?? "Replay comparison queued",
|
|
5649
|
+
plannedPlayouts
|
|
5610
5650
|
});
|
|
5611
5651
|
return;
|
|
5612
5652
|
}
|
|
5613
|
-
const timeoutMs = parsePositiveInt(flags.timeout, DEFAULT_WAIT_TIMEOUT_MS, 24 * 60 * 60 * 1000);
|
|
5614
|
-
const pollIntervalMs = parsePositiveInt(flags["poll-interval"], DEFAULT_POLL_INTERVAL_MS, 120000);
|
|
5615
5653
|
const deadline = Date.now() + timeoutMs;
|
|
5616
5654
|
let latest = null;
|
|
5617
5655
|
while (Date.now() < deadline) {
|
|
@@ -5622,7 +5660,7 @@ async function runPromptAb(flags, profileOptions, context) {
|
|
|
5622
5660
|
break;
|
|
5623
5661
|
}
|
|
5624
5662
|
if (context.outputMode === "human") {
|
|
5625
|
-
process.stderr.write(`Waiting for replay run (${
|
|
5663
|
+
process.stderr.write(`Waiting for replay run (${waitLabel(latest.run)})...
|
|
5626
5664
|
`);
|
|
5627
5665
|
}
|
|
5628
5666
|
await sleep(pollIntervalMs);
|
|
@@ -5632,6 +5670,7 @@ async function runPromptAb(flags, profileOptions, context) {
|
|
|
5632
5670
|
}
|
|
5633
5671
|
const result = {
|
|
5634
5672
|
replaySetId,
|
|
5673
|
+
plannedPlayouts,
|
|
5635
5674
|
runId: latest.run.runId,
|
|
5636
5675
|
status: latest.run.status,
|
|
5637
5676
|
verdict: formatVerdict(latest),
|
|
@@ -5643,6 +5682,65 @@ async function runPromptAb(flags, profileOptions, context) {
|
|
|
5643
5682
|
}
|
|
5644
5683
|
context.output.writeData(result);
|
|
5645
5684
|
}
|
|
5685
|
+
function warnIfQueued(enqueue) {
|
|
5686
|
+
const { runsAhead, maxActiveRuns } = enqueue;
|
|
5687
|
+
if (typeof runsAhead !== "number" || typeof maxActiveRuns !== "number")
|
|
5688
|
+
return;
|
|
5689
|
+
if (runsAhead < maxActiveRuns)
|
|
5690
|
+
return;
|
|
5691
|
+
process.stderr.write(`Note: ${runsAhead} replay runs for this tenant are already running or queued ` + `(${maxActiveRuns} at a time); run ${enqueue.runId} waits in the queue until a slot frees.
|
|
5692
|
+
`);
|
|
5693
|
+
}
|
|
5694
|
+
function waitLabel(run) {
|
|
5695
|
+
if (!run?.status)
|
|
5696
|
+
return "pending";
|
|
5697
|
+
if (run.status === "accepted") {
|
|
5698
|
+
return typeof run.runsAhead === "number" ? `queued, ${run.runsAhead} run${run.runsAhead === 1 ? "" : "s"} ahead` : "queued";
|
|
5699
|
+
}
|
|
5700
|
+
return run.status;
|
|
5701
|
+
}
|
|
5702
|
+
var FILE_ARM_RE = /[\\/]|\.(?:md|txt|json|ya?ml|jinja2?|j2|hbs|tmpl|prompt|[cm]?[jt]sx?|py)$/i;
|
|
5703
|
+
var PROMPT_KEY_RE = /^[^\s"'`\x00-\x1f\x7f]+$/;
|
|
5704
|
+
function assertArmSourceShape(source, label) {
|
|
5705
|
+
const trimmed = source.trim();
|
|
5706
|
+
if (!trimmed) {
|
|
5707
|
+
throw new CliInputError(`Empty prompt source for ${label}, so nothing was sent.`);
|
|
5708
|
+
}
|
|
5709
|
+
const absPath = isAbsolute(trimmed) ? trimmed : resolve(process.cwd(), trimmed);
|
|
5710
|
+
const exists = existsSync(absPath);
|
|
5711
|
+
if (!exists && FILE_ARM_RE.test(trimmed)) {
|
|
5712
|
+
if (isDiscoveredKey(trimmed))
|
|
5713
|
+
return;
|
|
5714
|
+
throw new CliInputError(`--${label} file not found: ${trimmed}, so nothing was sent.`);
|
|
5715
|
+
}
|
|
5716
|
+
if (exists) {
|
|
5717
|
+
if (!statSync(absPath).isFile()) {
|
|
5718
|
+
throw new CliInputError(`--${label} is not a file: ${trimmed}, so nothing was sent.`);
|
|
5719
|
+
}
|
|
5720
|
+
try {
|
|
5721
|
+
accessSync(absPath, fsConstants.R_OK);
|
|
5722
|
+
} catch {
|
|
5723
|
+
throw new CliInputError(`--${label} file is not readable: ${trimmed}, so nothing was sent.`);
|
|
5724
|
+
}
|
|
5725
|
+
if (/\.(?:[cm]?[jt]sx?|py)$/i.test(absPath)) {
|
|
5726
|
+
throw new CliInputError(`--${label}=${trimmed} is a source module; source modules cannot be replayed as instructions, so nothing was sent.`, "Use the registered prompt key for this code source.");
|
|
5727
|
+
}
|
|
5728
|
+
return;
|
|
5729
|
+
}
|
|
5730
|
+
if (!PROMPT_KEY_RE.test(trimmed)) {
|
|
5731
|
+
throw new CliInputError(`--${label}=${JSON.stringify(trimmed)} is neither an existing prompt file nor a valid prompt key, so nothing was sent.`, "Pass a .prompt.md path or a prompt key (no spaces or quotes).");
|
|
5732
|
+
}
|
|
5733
|
+
}
|
|
5734
|
+
function isDiscoveredKey(key) {
|
|
5735
|
+
try {
|
|
5736
|
+
return discoverPrompts().some((prompt) => prompt.key === key);
|
|
5737
|
+
} catch {
|
|
5738
|
+
return false;
|
|
5739
|
+
}
|
|
5740
|
+
}
|
|
5741
|
+
|
|
5742
|
+
class UnresolvedPromptError extends Error {
|
|
5743
|
+
}
|
|
5646
5744
|
function resolvePromptArmSpec(source, label, opts = {}) {
|
|
5647
5745
|
const trimmed = source.trim();
|
|
5648
5746
|
if (!trimmed) {
|
|
@@ -5676,7 +5774,7 @@ function resolvePromptArmSpec(source, label, opts = {}) {
|
|
|
5676
5774
|
} else {
|
|
5677
5775
|
const match = fromDiscovery();
|
|
5678
5776
|
if (!match) {
|
|
5679
|
-
throw new
|
|
5777
|
+
throw new UnresolvedPromptError(`Could not resolve prompt '${trimmed}' — pass a .prompt.md path or a discovered prompt key`);
|
|
5680
5778
|
}
|
|
5681
5779
|
key = match.key;
|
|
5682
5780
|
content = match.content;
|
|
@@ -5689,20 +5787,86 @@ function resolvePromptArmSpec(source, label, opts = {}) {
|
|
|
5689
5787
|
prompt_label: label
|
|
5690
5788
|
};
|
|
5691
5789
|
}
|
|
5692
|
-
async function
|
|
5790
|
+
async function planReplaySet(flags, tenantId, profileOptions) {
|
|
5693
5791
|
const existingSetId = (flags["set-id"] || flags["replay-set-id"] || "").trim();
|
|
5694
5792
|
if (existingSetId) {
|
|
5695
|
-
|
|
5793
|
+
const snapshot = await fetchReplaySetCases(tenantId, existingSetId, profileOptions);
|
|
5794
|
+
return { kind: "existing", setId: existingSetId, cases: snapshot?.ids.length ?? null, caseIds: snapshot?.ids };
|
|
5696
5795
|
}
|
|
5697
5796
|
const conversations = parseCsv(flags.traces ?? flags.conversations ?? flags["conversation-ids"]);
|
|
5797
|
+
if (conversations.length > MAX_REPLAY_TRACES) {
|
|
5798
|
+
throw new CliInputError(`--traces has ${conversations.length} ids; the limit is ${MAX_REPLAY_TRACES} per replay set, so nothing was sent.`, "Split the traces across several runs, or use --auto-generate.");
|
|
5799
|
+
}
|
|
5698
5800
|
if (conversations.length) {
|
|
5699
|
-
return
|
|
5801
|
+
return { kind: "traces", traces: conversations, cases: conversations.length };
|
|
5700
5802
|
}
|
|
5701
|
-
if (flags["auto-generate"] === "false"
|
|
5803
|
+
if (flags["auto-generate"] === "false") {
|
|
5702
5804
|
throw new Error("Provide --set-id=, --traces= (legacy alias: --conversations=), or allow --auto-generate (default)");
|
|
5703
5805
|
}
|
|
5704
|
-
|
|
5705
|
-
|
|
5806
|
+
return {
|
|
5807
|
+
kind: "auto-generate",
|
|
5808
|
+
cases: parseCappedInt(flags.cases ?? flags["case-count"], DEFAULT_AUTO_GENERATE_CASES, MAX_AUTO_GENERATE_CASES, "--cases")
|
|
5809
|
+
};
|
|
5810
|
+
}
|
|
5811
|
+
async function fetchReplaySetCases(tenantId, setId, profileOptions) {
|
|
5812
|
+
try {
|
|
5813
|
+
const set = await callControlAPI(`/tenants/${encodeURIComponent(tenantId)}/replay-sets/${encodeURIComponent(setId)}`, { method: "GET" }, profileOptions);
|
|
5814
|
+
if (!Array.isArray(set?.cases))
|
|
5815
|
+
return null;
|
|
5816
|
+
return {
|
|
5817
|
+
ids: set.cases.filter((c) => String(c?.scenario ?? "").trim()).map((c, index) => String(c?.id ?? `case_${index}`))
|
|
5818
|
+
};
|
|
5819
|
+
} catch (error) {
|
|
5820
|
+
if (error?.statusCode === 404) {
|
|
5821
|
+
throw new CliInputError(`Replay set '${setId}' not found for tenant ${tenantId}, so nothing was sent.`, "Check --set-id (list sets in the dashboard, or omit it to auto-generate).");
|
|
5822
|
+
}
|
|
5823
|
+
return null;
|
|
5824
|
+
}
|
|
5825
|
+
}
|
|
5826
|
+
function replayCasePin(caseIds) {
|
|
5827
|
+
if (!caseIds)
|
|
5828
|
+
return {};
|
|
5829
|
+
return caseIds.length <= REPLAY_CASE_IDS_PIN_MAX ? { caseIds, expectedCases: caseIds.length } : { expectedCases: caseIds.length };
|
|
5830
|
+
}
|
|
5831
|
+
var REPLAY_CASE_IDS_PIN_MAX = 500;
|
|
5832
|
+
function describePlayouts(planned, cases, seeds) {
|
|
5833
|
+
return planned === null ? `Planned replay playouts: unknown (case count unavailable) × ${seeds} seeds × 2 arms` : `Planned replay playouts: ${planned} (${cases} cases × ${seeds} seeds × 2 arms)`;
|
|
5834
|
+
}
|
|
5835
|
+
function assertPlayoutBudget(planned, shape, yes, rerun) {
|
|
5836
|
+
if (yes)
|
|
5837
|
+
return;
|
|
5838
|
+
if (planned !== null && planned <= PLAYOUT_CONFIRM_THRESHOLD)
|
|
5839
|
+
return;
|
|
5840
|
+
const what = planned === null ? "an unknown number of replay playouts (the replay set's case count could not be read)" : `${planned} replay playouts (${shape.cases} cases × ${shape.seedsPerCase} seeds × 2 arms)`;
|
|
5841
|
+
throw new CliInputError(`This would run ${what}, above the ${PLAYOUT_CONFIRM_THRESHOLD}-playout limit that needs confirmation. Nothing was sent.`, `Re-run with --yes to confirm: ${rerun}`);
|
|
5842
|
+
}
|
|
5843
|
+
function assertAllowedModel(raw, flags, flagName) {
|
|
5844
|
+
const model = (raw ?? "").trim();
|
|
5845
|
+
if (!model || flags["allow-any-model"] === "true")
|
|
5846
|
+
return model;
|
|
5847
|
+
if (REPLAY_MODEL_ALLOWLIST.includes(model))
|
|
5848
|
+
return model;
|
|
5849
|
+
throw new CliInputError(`${flagName}=${model} is not in the replay model allowlist, so nothing was sent.`, `Allowed: ${REPLAY_MODEL_ALLOWLIST.join(", ")}. Pass --allow-any-model to use another OpenRouter slug.`);
|
|
5850
|
+
}
|
|
5851
|
+
function rerunWithYes(command, positionals, flags) {
|
|
5852
|
+
const quote = (v) => /^[\w@%+=:,./-]+$/.test(v) ? v : `'${v.replace(/'/g, `'\\''`)}'`;
|
|
5853
|
+
const parts = ["moda", ...command.split(" "), ...positionals.map(quote)];
|
|
5854
|
+
for (const [key, value] of Object.entries(flags)) {
|
|
5855
|
+
if (key === "yes")
|
|
5856
|
+
continue;
|
|
5857
|
+
parts.push(value === "true" ? `--${key}` : `--${key}=${quote(value)}`);
|
|
5858
|
+
}
|
|
5859
|
+
parts.push("--yes");
|
|
5860
|
+
return parts.join(" ");
|
|
5861
|
+
}
|
|
5862
|
+
async function ensureReplaySet(plan, flags, tenantId, profileOptions) {
|
|
5863
|
+
if (plan.kind === "existing")
|
|
5864
|
+
return plan.setId;
|
|
5865
|
+
if (plan.kind === "traces") {
|
|
5866
|
+
return createSetFromConversations(plan.traces, flags, tenantId, profileOptions);
|
|
5867
|
+
}
|
|
5868
|
+
const caseCount = plan.cases;
|
|
5869
|
+
const lookbackDays = parsePositiveInt(flags["lookback-days"], 30, 365, "--lookback-days");
|
|
5706
5870
|
const name = (flags.name ?? flags["set-name"] ?? `Prompt A/B ${new Date().toISOString().slice(0, 10)}`).trim();
|
|
5707
5871
|
const generated = await callControlAPI(`/tenants/${encodeURIComponent(tenantId)}/replay-sets/auto-generate`, {
|
|
5708
5872
|
method: "POST",
|
|
@@ -5826,9 +5990,12 @@ async function enqueueAndPollComparison(args, profileOptions) {
|
|
|
5826
5990
|
promptArms: args.promptArms,
|
|
5827
5991
|
seedsPerCase: args.seedsPerCase ?? DEFAULT_SEEDS_PER_CASE,
|
|
5828
5992
|
assistantModel: args.assistantModel || undefined,
|
|
5829
|
-
promotePrimary: args.promotePrimary ?? false
|
|
5993
|
+
promotePrimary: args.promotePrimary ?? false,
|
|
5994
|
+
...replayCasePin(args.caseIds)
|
|
5830
5995
|
})
|
|
5831
|
-
}, profileOptions, { timeoutMs: 120000, retries:
|
|
5996
|
+
}, profileOptions, { timeoutMs: 120000, retries: 0 });
|
|
5997
|
+
if (args.onWaiting)
|
|
5998
|
+
warnIfQueued(enqueue);
|
|
5832
5999
|
const timeoutMs = args.timeoutMs ?? DEFAULT_WAIT_TIMEOUT_MS;
|
|
5833
6000
|
const pollIntervalMs = args.pollIntervalMs ?? DEFAULT_POLL_INTERVAL_MS;
|
|
5834
6001
|
const deadline = Date.now() + timeoutMs;
|
|
@@ -5840,7 +6007,7 @@ async function enqueueAndPollComparison(args, profileOptions) {
|
|
|
5840
6007
|
if (TERMINAL_RUN_STATUSES.has(status) && runId === enqueue.runId) {
|
|
5841
6008
|
return latest;
|
|
5842
6009
|
}
|
|
5843
|
-
args.onWaiting?.(
|
|
6010
|
+
args.onWaiting?.(waitLabel(latest.run));
|
|
5844
6011
|
await sleep(pollIntervalMs);
|
|
5845
6012
|
}
|
|
5846
6013
|
if (!latest?.run || latest.run.runId !== enqueue.runId || !TERMINAL_RUN_STATUSES.has(latest.run.status)) {
|
|
@@ -5848,13 +6015,21 @@ async function enqueueAndPollComparison(args, profileOptions) {
|
|
|
5848
6015
|
}
|
|
5849
6016
|
return latest;
|
|
5850
6017
|
}
|
|
5851
|
-
function parsePositiveInt(raw, fallback, max) {
|
|
5852
|
-
if (
|
|
5853
|
-
return fallback;
|
|
5854
|
-
const parsed = Number.parseInt(raw, 10);
|
|
5855
|
-
if (!Number.isFinite(parsed) || parsed < 1)
|
|
6018
|
+
function parsePositiveInt(raw, fallback, max, flagName) {
|
|
6019
|
+
if (raw === undefined)
|
|
5856
6020
|
return fallback;
|
|
5857
|
-
|
|
6021
|
+
const text = String(raw).trim();
|
|
6022
|
+
const parsed = Number(text);
|
|
6023
|
+
if (!/^\d+$/.test(text) || !Number.isSafeInteger(parsed) || parsed < 1) {
|
|
6024
|
+
throw new CliInputError(`${flagName}=${raw} must be a whole number of at least 1, so nothing was sent.`, `Use ${flagName}=N with 1 <= N <= ${max}.`);
|
|
6025
|
+
}
|
|
6026
|
+
if (parsed > max) {
|
|
6027
|
+
throw new CliInputError(`${flagName}=${parsed} is above the maximum of ${max}, so nothing was sent.`, `Use ${flagName}=${max} or less.`);
|
|
6028
|
+
}
|
|
6029
|
+
return parsed;
|
|
6030
|
+
}
|
|
6031
|
+
function parseCappedInt(raw, fallback, max, flagName) {
|
|
6032
|
+
return parsePositiveInt(raw, fallback, max, flagName);
|
|
5858
6033
|
}
|
|
5859
6034
|
function parseCsv(raw) {
|
|
5860
6035
|
if (!raw?.trim())
|
|
@@ -5875,10 +6050,31 @@ async function runPromptPropose(positionals, flags, profileOptions, context) {
|
|
|
5875
6050
|
const fromRunId = flags["from-run"] || flags["from-run-id"] || flags["run-id"];
|
|
5876
6051
|
const replaySetId = flags["set-id"] || flags["replay-set-id"];
|
|
5877
6052
|
if (!promptKey || !fromRunId || !replaySetId) {
|
|
5878
|
-
throw new Error("Usage: moda prompts propose --prompt-key=<key> --from-run=<run_id> --set-id=<replay_set_id> [--max-dossiers=8] [--model=<slug>] [--out=path.prompt.md] [--gate] [--promote-on-win]");
|
|
6053
|
+
throw new Error("Usage: moda prompts propose --prompt-key=<key> --from-run=<run_id> --set-id=<replay_set_id> [--max-dossiers=8] [--model=<slug>] [--out=path.prompt.md] [--gate [--seeds=1-10] [--assistant-model=<slug>] [--yes]] [--promote-on-win] [--allow-any-model]");
|
|
6054
|
+
}
|
|
6055
|
+
const maxDossiers = parsePositiveInt2(flags["max-dossiers"], 8, 16, "--max-dossiers");
|
|
6056
|
+
parsePositiveInt2(flags["gate-timeout"] ?? flags.timeout, 2 * 60 * 60 * 1000, 24 * 60 * 60 * 1000, flags["gate-timeout"] !== undefined ? "--gate-timeout" : "--timeout");
|
|
6057
|
+
parsePositiveInt2(flags["poll-interval"], 15000, 120000, "--poll-interval");
|
|
6058
|
+
const model = assertAllowedModel(flags.model ?? flags["revise-model"], flags, "--model");
|
|
6059
|
+
const wantGate = flags.gate === "true";
|
|
6060
|
+
let gatePlan;
|
|
6061
|
+
if (wantGate) {
|
|
6062
|
+
const tenantId = flags["tenant-id"] || resolveApiTenantId(profileOptions);
|
|
6063
|
+
if (!tenantId) {
|
|
6064
|
+
throw new Error("Missing tenant id for --gate. Run `moda init` or pass --tenant-id=<id>.");
|
|
6065
|
+
}
|
|
6066
|
+
const seedsPerCase = parsePositiveInt2(flags.seeds ?? flags["seeds-per-case"], 3, 10, flags.seeds !== undefined ? "--seeds" : "--seeds-per-case");
|
|
6067
|
+
const assistantModel = assertAllowedModel(flags["assistant-model"], flags, "--assistant-model");
|
|
6068
|
+
const snapshot = await fetchGateSetCases(tenantId, replaySetId, profileOptions);
|
|
6069
|
+
const cases = snapshot?.ids.length ?? null;
|
|
6070
|
+
const plannedPlayouts = cases === null ? null : cases * seedsPerCase * 2;
|
|
6071
|
+
assertPlayoutBudget(plannedPlayouts, { seedsPerCase, cases }, context.yes, rerunWithYes("prompts propose", positionals.slice(1), flags));
|
|
6072
|
+
if (context.outputMode === "human") {
|
|
6073
|
+
process.stderr.write(`Gate: ${describePlayouts(plannedPlayouts, cases, seedsPerCase)}
|
|
6074
|
+
`);
|
|
6075
|
+
}
|
|
6076
|
+
gatePlan = { tenantId, seedsPerCase, assistantModel, plannedPlayouts, cases, caseIds: snapshot?.ids };
|
|
5879
6077
|
}
|
|
5880
|
-
const maxDossiers = parsePositiveInt2(flags["max-dossiers"], 8, 16);
|
|
5881
|
-
const model = (flags.model ?? flags["revise-model"] ?? "").trim();
|
|
5882
6078
|
const result = await callControlAPI(`/prompts/${encodeURIComponent(promptKey)}/propose`, {
|
|
5883
6079
|
method: "POST",
|
|
5884
6080
|
body: JSON.stringify({ fromRunId, replaySetId, maxDossiers, model: model || undefined })
|
|
@@ -5888,10 +6084,9 @@ async function runPromptPropose(positionals, flags, profileOptions, context) {
|
|
|
5888
6084
|
outPath = isAbsolute2(flags.out) ? flags.out : resolve2(process.cwd(), flags.out);
|
|
5889
6085
|
writeFileSync(outPath, result.content, "utf8");
|
|
5890
6086
|
}
|
|
5891
|
-
const wantGate = flags.gate === "true";
|
|
5892
6087
|
const promoteOnWin = flags["promote-on-win"] === "true";
|
|
5893
6088
|
let gate;
|
|
5894
|
-
if (
|
|
6089
|
+
if (gatePlan) {
|
|
5895
6090
|
if (result.status !== "proposed" || !result.content) {
|
|
5896
6091
|
gate = {
|
|
5897
6092
|
ran: false,
|
|
@@ -5900,7 +6095,8 @@ async function runPromptPropose(positionals, flags, profileOptions, context) {
|
|
|
5900
6095
|
verdict: "skipped",
|
|
5901
6096
|
outcome: "no-run",
|
|
5902
6097
|
runId: null,
|
|
5903
|
-
promoted: false
|
|
6098
|
+
promoted: false,
|
|
6099
|
+
plannedPlayouts: gatePlan.plannedPlayouts
|
|
5904
6100
|
};
|
|
5905
6101
|
} else {
|
|
5906
6102
|
gate = await runGate({
|
|
@@ -5908,6 +6104,7 @@ async function runPromptPropose(positionals, flags, profileOptions, context) {
|
|
|
5908
6104
|
promptKey,
|
|
5909
6105
|
replaySetId,
|
|
5910
6106
|
flags,
|
|
6107
|
+
plan: gatePlan,
|
|
5911
6108
|
promoteOnWin,
|
|
5912
6109
|
profileOptions,
|
|
5913
6110
|
context
|
|
@@ -5923,12 +6120,25 @@ async function runPromptPropose(positionals, flags, profileOptions, context) {
|
|
|
5923
6120
|
...gate ? { gate } : {}
|
|
5924
6121
|
});
|
|
5925
6122
|
}
|
|
5926
|
-
async function
|
|
5927
|
-
|
|
5928
|
-
|
|
5929
|
-
|
|
5930
|
-
|
|
6123
|
+
async function fetchGateSetCases(tenantId, replaySetId, profileOptions) {
|
|
6124
|
+
let set;
|
|
6125
|
+
try {
|
|
6126
|
+
set = await callControlAPI(`/tenants/${encodeURIComponent(tenantId)}/replay-sets/${encodeURIComponent(replaySetId)}`, { method: "GET" }, profileOptions);
|
|
6127
|
+
} catch (error) {
|
|
6128
|
+
if (error?.statusCode === 404) {
|
|
6129
|
+
throw new CliInputError(`Replay set '${replaySetId}' not found for tenant ${tenantId}, so nothing was sent.`, "Check --set-id: use the replay set the --from-run run was scored on (see `moda prompts replay-runs <key>`).");
|
|
6130
|
+
}
|
|
6131
|
+
return null;
|
|
5931
6132
|
}
|
|
6133
|
+
if (!Array.isArray(set?.cases))
|
|
6134
|
+
return null;
|
|
6135
|
+
return {
|
|
6136
|
+
ids: set.cases.filter((c) => String(c?.scenario ?? "").trim()).map((c, index) => String(c?.id ?? `case_${index}`))
|
|
6137
|
+
};
|
|
6138
|
+
}
|
|
6139
|
+
async function runGate(args) {
|
|
6140
|
+
const { result, promptKey, replaySetId, flags, plan, promoteOnWin, profileOptions, context } = args;
|
|
6141
|
+
const { tenantId, seedsPerCase, assistantModel } = plan;
|
|
5932
6142
|
const prodArm = resolvePromptArmSpec(promptKey, "baseline", { preferKey: true });
|
|
5933
6143
|
const proposedArm = {
|
|
5934
6144
|
content: result.content ?? "",
|
|
@@ -5936,10 +6146,8 @@ async function runGate(args) {
|
|
|
5936
6146
|
prompt_id: "",
|
|
5937
6147
|
prompt_label: "candidate"
|
|
5938
6148
|
};
|
|
5939
|
-
const
|
|
5940
|
-
const
|
|
5941
|
-
const timeoutMs = parsePositiveInt2(flags["gate-timeout"] ?? flags.timeout, 2 * 60 * 60 * 1000, 24 * 60 * 60 * 1000);
|
|
5942
|
-
const pollIntervalMs = parsePositiveInt2(flags["poll-interval"], 15000, 120000);
|
|
6149
|
+
const timeoutMs = parsePositiveInt2(flags["gate-timeout"] ?? flags.timeout, 2 * 60 * 60 * 1000, 24 * 60 * 60 * 1000, flags["gate-timeout"] !== undefined ? "--gate-timeout" : "--timeout");
|
|
6150
|
+
const pollIntervalMs = parsePositiveInt2(flags["poll-interval"], 15000, 120000, "--poll-interval");
|
|
5943
6151
|
const scopeNote = result.holdoutCaseIds.length ? `gated over the FULL replay set (${result.holdoutCaseIds.length} holdout case(s) could not be isolated — the run endpoint scores every case)` : "gated over the FULL replay set";
|
|
5944
6152
|
if (context.outputMode === "human") {
|
|
5945
6153
|
process.stderr.write(`
|
|
@@ -5953,6 +6161,7 @@ Gating candidate for '${promptKey}' — ${scopeNote}...
|
|
|
5953
6161
|
seedsPerCase,
|
|
5954
6162
|
assistantModel: assistantModel || undefined,
|
|
5955
6163
|
promotePrimary: false,
|
|
6164
|
+
caseIds: plan.caseIds,
|
|
5956
6165
|
timeoutMs,
|
|
5957
6166
|
pollIntervalMs,
|
|
5958
6167
|
onWaiting: context.outputMode === "human" ? (status) => process.stderr.write(`Waiting for gate run (${status})...
|
|
@@ -5966,7 +6175,8 @@ Gating candidate for '${promptKey}' — ${scopeNote}...
|
|
|
5966
6175
|
verdict: formatVerdict(payload),
|
|
5967
6176
|
outcome,
|
|
5968
6177
|
runId: payload.run?.runId ?? null,
|
|
5969
|
-
promoted: false
|
|
6178
|
+
promoted: false,
|
|
6179
|
+
plannedPlayouts: plan.plannedPlayouts
|
|
5970
6180
|
};
|
|
5971
6181
|
if (promoteOnWin) {
|
|
5972
6182
|
if (outcome === "candidate" && result.versionId) {
|
|
@@ -6005,6 +6215,8 @@ function printHuman(result, outPath, gate) {
|
|
|
6005
6215
|
w("Gate (automatic A/B of candidate vs baseline):");
|
|
6006
6216
|
w(` scope: ${gate.scopeNote}`);
|
|
6007
6217
|
w(` verdict: ${gate.verdict}`);
|
|
6218
|
+
if (gate.plannedPlayouts !== undefined)
|
|
6219
|
+
w(` playouts: ${gate.plannedPlayouts ?? "unknown"} planned`);
|
|
6008
6220
|
if (gate.runId)
|
|
6009
6221
|
w(` run id: ${gate.runId}`);
|
|
6010
6222
|
if (gate.promoted) {
|
|
@@ -6037,11 +6249,18 @@ function printHuman(result, outPath, gate) {
|
|
|
6037
6249
|
w(` # or: moda prompts propose ... --gate --promote-on-win (auto-A/B, promote on a strict win)`);
|
|
6038
6250
|
w(` moda prompts promote ${result.promptKey} --label=prod --version=${result.versionId}`);
|
|
6039
6251
|
}
|
|
6040
|
-
function parsePositiveInt2(raw, fallback, max) {
|
|
6041
|
-
|
|
6042
|
-
if (!Number.isFinite(n) || n <= 0)
|
|
6252
|
+
function parsePositiveInt2(raw, fallback, max, flagName) {
|
|
6253
|
+
if (raw === undefined)
|
|
6043
6254
|
return fallback;
|
|
6044
|
-
|
|
6255
|
+
const text = String(raw).trim();
|
|
6256
|
+
const parsed = Number(text);
|
|
6257
|
+
if (!/^\d+$/.test(text) || !Number.isSafeInteger(parsed) || parsed < 1) {
|
|
6258
|
+
throw new CliInputError(`${flagName}=${raw} must be a whole number of at least 1, so nothing was sent.`, `Use ${flagName}=N with 1 <= N <= ${max}.`);
|
|
6259
|
+
}
|
|
6260
|
+
if (parsed > max) {
|
|
6261
|
+
throw new CliInputError(`${flagName}=${parsed} is above the maximum of ${max}, so nothing was sent.`, `Use ${flagName}=${max} or less.`);
|
|
6262
|
+
}
|
|
6263
|
+
return parsed;
|
|
6045
6264
|
}
|
|
6046
6265
|
|
|
6047
6266
|
// src/prompt-attribution.ts
|
|
@@ -6179,6 +6398,77 @@ var DEFAULT_PROMPT_PATHS = [
|
|
|
6179
6398
|
"prompts/**/*.prompt.yaml",
|
|
6180
6399
|
"prompts/**/*.prompt.yml"
|
|
6181
6400
|
];
|
|
6401
|
+
var PROMPT_WRITE_SUBCOMMAND_FLAGS = {
|
|
6402
|
+
init: ["from-harness"],
|
|
6403
|
+
status: [],
|
|
6404
|
+
diff: [],
|
|
6405
|
+
sync: [
|
|
6406
|
+
"dry-run",
|
|
6407
|
+
"from-harness",
|
|
6408
|
+
"no-analyze",
|
|
6409
|
+
"allow-untrack",
|
|
6410
|
+
"watch",
|
|
6411
|
+
"interval",
|
|
6412
|
+
"analyst",
|
|
6413
|
+
"assist",
|
|
6414
|
+
"analyst-command",
|
|
6415
|
+
"analyst-model"
|
|
6416
|
+
],
|
|
6417
|
+
promote: ["label", "version", "version-id"],
|
|
6418
|
+
ab: [
|
|
6419
|
+
"baseline",
|
|
6420
|
+
"candidate",
|
|
6421
|
+
"cases",
|
|
6422
|
+
"conversations",
|
|
6423
|
+
"model",
|
|
6424
|
+
"name",
|
|
6425
|
+
"seeds",
|
|
6426
|
+
"sync",
|
|
6427
|
+
"timeout",
|
|
6428
|
+
"traces",
|
|
6429
|
+
"wait",
|
|
6430
|
+
"assistant-model",
|
|
6431
|
+
"auto-generate",
|
|
6432
|
+
"baseline-file",
|
|
6433
|
+
"baseline-key",
|
|
6434
|
+
"candidate-file",
|
|
6435
|
+
"candidate-key",
|
|
6436
|
+
"case-count",
|
|
6437
|
+
"conversation-ids",
|
|
6438
|
+
"lookback-days",
|
|
6439
|
+
"no-promote",
|
|
6440
|
+
"no-wait",
|
|
6441
|
+
"poll-interval",
|
|
6442
|
+
"replay-set-id",
|
|
6443
|
+
"seeds-per-case",
|
|
6444
|
+
"set-id",
|
|
6445
|
+
"set-name",
|
|
6446
|
+
"tenant-id",
|
|
6447
|
+
"allow-any-model"
|
|
6448
|
+
],
|
|
6449
|
+
propose: [
|
|
6450
|
+
"gate",
|
|
6451
|
+
"model",
|
|
6452
|
+
"out",
|
|
6453
|
+
"seeds",
|
|
6454
|
+
"timeout",
|
|
6455
|
+
"assistant-model",
|
|
6456
|
+
"from-run",
|
|
6457
|
+
"from-run-id",
|
|
6458
|
+
"gate-timeout",
|
|
6459
|
+
"max-dossiers",
|
|
6460
|
+
"poll-interval",
|
|
6461
|
+
"promote-on-win",
|
|
6462
|
+
"prompt-key",
|
|
6463
|
+
"replay-set-id",
|
|
6464
|
+
"revise-model",
|
|
6465
|
+
"run-id",
|
|
6466
|
+
"seeds-per-case",
|
|
6467
|
+
"set-id",
|
|
6468
|
+
"tenant-id",
|
|
6469
|
+
"allow-any-model"
|
|
6470
|
+
]
|
|
6471
|
+
};
|
|
6182
6472
|
async function runPromptsCommand(input, legacyFlags, legacyProfileOptions) {
|
|
6183
6473
|
const context = Array.isArray(input) ? createCommandContext({
|
|
6184
6474
|
command: "prompts",
|
|
@@ -6194,6 +6484,9 @@ async function runPromptsCommand(input, legacyFlags, legacyProfileOptions) {
|
|
|
6194
6484
|
const flags = context.flags;
|
|
6195
6485
|
const profileOptions = legacyProfileOptions ?? { profile: context.profile, env: context.env };
|
|
6196
6486
|
const subcommand = positionals[0] || "status";
|
|
6487
|
+
const writeFlags = PROMPT_WRITE_SUBCOMMAND_FLAGS[subcommand];
|
|
6488
|
+
if (writeFlags)
|
|
6489
|
+
assertFlags(`prompts ${subcommand}`, flags, writeFlags);
|
|
6197
6490
|
switch (subcommand) {
|
|
6198
6491
|
case "init": {
|
|
6199
6492
|
const imported = flags["from-harness"] ? importHarnessPromptSources(flags["from-harness"]) : undefined;
|
|
@@ -6219,9 +6512,27 @@ async function runPromptsCommand(input, legacyFlags, legacyProfileOptions) {
|
|
|
6219
6512
|
case "status":
|
|
6220
6513
|
context.output.writeData(await getPromptStatus());
|
|
6221
6514
|
break;
|
|
6222
|
-
case "diff":
|
|
6223
|
-
|
|
6515
|
+
case "diff": {
|
|
6516
|
+
const status = await getPromptStatus();
|
|
6517
|
+
const differing = status.prompts.filter((row) => row.state !== "unchanged");
|
|
6518
|
+
const tracked = existsSync2(MANIFEST_PATH);
|
|
6519
|
+
const untrackedWarning = tracked ? [] : [
|
|
6520
|
+
`No ${MANIFEST_PATH} here, so diff has no last sync to compare against: it only covers prompts tracked by \`moda prompts sync\`${differing.length > 0 ? ", and every discovered prompt file lists as new" : ""}. For a registry manifest, preview with \`moda registry push --dry-run\`.`
|
|
6521
|
+
];
|
|
6522
|
+
context.output.writeData({
|
|
6523
|
+
...status,
|
|
6524
|
+
unchanged: status.total - differing.length,
|
|
6525
|
+
prompts: differing,
|
|
6526
|
+
...tracked ? {} : { tracked: false }
|
|
6527
|
+
}, {
|
|
6528
|
+
summary: {
|
|
6529
|
+
text: !tracked ? differing.length === 0 ? `Not a prompts-sync project (no ${MANIFEST_PATH}) and no prompt files found; diff has nothing to compare.` : `Not a prompts-sync project (no ${MANIFEST_PATH}): ${differing.length} discovered prompt file(s) are untracked and would be new on \`moda prompts sync\`.` : differing.length === 0 ? "No prompt differs from the last sync." : `${differing.length} prompt(s) differ from the last sync (changed ${status.changed}, new ${status.new}, deleted ${status.deleted}).`,
|
|
6530
|
+
confidence: "high"
|
|
6531
|
+
},
|
|
6532
|
+
...untrackedWarning.length > 0 ? { warnings: untrackedWarning } : {}
|
|
6533
|
+
});
|
|
6224
6534
|
break;
|
|
6535
|
+
}
|
|
6225
6536
|
case "sync":
|
|
6226
6537
|
await syncPrompts(flags, profileOptions, context);
|
|
6227
6538
|
break;
|
|
@@ -6234,8 +6545,16 @@ async function runPromptsCommand(input, legacyFlags, legacyProfileOptions) {
|
|
|
6234
6545
|
case "propose":
|
|
6235
6546
|
await runPromptPropose(positionals, flags, profileOptions, context);
|
|
6236
6547
|
break;
|
|
6548
|
+
case "list":
|
|
6549
|
+
case "show":
|
|
6550
|
+
case "proposals":
|
|
6551
|
+
case "usage":
|
|
6552
|
+
case "replay-runs":
|
|
6553
|
+
case "decide":
|
|
6554
|
+
await runPromptRegistryRead(subcommand, positionals, flags, profileOptions, context);
|
|
6555
|
+
break;
|
|
6237
6556
|
default:
|
|
6238
|
-
throw new Error(`Unknown prompts command '${subcommand}'. Use init, status, diff, sync, promote, ab, or
|
|
6557
|
+
throw new Error(`Unknown prompts command '${subcommand}'. Use init, status, diff, sync, promote, ab, propose, list, show, proposals, usage, replay-runs, or decide.`);
|
|
6239
6558
|
}
|
|
6240
6559
|
}
|
|
6241
6560
|
async function initPrompts(options = {}) {
|
|
@@ -6320,7 +6639,7 @@ async function runPromptSyncExclusive(flags, profileOptions, options) {
|
|
|
6320
6639
|
const analyzedDefinitions = new Map;
|
|
6321
6640
|
if (flags["no-analyze"] !== "true") {
|
|
6322
6641
|
options.onProgress?.("Analyzing prompt sources before sync.");
|
|
6323
|
-
const { analyzeHarnessPrompts } = await import("./harness-
|
|
6642
|
+
const { analyzeHarnessPrompts } = await import("./harness-191xt7bq.js");
|
|
6324
6643
|
const result = await analyzeHarnessPrompts({
|
|
6325
6644
|
rootDir: process.cwd(),
|
|
6326
6645
|
existing: established.flatMap((prompt) => prompt.sourceDefinition ? [prompt.sourceDefinition.source] : []),
|
|
@@ -6800,6 +7119,80 @@ function sha256(input) {
|
|
|
6800
7119
|
function printJson(value) {
|
|
6801
7120
|
console.log(JSON.stringify(value, null, 2));
|
|
6802
7121
|
}
|
|
7122
|
+
var PROMPT_LABELS = ["prod", "staging", "dev"];
|
|
7123
|
+
async function runPromptRegistryRead(subcommand, positionals, flags, profileOptions, context) {
|
|
7124
|
+
const allowed = {
|
|
7125
|
+
list: ["label", "search"],
|
|
7126
|
+
show: [],
|
|
7127
|
+
proposals: [],
|
|
7128
|
+
usage: [],
|
|
7129
|
+
"replay-runs": [],
|
|
7130
|
+
decide: ["action", "dry-run"]
|
|
7131
|
+
};
|
|
7132
|
+
assertFlags(`prompts ${subcommand}`, flags, allowed[subcommand] ?? []);
|
|
7133
|
+
const maxPositionals = subcommand === "list" ? 1 : subcommand === "decide" ? 3 : 2;
|
|
7134
|
+
assertPositionals(`prompts ${subcommand}`, positionals, maxPositionals, `moda prompts ${subcommand}${subcommand === "list" ? "" : " <key>"}${subcommand === "decide" ? " <proposal_id> --action=merge|reject|reopen" : ""}`);
|
|
7135
|
+
const keyOrId = positionals[1];
|
|
7136
|
+
const needKey = (usage) => {
|
|
7137
|
+
if (!keyOrId)
|
|
7138
|
+
throw new CliInputError("<prompt key or id> is required.", `Usage: ${usage}`);
|
|
7139
|
+
return encodeURIComponent(keyOrId);
|
|
7140
|
+
};
|
|
7141
|
+
switch (subcommand) {
|
|
7142
|
+
case "list": {
|
|
7143
|
+
const label = flags.label;
|
|
7144
|
+
if (label && label !== "unlabeled" && !PROMPT_LABELS.includes(label)) {
|
|
7145
|
+
throw new CliInputError(`--label must be one of ${PROMPT_LABELS.join(", ")}, or unlabeled.`);
|
|
7146
|
+
}
|
|
7147
|
+
const response = asRecordValue(await callControlAPI("/prompts", {}, profileOptions));
|
|
7148
|
+
const all = Array.isArray(response.prompts) ? response.prompts.map(asRecordValue) : [];
|
|
7149
|
+
const search = flags.search?.toLowerCase();
|
|
7150
|
+
const prompts = all.filter((prompt) => {
|
|
7151
|
+
if (search && !`${prompt.key ?? ""} ${prompt.name ?? ""} ${prompt.description ?? ""}`.toLowerCase().includes(search))
|
|
7152
|
+
return false;
|
|
7153
|
+
if (!label)
|
|
7154
|
+
return true;
|
|
7155
|
+
const labelField = (name) => name === "dev" ? prompt.currentVersion ?? prompt.current_version : prompt[`${name}Version`] ?? prompt[`${name}_version`];
|
|
7156
|
+
const labels = PROMPT_LABELS.filter((name) => labelField(name));
|
|
7157
|
+
return label === "unlabeled" ? labels.length === 0 : labels.includes(label);
|
|
7158
|
+
});
|
|
7159
|
+
context.output.writeData({ prompts, total: prompts.length, ...label || search ? { filtered_from: all.length } : {} });
|
|
7160
|
+
return;
|
|
7161
|
+
}
|
|
7162
|
+
case "show":
|
|
7163
|
+
context.output.writeData(await callControlAPI(`/prompts/${needKey("moda prompts show <key>")}`, {}, profileOptions));
|
|
7164
|
+
return;
|
|
7165
|
+
case "proposals":
|
|
7166
|
+
context.output.writeData(await callControlAPI(`/prompts/${needKey("moda prompts proposals <key>")}/proposals`, {}, profileOptions));
|
|
7167
|
+
return;
|
|
7168
|
+
case "usage":
|
|
7169
|
+
context.output.writeData(await callControlAPI(`/prompts/${needKey("moda prompts usage <key>")}/usage`, {}, profileOptions));
|
|
7170
|
+
return;
|
|
7171
|
+
case "replay-runs":
|
|
7172
|
+
context.output.writeData(await callControlAPI(`/prompts/${needKey("moda prompts replay-runs <key>")}/replay-runs`, {}, profileOptions));
|
|
7173
|
+
return;
|
|
7174
|
+
case "decide": {
|
|
7175
|
+
const usage = "moda prompts decide <key> <proposal_id> --action=merge|reject|reopen";
|
|
7176
|
+
const id = needKey(usage);
|
|
7177
|
+
const proposalId = positionals[2];
|
|
7178
|
+
const action = flags.action;
|
|
7179
|
+
if (!proposalId)
|
|
7180
|
+
throw new CliInputError("<proposal_id> is required.", `Usage: ${usage}`);
|
|
7181
|
+
if (!action || !["merge", "reject", "reopen"].includes(action)) {
|
|
7182
|
+
throw new CliInputError("--action must be merge, reject, or reopen.", `Usage: ${usage}`);
|
|
7183
|
+
}
|
|
7184
|
+
if (context.dryRun) {
|
|
7185
|
+
context.output.writeData({ dryRun: true, planned: { method: "PATCH", endpoint: `/api/prompts/${id}/proposals/${encodeURIComponent(proposalId)}`, body: { action } } });
|
|
7186
|
+
return;
|
|
7187
|
+
}
|
|
7188
|
+
context.output.writeData(await callControlAPI(`/prompts/${id}/proposals/${encodeURIComponent(proposalId)}`, { method: "PATCH", body: JSON.stringify({ action }) }, profileOptions, { retries: 0 }));
|
|
7189
|
+
return;
|
|
7190
|
+
}
|
|
7191
|
+
}
|
|
7192
|
+
}
|
|
7193
|
+
function asRecordValue(value) {
|
|
7194
|
+
return value && typeof value === "object" && !Array.isArray(value) ? value : {};
|
|
7195
|
+
}
|
|
6803
7196
|
|
|
6804
7197
|
// src/skills.ts
|
|
6805
7198
|
import { execFileSync as execFileSync2 } from "node:child_process";
|
|
@@ -7339,6 +7732,38 @@ function escapeYamlString(value) {
|
|
|
7339
7732
|
function formatProposalRate(value) {
|
|
7340
7733
|
return `${(Number(value || 0) * 100).toFixed(1)}%`;
|
|
7341
7734
|
}
|
|
7735
|
+
var SKILLS_SUBCOMMAND_FLAGS = {
|
|
7736
|
+
gen: [
|
|
7737
|
+
"replay",
|
|
7738
|
+
"source",
|
|
7739
|
+
"wait",
|
|
7740
|
+
"clustering-poll-seconds",
|
|
7741
|
+
"clustering-run-id",
|
|
7742
|
+
"clustering-timeout-seconds",
|
|
7743
|
+
"end-at",
|
|
7744
|
+
"eval-set-id",
|
|
7745
|
+
"force-reprocess",
|
|
7746
|
+
"harness-url",
|
|
7747
|
+
"improve-candidates",
|
|
7748
|
+
"improvement-rounds",
|
|
7749
|
+
"lookback-hours",
|
|
7750
|
+
"max-cluster-pct",
|
|
7751
|
+
"max-sessions",
|
|
7752
|
+
"max-traces",
|
|
7753
|
+
"no-clustering",
|
|
7754
|
+
"no-replay",
|
|
7755
|
+
"replay-set-id",
|
|
7756
|
+
"reprocess-segments",
|
|
7757
|
+
"start-at",
|
|
7758
|
+
"tenant-id",
|
|
7759
|
+
"wait-for-completion"
|
|
7760
|
+
],
|
|
7761
|
+
status: ["run-id"],
|
|
7762
|
+
pull: ["status"],
|
|
7763
|
+
proposals: ["status", "generation-run-id", "tenant-id"],
|
|
7764
|
+
proposal: ["proposal-id", "tenant-id"],
|
|
7765
|
+
install: ["list"]
|
|
7766
|
+
};
|
|
7342
7767
|
async function runSkillsCommand(input) {
|
|
7343
7768
|
const context = "output" in input ? input : createCommandContext({
|
|
7344
7769
|
command: input.command,
|
|
@@ -7351,6 +7776,8 @@ async function runSkillsCommand(input) {
|
|
|
7351
7776
|
const sub = context.positional;
|
|
7352
7777
|
const positionals = context.positionals ?? (sub ? [sub] : []);
|
|
7353
7778
|
const flags = context.flags;
|
|
7779
|
+
if (sub && SKILLS_SUBCOMMAND_FLAGS[sub])
|
|
7780
|
+
assertFlags(`skills ${sub}`, flags, SKILLS_SUBCOMMAND_FLAGS[sub]);
|
|
7354
7781
|
if (sub === "gen") {
|
|
7355
7782
|
const result = await generateSkills(flags);
|
|
7356
7783
|
if (context.outputMode !== "human") {
|
|
@@ -7433,7 +7860,11 @@ async function runSkillsCommand(input) {
|
|
|
7433
7860
|
}
|
|
7434
7861
|
return 0;
|
|
7435
7862
|
}
|
|
7863
|
+
if (sub === "list" || sub === "show" || sub === "policy") {
|
|
7864
|
+
return runSkillRegistryCommand(sub, positionals, flags, context);
|
|
7865
|
+
}
|
|
7436
7866
|
if (sub === "sync") {
|
|
7867
|
+
assertFlags("skills sync", flags, ["dry-run"]);
|
|
7437
7868
|
const result = await runSkillSync(flags, { profile: context.profile, env: context.env });
|
|
7438
7869
|
context.output.writeData({
|
|
7439
7870
|
synced: result.synced,
|
|
@@ -7446,7 +7877,7 @@ async function runSkillsCommand(input) {
|
|
|
7446
7877
|
return runSkillsInstall(context, positionals.slice(1));
|
|
7447
7878
|
}
|
|
7448
7879
|
if (sub !== "pull") {
|
|
7449
|
-
throw new CliInputError(sub ? `Unknown skills subcommand '${sub}'` : "A skills subcommand is required", "Usage: moda skills gen [--source=all|sdk] [--max-sessions=N] [--start-at=ISO] [--end-at=ISO] [--replay] [--replay-set-id=ID] [--reprocess-segments] [--force-reprocess] [--wait] | moda skills status [run-id] | moda skills proposals list [--status=ready_for_pr] | moda skills proposal apply <proposal-id> | moda skills pull [--status=approved|proposed|all] | moda skills sync [--dry-run] | moda skills install [--list] <id...>");
|
|
7880
|
+
throw new CliInputError(sub ? `Unknown skills subcommand '${sub}'` : "A skills subcommand is required", "Usage: moda skills gen [--source=all|sdk] [--max-sessions=N (default 200)] [--lookback-hours=N (default 720)] [--max-traces=N] [--improvement-rounds=1-3] [--start-at=ISO] [--end-at=ISO] [--replay] [--replay-set-id=ID] [--reprocess-segments] [--force-reprocess] [--wait] | moda skills status [run-id] | moda skills proposals list [--status=ready_for_pr] | moda skills proposal apply <proposal-id> | moda skills pull [--status=approved|proposed|all] | moda skills sync [--dry-run] | moda skills install [--list] <id...>");
|
|
7450
7881
|
}
|
|
7451
7882
|
const status = (flags.status ?? "approved").toLowerCase();
|
|
7452
7883
|
if (!VALID_STATUSES.has(status)) {
|
|
@@ -7487,8 +7918,8 @@ async function generateSkills(flags) {
|
|
|
7487
7918
|
}
|
|
7488
7919
|
const body = compactObject({
|
|
7489
7920
|
tenant_id: tenantId || undefined,
|
|
7490
|
-
max_sessions:
|
|
7491
|
-
lookback_hours:
|
|
7921
|
+
max_sessions: parseScopeInt(flags["max-sessions"], "max-sessions"),
|
|
7922
|
+
lookback_hours: parseScopeInt(flags["lookback-hours"], "lookback-hours"),
|
|
7492
7923
|
start_at: flags["start-at"] || undefined,
|
|
7493
7924
|
end_at: flags["end-at"] || undefined,
|
|
7494
7925
|
clustering_run_id: flags["clustering-run-id"] || undefined,
|
|
@@ -7500,9 +7931,9 @@ async function generateSkills(flags) {
|
|
|
7500
7931
|
max_cluster_pct: parseOptionalFloat(flags["max-cluster-pct"]),
|
|
7501
7932
|
reprocess_segments: flags["reprocess-segments"] === "true" || forceReprocess,
|
|
7502
7933
|
run_replay: flags.replay === "true" && flags["no-replay"] !== "true",
|
|
7503
|
-
max_traces:
|
|
7934
|
+
max_traces: parseScopeInt(flags["max-traces"], "max-traces"),
|
|
7504
7935
|
improve_candidates: flags["improve-candidates"] === "true",
|
|
7505
|
-
improvement_rounds:
|
|
7936
|
+
improvement_rounds: parseScopeInt(flags["improvement-rounds"], "improvement-rounds", MAX_IMPROVEMENT_ROUNDS) ?? 2,
|
|
7506
7937
|
force_reprocess: forceReprocess,
|
|
7507
7938
|
wait_for_completion: waitForCompletion
|
|
7508
7939
|
});
|
|
@@ -7523,6 +7954,19 @@ async function generateSkills(flags) {
|
|
|
7523
7954
|
}
|
|
7524
7955
|
return await response.json();
|
|
7525
7956
|
}
|
|
7957
|
+
var MAX_IMPROVEMENT_ROUNDS = 3;
|
|
7958
|
+
function parseScopeInt(value, flag, max) {
|
|
7959
|
+
if (value === undefined)
|
|
7960
|
+
return;
|
|
7961
|
+
const parsed = /^\d+$/.test(value.trim()) ? Number.parseInt(value, 10) : Number.NaN;
|
|
7962
|
+
if (!Number.isFinite(parsed) || parsed < 1) {
|
|
7963
|
+
throw new CliInputError(`--${flag}=${value} is not a positive integer, so nothing was sent.`, `0 would mean unlimited server-side; omit --${flag} for the server default, or pass a number of 1 or more.`);
|
|
7964
|
+
}
|
|
7965
|
+
if (max !== undefined && parsed > max) {
|
|
7966
|
+
throw new CliInputError(`--${flag}=${value} is above the maximum of ${max}, so nothing was sent.`);
|
|
7967
|
+
}
|
|
7968
|
+
return parsed;
|
|
7969
|
+
}
|
|
7526
7970
|
function parsePositiveInt3(value, fallback) {
|
|
7527
7971
|
if (!value)
|
|
7528
7972
|
return fallback;
|
|
@@ -7617,10 +8061,43 @@ async function runSkillsInstall(context, ids) {
|
|
|
7617
8061
|
}
|
|
7618
8062
|
return 0;
|
|
7619
8063
|
}
|
|
8064
|
+
var RESERVED_SKILL_PATHS = new Set(["inbox", "library", "observed", "runs", "sync", "surface", "families", "proposals"]);
|
|
8065
|
+
async function runSkillRegistryCommand(sub, positionals, flags, context) {
|
|
8066
|
+
const profileOptions = { profile: context.profile, env: context.env };
|
|
8067
|
+
assertFlags(`skills ${sub}`, flags, sub === "policy" ? ["live", "local", "dry-run"] : []);
|
|
8068
|
+
assertPositionals(`skills ${sub}`, positionals, sub === "list" ? 1 : 2, sub === "list" ? "moda skills list" : `moda skills ${sub} <key>${sub === "policy" ? " --live|--local" : ""}`);
|
|
8069
|
+
if (sub === "list") {
|
|
8070
|
+
context.output.writeData(await callControlAPI("/skills", {}, profileOptions));
|
|
8071
|
+
return 0;
|
|
8072
|
+
}
|
|
8073
|
+
const idOrKey = positionals[1];
|
|
8074
|
+
if (!idOrKey) {
|
|
8075
|
+
throw new CliInputError(`A skill key or id is required for skills ${sub}.`, sub === "show" ? "Usage: moda skills show <key>" : "Usage: moda skills policy <key> --live|--local");
|
|
8076
|
+
}
|
|
8077
|
+
if (RESERVED_SKILL_PATHS.has(idOrKey)) {
|
|
8078
|
+
throw new CliInputError(`'${idOrKey}' is a reserved path, so this skill must be addressed by its id.`, "Find the id (skill_…) with `moda skills list`, then pass that instead of the key.");
|
|
8079
|
+
}
|
|
8080
|
+
if (sub === "show") {
|
|
8081
|
+
context.output.writeData(await callControlAPI(`/skills/${encodeURIComponent(idOrKey)}`, {}, profileOptions));
|
|
8082
|
+
return 0;
|
|
8083
|
+
}
|
|
8084
|
+
const live = flags.live === "true";
|
|
8085
|
+
const local = flags.local === "true";
|
|
8086
|
+
if (live === local) {
|
|
8087
|
+
throw new CliInputError("Pass exactly one of --live or --local.", "Usage: moda skills policy <key> --live|--local");
|
|
8088
|
+
}
|
|
8089
|
+
const surfacePolicy = live ? "include" : "local_only";
|
|
8090
|
+
if (context.dryRun) {
|
|
8091
|
+
context.output.writeData({ dryRun: true, planned: { method: "PATCH", endpoint: `/api/skills/${encodeURIComponent(idOrKey)}/surface-policy`, body: { surfacePolicy } } });
|
|
8092
|
+
return 0;
|
|
8093
|
+
}
|
|
8094
|
+
context.output.writeData(await callControlAPI(`/skills/${encodeURIComponent(idOrKey)}/surface-policy`, { method: "PATCH", body: JSON.stringify({ surfacePolicy }) }, profileOptions, { retries: 0 }));
|
|
8095
|
+
return 0;
|
|
8096
|
+
}
|
|
7620
8097
|
|
|
7621
8098
|
// src/doctor.ts
|
|
7622
8099
|
import { createHash as createHash3 } from "node:crypto";
|
|
7623
|
-
import { existsSync as existsSync5, readFileSync as readFileSync4, statSync } from "node:fs";
|
|
8100
|
+
import { existsSync as existsSync5, readFileSync as readFileSync4, statSync as statSync2 } from "node:fs";
|
|
7624
8101
|
import { join as join4 } from "node:path";
|
|
7625
8102
|
async function runDoctorCommand(context) {
|
|
7626
8103
|
const report = await buildDoctorReport(context, {
|