@withgauge/cli 0.11.0 → 0.11.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/README.md +5 -1
  2. package/dist/index.js +106 -124
  3. package/package.json +4 -2
package/README.md CHANGED
@@ -98,6 +98,10 @@ immediate launches. They use the stored configuration and ask before spending
98
98
  (`--yes` skips the prompt for scripts).
99
99
  `gauge preference run <id> --scenario <scenarioId>` (repeatable) runs a subset;
100
100
  omit it and every scenario runs. Neither launch touches the cadence clock.
101
+ Superusers can use `--org <slug>` to launch in any org. Their runs use PLATFORM
102
+ funding by default, like Run Now in the web UI; `--bill-to-org` uses that org's
103
+ credits instead. Other users always use org credits and can launch only in orgs
104
+ they belong to.
101
105
 
102
106
  ## Optimizations: Gauge measures, you author
103
107
 
@@ -125,7 +129,7 @@ gauge optimizations trials add cl_1 \
125
129
  gauge optimizations watch cl_1
126
130
  gauge optimizations trial cl_1 R1 # score, verdict, sessions
127
131
  gauge optimizations change cl_1 R1 --body # the winning text, to paste
128
- gauge optimizations adopt cl_1 R1 --yes # the new moving baseline
132
+ gauge optimizations adopt cl_1 R1 --yes # mark the winner; baseline stays fixed
129
133
  gauge optimizations complete cl_1 --yes
130
134
  ```
131
135
 
package/dist/index.js CHANGED
@@ -6,7 +6,7 @@ var __export = (target, all) => {
6
6
  };
7
7
 
8
8
  // src/index.ts
9
- import { Command, CommanderError, Option as Option2 } from "commander";
9
+ import { Command, CommanderError, Option as Option4 } from "commander";
10
10
 
11
11
  // src/commands/actions.ts
12
12
  var actions_exports2 = {};
@@ -1576,6 +1576,11 @@ var cancelRunResponseSchema = z17.object({
1576
1576
  cancelRequested: z17.boolean(),
1577
1577
  note: z17.string().optional()
1578
1578
  });
1579
+ var regradeRunResponseSchema = z17.object({
1580
+ runId: z17.string(),
1581
+ evalSetId: z17.string(),
1582
+ criteria: z17.number().int()
1583
+ });
1579
1584
  var forkRunBodySchema = z17.object({
1580
1585
  prompt: z17.string().trim().min(1).max(2e4),
1581
1586
  turn: z17.number().int().min(0).optional(),
@@ -1718,6 +1723,7 @@ function totalRunsPerCycle(owners, samplesPerCycle) {
1718
1723
 
1719
1724
  // ../packages/api-schemas/src/ownedConfigurations.ts
1720
1725
  var sampleCountSchema = z19.number().int().min(1).max(50);
1726
+ var nextRunAtInputSchema = z19.string().regex(/^\d{4}-\d{2}-\d{2}/, "Next run must be a calendar date").nullable();
1721
1727
  var ownedCadenceSchema = z19.enum([
1722
1728
  "NONE",
1723
1729
  "DAILY",
@@ -1733,6 +1739,7 @@ var ownedConfigurationInputShape = {
1733
1739
  mcpRefs: assetRefListSchema.optional().default([]),
1734
1740
  connectionIds: z19.array(z19.string().min(1)).max(32).optional().default([]),
1735
1741
  cadence: ownedCadenceSchema.optional().default("NONE"),
1742
+ nextRunAt: nextRunAtInputSchema.optional(),
1736
1743
  sampleCount: sampleCountSchema.optional().default(1)
1737
1744
  };
1738
1745
  var ownedConfigurationInputSchema = z19.object(ownedConfigurationInputShape).strict();
@@ -1745,6 +1752,7 @@ var ownedConfigurationPatchShape = {
1745
1752
  mcpRefs: assetRefListSchema.optional(),
1746
1753
  connectionIds: z19.array(z19.string().min(1)).max(32).optional(),
1747
1754
  cadence: ownedCadenceSchema.optional(),
1755
+ nextRunAt: nextRunAtInputSchema.optional(),
1748
1756
  sampleCount: sampleCountSchema.optional()
1749
1757
  };
1750
1758
  var ownedConfigurationSchema = z19.object({
@@ -1757,7 +1765,8 @@ var ownedConfigurationSchema = z19.object({
1757
1765
  connectionIds: z19.array(z19.string()),
1758
1766
  cadence: ownedCadenceSchema,
1759
1767
  sampleCount: sampleCountSchema,
1760
- lastScheduledAt: z19.string().nullable()
1768
+ lastScheduledAt: z19.string().nullable(),
1769
+ nextRunAt: z19.string().nullable()
1761
1770
  });
1762
1771
  var visibilityScenarioSchema = z19.object({
1763
1772
  id: z19.string(),
@@ -1776,6 +1785,7 @@ var visibilityPromptSettingsInputShape = {
1776
1785
  mcpRefs: ownedConfigurationInputShape.mcpRefs,
1777
1786
  connectionIds: ownedConfigurationInputShape.connectionIds,
1778
1787
  cadence: ownedConfigurationInputShape.cadence,
1788
+ nextRunAt: ownedConfigurationInputShape.nextRunAt,
1779
1789
  sampleCount: ownedConfigurationInputShape.sampleCount
1780
1790
  };
1781
1791
  var visibilityPromptSettingsPatchShape = {
@@ -1784,6 +1794,7 @@ var visibilityPromptSettingsPatchShape = {
1784
1794
  mcpRefs: ownedConfigurationPatchShape.mcpRefs,
1785
1795
  connectionIds: ownedConfigurationPatchShape.connectionIds,
1786
1796
  cadence: ownedConfigurationPatchShape.cadence,
1797
+ nextRunAt: ownedConfigurationPatchShape.nextRunAt,
1787
1798
  sampleCount: ownedConfigurationPatchShape.sampleCount
1788
1799
  };
1789
1800
  var visibilityPromptSettingsSchema = z19.object({
@@ -1793,11 +1804,13 @@ var visibilityPromptSettingsSchema = z19.object({
1793
1804
  connectionIds: z19.array(z19.string()),
1794
1805
  cadence: ownedCadenceSchema,
1795
1806
  sampleCount: sampleCountSchema,
1796
- lastScheduledAt: z19.string().nullable()
1807
+ lastScheduledAt: z19.string().nullable(),
1808
+ nextRunAt: z19.string().nullable()
1797
1809
  });
1798
1810
  var runAgentsOverrideSchema = z19.array(agentConfigInputSchema).min(1, "Select at least one agent");
1799
- var runEvalBodySchema = z19.object({}).strict();
1811
+ var runEvalBodySchema = z19.object({ billToOrg: z19.boolean().optional() }).strict();
1800
1812
  var runVisibilityBodySchema = z19.object({
1813
+ billToOrg: z19.boolean().optional(),
1801
1814
  visibilityScenarioIds: z19.array(z19.string().min(1)).min(1).optional()
1802
1815
  }).strict();
1803
1816
 
@@ -2690,8 +2703,6 @@ var climbSettingsSchema = z29.object({
2690
2703
  maxRounds: z29.number().int().positive().default(5),
2691
2704
  sampleCount: z29.number().int().positive().default(1),
2692
2705
  creditCap: z29.number().int().nonnegative(),
2693
- exploreModel: z29.string().min(1).optional(),
2694
- autoAdopt: z29.boolean().default(false),
2695
2706
  targetScore: z29.number().min(0).max(1).default(1),
2696
2707
  // Narrow the subject's whole roster to one slot, for the baseline and every
2697
2708
  // trial alike. A subject rostering ten models otherwise multiplies every
@@ -2699,11 +2710,11 @@ var climbSettingsSchema = z29.object({
2699
2710
  // is not a comparison. `pinnedModel` empty means the agent's default model.
2700
2711
  pinnedAgent: schedulableAgentSchema.optional(),
2701
2712
  pinnedModel: z29.string().optional(),
2702
- maxTrialsPerRound: z29.number().int().min(1).max(3).default(3),
2713
+ maxTrialsPerRound: z29.number().int().min(1).max(10).default(3),
2703
2714
  // The round a continuation started at; the no-improvement counter
2704
2715
  // restarts here.
2705
2716
  continuedAtRound: z29.number().int().nonnegative().optional()
2706
- }).strict().refine((value) => value.pinnedAgent !== void 0 || !value.pinnedModel, {
2717
+ }).strip().refine((value) => value.pinnedAgent !== void 0 || !value.pinnedModel, {
2707
2718
  path: ["pinnedAgent"],
2708
2719
  message: "Pin an agent with the model"
2709
2720
  });
@@ -2727,7 +2738,6 @@ var CLIMB_ACTIVITY_KINDS = [
2727
2738
  "EXCLUDED",
2728
2739
  "LAUNCHED",
2729
2740
  "SETTLED",
2730
- "CONFIRMING",
2731
2741
  "DONE",
2732
2742
  "FAILED",
2733
2743
  "DIRECTION",
@@ -2882,9 +2892,9 @@ var optimizationSummarySchema = z30.object({
2882
2892
  maxRounds: z30.number().int(),
2883
2893
  /** Starting baseline score, 0..1. */
2884
2894
  before: z30.number().nullable(),
2885
- /** Score of the adopted tip, 0..1. */
2895
+ /** Score of the marked winner, or best BETTER trial, 0..1. */
2886
2896
  after: z30.number().nullable(),
2887
- /** Percentage-point change from the starting baseline to the adopted tip. */
2897
+ /** Percentage-point change from the baseline to the winner or best BETTER trial. */
2888
2898
  delta: z30.number().nullable(),
2889
2899
  /** Score after each round, baseline first. */
2890
2900
  path: z30.array(z30.number()),
@@ -2907,7 +2917,7 @@ var planShape = {
2907
2917
  effort: optimizationEffortSchema.optional(),
2908
2918
  sampleCount: z30.number().int().min(1).max(100).optional(),
2909
2919
  maxRounds: z30.number().int().min(1).max(100).optional(),
2910
- maxTrialsPerRound: z30.number().int().min(1).max(3).optional()
2920
+ maxTrialsPerRound: z30.number().int().min(1).max(10).optional()
2911
2921
  };
2912
2922
  function refinePlan(value, ctx) {
2913
2923
  if (value.effort && (value.sampleCount || value.maxRounds))
@@ -2929,7 +2939,6 @@ var startOptimizationBodySchema = z30.object({
2929
2939
  name: z30.string().trim().min(1).max(200).optional(),
2930
2940
  ...planShape,
2931
2941
  creditCap: z30.number().int().nonnegative().optional(),
2932
- autoAdopt: z30.boolean().default(true),
2933
2942
  targetScore: z30.number().min(0).max(1).optional(),
2934
2943
  target: targetSchema.extend({ agent: schedulableAgentSchema })
2935
2944
  }).strict().superRefine(refinePlan);
@@ -3010,7 +3019,9 @@ var trialSchema = z30.object({
3010
3019
  climbId: z30.string(),
3011
3020
  roundNumber: z30.number().int(),
3012
3021
  key: z30.string(),
3022
+ /** @deprecated All trials branch from the baseline. */
3013
3023
  parentTrialId: z30.string().nullable(),
3024
+ /** @deprecated Always the baseline key for non-baseline trials. */
3014
3025
  parentKey: z30.string().nullable(),
3015
3026
  title: z30.string(),
3016
3027
  hypothesis: z30.string().nullable(),
@@ -3022,8 +3033,8 @@ var trialSchema = z30.object({
3022
3033
  baselineScore: z30.number().nullable(),
3023
3034
  /** Judged sessions behind `score`. */
3024
3035
  sessions: z30.number().int(),
3025
- explore: z30.boolean(),
3026
3036
  isBaseline: z30.boolean(),
3037
+ /** The user marked this trial as the winner; it does not move the baseline. */
3027
3038
  adopted: z30.boolean(),
3028
3039
  adoptedAt: z30.string().nullable(),
3029
3040
  outcome: z30.string().nullable(),
@@ -3059,9 +3070,7 @@ var optimizationDetailSchema = optimizationSummarySchema.extend({
3059
3070
  maxRounds: z30.number().int(),
3060
3071
  maxTrialsPerRound: z30.number().int(),
3061
3072
  creditCap: z30.number().int(),
3062
- autoAdopt: z30.boolean(),
3063
3073
  targetScore: z30.number(),
3064
- exploreModel: z30.string().nullable(),
3065
3074
  pinnedAgent: z30.string().nullable(),
3066
3075
  pinnedModel: z30.string().nullable(),
3067
3076
  continuedAtRound: z30.number().int().nullable()
@@ -3138,9 +3147,8 @@ var resolveChatResponseSchema = z30.object({
3138
3147
  href: z30.string()
3139
3148
  });
3140
3149
  var continueOptimizationBodySchema = z30.object({
3141
- fromTrialId: z30.string().min(1),
3142
3150
  rounds: z30.number().int().min(1).max(20).default(1),
3143
- maxTrialsPerRound: z30.number().int().min(1).max(3).optional()
3151
+ maxTrialsPerRound: z30.number().int().min(1).max(10).optional()
3144
3152
  }).strict();
3145
3153
  var patchOptimizationBodySchema = z30.object({
3146
3154
  ...planPatchShape,
@@ -5182,6 +5190,7 @@ __export(evals_exports2, {
5182
5190
  register: () => register11
5183
5191
  });
5184
5192
  import { readFileSync as readFileSync6 } from "fs";
5193
+ import { Option as Option2 } from "commander";
5185
5194
 
5186
5195
  // src/lib/confirm.ts
5187
5196
  import { createInterface as createInterface2 } from "readline";
@@ -5203,6 +5212,11 @@ async function confirmOrAbort(question, yes) {
5203
5212
  }
5204
5213
  }
5205
5214
 
5215
+ // src/lib/staff.ts
5216
+ function staffCommandsVisible(config = readConfig()) {
5217
+ return config.staff === true;
5218
+ }
5219
+
5206
5220
  // src/lib/ownedConfig.ts
5207
5221
  function parseCadence(raw) {
5208
5222
  const cadence = (raw ?? "NONE").trim().toUpperCase();
@@ -5590,23 +5604,30 @@ function register11(program2) {
5590
5604
  if (resolveOutput(cmd) !== "json") console.log(`Deleted eval set ${id}`);
5591
5605
  });
5592
5606
  group.command("run <id>").description(
5593
- "Run an eval set now with its stored configuration (ORG funding, credit-gated)"
5594
- ).option("--yes", "skip the confirmation").action(async (id, opts, cmd) => {
5595
- const org = resolveOrg(cmd);
5596
- await confirmOrAbort(
5597
- "Run this eval's stored configuration now?",
5598
- opts.yes
5599
- );
5600
- const result = await createClient().post(
5601
- `${evalSetsPath(org)}/${encodeURIComponent(id)}/run`,
5602
- {}
5603
- );
5604
- if (resolveOutput(cmd) === "json") printJson(result);
5605
- else
5606
- console.log(
5607
- `Started ${result.runIds.length} run${result.runIds.length === 1 ? "" : "s"} in batch ${result.batchIds[0] ?? ""}`
5607
+ staffCommandsVisible() ? "Run an eval set now with its stored configuration (superusers run without billing the org by default)" : "Run an eval set now with its stored configuration"
5608
+ ).addOption(
5609
+ new Option2(
5610
+ "--bill-to-org",
5611
+ "bill this run to the org's credits instead (superusers)"
5612
+ ).hideHelp(!staffCommandsVisible())
5613
+ ).option("--yes", "skip the confirmation").action(
5614
+ async (id, opts, cmd) => {
5615
+ const org = resolveOrg(cmd);
5616
+ await confirmOrAbort(
5617
+ "Run this eval's stored configuration now?",
5618
+ opts.yes
5608
5619
  );
5609
- });
5620
+ const result = await createClient().post(
5621
+ `${evalSetsPath(org)}/${encodeURIComponent(id)}/run`,
5622
+ opts.billToOrg ? { billToOrg: true } : {}
5623
+ );
5624
+ if (resolveOutput(cmd) === "json") printJson(result);
5625
+ else
5626
+ console.log(
5627
+ `Started ${result.runIds.length} run${result.runIds.length === 1 ? "" : "s"} in batch ${result.batchIds[0] ?? ""}`
5628
+ );
5629
+ }
5630
+ );
5610
5631
  group.command("runs <id>").description("The eval set's run log with per-criterion verdicts").option("--limit <n>", "max runs (default 100)").action(async (id, opts, cmd) => {
5611
5632
  const org = resolveOrg(cmd);
5612
5633
  const res = await createClient().get(
@@ -5723,7 +5744,7 @@ function printDetail3(e) {
5723
5744
  }
5724
5745
  function register12(program2) {
5725
5746
  const group = program2.command("experiments").description(
5726
- "Author, launch, and read Fixer content experiments (legacy: orgs with Optimizations answer 409 on writes; use `gauge optimizations`)"
5747
+ "Author, launch, and read content experiments (legacy: orgs with Optimizations reject writes; use `gauge optimizations`)"
5727
5748
  );
5728
5749
  group.command("list").description("List the org's experiments").option(
5729
5750
  "--status <statuses>",
@@ -5769,13 +5790,13 @@ function register12(program2) {
5769
5790
  );
5770
5791
  });
5771
5792
  group.command("create").description(
5772
- "Open (or resume) an experiment draft and start the Fixer plan round"
5793
+ "Open (or resume) an experiment draft and start the plan round"
5773
5794
  ).requiredOption("--run <runId>", "source run id").requiredOption(
5774
5795
  "--boundary <seq>",
5775
5796
  "opaque boundarySeq from `experiments boundaries`"
5776
5797
  ).option("--problem <text>", "problem statement (derived if omitted)").option(
5777
5798
  "--mode <mode>",
5778
- "who writes the rewrites: propose (default, Fixer proposes directions then writes each), direct (one --change instruction, one write), manual (no generation \u2014 the rewrite is seeded with the page the agent read)",
5799
+ "who writes the rewrites: propose (default, directions are proposed then each is written), direct (one --change instruction, one write), manual (no generation \u2014 the rewrite is seeded with the page the agent read)",
5779
5800
  "propose"
5780
5801
  ).option(
5781
5802
  "--change <text>",
@@ -5959,13 +5980,13 @@ function register12(program2) {
5959
5980
  }
5960
5981
  }
5961
5982
  );
5962
- group.command("replan <id>").description("Discard the directions and have the Fixer propose fresh ones").action(async (id, _opts, cmd) => {
5983
+ group.command("replan <id>").description("Discard the directions and propose fresh ones").action(async (id, _opts, cmd) => {
5963
5984
  const org = resolveOrg(cmd);
5964
5985
  await createClient().post(`${itemPath(org, id)}/replan`, {});
5965
5986
  if (resolveOutput(cmd) !== "json")
5966
5987
  console.log("Replanning \u2014 follow with `gauge experiments watch`");
5967
5988
  });
5968
- group.command("write <id>").description("Have the Fixer write rewrites for directions that lack one").action(async (id, _opts, cmd) => {
5989
+ group.command("write <id>").description("Write rewrites for directions that lack one").action(async (id, _opts, cmd) => {
5969
5990
  const org = resolveOrg(cmd);
5970
5991
  const result = await createClient().post(
5971
5992
  `${itemPath(org, id)}/write`,
@@ -6411,7 +6432,7 @@ function register15(program2) {
6411
6432
  });
6412
6433
  group.command("add").description(
6413
6434
  "Add a server, or append a version to an existing --name (versions are immutable)"
6414
- ).requiredOption("--name <handle>", "our handle (the agent config key)").option("--command <cmd>", "stdio launch command, e.g. npx").option(
6435
+ ).requiredOption("--name <handle>", "the handle (the agent config key)").option("--command <cmd>", "stdio launch command, e.g. npx").option(
6415
6436
  "--arg <arg>",
6416
6437
  "stdio argument (repeatable)",
6417
6438
  collect,
@@ -6484,7 +6505,7 @@ function register15(program2) {
6484
6505
  }
6485
6506
  );
6486
6507
  group.command("rm <ref>").description(
6487
- "Soft-delete a server (handle stays reserved; finished runs keep their snapshot)"
6508
+ "Remove a server; the handle stays reserved and finished runs keep their snapshot"
6488
6509
  ).action(async (ref, _opts, cmd) => {
6489
6510
  const org = resolveOrg(cmd);
6490
6511
  const result = await createClient().delete(serverPath(org, ref));
@@ -6883,7 +6904,6 @@ function trialBodyFromOpts(opts, read = readText2, walk) {
6883
6904
  }
6884
6905
  if (!opts.title) throw new CliError("usage", "--title is required");
6885
6906
  const lever = opts.lever?.toUpperCase();
6886
- const kind = opts.kind?.toUpperCase();
6887
6907
  let intervention;
6888
6908
  let files;
6889
6909
  if (opts.page || opts.bodyFile) {
@@ -6893,12 +6913,6 @@ function trialBodyFromOpts(opts, read = readText2, walk) {
6893
6913
  kind: "REWRITE_DOCS",
6894
6914
  after: { pages: [{ url: opts.page, body: read(opts.bodyFile) }] }
6895
6915
  };
6896
- } else if (opts.removeSkill) {
6897
- intervention = {
6898
- kind: "REMOVE_SKILL",
6899
- before: { skillVersionId: opts.removeSkill },
6900
- after: {}
6901
- };
6902
6916
  } else if (opts.editSkill) {
6903
6917
  if (!opts.skillDir)
6904
6918
  throw new CliError(
@@ -6911,21 +6925,10 @@ function trialBodyFromOpts(opts, read = readText2, walk) {
6911
6925
  after: { skillVersionId: opts.editSkill }
6912
6926
  };
6913
6927
  files = readSkillDir(opts.skillDir, read, walk);
6914
- } else if (opts.skillVersion) {
6915
- if (opts.skillDir)
6916
- throw new CliError(
6917
- "usage",
6918
- "--skill-dir edits a skill: pass --edit-skill <id>, not --skill-version"
6919
- );
6920
- intervention = {
6921
- kind: kind ?? "ADD_SKILL",
6922
- after: { skillVersionId: opts.skillVersion },
6923
- ...kind === "EDIT_SKILL" ? { before: { skillVersionId: opts.skillVersion } } : {}
6924
- };
6925
6928
  } else {
6926
6929
  throw new CliError(
6927
6930
  "usage",
6928
- "describe the change: --skill-version <id>, --edit-skill <id> --skill-dir <path>, --remove-skill <id>, --page <url> --body-file <path>, or --file <trial.json>"
6931
+ "describe the change: --edit-skill <id> --skill-dir <path>, --page <url> --body-file <path>, or --file <trial.json>"
6929
6932
  );
6930
6933
  }
6931
6934
  const trial = {
@@ -6944,18 +6947,6 @@ function trialBodyFromOpts(opts, read = readText2, walk) {
6944
6947
  };
6945
6948
  }
6946
6949
  var TRIAL_TEMPLATES = {
6947
- ADD_SKILL: {
6948
- trial: {
6949
- title: "Add the troubleshooting skill",
6950
- hypothesis: "The agent misses the retry rule without it",
6951
- lever: "SKILL",
6952
- intervention: {
6953
- kind: "ADD_SKILL",
6954
- after: { skillVersionId: "sv_REPLACE_ME" }
6955
- },
6956
- authorMode: "MANUAL"
6957
- }
6958
- },
6959
6950
  EDIT_SKILL: {
6960
6951
  trial: {
6961
6952
  title: "Name the env var in the first line",
@@ -6972,19 +6963,6 @@ var TRIAL_TEMPLATES = {
6972
6963
  { path: "SKILL.md", body: "# Skill\n\nThe complete rewritten file.\n" }
6973
6964
  ]
6974
6965
  },
6975
- REMOVE_SKILL: {
6976
- trial: {
6977
- title: "Drop the legacy skill",
6978
- hypothesis: "It contradicts the quickstart",
6979
- lever: "SKILL",
6980
- intervention: {
6981
- kind: "REMOVE_SKILL",
6982
- before: { skillVersionId: "sv_REPLACE_ME" },
6983
- after: {}
6984
- },
6985
- authorMode: "MANUAL"
6986
- }
6987
- },
6988
6966
  REWRITE_DOCS: {
6989
6967
  trial: {
6990
6968
  title: "Move the auth note above the example",
@@ -7041,12 +7019,11 @@ function trialRow(t) {
7041
7019
  title: t.title.length > 48 ? `${t.title.slice(0, 45)}...` : t.title,
7042
7020
  surface: t.lever ?? "",
7043
7021
  change: t.kind,
7044
- "builds on": t.parentKey ?? "",
7045
7022
  status: t.status,
7046
7023
  score: pct(t.score),
7047
7024
  result: t.verdict ?? "",
7048
7025
  sessions: t.sessions,
7049
- adopted: t.adopted ? "yes" : "",
7026
+ winner: t.adopted ? "yes" : "",
7050
7027
  id: t.id
7051
7028
  };
7052
7029
  }
@@ -7066,7 +7043,7 @@ function printDetail4(o) {
7066
7043
  `target: ${o.settings.pinnedAgent ?? "(subject roster)"}${o.settings.pinnedModel ? `:${o.settings.pinnedModel}` : ""}`
7067
7044
  );
7068
7045
  console.log(
7069
- `plan: ${o.settings.sampleCount} session${o.settings.sampleCount === 1 ? "" : "s"} per trial, up to ${o.settings.maxRounds} rounds, auto-adopt ${o.settings.autoAdopt ? "on" : "off"}`
7046
+ `plan: ${o.settings.sampleCount} session${o.settings.sampleCount === 1 ? "" : "s"} per trial, up to ${o.settings.maxRounds} rounds`
7070
7047
  );
7071
7048
  console.log(
7072
7049
  `credits: ${o.creditsSpent} spent; ${o.settings.creditCap} estimated (not a spending cap)`
@@ -7089,9 +7066,8 @@ function printTrial(t) {
7089
7066
  console.log(`title: ${t.title}`);
7090
7067
  if (t.hypothesis) console.log(`hypothesis: ${t.hypothesis}`);
7091
7068
  console.log(`change: ${t.kind}${t.lever ? ` on ${t.lever}` : ""}`);
7092
- console.log(`builds on: ${t.parentKey ?? "\u2014"}`);
7093
7069
  console.log(
7094
- `status: ${t.status}${t.verdict ? ` \xB7 ${t.verdict}` : ""}${t.adopted ? " \xB7 adopted" : ""}`
7070
+ `status: ${t.status}${t.verdict ? ` \xB7 ${t.verdict}` : ""}${t.adopted ? " \xB7 winner" : ""}`
7095
7071
  );
7096
7072
  console.log(
7097
7073
  `score: ${pct(t.baselineScore)} \u2192 ${pct(t.score)} over ${t.sessions} session${t.sessions === 1 ? "" : "s"}`
@@ -7165,7 +7141,7 @@ function addPlanOptions(command) {
7165
7141
  }
7166
7142
  function register19(program2) {
7167
7143
  const group = program2.command("optimizations").alias("optimize").description(
7168
- "Start, steer, and read optimizations: Gauge changes one owned surface per trial and adopts what improves an eval or preference prompt"
7144
+ "Start, steer, and read optimizations: Gauge changes one owned surface per trial and reports the best measured change against the baseline"
7169
7145
  );
7170
7146
  group.command("list").description("List the org's optimizations, newest activity first").option(
7171
7147
  "--status <statuses>",
@@ -7226,10 +7202,10 @@ function register19(program2) {
7226
7202
  "Measure a baseline; author and explicitly launch candidate rounds afterward"
7227
7203
  )
7228
7204
  )
7229
- ).option("--name <name>", "climb name (default: from the subject)").option(
7205
+ ).option("--name <name>", "optimization name (default: from the subject)").option(
7230
7206
  "--credit-cap <n>",
7231
7207
  "record this credit estimate instead (not an enforced cap)"
7232
- ).option("--target-score <0-1>", "stop once the score reaches this").option("--no-auto-adopt", "keep improvements pending until you adopt them").option("--yes", "skip the confirmation").action(
7208
+ ).option("--target-score <0-1>", "stop once the score reaches this").option("--yes", "skip the confirmation").action(
7233
7209
  async (opts, cmd) => {
7234
7210
  const org = resolveOrg(cmd);
7235
7211
  if (!opts.target)
@@ -7258,7 +7234,6 @@ function register19(program2) {
7258
7234
  ...plan,
7259
7235
  name: opts.name,
7260
7236
  creditCap,
7261
- autoAdopt: opts.autoAdopt,
7262
7237
  targetScore
7263
7238
  });
7264
7239
  if (resolveOutput(cmd) === "json") printJson(detail);
@@ -7392,9 +7367,8 @@ function register19(program2) {
7392
7367
  result.alreadyLaunched ? "Round already launched." : `Launched ${result.launched.length} trials in round ${result.roundNumber}.`
7393
7368
  );
7394
7369
  });
7395
- group.command("continue <id>").description("Continue a finished optimization from a settled trial").requiredOption(
7396
- "--from <trialId>",
7397
- "starting trial id or key, e.g. B0 or R1.1"
7370
+ group.command("continue <id>").description(
7371
+ "Add rounds to a finished optimization against its fixed baseline"
7398
7372
  ).option("--rounds <n>", "additional rounds (1\u201320, default 1)").option("--trials-per-round <n>", "maximum candidates per round (1\u20133)").option("--yes", "skip the confirmation").action(
7399
7373
  async (id, opts, cmd) => {
7400
7374
  const rounds = parseCount(opts.rounds, "--rounds", 20) ?? 1;
@@ -7404,12 +7378,12 @@ function register19(program2) {
7404
7378
  3
7405
7379
  );
7406
7380
  await confirmOrAbort(
7407
- `Continue ${id} from ${opts.from} for up to ${rounds} additional rounds? Saved trials still require launch.`,
7381
+ `Continue ${id} for up to ${rounds} additional rounds? Saved trials still require launch.`,
7408
7382
  opts.yes
7409
7383
  );
7410
7384
  const detail = await createClient().post(
7411
7385
  `${itemPath2(resolveOrg(cmd), id)}/continue`,
7412
- { fromTrialId: opts.from, rounds, maxTrialsPerRound }
7386
+ { rounds, maxTrialsPerRound }
7413
7387
  );
7414
7388
  if (resolveOutput(cmd) === "json") printJson(detail);
7415
7389
  else printDetail4(detail);
@@ -7461,7 +7435,7 @@ function register19(program2) {
7461
7435
  if (resolveOutput(cmd) === "json") printJson({ deleted: id });
7462
7436
  else console.log(`Deleted ${id}.`);
7463
7437
  });
7464
- group.command("complete <id>").description("Complete an optimization: it reads as Finished, not canceled").option("--adopt <trialId>", "adopt this settled trial as the winner first").option("--yes", "skip the confirmation").action(
7438
+ group.command("complete <id>").description("Complete an optimization: it reads as Finished, not canceled").option("--adopt <trialId>", "mark this settled trial as the winner").option("--yes", "skip the confirmation").action(
7465
7439
  async (id, opts, cmd) => {
7466
7440
  const org = resolveOrg(cmd);
7467
7441
  await confirmOrAbort(
@@ -7525,15 +7499,12 @@ function register19(program2) {
7525
7499
  "--file <path>",
7526
7500
  "authored trial JSON, as `trials template` prints it: one {trial, files?} or an array; '-' for stdin"
7527
7501
  ).option("--pick <n>", "which trial to take from an array (default 1)").option("--title <text>", "what the change is").option("--hypothesis <text>", "why it should help").option("--lever <lever>", "DOCS, MARKETING_SITE, SKILL, SDK, CLI").option(
7528
- "--kind <kind>",
7529
- "override the change kind the other flags imply; rarely needed"
7530
- ).option("--skill-version <id>", "skill version to add").option(
7531
7502
  "--edit-skill <id>",
7532
7503
  "skill version to rewrite (with --skill-dir); see `gauge optimizations skill-files <id>`"
7533
7504
  ).option(
7534
7505
  "--skill-dir <path>",
7535
7506
  "local directory holding the complete edited bundle, SKILL.md included"
7536
- ).option("--remove-skill <id>", "skill version to remove from the baseline").option("--page <url>", "docs page to rewrite (with --body-file)").option("--body-file <path>", "the rewritten page body; '-' for stdin").option("--fork-run <runId>", "fork-at-fetch source run for a docs rewrite").option(
7507
+ ).option("--page <url>", "docs page to rewrite (with --body-file)").option("--body-file <path>", "the rewritten page body; '-' for stdin").option("--fork-run <runId>", "fork-at-fetch source run for a docs rewrite").option(
7537
7508
  "--fork-boundary <seq>",
7538
7509
  "fork-at-fetch boundary from `gauge experiments boundaries`"
7539
7510
  ).option(
@@ -7563,10 +7534,7 @@ function register19(program2) {
7563
7534
  );
7564
7535
  }
7565
7536
  );
7566
- trials.command("template").description("Print a trial JSON skeleton for `trials add --file`").requiredOption(
7567
- "--kind <kind>",
7568
- "ADD_SKILL, EDIT_SKILL, REMOVE_SKILL, or REWRITE_DOCS"
7569
- ).action((opts) => {
7537
+ trials.command("template").description("Print a trial JSON skeleton for `trials add --file`").requiredOption("--kind <kind>", "EDIT_SKILL or REWRITE_DOCS").action((opts) => {
7570
7538
  printJson(trialTemplate(opts.kind));
7571
7539
  });
7572
7540
  trials.command("run <id> <trialId>").description("Run a draft trial now (spends credits)").option("--yes", "skip the confirmation").action(
@@ -7615,12 +7583,12 @@ function register19(program2) {
7615
7583
  }
7616
7584
  );
7617
7585
  group.command("adopt <id> <trialId>").description(
7618
- "Make a settled trial the moving baseline (the eval itself is unchanged)"
7586
+ "Mark a settled trial as the winner without changing the baseline"
7619
7587
  ).option("--yes", "skip the confirmation").action(
7620
7588
  async (id, trialId, opts, cmd) => {
7621
7589
  const org = resolveOrg(cmd);
7622
7590
  await confirmOrAbort(
7623
- `Adopt ${trialId} on ${id}? Later trials build on it.`,
7591
+ `Mark ${trialId} as the winner on ${id}?`,
7624
7592
  opts.yes
7625
7593
  );
7626
7594
  const detail = await createClient().post(
@@ -7628,14 +7596,11 @@ function register19(program2) {
7628
7596
  {}
7629
7597
  );
7630
7598
  if (resolveOutput(cmd) === "json") printJson(detail);
7631
- else
7632
- console.log(
7633
- `Adopted ${trialId}; baseline is now ${pct(detail.after)}`
7634
- );
7599
+ else console.log(`Marked ${trialId} as the winner.`);
7635
7600
  }
7636
7601
  );
7637
7602
  group.command("skill-files <versionId>").description(
7638
- "The text files of a skill version, the base an EDIT_SKILL trial edits"
7603
+ "The text files of a skill version, the base a skill-edit trial edits"
7639
7604
  ).action(async (versionId, _opts, cmd) => {
7640
7605
  const org = resolveOrg(cmd);
7641
7606
  const res = await createClient().get(
@@ -7649,7 +7614,7 @@ function register19(program2) {
7649
7614
  }
7650
7615
  });
7651
7616
  group.command("page <id>").description(
7652
- "The baseline body of a docs page, the base a REWRITE_DOCS trial edits"
7617
+ "The baseline body of a docs page, the base a docs-rewrite trial edits"
7653
7618
  ).requiredOption("--url <url>", "the page URL").action(async (id, opts, cmd) => {
7654
7619
  const org = resolveOrg(cmd);
7655
7620
  const res = await createClient().get(
@@ -8510,6 +8475,21 @@ function register25(program2) {
8510
8475
  }
8511
8476
  printInsight(insight);
8512
8477
  });
8478
+ runs.command("regrade <id>").description(
8479
+ "Re-judge the run against its eval set's current criteria (no new session)"
8480
+ ).action(async (id, _opts, command) => {
8481
+ const result = await createClient().post(
8482
+ `${runPath(id)}/regrade`
8483
+ );
8484
+ if (resolveOutput(command) === "json") {
8485
+ printJson(result);
8486
+ return;
8487
+ }
8488
+ console.log(
8489
+ `Re-judging run ${result.runId} against ${result.criteria} ${result.criteria === 1 ? "criterion" : "criteria"} of eval ${result.evalSetId}.`
8490
+ );
8491
+ console.log("Verdicts land on the run; `gauge runs get <id>` to check.");
8492
+ });
8513
8493
  runs.command("watch <id>").description(
8514
8494
  "Stream normalized trace events live (SSE; exits when the run finishes)"
8515
8495
  ).action(async (id, _opts, command) => {
@@ -8682,7 +8662,7 @@ function register26(program2) {
8682
8662
  ` ${skill2.name}@${v.label} ${v.contentDigest.slice(0, 19)} ${v.createdAt}`
8683
8663
  );
8684
8664
  });
8685
- group.command("add <reference>").description("Ingest a skill from owner/repo[/skill][#ref] or a GitHub URL").option("--as <handle>", "our handle (default: the directory basename)").option(
8665
+ group.command("add <reference>").description("Ingest a skill from owner/repo[/skill][#ref] or a GitHub URL").option("--as <handle>", "the handle (default: the directory basename)").option(
8686
8666
  "--label <label>",
8687
8667
  "version label (a typed label always mints; omitted dedups by digest)"
8688
8668
  ).option("--skill <name>", "select one skill from a source holding several").option("--note <text>", "note shown on the index").option(
@@ -8747,7 +8727,7 @@ function register26(program2) {
8747
8727
  );
8748
8728
  });
8749
8729
  group.command("rm <ref>").description(
8750
- "Soft-delete a skill (handle stays reserved; finished runs keep their snapshot)"
8730
+ "Remove a skill; the handle stays reserved and finished runs keep their snapshot"
8751
8731
  ).action(async (ref, _opts, cmd) => {
8752
8732
  const org = resolveOrg(cmd);
8753
8733
  const result = await createClient().delete(skillPath(org, ref));
@@ -9071,6 +9051,7 @@ __export(visibility_exports2, {
9071
9051
  register: () => register31,
9072
9052
  visibilityPath: () => visibilityPath
9073
9053
  });
9054
+ import { Option as Option3 } from "commander";
9074
9055
  function visibilityPath(org) {
9075
9056
  return `/api/v1/orgs/${encodeURIComponent(org)}/visibility-prompts`;
9076
9057
  }
@@ -9298,6 +9279,11 @@ function registerPreferenceGroup(group) {
9298
9279
  "run only this scenario (repeatable; default: all)",
9299
9280
  collect,
9300
9281
  []
9282
+ ).addOption(
9283
+ new Option3(
9284
+ "--bill-to-org",
9285
+ "bill this run to the org's credits instead (superusers)"
9286
+ ).hideHelp(!staffCommandsVisible())
9301
9287
  ).option("--yes", "skip the confirmation").action(
9302
9288
  async (id, opts, cmd) => {
9303
9289
  const org = resolveOrg(cmd);
@@ -9308,6 +9294,7 @@ function registerPreferenceGroup(group) {
9308
9294
  const result = await createClient().post(
9309
9295
  `${visibilityPath(org)}/${encodeURIComponent(id)}/run`,
9310
9296
  {
9297
+ ...opts.billToOrg ? { billToOrg: true } : {},
9311
9298
  visibilityScenarioIds: opts.scenario.length > 0 ? opts.scenario : void 0
9312
9299
  }
9313
9300
  );
@@ -9424,11 +9411,6 @@ function register31(program2) {
9424
9411
  registerPreferenceGroup(legacy);
9425
9412
  }
9426
9413
 
9427
- // src/lib/staff.ts
9428
- function staffCommandsVisible(config = readConfig()) {
9429
- return config.staff === true;
9430
- }
9431
-
9432
9414
  // src/index.ts
9433
9415
  var program = new Command();
9434
9416
  program.name("gauge").description(
@@ -9437,7 +9419,7 @@ program.name("gauge").description(
9437
9419
  "--org <org>",
9438
9420
  "organization slug or public id (default: `gauge config get defaultOrg`)"
9439
9421
  ).addOption(
9440
- new Option2("-o, --output <format>", "output format").choices([...OUTPUT_FORMATS]).default("table")
9422
+ new Option4("-o, --output <format>", "output format").choices([...OUTPUT_FORMATS]).default("table")
9441
9423
  );
9442
9424
  for (const group of [
9443
9425
  instructions_exports,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@withgauge/cli",
3
- "version": "0.11.0",
3
+ "version": "0.11.2",
4
4
  "description": "Gauge customer CLI — Agent Preference, evals, runs, and organization settings over the Gauge API",
5
5
  "private": false,
6
6
  "author": "Gauge <support@withgauge.com> (https://withgauge.com)",
@@ -22,7 +22,9 @@
22
22
  "scripts": {
23
23
  "build": "tsup",
24
24
  "dev": "tsx src/index.ts",
25
- "test": "tsx --test src/**/*.test.ts",
25
+ "prepublishOnly": "node scripts/validate-release.mjs",
26
+ "release:validate": "node scripts/validate-release.mjs",
27
+ "test": "tsx --test src/**/*.test.ts scripts/*.test.mjs",
26
28
  "typecheck": "tsc --noEmit"
27
29
  },
28
30
  "dependencies": {