@remits/remits-cli 0.1.136 → 0.1.137

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/index.js CHANGED
@@ -4,18 +4,18 @@
4
4
  ## Table of Contents
5
5
 
6
6
  - L22 Runtime Bootstrap And Shared State
7
- - L119 Sessions, Account Resolution, And Production Guards
8
- - L603 Local State, Workspaces, And Verification Evidence
9
- - L2624 Account Repos, Guide Sync, And Platform Repo
10
- - L3432 Component Discovery And HTTP Logging
11
- - L4073 Skill Delivery And TOC Resolution
12
- - L4383 Auth And Component Staging
13
- - L4998 Component Summaries, Status, And Sync Gates
14
- - L7067 Branches, Promotion, Commit, And Test Runs
15
- - L8750 Tokens, Tools, Verification, And Config
16
- - L10177 Service Dashboard And WebSocket Listener
17
- - L12298 Agent And Ticket Workflows
18
- - L15888 Help, Auto Update, And Command Dispatch
7
+ - L123 Sessions, Account Resolution, And Production Guards
8
+ - L615 Local State, Workspaces, And Verification Evidence
9
+ - L2652 Account Repos, Guide Sync, And Platform Repo
10
+ - L3460 Component Discovery And HTTP Logging
11
+ - L4107 Skill Delivery And TOC Resolution
12
+ - L4417 Auth And Component Staging
13
+ - L5032 Component Summaries, Status, And Sync Gates
14
+ - L7104 Branches, Promotion, Commit, And Test Runs
15
+ - L9193 Tokens, Tools, Verification, And Config
16
+ - L10620 Service Dashboard And WebSocket Listener
17
+ - L12742 Agent And Ticket Workflows
18
+ - L16337 Help, Auto Update, And Command Dispatch
19
19
  */
20
20
 
21
21
  /*
@@ -58,6 +58,10 @@ const SERVICE_STATE_FILE = path.join(SESSION_DIR, 'service-state.json');
58
58
  const ISSUES_DIR = path.join(SESSION_DIR, 'issues');
59
59
  const WEBSOCKET_STATE_FILE = path.join(SESSION_DIR, 'websocket-state.json');
60
60
  const AUTO_UPDATE_LOCK_DIR = path.join(SESSION_DIR, 'auto-update.lock');
61
+ const AUTO_UPDATE_STAMP_FILE = path.join(SESSION_DIR, 'auto-update-check.json');
62
+ // `npm view` is a network round trip. It used to run on EVERY command, which is a per-command latency tax
63
+ // on the agents that issue the most commands.
64
+ const AUTO_UPDATE_CHECK_INTERVAL_MS = 6 * 60 * 60 * 1000;
61
65
  const AUTO_UPDATE_DISABLED_VALUES = new Set(['0', 'false', 'no', 'off']);
62
66
  const REMITS_CLI_PACKAGE_NAME = '@remits/remits-cli';
63
67
  const DEFAULT_DATA_MODE = 'test';
@@ -154,6 +158,14 @@ function flagEnabled(value) {
154
158
  return value === true || value === 'true' || value === '1' || value === 'yes';
155
159
  }
156
160
 
161
+ function flagDisabled(value) {
162
+ return value === false || value === 'false' || value === '0' || value === 'no';
163
+ }
164
+
165
+ function waitForTestCompletion(flags = {}) {
166
+ return !(flagDisabled(flags.wait) || flagEnabled(flags['no-wait']) || flagEnabled(flags.noWait));
167
+ }
168
+
157
169
  function withStdoutRoutedToStderr(enabled, fn) {
158
170
  if (!enabled) return fn();
159
171
  const originalLog = console.log;
@@ -898,6 +910,22 @@ function printLocalStateWarnings(cwd, flags = {}) {
898
910
  warnings.forEach((warning) => console.log(' - ' + warning));
899
911
  }
900
912
 
913
+ /**
914
+ * "Your remits-cli is older than this platform's" - printed once per process, from ONE place.
915
+ *
916
+ * The platform attaches `cliAdvisory` to the two responses every working loop passes through (stage and
917
+ * test-run start) and returns nothing at all when the caller is current, so this renders only when there
918
+ * is something to act on. stderr, because a `--json` caller's stdout is a document.
919
+ */
920
+ let cliAdvisoryPrinted = false;
921
+ function printCliAdvisory(response) {
922
+ const advisory = response && response.cliAdvisory;
923
+ if (!advisory || cliAdvisoryPrinted) return;
924
+ cliAdvisoryPrinted = true;
925
+ console.error('');
926
+ console.error('[remits-cli ' + (advisory.current || '?') + ' -> ' + (advisory.expected || '?') + '] ' + advisory.message);
927
+ }
928
+
901
929
  function printLocalCommandContext(cwd, flags = {}, context = {}) {
902
930
  const paths = localStatePaths(cwd);
903
931
  const branchName = context.branchName || flags.branch || safeGitValue(cwd, 'git rev-parse --abbrev-ref HEAD') || 'unknown';
@@ -3834,6 +3862,12 @@ function stageTimeoutMs(components, flags = {}) {
3834
3862
 
3835
3863
  function buildAxios(baseUrl, token, timeoutMs = 60000) {
3836
3864
  const headers = token ? { Authorization: 'Bearer ' + token } : {};
3865
+ // The CLI version on EVERY request, from the one place every request is built. It used to travel only on
3866
+ // `agent register`, so the platform knew the version of the sessions that registered - and the agents that
3867
+ // loop hardest are exactly the ones that never do. Without it the platform cannot tell an agent that the
3868
+ // diagnostics it is missing exist, and every output improvement lands invisibly.
3869
+ const cliVersion = readCliVersion();
3870
+ if (cliVersion) headers['X-Remits-Cli-Version'] = cliVersion;
3837
3871
  return axios.create({ baseURL: baseUrl, timeout: parsePositiveInt(timeoutMs, 60000), headers });
3838
3872
  }
3839
3873
 
@@ -5232,6 +5266,9 @@ function printStageSummary(response, flags) {
5232
5266
  if (typeof printAccountLanes === 'function') {
5233
5267
  printAccountLanes(response);
5234
5268
  }
5269
+ if (typeof printCliAdvisory === 'function') {
5270
+ printCliAdvisory(response);
5271
+ }
5235
5272
  printComponentCommandResponse('Components stage', response, flags);
5236
5273
  }
5237
5274
 
@@ -7083,6 +7120,10 @@ async function branchesComponentsCommand(flags) {
7083
7120
  // after the subcommand is the branch name when present.
7084
7121
  const positional = flags._ && flags._[2];
7085
7122
  const branchName = flags.branch || positional;
7123
+ const copyToBranch = flags['copy-to'] || flags.copyTo;
7124
+ if ((flags['copy-to'] !== undefined || flags.copyTo !== undefined) && !branchName) {
7125
+ throw new Error('components branch <source> --copy-to <target> requires a source branch name');
7126
+ }
7086
7127
 
7087
7128
  // Without a branch there is nothing to detail, so always list. With one, the mutating flags
7088
7129
  // (--subscribe/--unsubscribe/--retire) win, then the narrower read views, and the default is the
@@ -7092,10 +7133,14 @@ async function branchesComponentsCommand(flags) {
7092
7133
  if (flags.subscribe !== undefined) mode = 'subscribe';
7093
7134
  else if (flags.unsubscribe !== undefined) mode = 'unsubscribe';
7094
7135
  else if (flagEnabled(flags.retire)) mode = 'retire';
7136
+ else if (flags['copy-to'] !== undefined || flags.copyTo !== undefined) mode = 'copy';
7095
7137
  else if (flags.diff !== undefined) mode = 'diff';
7096
7138
  else if (flagEnabled(flags.subscribers)) mode = 'subscribers';
7097
7139
  else mode = 'status';
7098
7140
  }
7141
+ if (mode === 'copy' && (copyToBranch === true || !String(copyToBranch || '').trim())) {
7142
+ throw new Error('components branch <source> --copy-to <target> requires a target branch name');
7143
+ }
7099
7144
 
7100
7145
  // `--diff 42` carries the component id inline; a bare `--diff` falls back to --component-id/--name.
7101
7146
  const componentId = mode === 'diff'
@@ -7118,6 +7163,7 @@ async function branchesComponentsCommand(flags) {
7118
7163
  componentType: flags['component-type'] || flags.type,
7119
7164
  componentId,
7120
7165
  componentName: flags['component-name'] || flags.name,
7166
+ targetBranchName: copyToBranch,
7121
7167
  subscribeAccountId,
7122
7168
  parentAccountId: flags['parent-account'],
7123
7169
  // Optional branch-scoped custom host set on the same edge as the subscription.
@@ -7241,6 +7287,33 @@ function printBranchesSummary(response) {
7241
7287
  return;
7242
7288
  }
7243
7289
 
7290
+ if (response.mode === 'copy') {
7291
+ console.log(response.message || ('Copied branch overlays to ' + response.targetBranch));
7292
+ console.log('Source branch:', response.branch);
7293
+ console.log('Target branch:', response.targetBranch);
7294
+ // A dry run writes nothing, so `copied` is 0 by contract — printing it as "Copied overlays: 0"
7295
+ // under a "Would copy 2" message reads as a failed copy. Name the plan instead.
7296
+ if (response.dryRun) {
7297
+ console.log('Plan only (dry run) — overlays that would be copied:', response.plannedCopies || 0);
7298
+ if (response.existingCount) console.log('Existing overlays that would be replaced:', response.existingCount);
7299
+ } else {
7300
+ console.log('Copied overlays:', response.copied || 0);
7301
+ if (response.overwritten) console.log('Replaced existing overlays:', response.overwritten);
7302
+ }
7303
+ // A source overlay that is identical to trunk in both content and metadata is sparse and is never
7304
+ // stored, so it is named rather than silently missing from the count.
7305
+ const skippedSparse = response.skippedIdenticalToTrunk || [];
7306
+ if (skippedSparse.length) {
7307
+ console.log('Skipped as identical to trunk:', skippedSparse.join(', '));
7308
+ }
7309
+ // The number an operator needs BEFORE deciding to pass --force: a target branch with live
7310
+ // subscribers is code those accounts are running right now.
7311
+ if (response.targetSubscriberCount) {
7312
+ console.log('Live subscribers on the target branch:', response.targetSubscriberCount);
7313
+ }
7314
+ return;
7315
+ }
7316
+
7244
7317
  if (response.mode === 'subscribe' || response.mode === 'unsubscribe' || response.mode === 'retire') {
7245
7318
  console.log(response.message);
7246
7319
  if (response.requiresConfirmation) {
@@ -7852,7 +7925,7 @@ async function waitForStatus(api, cwd, accountId, branchName, taskId, token, dat
7852
7925
  dataMode
7853
7926
  }).then((r) => r.data);
7854
7927
 
7855
- if (status.status === 'completed' || status.status === 'failed') {
7928
+ if (testStatusIsTerminal(status)) {
7856
7929
  return status;
7857
7930
  }
7858
7931
  await new Promise((r) => setTimeout(r, pollDelayMs));
@@ -7860,6 +7933,10 @@ async function waitForStatus(api, cwd, accountId, branchName, taskId, token, dat
7860
7933
  }
7861
7934
  }
7862
7935
 
7936
+ function testStatusIsTerminal(status = {}) {
7937
+ return ['completed', 'failed', 'interrupted'].includes(String(status.status || '').toLowerCase());
7938
+ }
7939
+
7863
7940
  function testEvidenceCategories(status = {}, selectedNames = []) {
7864
7941
  const categories = new Set(['test_run']);
7865
7942
  const test = status.test || {};
@@ -7989,6 +8066,108 @@ function detectNondeterministicTestRun(priorPackets, currentStatus, currentProve
7989
8066
  return null;
7990
8067
  }
7991
8068
 
8069
+ async function appendTerminalTestRunEvidence(options = {}) {
8070
+ const {
8071
+ api,
8072
+ cwd,
8073
+ session,
8074
+ accountId,
8075
+ flags,
8076
+ status,
8077
+ testRef,
8078
+ selectedNames = [],
8079
+ dataMode,
8080
+ dataModeSource,
8081
+ branchName,
8082
+ workspace,
8083
+ baseUrl,
8084
+ activeEnvelopeId,
8085
+ quiet
8086
+ } = options;
8087
+ const unmatched = (status.result && status.result.unmatchedTestNames) || [];
8088
+ const componentProvenance = testComponentProvenance(status);
8089
+ const sourceRevision = collectVerificationSource(cwd, flags);
8090
+ const priorPackets = await readVerificationPacketsForDiagnostics(api, cwd, session, accountId, activeEnvelopeId, {
8091
+ packetType: 'test_run',
8092
+ suite: status.test && status.test.name,
8093
+ max: 50
8094
+ });
8095
+ const nondeterminism = detectNondeterministicTestRun(priorPackets, status, componentProvenance, {
8096
+ laneContentHash: status.staging && status.staging.laneSummary && status.staging.laneSummary.contentHash,
8097
+ gitHead: sourceRevision.gitHead
8098
+ });
8099
+
8100
+ await appendVerificationPacket(api, cwd, session, accountId, flags, {
8101
+ type: 'test_run',
8102
+ success: status.status === 'completed' && !(status.result && status.result.failed > 0) && !unmatched.length,
8103
+ claim: 'Test run ' + String(testRef || (status.test && (status.test.name || status.test.id)) || status.taskId || 'unknown'),
8104
+ world: buildCommandWorld(status, {
8105
+ accountId,
8106
+ dataMode: status.dataMode || dataMode,
8107
+ dataModeSource,
8108
+ branchName,
8109
+ workspace,
8110
+ host: normalizeBaseUrl(baseUrl),
8111
+ sourceLayer: status.staging && status.staging.testComponentSource
8112
+ }),
8113
+ revision: Object.assign(sourceRevision, {
8114
+ compileSignatures: testCompileSignatures(status),
8115
+ componentProvenance
8116
+ }),
8117
+ nondeterministic: nondeterminism ? true : undefined,
8118
+ nondeterminism: nondeterminism || undefined,
8119
+ test: {
8120
+ taskId: status.taskId || (status.result && status.result.taskId),
8121
+ testId: status.test && status.test.id,
8122
+ testName: status.test && status.test.name,
8123
+ selectedCases: selectedNames,
8124
+ passed: status.result && status.result.passed,
8125
+ failed: status.result && status.result.failed,
8126
+ tests: status.result && status.result.tests,
8127
+ dataModeSource: status.dataModeSource || dataModeSource,
8128
+ unmatchedTestNames: unmatched
8129
+ },
8130
+ evidenceCategories: testEvidenceCategories(status, selectedNames),
8131
+ // What the run spent and whether that live AI was declared (withAiBudget), so an envelope can tell a
8132
+ // measurement run from a suite with missing mocks without re-reading every case.
8133
+ dependencies: testRunDependencies(status),
8134
+ summary: status.result && status.result.summary,
8135
+ limitations: []
8136
+ .concat(unmatched.length ? ['One or more requested test case selectors matched no case.'] : [])
8137
+ .concat(nondeterminism ? [nondeterminism.message] : []),
8138
+ rawRefs: { testStatusKey: status.taskId, durableTestRun: status.result && status.result.durableRecord ? status.taskId : undefined },
8139
+ status
8140
+ }, { quiet });
8141
+
8142
+ return { unmatched, nondeterminism };
8143
+ }
8144
+
8145
+ /**
8146
+ * One failed case per distinct failure root, largest root first, capped.
8147
+ *
8148
+ * A pivot block per failed case is one problem stated N times: the thirty-three failures of a real suite
8149
+ * are twenty-one roots, and the twelve repeats carry the same trace shape, the same components and the
8150
+ * same error. The grouped roll-up above already names every case; this picks the ones worth a full block.
8151
+ */
8152
+ /** Truncate, and SAY that it was truncated — a silent cut reads as the whole message. */
8153
+ function truncateForPivot(text, max) {
8154
+ if (text.length <= max) return text;
8155
+ return text.slice(0, max) + '… (truncated; --json for the full text)';
8156
+ }
8157
+
8158
+ function selectPivotCases(tests, limit = 6) {
8159
+ const failed = (tests || []).filter((test) => test && test.passed === false);
8160
+ const seen = new Set();
8161
+ const representatives = [];
8162
+ failed.forEach((test) => {
8163
+ const root = testFailureRoot(test);
8164
+ if (seen.has(root)) return;
8165
+ seen.add(root);
8166
+ representatives.push(test);
8167
+ });
8168
+ return { shown: representatives.slice(0, limit), failedCount: failed.length, rootCount: representatives.length };
8169
+ }
8170
+
7992
8171
  function printTestRunPivots(status = {}) {
7993
8172
  const tests = status.result && Array.isArray(status.result.tests) ? status.result.tests : [];
7994
8173
  if (!tests.length) return;
@@ -7996,7 +8175,8 @@ function printTestRunPivots(status = {}) {
7996
8175
  if (slowest && slowest.duration != null) {
7997
8176
  console.log('Slowest case:', (slowest.name || '(unnamed)') + ' in ' + formatDurationMs(slowest.duration));
7998
8177
  }
7999
- tests.filter((test) => test && test.passed === false).forEach((test) => {
8178
+ const selection = selectPivotCases(tests);
8179
+ selection.shown.forEach((test) => {
8000
8180
  console.log('Pivots for failed case:', test.name || '(unnamed)');
8001
8181
  if (test.outcome) console.log(' Outcome:', test.outcome + (test.outcomeReason && test.outcomeReason !== test.error ? ' - ' + String(test.outcomeReason).slice(0, 300) : ''));
8002
8182
  if (test.duration != null) console.log(' Duration:', formatDurationMs(test.duration));
@@ -8007,7 +8187,9 @@ function printTestRunPivots(status = {}) {
8007
8187
  if (test.threadGroupingId || test.threadGroupId || test.traceId) {
8008
8188
  console.log(' Trace:', test.traceId || test.threadGroupingId || test.threadGroupId);
8009
8189
  }
8010
- if (test.error) console.log(' Error:', String(test.error).slice(0, 500));
8190
+ // 500 chars cut the replay diagnosis mid-sentence, losing the half that says what to DO. A pivot is the
8191
+ // block a reader acts on; truncating its one actionable sentence to save four lines is a bad trade.
8192
+ if (test.error) console.log(' Error:', truncateForPivot(String(test.error), 1200));
8011
8193
  const diagnostics = test.diagnostics && typeof test.diagnostics === 'object' ? test.diagnostics : null;
8012
8194
  if (diagnostics && Object.keys(diagnostics).length) {
8013
8195
  console.log(' Diagnostics:', JSON.stringify(diagnostics).slice(0, 1200));
@@ -8023,6 +8205,12 @@ function printTestRunPivots(status = {}) {
8023
8205
  console.log(' Live HTTP calls:', test.liveHttpCalls.length + (intentional ? ' (' + intentional + ' intentional, inside withAiBudget)' : ''));
8024
8206
  }
8025
8207
  });
8208
+ // Named, not silently dropped: a reader must be able to tell "this is everything" from "this is a sample".
8209
+ if (selection.rootCount > selection.shown.length) {
8210
+ console.log('Pivots shown for ' + selection.shown.length + ' of ' + selection.rootCount +
8211
+ ' failure root(s) (' + selection.failedCount + ' failed case(s)). Every root is listed above; ' +
8212
+ 'use --names "<case>" for one, or --json for all.');
8213
+ }
8026
8214
  }
8027
8215
 
8028
8216
  function formatDurationMs(value) {
@@ -8236,6 +8424,114 @@ function stagedMaskedByVariantLines(masked, accountId) {
8236
8424
  return lines;
8237
8425
  }
8238
8426
 
8427
+ /**
8428
+ * The assertion ROOT of one failed case: what broke, with the particulars of this case removed.
8429
+ *
8430
+ * Deliberately built only from properties of the JVM/Groovy failure format - a power-assert's
8431
+ * `Expression:`/`Values:` decoration, an `assert <expr>` head, an exception class prefix - and never from
8432
+ * any vocabulary a suite happens to use. Identifiers and numbers are replaced because two cases failing
8433
+ * the same way differ exactly in those.
8434
+ */
8435
+ function testFailureRoot(test = {}) {
8436
+ let text = String(test.error || test.outcomeReason || test.outcome || 'failed').trim();
8437
+ // Groovy's power assert appends the rendered expression and every intermediate value.
8438
+ text = text.split(/\.\s+(?:Expression|Values):/)[0];
8439
+ const assertion = text.match(/assert\s+(.+)$/);
8440
+ if (assertion) text = 'assert ' + assertion[1];
8441
+ text = text
8442
+ .replace(/['"][0-9a-fA-F]{8}-[0-9a-fA-F-]{4,}['"]/g, "'<id>'")
8443
+ .replace(/\b[0-9a-fA-F]{8}-[0-9a-fA-F-]{27,}\b/g, '<id>')
8444
+ .replace(/\b\d[\d.,]*\b/g, '<n>')
8445
+ .replace(/\s+/g, ' ')
8446
+ .trim();
8447
+ if (text.length <= FAILURE_ROOT_LABEL_CHARS) return text;
8448
+ // Cut on a word boundary: "found no usable stored provid" reads as a different error than the one it is.
8449
+ const cut = text.slice(0, FAILURE_ROOT_LABEL_CHARS);
8450
+ const lastSpace = cut.lastIndexOf(' ');
8451
+ return (lastSpace > FAILURE_ROOT_LABEL_CHARS * 0.6 ? cut.slice(0, lastSpace) : cut) + '\u2026';
8452
+ }
8453
+
8454
+ /** Long enough to tell two failures apart, short enough that a root list stays a list. */
8455
+ const FAILURE_ROOT_LABEL_CHARS = 120;
8456
+
8457
+ /**
8458
+ * Failed cases grouped by that root, largest first.
8459
+ *
8460
+ * Thirty-three failures printed in run order read as thirty-three problems; the same run grouped is four.
8461
+ * This is the line that decides whether the next move is a narrow root-cause pass or another broad rerun,
8462
+ * so it is computed from the run itself rather than left to the reader.
8463
+ */
8464
+ function testFailureGroups(status = {}) {
8465
+ const tests = status.result && Array.isArray(status.result.tests) ? status.result.tests : [];
8466
+ const groups = new Map();
8467
+ tests.filter((test) => test && test.passed === false && !test.interrupted).forEach((test) => {
8468
+ const root = testFailureRoot(test);
8469
+ if (!groups.has(root)) groups.set(root, { root, count: 0, cases: [] });
8470
+ const group = groups.get(root);
8471
+ group.count += 1;
8472
+ group.cases.push(test.name || '(unnamed)');
8473
+ });
8474
+ return Array.from(groups.values()).sort((left, right) => right.count - left.count || left.root.localeCompare(right.root));
8475
+ }
8476
+
8477
+ const FAILURE_GROUPS_SHOWN = 8;
8478
+
8479
+ function testFailureGroupLines(status = {}, options = {}) {
8480
+ const groups = testFailureGroups(status);
8481
+ if (!groups.length) return [];
8482
+ const failed = groups.reduce((sum, group) => sum + group.count, 0);
8483
+ const lines = [''];
8484
+ lines.push('Failure roots (' + failed + ' failed case(s), ' + groups.length + ' distinct root(s)):');
8485
+ groups.slice(0, FAILURE_GROUPS_SHOWN).forEach((group) => {
8486
+ lines.push(' ' + String(group.count) + 'x ' + group.root);
8487
+ // The representative case is what `--names` takes, so the next command is a copy of this line.
8488
+ lines.push(' e.g. ' + group.cases[0] + (group.count > 1 ? ' (+' + (group.count - 1) + ' more)' : ''));
8489
+ });
8490
+ if (groups.length > FAILURE_GROUPS_SHOWN) {
8491
+ lines.push(' ...' + (groups.length - FAILURE_GROUPS_SHOWN) + ' more root(s); --json for all');
8492
+ }
8493
+ const testRef = options.testRef || (status.test && (status.test.id || status.test.name)) ||
8494
+ (status.result && (status.result.testId || status.result.testName));
8495
+ if (groups[0] && groups[0].count > 1 && testRef) {
8496
+ lines.push('Largest root first: remits-cli test run --test ' + JSON.stringify(String(testRef)) +
8497
+ ' --names ' + JSON.stringify(groups[0].cases[0]));
8498
+ }
8499
+ return lines;
8500
+ }
8501
+
8502
+ /**
8503
+ * The verdict of a run in four lines, shared by `test run` and `test status`.
8504
+ *
8505
+ * `test run` used to print the whole run object as pretty JSON before its human summary. Measured on one
8506
+ * 92-case suite that is 220 KB - most of it per-case ids repeated once per case - spent by the command an
8507
+ * agent runs most often, on the turn where it has the least room left to reason. Every byte is still one
8508
+ * `--json` away.
8509
+ */
8510
+ function testRunHeadlineLines(status = {}, options = {}) {
8511
+ const result = status.result || {};
8512
+ const lines = [];
8513
+ lines.push('Test run status: ' + (status.status || 'unknown'));
8514
+ if (options.taskId || status.taskId) lines.push('Task ID: ' + (options.taskId || status.taskId));
8515
+ if (options.source) lines.push('Source: ' + options.source);
8516
+ const actualDataMode = status.dataMode || result.dataMode || options.dataMode;
8517
+ let dataModeLine = 'Data mode: ' + (actualDataMode || 'unknown');
8518
+ // A durable run answers about the lane it RAN in, which need not be the lane this command asked for.
8519
+ // Reporting only the run's own lane is correct and reads as if the request had been honoured.
8520
+ if (options.requestedDataMode && actualDataMode && options.requestedDataMode !== actualDataMode) {
8521
+ dataModeLine += ' (you asked for ' + options.requestedDataMode + '; this run was recorded in the ' +
8522
+ actualDataMode + ' lane, and that is what it proves)';
8523
+ }
8524
+ lines.push(dataModeLine);
8525
+ if (result.total != null || result.passed != null) {
8526
+ lines.push('Cases: ' + (result.passed || 0) + ' passed, ' + (result.failed || 0) + ' failed, ' +
8527
+ (result.total || 0) + ' total');
8528
+ }
8529
+ if (status.error || status.message || result.error) {
8530
+ lines.push('Message: ' + (status.error || status.message || result.error));
8531
+ }
8532
+ return lines;
8533
+ }
8534
+
8239
8535
  // The corpus-style roll-up printed after a run: outcome counts, case duration percentiles, AI usage split
8240
8536
  // live/mocked, and the durable record. Built only from the result the platform returned.
8241
8537
  function testRunSummaryLines(status = {}) {
@@ -8491,6 +8787,7 @@ async function testCommand(flags) {
8491
8787
  printLocalStateWarnings(cwd, flags);
8492
8788
  printStagingLane(branchName, workspace, workspaceSource(cwd, flags));
8493
8789
  printStagingLaneOwnerNotice(start.staging || {});
8790
+ printCliAdvisory(start);
8494
8791
  if (start.staging && Array.isArray(start.staging.accountLanes)) {
8495
8792
  printOrphanedWorkspaceWarning(start.staging);
8496
8793
  printAccountLanes({ accountLanes: start.staging.accountLanes, branchName, workspace });
@@ -8506,6 +8803,48 @@ async function testCommand(flags) {
8506
8803
  });
8507
8804
  runtimeState.currentTestTaskId = start.taskId;
8508
8805
 
8806
+ const waitForCompletion = waitForTestCompletion(flags);
8807
+ if (!waitForCompletion) {
8808
+ recordEvidenceEntry(cwd, {
8809
+ type: 'test_run',
8810
+ success: null,
8811
+ pending: true,
8812
+ claim: 'Test run ' + String(testRef),
8813
+ world: buildCommandWorld(start, {
8814
+ accountId,
8815
+ dataMode,
8816
+ dataModeSource,
8817
+ branchName,
8818
+ workspace,
8819
+ host: normalizeBaseUrl(baseUrl),
8820
+ sourceLayer: start.staging && start.staging.testComponentSource
8821
+ }),
8822
+ test: {
8823
+ taskId: start.taskId,
8824
+ testId: start.test && start.test.id,
8825
+ testName: start.test && start.test.name,
8826
+ selectedCases: names,
8827
+ dataModeSource
8828
+ },
8829
+ evidenceCategories: testEvidenceCategories(start, names),
8830
+ rawRefs: { testStatusKey: start.taskId },
8831
+ status: start
8832
+ }, { accountId, dataMode, branchName, workspace, baseUrl: normalizeBaseUrl(baseUrl), envelopeId: activeEnvelopeId });
8833
+ runtimeState.currentTestTaskId = null;
8834
+ if (jsonOutput) {
8835
+ console.log(JSON.stringify(Object.assign({}, start, {
8836
+ pending: true,
8837
+ wait: false,
8838
+ statusCommand: 'remits-cli test status --task-id ' + start.taskId
8839
+ }), null, 2));
8840
+ } else {
8841
+ console.log('Not waiting (--wait false).');
8842
+ console.log('Poll this run: remits-cli test status --task-id ' + start.taskId);
8843
+ console.log('Terminal evidence attaches when `test status` reads a completed, failed or interrupted run.');
8844
+ }
8845
+ return;
8846
+ }
8847
+
8509
8848
  let stopWs = null;
8510
8849
  if (flags.watch !== 'false') {
8511
8850
  const topic = session.websocketTopic || start.websocketTopic || (session.user && String(session.user.uuid || '').replace(/-/g, ''));
@@ -8526,10 +8865,16 @@ async function testCommand(flags) {
8526
8865
  if (jsonOutput) {
8527
8866
  console.log(JSON.stringify(status, null, 2));
8528
8867
  } else {
8529
- console.log('Final status:', JSON.stringify(status, null, 2));
8530
8868
  printStagingLaneOwnerNotice(status.staging || {});
8869
+ testRunHeadlineLines(status, { taskId: start.taskId, dataMode, requestedDataMode: dataMode })
8870
+ .forEach((line) => console.log(line));
8531
8871
  testRunSummaryLines(status).forEach((line) => console.log(line));
8872
+ testFailureGroupLines(status, { testRef }).forEach((line) => console.log(line));
8532
8873
  printTestRunPivots(status);
8874
+ // The whole run object used to be printed here as pretty JSON. One 92-case suite measured 220 KB,
8875
+ // most of it identifiers repeated once per case, spent on the turn with the least room left. It is
8876
+ // still one flag away, and the durable record keeps it after the live status expires.
8877
+ console.log('Full run payload: remits-cli test status --task-id ' + start.taskId + ' --json');
8533
8878
  }
8534
8879
 
8535
8880
  // A selector that matched no case is a mis-specified run, not a passing one. Say so in the terminal
@@ -8554,17 +8899,24 @@ async function testCommand(flags) {
8554
8899
  process.exitCode = 1;
8555
8900
  }
8556
8901
 
8557
- const componentProvenance = testComponentProvenance(status);
8558
- const sourceRevision = collectVerificationSource(cwd, flags);
8559
- const priorPackets = await readVerificationPacketsForDiagnostics(api, cwd, session, accountId, activeEnvelopeId, {
8560
- packetType: 'test_run',
8561
- suite: status.test && status.test.name,
8562
- max: 50
8563
- });
8564
- const nondeterminism = detectNondeterministicTestRun(priorPackets, status, componentProvenance, {
8565
- laneContentHash: status.staging && status.staging.laneSummary && status.staging.laneSummary.contentHash,
8566
- gitHead: sourceRevision.gitHead
8902
+ const evidence = await appendTerminalTestRunEvidence({
8903
+ api,
8904
+ cwd,
8905
+ session,
8906
+ accountId,
8907
+ flags: verification.evidenceFlags,
8908
+ status,
8909
+ testRef,
8910
+ selectedNames: names,
8911
+ dataMode,
8912
+ dataModeSource,
8913
+ branchName,
8914
+ workspace,
8915
+ baseUrl,
8916
+ activeEnvelopeId,
8917
+ quiet: jsonOutput
8567
8918
  });
8919
+ const nondeterminism = evidence.nondeterminism;
8568
8920
  if (nondeterminism && !jsonOutput) {
8569
8921
  console.log('Nondeterministic signal:', nondeterminism.message);
8570
8922
  nondeterminism.flips.slice(0, 6).forEach((flip) => {
@@ -8572,40 +8924,6 @@ async function testCommand(flags) {
8572
8924
  ' (previous packet ' + (flip.previousPacketId || 'unknown') + ')');
8573
8925
  });
8574
8926
  }
8575
-
8576
- await appendVerificationPacket(api, cwd, session, accountId, verification.evidenceFlags, {
8577
- type: 'test_run',
8578
- success: status.status === 'completed' && !(status.result && status.result.failed > 0) && !unmatched.length,
8579
- claim: 'Test run ' + String(testRef),
8580
- world: buildCommandWorld(status, { accountId, dataMode: status.dataMode || dataMode, dataModeSource, branchName, workspace, host: normalizeBaseUrl(baseUrl), sourceLayer: status.staging && status.staging.testComponentSource }),
8581
- revision: Object.assign(sourceRevision, {
8582
- compileSignatures: testCompileSignatures(status),
8583
- componentProvenance
8584
- }),
8585
- nondeterministic: nondeterminism ? true : undefined,
8586
- nondeterminism: nondeterminism || undefined,
8587
- test: {
8588
- taskId: start.taskId,
8589
- testId: status.test && status.test.id,
8590
- testName: status.test && status.test.name,
8591
- selectedCases: names,
8592
- passed: status.result && status.result.passed,
8593
- failed: status.result && status.result.failed,
8594
- tests: status.result && status.result.tests,
8595
- dataModeSource: status.dataModeSource || dataModeSource,
8596
- unmatchedTestNames: unmatched
8597
- },
8598
- evidenceCategories: testEvidenceCategories(status, names),
8599
- // What the run spent and whether that live AI was declared (withAiBudget), so an envelope can tell a
8600
- // measurement run from a suite with missing mocks without re-reading every case.
8601
- dependencies: testRunDependencies(status),
8602
- summary: status.result && status.result.summary,
8603
- limitations: []
8604
- .concat(unmatched.length ? ['One or more requested test case selectors matched no case.'] : [])
8605
- .concat(nondeterminism ? [nondeterminism.message] : []),
8606
- rawRefs: { testStatusKey: start.taskId, durableTestRun: status.result && status.result.durableRecord ? start.taskId : undefined },
8607
- status
8608
- }, { quiet: jsonOutput });
8609
8927
  }
8610
8928
 
8611
8929
  async function testStatusCommand(flags) {
@@ -8615,9 +8933,11 @@ async function testStatusCommand(flags) {
8615
8933
  const { session, accountId } = sessionContext;
8616
8934
  const baseUrl = flags['base-url'] || session.baseUrl || DEFAULT_BASE_URL;
8617
8935
  const branchName = flags.branch || currentBranch(cwd);
8936
+ const workspace = resolveWorkspace(cwd, flags);
8618
8937
  const dataMode = hasExplicitDataModeFlag(flags)
8619
8938
  ? resolveDataMode(flags, null)
8620
8939
  : DEFAULT_DATA_MODE;
8940
+ const dataModeSource = dataModeFlagSource(flags);
8621
8941
  const taskId = flags['task-id'] || flags.taskId || flags.id || (flags._ && flags._[2]);
8622
8942
  const jsonOutput = flagEnabled(flags.json);
8623
8943
  if (!taskId) throw new Error('Missing --task-id <taskId>');
@@ -8631,31 +8951,94 @@ async function testStatusCommand(flags) {
8631
8951
  taskId,
8632
8952
  dataMode
8633
8953
  }).then((r) => r.data);
8954
+ if (!status.taskId) status.taskId = taskId;
8955
+
8956
+ let statusVerification = { envelopeId: null, evidenceFlags: flags };
8957
+ if (testStatusIsTerminal(status)) {
8958
+ statusVerification = await verificationPreflight(api, cwd, session, accountId, flags, {
8959
+ packetType: 'test_run',
8960
+ world: buildCommandWorld(status, {
8961
+ accountId,
8962
+ dataMode: status.dataMode || dataMode,
8963
+ dataModeSource,
8964
+ branchName,
8965
+ workspace,
8966
+ host: normalizeBaseUrl(baseUrl),
8967
+ sourceLayer: status.staging && status.staging.testComponentSource
8968
+ }),
8969
+ test: status.test && status.test.id !== undefined && status.test.id !== null
8970
+ ? { testId: status.test.id, testName: status.test.name }
8971
+ : { testName: status.test && status.test.name }
8972
+ }, { command: 'test status', neverRefuse: true, quiet: jsonOutput });
8973
+ }
8634
8974
 
8635
8975
  if (jsonOutput) {
8976
+ if (testStatusIsTerminal(status)) {
8977
+ await appendTerminalTestRunEvidence({
8978
+ api,
8979
+ cwd,
8980
+ session,
8981
+ accountId,
8982
+ flags: statusVerification.evidenceFlags,
8983
+ status,
8984
+ testRef: status.test && (status.test.name || status.test.id) || taskId,
8985
+ selectedNames: status.result && Array.isArray(status.result.tests) ? status.result.tests.map((t) => t && t.name).filter(Boolean) : [],
8986
+ dataMode,
8987
+ dataModeSource,
8988
+ branchName,
8989
+ workspace,
8990
+ baseUrl,
8991
+ activeEnvelopeId: statusVerification.envelopeId,
8992
+ quiet: true
8993
+ });
8994
+ }
8636
8995
  console.log(JSON.stringify(status, null, 2));
8637
8996
  return status;
8638
8997
  }
8639
8998
  printSessionResolutionWarning(sessionContext);
8640
8999
  printResolvedBaseUrl(baseUrl);
8641
- console.log('Test run status:', status.status || 'unknown');
8642
- console.log('Task ID:', taskId);
8643
9000
  // Say where this answer came from. A durable record is a finished snapshot rebuilt from the database after the
8644
9001
  // live status was gone; without this line a reconstructed run is indistinguishable from one still being watched.
8645
- console.log('Source:', status.durable
8646
- ? 'durable run record (the live status has expired; this run is final)'
8647
- : 'live run status');
8648
- console.log('Data mode:', status.dataMode || dataMode);
8649
- if (status.result) {
8650
- console.log('Cases:', (status.result.passed || 0) + ' passed, ' + (status.result.failed || 0) + ' failed, ' + (status.result.total || 0) + ' total');
8651
- }
8652
- if (status.error || status.message) {
8653
- console.log('Message:', status.error || status.message);
8654
- }
9002
+ testRunHeadlineLines(status, {
9003
+ taskId,
9004
+ source: status.durable
9005
+ ? 'durable run record (the live status has expired; this run is final)'
9006
+ : 'live run status',
9007
+ dataMode,
9008
+ // A durable run reports the lane it RAN in. When that differs from the lane this command asked for,
9009
+ // say so: auditing prod activity and being handed a test-lane run is correct and reads as if it were not.
9010
+ requestedDataMode: hasExplicitDataModeFlag(flags) ? dataMode : null
9011
+ }).forEach((line) => console.log(line));
8655
9012
  if (status.world) runWorldLines(status.world, { host: normalizeBaseUrl(baseUrl) }).forEach((line) => console.log(line));
8656
9013
  testRunSummaryLines(status).forEach((line) => console.log(line));
9014
+ testFailureGroupLines(status).forEach((line) => console.log(line));
8657
9015
  printTestRunPivots(status);
8658
- if (status.status === 'failed' || (status.result && status.result.failed > 0)) {
9016
+ if (testStatusIsTerminal(status)) {
9017
+ const evidence = await appendTerminalTestRunEvidence({
9018
+ api,
9019
+ cwd,
9020
+ session,
9021
+ accountId,
9022
+ flags: statusVerification.evidenceFlags,
9023
+ status,
9024
+ testRef: status.test && (status.test.name || status.test.id) || taskId,
9025
+ selectedNames: status.result && Array.isArray(status.result.tests) ? status.result.tests.map((t) => t && t.name).filter(Boolean) : [],
9026
+ dataMode,
9027
+ dataModeSource,
9028
+ branchName,
9029
+ workspace,
9030
+ baseUrl,
9031
+ activeEnvelopeId: statusVerification.envelopeId
9032
+ });
9033
+ if (evidence.nondeterminism) {
9034
+ console.log('Nondeterministic signal:', evidence.nondeterminism.message);
9035
+ evidence.nondeterminism.flips.slice(0, 6).forEach((flip) => {
9036
+ console.log(' - ' + flip.caseName + ': ' + (flip.previousPassed ? 'passed' : 'failed') + ' -> ' + (flip.currentPassed ? 'passed' : 'failed') +
9037
+ ' (previous packet ' + (flip.previousPacketId || 'unknown') + ')');
9038
+ });
9039
+ }
9040
+ }
9041
+ if (status.status === 'failed' || status.status === 'interrupted' || (status.result && status.result.failed > 0)) {
8659
9042
  process.exitCode = 1;
8660
9043
  }
8661
9044
  return status;
@@ -8671,10 +9054,13 @@ async function testRunsCommand(flags) {
8671
9054
  const baseUrl = flags['base-url'] || session.baseUrl || DEFAULT_BASE_URL;
8672
9055
  const api = buildAxios(baseUrl, session.token);
8673
9056
  const testRef = flags.test || flags['test-id'] || flags.name;
9057
+ const allDataLanes = flagEnabled(flags['all-lanes']) || flagEnabled(flags['all-data-lanes']);
8674
9058
  const data = await loggedPost(api, cwd, '/cli/testRuns', {
8675
9059
  token: session.token,
8676
9060
  accountId,
8677
9061
  asAccountId: flags['as-account'] || flags['as-account-id'],
9062
+ dataMode: resolveDataMode(flags, session),
9063
+ allDataLanes: allDataLanes || undefined,
8678
9064
  testId: testRef && /^\d+$/.test(String(testRef)) ? Number(testRef) : undefined,
8679
9065
  testName: testRef && !/^\d+$/.test(String(testRef)) ? String(testRef) : undefined,
8680
9066
  max: flags.limit || flags.max || 20
@@ -8682,9 +9068,17 @@ async function testRunsCommand(flags) {
8682
9068
  if (!data.success) throw new Error(data.message || 'Could not list test runs');
8683
9069
 
8684
9070
  if (flagEnabled(flags.compare)) {
8685
- const runs = data.runs || [];
8686
- if (runs.length < 2) throw new Error('--compare needs at least two recorded runs' + (testRef ? ' of ' + testRef : '') + '; found ' + runs.length);
8687
- return testCompareCommand(Object.assign({}, flags, { base: runs[1].taskId, head: runs[0].taskId }));
9071
+ const pair = selectComparableRuns(data.runs || []);
9072
+ if (!pair) {
9073
+ throw new Error('--compare needs two COMPARABLE recorded runs' + (testRef ? ' of ' + testRef : '') +
9074
+ ' — same data lane, branch, workspace and case count. Found ' + (data.runs || []).length +
9075
+ ' run(s); pick two explicitly with: remits-cli test compare --base <taskId> --head <taskId>');
9076
+ }
9077
+ if (pair.skipped) {
9078
+ console.error('Comparing the latest two comparable runs (' + describeRunShape(pair.head) + '); ' +
9079
+ pair.skipped + ' newer run(s) differ in lane, branch, workspace or case count and were skipped.');
9080
+ }
9081
+ return testCompareCommand(Object.assign({}, flags, { base: pair.base.taskId, head: pair.head.taskId }));
8688
9082
  }
8689
9083
  if (flagEnabled(flags.json)) {
8690
9084
  console.log(JSON.stringify(data, null, 2));
@@ -8693,6 +9087,9 @@ async function testRunsCommand(flags) {
8693
9087
  printSessionResolutionWarning(sessionContext);
8694
9088
  printResolvedBaseUrl(baseUrl);
8695
9089
  console.log('Recorded test runs' + (testRef ? ' for ' + testRef : '') + ' (account ' + data.accountId + '), newest first:');
9090
+ // Which lane this list is, said once. Pass counts from two lanes are two different facts.
9091
+ console.log('Data lane: ' + (data.dataLaneScope || 'all-data-lanes') +
9092
+ (data.otherLaneCount ? ' (' + data.otherLaneCount + ' more run(s) in the other lane; --all-lanes to include)' : ''));
8696
9093
  (data.runs || []).forEach((run) => {
8697
9094
  console.log('- ' + run.taskId + ' ' + (run.status || 'unknown') + ' ' + (run.passed || 0) + '/' + (run.total || 0) + ' passed' +
8698
9095
  ' ' + corpusAiLabel(run) +
@@ -8705,6 +9102,40 @@ async function testRunsCommand(flags) {
8705
9102
  return data;
8706
9103
  }
8707
9104
 
9105
+ /** The world fields that decide whether two runs are the same experiment repeated. */
9106
+ function runShapeKey(run = {}) {
9107
+ return [run.dataMode || '?', run.branchName || '?', run.workspace || 'shared',
9108
+ run.variantBranch || 'none', run.total == null ? '?' : run.total].join('|');
9109
+ }
9110
+
9111
+ function describeRunShape(run = {}) {
9112
+ return [(run.dataMode || '?') + ' lane', run.branchName || '?',
9113
+ run.workspace ? 'ws:' + run.workspace : 'shared lane',
9114
+ (run.total == null ? '?' : run.total) + ' case(s)'].join(' · ');
9115
+ }
9116
+
9117
+ /**
9118
+ * The newest two runs that are actually comparable, and how many newer ones were passed over.
9119
+ *
9120
+ * `--compare` used to take runs[0] and runs[1] unconditionally. On a real history that pairs a 92-case
9121
+ * suite with a 1-case `--names` probe, or a prod-lane run with a test-lane one, and reports the difference
9122
+ * as regressions. A comparison across worlds is not a comparison.
9123
+ */
9124
+ function selectComparableRuns(runs) {
9125
+ const list = Array.isArray(runs) ? runs.filter(Boolean) : [];
9126
+ for (let head = 0; head < list.length; head += 1) {
9127
+ const key = runShapeKey(list[head]);
9128
+ for (let base = head + 1; base < list.length; base += 1) {
9129
+ if (runShapeKey(list[base]) === key) {
9130
+ // Every run passed over, including the ones BETWEEN head and base — a 1-case probe sitting between
9131
+ // two full suites is exactly the run whose absence from the comparison needs saying.
9132
+ return { head: list[head], base: list[base], skipped: base - 1 };
9133
+ }
9134
+ }
9135
+ }
9136
+ return null;
9137
+ }
9138
+
8708
9139
  // `test compare --base <taskId> --head <taskId>`: per-case outcome, cost and duration deltas between two runs,
8709
9140
  // read from the durable records.
8710
9141
  async function testCompareCommand(flags) {
@@ -8735,6 +9166,18 @@ async function testCompareCommand(flags) {
8735
9166
  .filter((key) => comparison.baseWorld && comparison.headWorld && String(comparison.baseWorld[key] || '') !== String(comparison.headWorld[key] || ''))
8736
9167
  .map((key) => key + ' ' + short(String(comparison.baseWorld[key] || 'none')) + ' -> ' + short(String(comparison.headWorld[key] || 'none')));
8737
9168
  if (worldDiff.length) console.log('World changed: ' + worldDiff.join('; '));
9169
+ // A lane or case-count difference is not a delta to interpret, it is two different experiments. Say so
9170
+ // rather than letting "regressed (33)" stand for "these runs never measured the same thing".
9171
+ const baseLane = comparison.baseWorld && comparison.baseWorld.dataMode;
9172
+ const headLane = comparison.headWorld && comparison.headWorld.dataMode;
9173
+ if (baseLane && headLane && baseLane !== headLane) {
9174
+ console.log('NOT COMPARABLE: these runs are in different DATA LANES (' + baseLane + ' vs ' + headLane +
9175
+ '). A record\'s dataMode is the lane it was written in; the pass counts below are two different facts.');
9176
+ }
9177
+ if (comparison.base.total && comparison.head.total && comparison.base.total !== comparison.head.total) {
9178
+ console.log('NOT COMPARABLE like-for-like: ' + comparison.base.total + ' case(s) vs ' + comparison.head.total +
9179
+ '. One of these is a narrowed --names run; "absent" below means the case did not run, not that it broke.');
9180
+ }
8738
9181
  if (comparison.improved.length) console.log('Improved (' + comparison.improved.length + '): ' + comparison.improved.join(', '));
8739
9182
  if (comparison.regressed.length) console.log('Regressed (' + comparison.regressed.length + '): ' + comparison.regressed.join(', '));
8740
9183
  comparison.changed.filter((c) => c.direction === 'changed').forEach((c) => console.log('Changed: ' + c.name + ' ' + c.base + ' -> ' + c.head));
@@ -10898,6 +11341,7 @@ function buildRepoSnapshot(entry) {
10898
11341
  const repoFiles = [
10899
11342
  { label: 'account-info.json', path: entry.accountInfoPath || path.join(directory, 'account-info.json'), mode: 'json' },
10900
11343
  { label: 'account-hierarchy.json', path: path.join(directory, 'account-hierarchy.json'), mode: 'json' },
11344
+ { label: 'account-analytics.json', path: path.join(directory, 'account-analytics.json'), mode: 'json' },
10901
11345
  { label: '.remits-cli/active-actor/current-session.txt', path: localPaths.currentSessionFile, mode: 'text' },
10902
11346
  { label: '.remits-cli/shared/tools/tools.json', path: path.join(localPaths.toolsDir, 'tools.json'), mode: 'json' },
10903
11347
  { label: '.remits-cli/active-actor/session log (tail)', path: currentSessionLog, mode: 'tail' },
@@ -15143,7 +15587,12 @@ function printActivityInspectSummary(response) {
15143
15587
  if (env.noRequiredEvidence) {
15144
15588
  console.log(' evidence log — no verdict requested; add a claim only if someone needs one: remits-cli verify claim <id> --text "..." --envelope ' + env.envelopeId);
15145
15589
  }
15146
- if (env.currentFailureCount) console.log(' current failures: ' + env.currentFailureCount);
15590
+ // A failure recorded in an envelope that promised nothing is HISTORY, not an outstanding task — the
15591
+ // same rule the smells already apply (`promises` in activitySmells). Printing it as "current failures"
15592
+ // re-created, one line lower, exactly the unfinishable-looking work the evidence/verdict split removed.
15593
+ if (env.currentFailureCount) {
15594
+ console.log(' ' + (env.noRequiredEvidence ? 'failed evidence recorded: ' : 'current failures: ') + env.currentFailureCount);
15595
+ }
15147
15596
  });
15148
15597
  }
15149
15598
 
@@ -15968,10 +16417,18 @@ function releaseAutoUpdateLock() {
15968
16417
  } catch (_) {}
15969
16418
  }
15970
16419
 
16420
+ /**
16421
+ * May this invocation check for, and install, a newer CLI?
16422
+ *
16423
+ * `--json` used to disable it outright, to keep update chatter out of a document a program is parsing.
16424
+ * The effect was that the callers who pass `--json` on every command - which is every AI agent - never
16425
+ * upgraded at all, so each release of better diagnostics reached the population that needed it least.
16426
+ * The chatter now goes to stderr (see `autoUpdateIfNeeded`), which is where a program's stdout contract
16427
+ * says it belongs, and the check stays on.
16428
+ */
15971
16429
  function shouldAutoUpdate(command, flags) {
15972
16430
  if (!command || command === 'help' || command === '--help') return false;
15973
16431
  if (flags && flags['no-auto-update']) return false;
15974
- if (flags && flagEnabled(flags.json)) return false;
15975
16432
  if (process.env.REMITS_CLI_AUTO_UPDATE && AUTO_UPDATE_DISABLED_VALUES.has(String(process.env.REMITS_CLI_AUTO_UPDATE).toLowerCase())) {
15976
16433
  return false;
15977
16434
  }
@@ -15979,10 +16436,31 @@ function shouldAutoUpdate(command, flags) {
15979
16436
  return true;
15980
16437
  }
15981
16438
 
16439
+ /** True when the last registry check is recent enough that another one would only cost latency. */
16440
+ function autoUpdateCheckedRecently() {
16441
+ try {
16442
+ const stamp = JSON.parse(fs.readFileSync(AUTO_UPDATE_STAMP_FILE, 'utf8'));
16443
+ return Number(stamp.checkedAtMs) > Date.now() - AUTO_UPDATE_CHECK_INTERVAL_MS;
16444
+ } catch (_) {
16445
+ return false;
16446
+ }
16447
+ }
16448
+
16449
+ function recordAutoUpdateCheck(latest) {
16450
+ try {
16451
+ ensureSessionDir();
16452
+ fs.writeFileSync(AUTO_UPDATE_STAMP_FILE,
16453
+ JSON.stringify({ checkedAtMs: Date.now(), latest: latest || null }, null, 2));
16454
+ } catch (_) {}
16455
+ }
16456
+
15982
16457
  function autoUpdateIfNeeded(originalArgv, options = {}) {
15983
16458
  const requireSuccess = Boolean(options.requireSuccess);
16459
+ // Update chatter is diagnostics, never part of a `--json` document. stderr keeps both promises at once.
16460
+ const note = (line) => console.error(line);
15984
16461
  let lockAcquired = false;
15985
16462
  try {
16463
+ if (!requireSuccess && autoUpdateCheckedRecently()) return false;
15986
16464
  lockAcquired = acquireAutoUpdateLock();
15987
16465
  if (!lockAcquired) {
15988
16466
  if (requireSuccess) {
@@ -15996,20 +16474,22 @@ function autoUpdateIfNeeded(originalArgv, options = {}) {
15996
16474
  encoding: 'utf8',
15997
16475
  stdio: ['ignore', 'pipe', 'ignore']
15998
16476
  }).trim();
16477
+ recordAutoUpdateCheck(latest);
15999
16478
  if (latest && compareSemver(latest, currentVersion) > 0) {
16000
- console.log('[update] New version available: ' + currentVersion + ' -> ' + latest + '. Installing...');
16479
+ note('[update] New version available: ' + currentVersion + ' -> ' + latest + '. Installing...');
16480
+ // npm's own progress output goes to stderr too: a `--json` caller's stdout must stay one document.
16001
16481
  const install = spawnSync(npmCommand(), ['install', '-g', REMITS_CLI_PACKAGE_NAME + '@latest'], {
16002
- stdio: 'inherit',
16482
+ stdio: ['ignore', process.stderr, process.stderr],
16003
16483
  env: process.env
16004
16484
  });
16005
16485
  if (install.error || install.status !== 0) {
16006
16486
  if (requireSuccess) {
16007
16487
  throw new Error('Auto-update failed; remits-cli start requires the latest published version before continuing.');
16008
16488
  }
16009
- console.log('[update] Update failed; continuing with version ' + currentVersion + '.');
16489
+ note('[update] Update failed; continuing with version ' + currentVersion + '.');
16010
16490
  return false;
16011
16491
  }
16012
- console.log('[update] Updated to ' + latest + '. Re-running command...');
16492
+ note('[update] Updated to ' + latest + '. Re-running command...');
16013
16493
  const rerun = spawnSync(remitsCliCommand(), originalArgv, {
16014
16494
  stdio: 'inherit',
16015
16495
  env: Object.assign({}, process.env, { REMITS_CLI_AUTO_UPDATED: '1' })
@@ -16018,7 +16498,7 @@ function autoUpdateIfNeeded(originalArgv, options = {}) {
16018
16498
  if (requireSuccess) {
16019
16499
  throw new Error('Auto-update succeeded but re-running the updated remits-cli command failed.');
16020
16500
  }
16021
- console.log('[update] Re-run failed; continuing with version ' + currentVersion + '.');
16501
+ note('[update] Re-run failed; continuing with version ' + currentVersion + '.');
16022
16502
  return false;
16023
16503
  }
16024
16504
  process.exitCode = rerun.status === null ? 1 : rerun.status;
@@ -16177,13 +16657,14 @@ function printComponentsHelp(subcommand) {
16177
16657
  console.log(' remits-cli components branch <name> [--json] # overridden/added/removed + drift');
16178
16658
  console.log(' remits-cli components branch <name> --diff <componentId> --component-type <kind> [--json]');
16179
16659
  console.log(' remits-cli components branch <name> --subscribers [--json]');
16660
+ console.log(' remits-cli components branch <name> --copy-to <newBranch> [--dry-run] [--force] [--json]');
16180
16661
  console.log(' remits-cli components branch <name> --subscribe <accountId> [--parent-account <id>] [--domain <host>] [--dry-run] [--confirm-primary-edge]');
16181
16662
  console.log(' remits-cli components branch <name> --unsubscribe <accountId>');
16182
16663
  console.log(' remits-cli components branch <name> --retire [--force]');
16183
16664
  }
16184
16665
 
16185
16666
  function printTestHelp() {
16186
- console.log('Usage: remits-cli test run --test <id|name> [--base-url URL] [--branch stagingScope] [--names "a|b"] [--watch true|false] [--data-mode test|prod] [--as-account ID] [--variant-branch NAME|none] [--json]');
16667
+ console.log('Usage: remits-cli test run --test <id|name> [--base-url URL] [--branch stagingScope] [--names "a|b"] [--watch true|false] [--wait true|false] [--data-mode test|prod] [--as-account ID] [--variant-branch NAME|none] [--json]');
16187
16668
  console.log(' remits-cli test status --task-id <taskId> [--base-url URL] [--account-id ID] [--branch stagingScope] [--data-mode test|prod] [--json]');
16188
16669
  console.log(' remits-cli test runs [--test <id|name>] [--limit 20] [--compare] [--json] # durable run history');
16189
16670
  console.log(' remits-cli test compare --base <taskId> --head <taskId> [--json] # per-case outcome/cost deltas');
@@ -16202,6 +16683,8 @@ function printTestHelp() {
16202
16683
  console.log(' to force production/subscription semantics from a variant checkout.');
16203
16684
  console.log(' --branch changes only the CLI staging namespace for test execution. Pair an unused');
16204
16685
  console.log(' value with --variant-branch none when existing staged entries would shadow DB rows.');
16686
+ console.log(' --watch false disables websocket progress streaming; the CLI still waits for final status.');
16687
+ console.log(' --wait false returns after launch with a task id; test status attaches terminal evidence.');
16205
16688
  console.log(' --json prints only the final status JSON to stdout; banners and progress go to stderr.');
16206
16689
  console.log(' --data-mode prod intentionally targets live production data.');
16207
16690
  console.log(' Every finished run is recorded durably: test status keeps working after the live status');
@@ -16473,10 +16956,11 @@ async function main() {
16473
16956
  console.log(' remits-cli components promotion [<branch>] [--json] [--no-fail] # promotion readiness + ordered next steps');
16474
16957
  console.log(' remits-cli components branches [--json] # committed branch variants for this account');
16475
16958
  console.log(' remits-cli components branch <name> [--diff <componentId> --component-type <kind>] [--subscribers] [--json]');
16959
+ console.log(' remits-cli components branch <name> --copy-to <newBranch> [--dry-run] [--force] [--json]');
16476
16960
  console.log(' remits-cli components branch <name> --subscribe <accountId> [--dry-run] [--confirm-primary-edge]');
16477
16961
  console.log(' remits-cli components branch <name> --unsubscribe <accountId> # return that account to trunk');
16478
16962
  console.log(' remits-cli components branch <name> --retire [--force] # delete the branch\'s overlays');
16479
- console.log(' remits-cli test run --test <id|name> [--base-url URL] [--branch stagingScope] [--names "a|b"] [--watch true|false] [--data-mode test|prod] [--as-account ID] [--variant-branch NAME|none] [--json]');
16963
+ console.log(' remits-cli test run --test <id|name> [--base-url URL] [--branch stagingScope] [--names "a|b"] [--watch true|false] [--wait true|false] [--data-mode test|prod] [--as-account ID] [--variant-branch NAME|none] [--json]');
16480
16964
  console.log(' remits-cli test runs [--test <id|name>] [--compare] # durable run history; compare latest two');
16481
16965
  console.log(' remits-cli test compare --base <taskId> --head <taskId>');
16482
16966
  console.log(' remits-cli corpus import --manifest corpus-manifest.json # seed a test corpus: cases + artifacts');
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@remits/remits-cli",
3
- "version": "0.1.136",
3
+ "version": "0.1.137",
4
4
  "description": "Local CLI for auth, component sync, and live test execution against Remits",
5
5
  "license": "MIT",
6
6
  "private": false,
@@ -48,6 +48,7 @@ reference below, load it before you act, not after something surprises you.
48
48
  | register this session as an agent, or ask why a routed ticket never started | `agent-sessions.md` | `serve` vs `register`, what registration does, worker spawning per agent kind, routing order, the control center, multi-session auth |
49
49
  | investigate live behavior | `investigation.md` | which tool reads which record, correlation keys, reading a record's `content`, HTTP audits, AI activity, node/`localMode`, the production support flows |
50
50
  | call any `mcp_*` tool | `tool-reference.md` | every tool's parameters, semantics, and traps — read the tool's entry before building its input |
51
+ | run, describe, poll, or interrupt an Action from the CLI | `tool-reference.md` → `mcp_run_action` | Action execution through `remits-cli tool --name mcp_run_action`; there is no separate top-level `remits-cli action` / `run action` wrapper |
51
52
  | reason about repo discovery, auth state, or what a command actually sent | `cli-state.md` | the global `~/.remits-cli/` control plane vs per-repo `./.remits-cli/`, and which file answers which question |
52
53
  | need an exact flag, or the auth / host / async surface | `command-reference.md` | authentication, host vs data mode, the two async mechanisms, hierarchy-scoped reads, the full command list, prod banners |
53
54
  | give up on something that misbehaved | `troubleshooting.md` | the symptom→fix table, the two kinds of escalation, and the escalation bundle |
@@ -408,6 +408,31 @@ from that checkout). Its first real sync (or commit) needs `--create-variant-bra
408
408
  variants or a subscriber the platform treats it as a feature branch and refuses to land it, so creating a
409
409
  new variant branch is always a stated decision, never a side effect of a feature branch's name.
410
410
 
411
+ For a nested branch cut from an existing variant branch, seed the platform overlays explicitly after the git
412
+ branch is created and pushed:
413
+
414
+ ```bash
415
+ remits-cli components branch forked --copy-to sandbox --dry-run
416
+ remits-cli components branch forked --copy-to sandbox
417
+ remits-cli components sync --safe --branch sandbox
418
+ ```
419
+
420
+ The copy is a DB overlay bootstrap, not a git write. It makes inherited overlays from `forked` already
421
+ `storedCurrent` on the first `sandbox` sync, so the safe diff gate only has to account for the new branch's
422
+ real file changes.
423
+
424
+ **Copy before you subscribe, or the copy needs `--force`.** The copy refuses a target that already has
425
+ stored overlays *or* live subscribers, and that second guard fires even when the target has no overlays at
426
+ all — a branch somebody is already resolving is code those accounts are running right now, and replacing it
427
+ has to be a stated decision. Seeding first and subscribing after keeps the plain form working. The copy is
428
+ also all-or-nothing: the delete of any replaced overlays and every inserted row share one transaction, so a
429
+ failure leaves the target exactly as it was rather than half-populated.
430
+
431
+ Run it as `--dry-run` first. The plan reports `plannedCopies` — the number the real run will write — which
432
+ is not always the source branch's row count: a source overlay that has become identical to trunk in both
433
+ content and metadata is sparse, is not stored, and is listed as skipped instead of being silently dropped
434
+ from the total.
435
+
411
436
  **Trunk moving also invalidates a branch.** Variant sparseness compares branch content against *current*
412
437
  trunk, so a trunk change can make an overlay obsolete without the branch changing at all. A trunk sync
413
438
  drops the affected branches' cached sync verdicts, so the next `components sync` on the branch really
@@ -15,6 +15,8 @@
15
15
  - [Hierarchy-scoped tool reads](#hierarchy-scoped-tool-reads)
16
16
  - [Data Mode](#data-mode)
17
17
  - [Command Reference](#command-reference)
18
+ - [Reading a failing run](#reading-a-failing-run)
19
+ - [A pass count only means something within one data lane](#a-pass-count-only-means-something-within-one-data-lane)
18
20
  - [Verification envelopes](#verification-envelopes)
19
21
  - [Staging modes: workset vs full snapshot](#staging-modes-workset-vs-full-snapshot)
20
22
  - [Prod banners and retryable failures](#prod-banners-and-retryable-failures)
@@ -236,12 +238,13 @@ remits-cli components branches [--json] # branche
236
238
  remits-cli components branch <name> [--json] # one branch: owner account, overridden / added / removed, drift flags, subscribers
237
239
  remits-cli components branch <name> --diff <componentId> --component-type <kind> [--json]
238
240
  remits-cli components branch <name> --subscribers [--json]
241
+ remits-cli components branch <name> --copy-to <newBranch> [--dry-run] [--force] [--json] # seed a new variant branch with this branch's stored overlays; force if target has overlays/subscribers
239
242
  remits-cli components branch <name> --subscribe <accountId> [--parent-account <id>] [--domain <host>] [--dry-run] [--confirm-primary-edge] # make an account resolve this branch
240
243
  remits-cli components branch <name> --unsubscribe <accountId> # return that account to trunk
241
244
  remits-cli components branch <name> --retire [--force] # delete the branch's overlays
242
- remits-cli test run --test <id|name> [--branch <stagingScope>] [--names "a|b"] [--watch true|false] [--data-mode test|prod] [--as-account <ID>] [--variant-branch <name|none>] [--json]
245
+ remits-cli test run --test <id|name> [--branch <stagingScope>] [--names "a|b"] [--watch true|false] [--wait true|false] [--data-mode test|prod] [--as-account <ID>] [--variant-branch <name|none>] [--json]
243
246
  remits-cli test status --task-id <taskId> [--branch <stagingScope>] [--data-mode test|prod] [--json] # falls back to the DURABLE run record once the live status expires
244
- remits-cli test runs [--test <id|name>] [--limit 20] [--compare] [--as-account <ID>] [--json] # durable run history: pass counts, live AI cost, lane, content hash
247
+ remits-cli test runs [--test <id|name>] [--limit 20] [--compare] [--all-lanes] [--as-account <ID>] [--json] # durable run history: pass counts, live AI cost, lane, content hash
245
248
  remits-cli test compare --base <taskId> --head <taskId> [--json] # per-case improved / regressed / changed, cost and world deltas
246
249
  remits-cli corpus import --manifest corpus-manifest.json [--corpus <name>] [--as-account <ID>] [--data-mode test|prod --confirm-prod] [--json] # seed an evaluation corpus: cases + immutable artifacts; idempotent by caseKey
247
250
  remits-cli corpus cases --corpus <name> [--split S] [--tag T|--tags T,U] [--key K|--keys K,L] [--include-values] [--include-retired] [--limit N] [--json]
@@ -252,6 +255,54 @@ remits-cli corpus consistency --corpus <name> --case <caseKey> [--json]
252
255
  remits-cli corpus retire --corpus <name> --case <caseKey> [--case ...] [--restore] [--json] # drop a case from future runs; past measurements stay readable
253
256
  ```
254
257
 
258
+ `test run` starts a server-side task immediately. By default the CLI process polls until the task is
259
+ terminal; `--watch false` disables websocket progress streaming but still waits. Pass `--wait false` to
260
+ return after launch with the task id. A single run with `--names "a|b|c"` selects cases into one suite task,
261
+ and those cases execute sequentially in declaration order. To overlap independent slow cases, launch
262
+ separate `remits-cli test run --names "<case>" --wait false` commands from the same staged lane and keep the
263
+ printed task ids; complete each proof with `test status --task-id <id>`. Avoid parallel runs for cases that
264
+ share mutable fixtures, suite-level side effects, or undeclared live AI/provider calls.
265
+
266
+ ### Reading a failing run
267
+
268
+ `test run` and `test status` print a **verdict**, not the run. The whole run object used to be dumped as
269
+ pretty JSON in human mode - 220 KB for one 92-case suite, most of it identifiers repeated once per case - so
270
+ the command an agent runs most often was the one that spent its remaining room to think. Every byte is still
271
+ there behind `--json`, and behind `test status --task-id <id> --json` once the live status expires.
272
+
273
+ What you get instead, and what to do with it:
274
+
275
+ ```text
276
+ Cases: 59 passed, 33 failed, 92 total
277
+
278
+ Failure roots (33 failed case(s), 21 distinct root(s)):
279
+ 10x assert statement.data.processingStatus == 'Analyzed'
280
+ e.g. bundled UK statements (+9 more)
281
+ 2x assert statement.data.feeBreakdownChecked == true
282
+ e.g. 7851 Flat Rate (+1 more)
283
+ Largest root first: remits-cli test run --test "Statement Reader Calculations" --names "bundled UK statements"
284
+ ```
285
+
286
+ **Thirty-three failures are not thirty-three problems.** Cases are grouped by their assertion root - the
287
+ failure with the particulars (ids, numbers, Groovy's `Expression:`/`Values:` decoration) removed - largest
288
+ group first, and one pivot block is printed per root rather than per case. Fix the largest root against one
289
+ named case, then re-run the whole suite. Re-running the suite before you have a root is how a session spends
290
+ an afternoon at the same pass count.
291
+
292
+ ### A pass count only means something within one data lane
293
+
294
+ `test runs` is scoped to the data lane of the command by default, and says so:
295
+
296
+ ```text
297
+ Data lane: test (6 more run(s) in the other lane; --all-lanes to include)
298
+ ```
299
+
300
+ A run's `dataMode` **is** the lane it was written in, so `64/92` in the prod lane and `59/92` in the test lane
301
+ are two different facts, not a regression. `--compare` (and `test runs --compare`) picks the newest two runs
302
+ of the same **shape** - same data lane, branch, workspace and case count - and says how many newer runs it
303
+ skipped; it refuses rather than comparing across worlds. `test status --data-mode prod` on a run recorded in
304
+ the test lane returns the run's real lane and now says that is what happened.
305
+
255
306
  A run labelled **`AI MOCKED`** replayed every AI turn from `aiMock`: it produces the same outcomes, scores and
256
307
  metrics a measured run would, so read that label before treating green as evidence. `compare` warns when base and
257
308
  head used different AI modes; `consistency` warns before calling a case unstable when its runs mixed modes.
@@ -16,6 +16,7 @@
16
16
  - [When staged overrides apply](#when-staged-overrides-apply)
17
17
  - [Diagnosing which version is in play](#diagnosing-which-version-is-in-play)
18
18
  - [Working alongside other agents: the staging WORKSPACE](#working-alongside-other-agents-the-staging-workspace)
19
+ - [Keeping your remits-cli current](#keeping-your-remits-cli-current)
19
20
  - [A lane holds an OVERLAY; your workset is a different number](#a-lane-holds-an-overlay-your-workset-is-a-different-number)
20
21
  - [Stage / sync / clear with remits-cli](#stage--sync--clear-with-remits-cli)
21
22
  - [Stale after sync / commit (the in-memory compile cache)](#stale-after-sync--commit-the-in-memory-compile-cache)
@@ -250,6 +251,18 @@ index with `remits-cli workstream status`.
250
251
  `null` means unknown — a lane staged by an older CLI, or a branch whose last sync is not recorded —
251
252
  never "fresh".
252
253
 
254
+ ### Keeping your remits-cli current
255
+
256
+ The diagnostics in these references only exist in the CLI that prints them. An older install does not print a
257
+ worse version of them — it prints nothing, with no error, so nothing tells you what you are not being shown.
258
+ `components stage` and `test run` warn when yours is behind:
259
+
260
+ ```text
261
+ [remits-cli 0.1.120 -> 0.1.136] ... npm install -g @remits/remits-cli@latest
262
+ ```
263
+
264
+ Upgrade when you see it. Auto-update handles it for you on most commands, including `--json` ones.
265
+
253
266
  ### A lane holds an OVERLAY; your workset is a different number
254
267
 
255
268
  This is the distinction that decides whether a lane is legible to anyone but you.
@@ -20,6 +20,7 @@
20
20
  - [Stage your workset, not the whole repo](#stage-your-workset-not-the-whole-repo)
21
21
  - [Three numbers, three questions](#three-numbers-three-questions)
22
22
  - [Step 4: Verify the Change](#step-4-verify-the-change)
23
+ - [Many failures are usually few causes](#many-failures-are-usually-few-causes)
23
24
  - [Step 5: Iterate If Needed](#step-5-iterate-if-needed)
24
25
  - [Step 6: Update Documentation](#step-6-update-documentation)
25
26
  - [Temporary Experiment Workflow](#temporary-experiment-workflow)
@@ -411,14 +412,54 @@ Tests run on the platform against your staged snapshot. They stream results in r
411
412
  id + content hash, platform sync vs local HEAD. If it is not the world you meant, stop: the result will be
412
413
  about a different world.
413
414
 
415
+ The platform launch is asynchronous. By default `remits-cli test run` waits for its task to finish, and any
416
+ cases selected by `--names "a|b"` execute sequentially inside that suite task. If several slow cases are
417
+ independent, stage once, then start separate `test run --names "<case>" --wait false` commands from the same
418
+ lane and keep the task ids they print. Complete each proof with
419
+ `remits-cli test status --task-id <id>`; that terminal status read records and attaches the final `test_run`
420
+ evidence. Do not parallelize cases that mutate the same fixture, rely on shared suite setup state, or make
421
+ undeclared live AI/provider calls.
422
+
414
423
  If cases fail, read the printed summary and pivots first: each case has an `outcome`
415
424
  (`failed`, `error`, `budget_exceeded`, `provider_unavailable`, …) with its reason, timing, trace id, AI usage
416
425
  (live vs mocked, cost), bounded `report(...)` diagnostics, live HTTP signals, and resolved component
417
426
  provenance. Fix the code, re-stage, and re-run only after those pivots explain the failure.
418
427
 
428
+ ##### Many failures are usually few causes
429
+
430
+ When several cases fail, the run leads with **failure roots** — the failures grouped by their assertion with
431
+ the particulars (ids, numbers, Groovy's `Expression:`/`Values:` decoration) stripped out, largest group first,
432
+ one pivot block per root rather than per case:
433
+
434
+ ```text
435
+ Failure roots (33 failed case(s), 21 distinct root(s)):
436
+ 10x assert statement.data.processingStatus == 'Analyzed'
437
+ e.g. bundled UK statements (+9 more)
438
+ Largest root first: remits-cli test run --test "Statement Reader Calculations" --names "bundled UK statements"
439
+ ```
440
+
441
+ **Work the largest root against ONE named case, then re-run the suite.** A full re-run after every edit is the
442
+ most expensive way to learn nothing: a real account spent two days and twenty-five runs holding a suite at
443
+ 56–59 of 92 while ten of its thirty-three failures were one cause. The narrow run is seconds, tells you
444
+ whether the cause moved, and leaves your context for the actual reasoning.
445
+
446
+ Two failure roots mean "stop and look elsewhere", not "iterate harder":
447
+
448
+ - **`aiMock ctx.replay(...) found no usable stored provider response`** — the message says which of three
449
+ things happened. *No rows at all* for that sessionId means the stored session this fixture borrows is not
450
+ on this platform and will not come back; the case cannot pass until the mock builds its own response with
451
+ `ctx.toolCall(...)` / `ctx.content(...)`. Do not re-run it. (Replaying a session that IS there renews it,
452
+ so a suite that runs regularly keeps its fixtures.)
453
+ - **`Method too large` / `Class too large` / a synthetic `_closureNN`** — a JVM limit on one method body, not
454
+ a bug in the line it names. Staging now warns *before* the refusal and names the closure's source line span.
455
+ Split that body; see `development-guide.md` → *Keep Component Bodies Split*.
456
+
419
457
  Every finished run is recorded durably: `remits-cli test status --task-id <id>` answers after the live status
420
- expires, `remits-cli test runs --test <name> --compare` compares the latest run with the previous one, and
421
- `remits-cli test compare --base <id> --head <id>` compares any two. For evaluation suites (a `corpus(name)` of
458
+ expires, `remits-cli test runs --test <name> --compare` compares the latest run with the newest **comparable**
459
+ one, and `remits-cli test compare --base <id> --head <id>` compares any two. Comparable means the same data
460
+ lane, branch, workspace and case count: a run's `dataMode` is the lane it was written in, so `64/92` in prod
461
+ and `59/92` in test are two facts and not a trend. `test runs` lists one lane at a time and names it
462
+ (`--all-lanes` to see both). For evaluation suites (a `corpus(name)` of
422
463
  cases seeded with `remits-cli corpus import`, one case per corpus case, intentional live AI inside
423
464
  `withAiBudget(...)`, measurements compared with `remits-cli corpus compare` / `corpus consistency`), read
424
465
  `guides/test-components.md` → *Evaluation Suites And Corpora*.
@@ -76,6 +76,9 @@ remits-cli tool --account-id 21 --as-account 37 --target-account 37 --name mcp_r
76
76
  # poll by run id: --account-id 21 --as-account 37 --target-account 37 --input '{"controlAction":"status","actionRunId":"my-stable-run-id"}'
77
77
  ```
78
78
 
79
+ That is the CLI Action-run surface. The command is still `remits-cli tool`; this CLI does not have a
80
+ separate top-level `remits-cli action`, `remits-cli actions`, or `remits-cli run action` wrapper.
81
+
79
82
  For long tools that lack their own async mode, use the CLI transport async (`--async true`), optionally with
80
83
  `--wait true` to poll locally, and `remits-cli tool status --call-id <callId>`. Do not stack both mechanisms
81
84
  (see `command-reference.md` → *Tool Execution Lifecycle*). Use `--timeout-ms <ms>` only to adjust the per-request client timeout; it is
@@ -403,12 +406,18 @@ Front-stage references:
403
406
  | `page` | no | 1-based page number. Default: `1` |
404
407
  | `pageSize` | no | Results per page. Default: `25`, max: `100` |
405
408
  | `summaryOnly` | no | Detail mode: return the MAP (record index + stats + timeline) with no payloads. Same as `parts:["index"]` |
406
- | `parts` | no | Detail parts: `index`, `stats`, `transcript`, `tool_calls`, and record sections `full`, `conversation_messages`, `system`, `response`, `tools` |
409
+ | `compact` | no | Detail mode: after reading the MAP, open selected records without repeating the `records` index and `timeline`; keeps `groupingSummary`, `representativeSession`, `stats`, selected ids, and requested payload sections |
410
+ | `payloadOnly` | no | Alias for `compact` |
411
+ | `parts` | no | Detail parts: `index`, `map`, `records`, `stats`, `transcript`, `tool_calls`, and record sections `full`, `conversation_messages`, `system`, `response`, `tools` |
407
412
  | `sections` | no | Alias for `parts` |
408
413
  | `recordIds` | no | Detail mode: open only these request/response record IDs |
409
414
  | `toolCallIds` | no | Detail mode: return the FULL exact input/result from `ai_tool_call` for these tool-call ids (what the tool PRODUCED — see the lens caveat) |
410
415
  | `consolidateContext` | no | When `true`, collapses repeated XML-like prompt context into a consolidated section |
411
416
 
417
+ Detail responses expose both lenses: `groupingSummary` is the authoritative grouping-wide summary
418
+ (counts, lanes, cost, status), while `representativeSession` names the concrete session used for
419
+ owner/runtime metadata. `session` remains only as a compatibility alias for older callers.
420
+
412
421
  Spend: each search row carries `liveRequestCount`, `mockedRequestCount`, `lanes`, `testTaskId` and
413
422
  `estimatedCost`/`estimatedCostUsd` = **live spend only** (a mocked turn is never priced, even when it replays a
414
423
  recording with a cost). `pageTotals` sums the page. Turns recorded before the platform stored the mock flag are
@@ -416,12 +425,16 @@ recording with a cost). `pageTotals` sums the page. Turns recorded before the pl
416
425
  is refused rather than silently widened; the applied bounds are echoed as `filters.rangeStart`/`rangeEnd`.
417
426
  To audit one Test run: `{"action":"search","testTaskId":"<taskId>","pageSize":100}`.
418
427
 
419
- Audit flow: `action:"search"` to find the grouping → `action:"detail"` + `summaryOnly:true` for the MAP → re-call detail with `recordIds`/`toolCallIds` + `parts` to open exactly what you need. Prefer the map → open flow over a full-detail dump. Remember the two-lens rule: `tool_calls`/`toolCallIds` is what the tool PRODUCED; `conversation_messages`/`transcript` is what the AI CONSUMED (after any `_offload`/`_hideResult`/`_message`/supersede/evict transform).
428
+ Audit flow: `action:"search"` to find the grouping → `action:"detail"` + `summaryOnly:true` for the MAP → re-call detail with `recordIds`/`toolCallIds` + `parts` to open exactly what you need, usually with `compact:true` once the map has chosen the row. Prefer the map → open flow over a full-detail dump. Remember the two-lens rule: `tool_calls`/`toolCallIds` is what the tool PRODUCED; `conversation_messages`/`transcript` is what the AI CONSUMED (after any `_offload`/`_hideResult`/`_message`/supersede/evict transform).
420
429
 
421
430
  ### `mcp_run_action`
422
431
  Run an Action on a target account, with explicit prod/test data mode, optional staged branch resolution, and
423
432
  staged-vs-DB provenance in the result.
424
433
 
434
+ Invoke it with `remits-cli tool --name mcp_run_action`. Despite the natural shorthand "run an Action", there
435
+ is no separate top-level `remits-cli action` / `actions` command and no `remits-cli run action` wrapper in
436
+ this CLI build.
437
+
425
438
  Describe the Action first when the input shape is not obvious. This does not execute the Action:
426
439
 
427
440
  ```bash
@@ -832,6 +845,10 @@ Common uses:
832
845
  - `remits-cli components branch <name> --diff <id> --component-type <kind>` — compare one variant against
833
846
  current trunk.
834
847
  - `remits-cli components branch <name> --subscribers` — list accounts resolving that branch.
848
+ - `remits-cli components branch <name> --copy-to <newBranch> [--dry-run] [--force]` — seed a new variant
849
+ branch with the source branch's stored overlays before the first safe sync of the new branch. Dry-run
850
+ reports the copy plan without writes; force is required when the target already has overlays or live
851
+ subscribers.
835
852
 
836
853
  ### `mcp_cache`
837
854
  Bounded read-only investigation of the platform Redis keyspace — the way to see exactly what a staged