@remits/remits-cli 0.1.135 → 0.1.137

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/index.js CHANGED
@@ -4,18 +4,18 @@
4
4
  ## Table of Contents
5
5
 
6
6
  - L22 Runtime Bootstrap And Shared State
7
- - L119 Sessions, Account Resolution, And Production Guards
8
- - L603 Local State, Workspaces, And Verification Evidence
9
- - L2474 Account Repos, Guide Sync, And Platform Repo
10
- - L3282 Component Discovery And HTTP Logging
11
- - L3923 Skill Delivery And TOC Resolution
12
- - L4233 Auth And Component Staging
13
- - L4848 Component Summaries, Status, And Sync Gates
14
- - L6917 Branches, Promotion, Commit, And Test Runs
15
- - L8600 Tokens, Tools, Verification, And Config
16
- - L9980 Service Dashboard And WebSocket Listener
17
- - L12101 Agent And Ticket Workflows
18
- - L15573 Help, Auto Update, And Command Dispatch
7
+ - L123 Sessions, Account Resolution, And Production Guards
8
+ - L615 Local State, Workspaces, And Verification Evidence
9
+ - L2652 Account Repos, Guide Sync, And Platform Repo
10
+ - L3460 Component Discovery And HTTP Logging
11
+ - L4107 Skill Delivery And TOC Resolution
12
+ - L4417 Auth And Component Staging
13
+ - L5032 Component Summaries, Status, And Sync Gates
14
+ - L7104 Branches, Promotion, Commit, And Test Runs
15
+ - L9193 Tokens, Tools, Verification, And Config
16
+ - L10620 Service Dashboard And WebSocket Listener
17
+ - L12742 Agent And Ticket Workflows
18
+ - L16337 Help, Auto Update, And Command Dispatch
19
19
  */
20
20
 
21
21
  /*
@@ -58,6 +58,10 @@ const SERVICE_STATE_FILE = path.join(SESSION_DIR, 'service-state.json');
58
58
  const ISSUES_DIR = path.join(SESSION_DIR, 'issues');
59
59
  const WEBSOCKET_STATE_FILE = path.join(SESSION_DIR, 'websocket-state.json');
60
60
  const AUTO_UPDATE_LOCK_DIR = path.join(SESSION_DIR, 'auto-update.lock');
61
+ const AUTO_UPDATE_STAMP_FILE = path.join(SESSION_DIR, 'auto-update-check.json');
62
+ // `npm view` is a network round trip. It used to run on EVERY command, which is a per-command latency tax
63
+ // on the agents that issue the most commands.
64
+ const AUTO_UPDATE_CHECK_INTERVAL_MS = 6 * 60 * 60 * 1000;
61
65
  const AUTO_UPDATE_DISABLED_VALUES = new Set(['0', 'false', 'no', 'off']);
62
66
  const REMITS_CLI_PACKAGE_NAME = '@remits/remits-cli';
63
67
  const DEFAULT_DATA_MODE = 'test';
@@ -154,6 +158,14 @@ function flagEnabled(value) {
154
158
  return value === true || value === 'true' || value === '1' || value === 'yes';
155
159
  }
156
160
 
161
+ function flagDisabled(value) {
162
+ return value === false || value === 'false' || value === '0' || value === 'no';
163
+ }
164
+
165
+ function waitForTestCompletion(flags = {}) {
166
+ return !(flagDisabled(flags.wait) || flagEnabled(flags['no-wait']) || flagEnabled(flags.noWait));
167
+ }
168
+
157
169
  function withStdoutRoutedToStderr(enabled, fn) {
158
170
  if (!enabled) return fn();
159
171
  const originalLog = console.log;
@@ -626,6 +638,7 @@ function localStatePaths(cwd) {
626
638
  toolResponsesDir: path.join(actorDir, 'tool-responses'),
627
639
  verificationDir: path.join(actorDir, 'verification'),
628
640
  diagnosticsDir: path.join(actorDir, 'diagnostics'),
641
+ evidenceFile: path.join(actorDir, 'evidence.jsonl'),
629
642
  activeVerificationFile: path.join(actorDir, 'verification', 'active'),
630
643
  activeVerificationContextsFile: path.join(actorDir, 'verification', 'active-contexts.json'),
631
644
  currentSessionFile: path.join(actorDir, 'current-session.txt'),
@@ -897,6 +910,22 @@ function printLocalStateWarnings(cwd, flags = {}) {
897
910
  warnings.forEach((warning) => console.log(' - ' + warning));
898
911
  }
899
912
 
913
+ /**
914
+ * "Your remits-cli is older than this platform's" - printed once per process, from ONE place.
915
+ *
916
+ * The platform attaches `cliAdvisory` to the two responses every working loop passes through (stage and
917
+ * test-run start) and returns nothing at all when the caller is current, so this renders only when there
918
+ * is something to act on. stderr, because a `--json` caller's stdout is a document.
919
+ */
920
+ let cliAdvisoryPrinted = false;
921
+ function printCliAdvisory(response) {
922
+ const advisory = response && response.cliAdvisory;
923
+ if (!advisory || cliAdvisoryPrinted) return;
924
+ cliAdvisoryPrinted = true;
925
+ console.error('');
926
+ console.error('[remits-cli ' + (advisory.current || '?') + ' -> ' + (advisory.expected || '?') + '] ' + advisory.message);
927
+ }
928
+
900
929
  function printLocalCommandContext(cwd, flags = {}, context = {}) {
901
930
  const paths = localStatePaths(cwd);
902
931
  const branchName = context.branchName || flags.branch || safeGitValue(cwd, 'git rev-parse --abbrev-ref HEAD') || 'unknown';
@@ -1999,9 +2028,145 @@ async function printVerificationRunProbe(api, cwd, session, accountId, envelopeI
1999
2028
  return result;
2000
2029
  }
2001
2030
 
2031
+ /**
2032
+ * The always-on evidence trail.
2033
+ *
2034
+ * <p>One world-stamped line per evidence-producing command, written whether or not any verification envelope
2035
+ * exists. This is the answer to "what have I actually run, and in which world?" - the question agents were
2036
+ * opening envelopes to answer, and an envelope is a far heavier instrument than that question needs. It is
2037
+ * per-actor, so two agents in one checkout never read each other's trail, and it is append-only with a bounded
2038
+ * tail so it cannot grow without limit.</p>
2039
+ *
2040
+ * <p>Recording is deliberately decision-free: there is no flag to remember, no state to start, and nothing an
2041
+ * agent can fail to satisfy.</p>
2042
+ */
2043
+ const EVIDENCE_TRAIL_MAX_LINES = 2000;
2044
+
2045
+ function recordEvidenceEntry(cwd, packet, context = {}) {
2046
+ try {
2047
+ const paths = ensureLocalState(cwd);
2048
+ const world = (packet && packet.world && typeof packet.world === 'object') ? packet.world : {};
2049
+ const hasSuccess = packet && Object.prototype.hasOwnProperty.call(packet, 'success');
2050
+ const entry = {
2051
+ ts: new Date().toISOString(),
2052
+ type: packet && packet.type || 'unknown',
2053
+ claim: packet && packet.claim || null,
2054
+ success: hasSuccess ? packet.success : true,
2055
+ pending: packet && packet.pending === true || (hasSuccess && packet.success === null),
2056
+ world: {
2057
+ repoAccountId: world.repoAccountId != null ? world.repoAccountId : context.accountId || null,
2058
+ executionAccountId: world.executionAccountId != null ? world.executionAccountId : (world.accountId != null ? world.accountId : null),
2059
+ targetAccountId: world.targetAccountId != null ? world.targetAccountId : null,
2060
+ dataMode: world.dataMode || context.dataMode || null,
2061
+ branchName: world.branchName || world.gitBranch || context.branchName || null,
2062
+ componentBranch: world.componentBranch || world.variantBranch || null,
2063
+ workspace: world.workspace !== undefined ? world.workspace : (context.workspace !== undefined ? context.workspace : null),
2064
+ sourceLayer: world.sourceLayer || null,
2065
+ laneContentHash: world.laneContentHash || world.stagedOverlayHash || null,
2066
+ host: world.host || context.baseUrl || null
2067
+ },
2068
+ ref: {
2069
+ packetId: packet && packet.packetId || null,
2070
+ taskId: (packet && packet.test && packet.test.taskId) || (packet && packet.status && packet.status.taskId) || null,
2071
+ callId: (packet && packet.tool && packet.tool.callId) || null,
2072
+ responseFile: (packet && packet.tool && packet.tool.responseFile) || null
2073
+ },
2074
+ // Present when an envelope happened to be active. Never required, and never something to chase.
2075
+ envelopeId: context.envelopeId || null,
2076
+ claims: Array.isArray(packet && packet.claims) ? packet.claims : undefined
2077
+ };
2078
+ fs.appendFileSync(paths.evidenceFile, JSON.stringify(entry) + '\n');
2079
+ trimEvidenceTrail(paths.evidenceFile);
2080
+ return entry;
2081
+ } catch (_) {
2082
+ // A trail write must never fail the command it is describing.
2083
+ return null;
2084
+ }
2085
+ }
2086
+
2087
+ function evidenceOutcome(row) {
2088
+ if (row && (row.pending === true || row.success === null)) return 'pending';
2089
+ return row && row.success === false ? 'failed' : 'ok';
2090
+ }
2091
+
2092
+ function evidenceOutcomeLabel(row) {
2093
+ const outcome = evidenceOutcome(row);
2094
+ if (outcome === 'failed') return 'FAILED ';
2095
+ if (outcome === 'pending') return 'PENDING ';
2096
+ return '';
2097
+ }
2098
+
2099
+ function trimEvidenceTrail(file) {
2100
+ try {
2101
+ const lines = fs.readFileSync(file, 'utf8').split('\n').filter(Boolean);
2102
+ if (lines.length <= EVIDENCE_TRAIL_MAX_LINES) return;
2103
+ atomicWriteFile(file, lines.slice(lines.length - EVIDENCE_TRAIL_MAX_LINES).join('\n') + '\n');
2104
+ } catch (_) { /* best effort */ }
2105
+ }
2106
+
2107
+ function readEvidenceTrail(cwd, limit = 50) {
2108
+ try {
2109
+ const paths = localStatePaths(cwd);
2110
+ if (!fs.existsSync(paths.evidenceFile)) return [];
2111
+ const lines = fs.readFileSync(paths.evidenceFile, 'utf8').split('\n').filter(Boolean);
2112
+ return lines.slice(Math.max(0, lines.length - limit)).map((line) => {
2113
+ try { return JSON.parse(line); } catch (_) { return null; }
2114
+ }).filter(Boolean);
2115
+ } catch (_) {
2116
+ return [];
2117
+ }
2118
+ }
2119
+
2120
+ /**
2121
+ * WHERE a command ran: accounts, lane, branch, workspace. Deliberately NOT `sourceLayer` or
2122
+ * `laneContentHash` - those legitimately differ between two commands in the same world (a stage is `staged`,
2123
+ * the test that follows resolves `db`), so keying the grouping on them split one world into several and made
2124
+ * two entries from the same context look like two different places. They belong on the entry.
2125
+ */
2126
+ function describeEvidenceWorld(world) {
2127
+ const w = world || {};
2128
+ const accounts = w.executionAccountId && String(w.executionAccountId) !== String(w.repoAccountId)
2129
+ ? 'account ' + w.repoAccountId + ' -> runs as ' + w.executionAccountId
2130
+ : 'account ' + (w.repoAccountId == null ? '?' : w.repoAccountId);
2131
+ return [
2132
+ accounts,
2133
+ w.targetAccountId && String(w.targetAccountId) !== String(w.executionAccountId) ? 'targets ' + w.targetAccountId : null,
2134
+ w.dataMode ? w.dataMode + ' lane' : null,
2135
+ w.branchName ? 'branch ' + w.branchName : null,
2136
+ w.componentBranch ? 'components ' + w.componentBranch : null,
2137
+ 'ws:' + (w.workspace || 'shared')
2138
+ ].filter(Boolean).join(' · ');
2139
+ }
2140
+
2141
+ function claimIdsFromFlags(flags) {
2142
+ return []
2143
+ .concat((flags && flags.claim) || [])
2144
+ .concat((flags && flags.claims) || [])
2145
+ .filter(Boolean)
2146
+ .flatMap((value) => String(value).split(','))
2147
+ .map((value) => value.trim())
2148
+ .filter(Boolean);
2149
+ }
2150
+
2151
+ function stampPacketClaims(packet, flags) {
2152
+ const claims = claimIdsFromFlags(flags);
2153
+ if (!claims.length) return packet;
2154
+ if (!Array.isArray(packet.evidenceCategories)) packet.evidenceCategories = [];
2155
+ claims.forEach((claim) => {
2156
+ if (!packet.evidenceCategories.includes(claim)) packet.evidenceCategories.push(claim);
2157
+ });
2158
+ packet.claims = claims;
2159
+ return packet;
2160
+ }
2161
+
2002
2162
  async function appendVerificationPacket(api, cwd, session, accountId, flags, packet, options = {}) {
2003
2163
  const context = activeVerificationContext(cwd, flags, session, accountId);
2004
2164
  const envelopeId = verificationEnvelopeIdForCommand(cwd, flags, context);
2165
+ // The trail is written FIRST and unconditionally. Evidence recording must not depend on an agent having
2166
+ // remembered to start anything: an envelope is an optional verdict on top of this, never the thing that
2167
+ // makes a command's world durable.
2168
+ stampPacketClaims(packet, flags);
2169
+ recordEvidenceEntry(cwd, packet, Object.assign({}, context, { accountId, envelopeId }));
2005
2170
  if (!envelopeId) return null;
2006
2171
  const finalPacket = Object.assign({
2007
2172
  packetId: crypto.randomUUID(),
@@ -2067,7 +2232,7 @@ function printVerificationSummaryLine(env) {
2067
2232
  const health = env.health ? ' health=' + env.health : '';
2068
2233
  const failures = env.currentFailureCount ? ' currentFailures=' + env.currentFailureCount : (env.failedCount ? ' failed=' + env.failedCount : '');
2069
2234
  const required = ' required=' + (env.satisfiedCount || 0) + '/' + (env.requiredCount || 0);
2070
- const noManifest = env.noRequiredEvidence ? ' no-manifest' : '';
2235
+ const noManifest = env.noRequiredEvidence ? ' evidence-log' : '';
2071
2236
  const stale = env.staleCount ? ' stale=' + env.staleCount : '';
2072
2237
  const changed = env.requirementsChangedAfterEvidence ? ' requirements-changed' : '';
2073
2238
  const lane = [env.branchName || 'unknown-branch', env.workspace ? 'ws:' + env.workspace : 'shared', env.dataMode || null, env.sourceLayer || null]
@@ -2084,7 +2249,7 @@ function printVerificationEnvelope(envelope, fallbackEnvelopeId) {
2084
2249
  const evaluation = envelope.evaluation || {};
2085
2250
  if (evaluation.health) console.log('Health:', evaluation.health);
2086
2251
  if (evaluation.noRequiredEvidence) {
2087
- console.log('Required evidence: none declared - evidence packets do not prove a complete acceptance contract.');
2252
+ console.log('Required evidence: none declared - evidence log only; no verdict is outstanding.');
2088
2253
  } else {
2089
2254
  console.log('Required evidence:', (evaluation.satisfiedCount || 0) + '/' + (evaluation.requiredCount || 0));
2090
2255
  }
@@ -2143,7 +2308,7 @@ function compactVerificationEnvelope(envelope, fallbackEnvelopeId) {
2143
2308
  const nextActions = [];
2144
2309
 
2145
2310
  if (evaluation.noRequiredEvidence) {
2146
- nextActions.push('Attach a manifest or required evidence contract; packets are context, not proof of completion.');
2311
+ nextActions.push('No verdict was requested. Leave this as an evidence log, or add `verify claim <id> --text "..."` only if someone needs a checkable verdict.');
2147
2312
  }
2148
2313
  if (failures.length) {
2149
2314
  nextActions.push('Inspect and replace the current failing evidence packet(s).');
@@ -2227,14 +2392,27 @@ function printEnvelopeWarnings(envelope) {
2227
2392
  console.log('Envelope warnings:');
2228
2393
  warnings.forEach((warning) => {
2229
2394
  console.log('- ' + (warning.message || warning.code || JSON.stringify(warning)));
2395
+ if (warning.command) console.log(' ' + warning.command);
2230
2396
  if (Array.isArray(warning.envelopes) && warning.envelopes.length) {
2231
2397
  warning.envelopes.forEach((env) => {
2232
- console.log(' ' + (env.envelopeId || '(no id)') + (env.status ? ' ' + env.status : ''));
2398
+ // Say whether the duplicate is in THIS world. A restart in a genuinely different lane or branch is
2399
+ // sometimes exactly right; a second envelope for the same goal in the same world is the loop.
2400
+ const world = env.world && typeof env.world === 'object'
2401
+ ? Object.keys(env.world).map((key) => key + '=' + env.world[key]).join(' ')
2402
+ : '';
2403
+ console.log(' ' + (env.envelopeId || '(no id)') + (env.status ? ' ' + env.status : '') +
2404
+ (env.sameWorld === true ? ' [same world]' : (world ? ' [' + world + ']' : '')));
2233
2405
  });
2406
+ console.log(' Continue one of those with `remits-cli verify use <id>`, or supersede it once this one supersedes its goal.');
2234
2407
  }
2235
2408
  });
2236
2409
  }
2237
2410
 
2411
+ function manifestClaims(manifest) {
2412
+ const claims = manifest && manifest.claims;
2413
+ return Array.isArray(claims) ? claims.filter(Boolean) : [];
2414
+ }
2415
+
2238
2416
  function manifestRequiredEvidence(manifest = {}) {
2239
2417
  const required = [];
2240
2418
  if (Array.isArray(manifest.requiredEvidence)) required.push(...manifest.requiredEvidence);
@@ -3684,6 +3862,12 @@ function stageTimeoutMs(components, flags = {}) {
3684
3862
 
3685
3863
  function buildAxios(baseUrl, token, timeoutMs = 60000) {
3686
3864
  const headers = token ? { Authorization: 'Bearer ' + token } : {};
3865
+ // The CLI version on EVERY request, from the one place every request is built. It used to travel only on
3866
+ // `agent register`, so the platform knew the version of the sessions that registered - and the agents that
3867
+ // loop hardest are exactly the ones that never do. Without it the platform cannot tell an agent that the
3868
+ // diagnostics it is missing exist, and every output improvement lands invisibly.
3869
+ const cliVersion = readCliVersion();
3870
+ if (cliVersion) headers['X-Remits-Cli-Version'] = cliVersion;
3687
3871
  return axios.create({ baseURL: baseUrl, timeout: parsePositiveInt(timeoutMs, 60000), headers });
3688
3872
  }
3689
3873
 
@@ -5082,6 +5266,9 @@ function printStageSummary(response, flags) {
5082
5266
  if (typeof printAccountLanes === 'function') {
5083
5267
  printAccountLanes(response);
5084
5268
  }
5269
+ if (typeof printCliAdvisory === 'function') {
5270
+ printCliAdvisory(response);
5271
+ }
5085
5272
  printComponentCommandResponse('Components stage', response, flags);
5086
5273
  }
5087
5274
 
@@ -6933,6 +7120,10 @@ async function branchesComponentsCommand(flags) {
6933
7120
  // after the subcommand is the branch name when present.
6934
7121
  const positional = flags._ && flags._[2];
6935
7122
  const branchName = flags.branch || positional;
7123
+ const copyToBranch = flags['copy-to'] || flags.copyTo;
7124
+ if ((flags['copy-to'] !== undefined || flags.copyTo !== undefined) && !branchName) {
7125
+ throw new Error('components branch <source> --copy-to <target> requires a source branch name');
7126
+ }
6936
7127
 
6937
7128
  // Without a branch there is nothing to detail, so always list. With one, the mutating flags
6938
7129
  // (--subscribe/--unsubscribe/--retire) win, then the narrower read views, and the default is the
@@ -6942,10 +7133,14 @@ async function branchesComponentsCommand(flags) {
6942
7133
  if (flags.subscribe !== undefined) mode = 'subscribe';
6943
7134
  else if (flags.unsubscribe !== undefined) mode = 'unsubscribe';
6944
7135
  else if (flagEnabled(flags.retire)) mode = 'retire';
7136
+ else if (flags['copy-to'] !== undefined || flags.copyTo !== undefined) mode = 'copy';
6945
7137
  else if (flags.diff !== undefined) mode = 'diff';
6946
7138
  else if (flagEnabled(flags.subscribers)) mode = 'subscribers';
6947
7139
  else mode = 'status';
6948
7140
  }
7141
+ if (mode === 'copy' && (copyToBranch === true || !String(copyToBranch || '').trim())) {
7142
+ throw new Error('components branch <source> --copy-to <target> requires a target branch name');
7143
+ }
6949
7144
 
6950
7145
  // `--diff 42` carries the component id inline; a bare `--diff` falls back to --component-id/--name.
6951
7146
  const componentId = mode === 'diff'
@@ -6968,6 +7163,7 @@ async function branchesComponentsCommand(flags) {
6968
7163
  componentType: flags['component-type'] || flags.type,
6969
7164
  componentId,
6970
7165
  componentName: flags['component-name'] || flags.name,
7166
+ targetBranchName: copyToBranch,
6971
7167
  subscribeAccountId,
6972
7168
  parentAccountId: flags['parent-account'],
6973
7169
  // Optional branch-scoped custom host set on the same edge as the subscription.
@@ -7091,6 +7287,33 @@ function printBranchesSummary(response) {
7091
7287
  return;
7092
7288
  }
7093
7289
 
7290
+ if (response.mode === 'copy') {
7291
+ console.log(response.message || ('Copied branch overlays to ' + response.targetBranch));
7292
+ console.log('Source branch:', response.branch);
7293
+ console.log('Target branch:', response.targetBranch);
7294
+ // A dry run writes nothing, so `copied` is 0 by contract — printing it as "Copied overlays: 0"
7295
+ // under a "Would copy 2" message reads as a failed copy. Name the plan instead.
7296
+ if (response.dryRun) {
7297
+ console.log('Plan only (dry run) — overlays that would be copied:', response.plannedCopies || 0);
7298
+ if (response.existingCount) console.log('Existing overlays that would be replaced:', response.existingCount);
7299
+ } else {
7300
+ console.log('Copied overlays:', response.copied || 0);
7301
+ if (response.overwritten) console.log('Replaced existing overlays:', response.overwritten);
7302
+ }
7303
+ // A source overlay that is identical to trunk in both content and metadata is sparse and is never
7304
+ // stored, so it is named rather than silently missing from the count.
7305
+ const skippedSparse = response.skippedIdenticalToTrunk || [];
7306
+ if (skippedSparse.length) {
7307
+ console.log('Skipped as identical to trunk:', skippedSparse.join(', '));
7308
+ }
7309
+ // The number an operator needs BEFORE deciding to pass --force: a target branch with live
7310
+ // subscribers is code those accounts are running right now.
7311
+ if (response.targetSubscriberCount) {
7312
+ console.log('Live subscribers on the target branch:', response.targetSubscriberCount);
7313
+ }
7314
+ return;
7315
+ }
7316
+
7094
7317
  if (response.mode === 'subscribe' || response.mode === 'unsubscribe' || response.mode === 'retire') {
7095
7318
  console.log(response.message);
7096
7319
  if (response.requiresConfirmation) {
@@ -7702,7 +7925,7 @@ async function waitForStatus(api, cwd, accountId, branchName, taskId, token, dat
7702
7925
  dataMode
7703
7926
  }).then((r) => r.data);
7704
7927
 
7705
- if (status.status === 'completed' || status.status === 'failed') {
7928
+ if (testStatusIsTerminal(status)) {
7706
7929
  return status;
7707
7930
  }
7708
7931
  await new Promise((r) => setTimeout(r, pollDelayMs));
@@ -7710,6 +7933,10 @@ async function waitForStatus(api, cwd, accountId, branchName, taskId, token, dat
7710
7933
  }
7711
7934
  }
7712
7935
 
7936
+ function testStatusIsTerminal(status = {}) {
7937
+ return ['completed', 'failed', 'interrupted'].includes(String(status.status || '').toLowerCase());
7938
+ }
7939
+
7713
7940
  function testEvidenceCategories(status = {}, selectedNames = []) {
7714
7941
  const categories = new Set(['test_run']);
7715
7942
  const test = status.test || {};
@@ -7839,6 +8066,108 @@ function detectNondeterministicTestRun(priorPackets, currentStatus, currentProve
7839
8066
  return null;
7840
8067
  }
7841
8068
 
8069
+ async function appendTerminalTestRunEvidence(options = {}) {
8070
+ const {
8071
+ api,
8072
+ cwd,
8073
+ session,
8074
+ accountId,
8075
+ flags,
8076
+ status,
8077
+ testRef,
8078
+ selectedNames = [],
8079
+ dataMode,
8080
+ dataModeSource,
8081
+ branchName,
8082
+ workspace,
8083
+ baseUrl,
8084
+ activeEnvelopeId,
8085
+ quiet
8086
+ } = options;
8087
+ const unmatched = (status.result && status.result.unmatchedTestNames) || [];
8088
+ const componentProvenance = testComponentProvenance(status);
8089
+ const sourceRevision = collectVerificationSource(cwd, flags);
8090
+ const priorPackets = await readVerificationPacketsForDiagnostics(api, cwd, session, accountId, activeEnvelopeId, {
8091
+ packetType: 'test_run',
8092
+ suite: status.test && status.test.name,
8093
+ max: 50
8094
+ });
8095
+ const nondeterminism = detectNondeterministicTestRun(priorPackets, status, componentProvenance, {
8096
+ laneContentHash: status.staging && status.staging.laneSummary && status.staging.laneSummary.contentHash,
8097
+ gitHead: sourceRevision.gitHead
8098
+ });
8099
+
8100
+ await appendVerificationPacket(api, cwd, session, accountId, flags, {
8101
+ type: 'test_run',
8102
+ success: status.status === 'completed' && !(status.result && status.result.failed > 0) && !unmatched.length,
8103
+ claim: 'Test run ' + String(testRef || (status.test && (status.test.name || status.test.id)) || status.taskId || 'unknown'),
8104
+ world: buildCommandWorld(status, {
8105
+ accountId,
8106
+ dataMode: status.dataMode || dataMode,
8107
+ dataModeSource,
8108
+ branchName,
8109
+ workspace,
8110
+ host: normalizeBaseUrl(baseUrl),
8111
+ sourceLayer: status.staging && status.staging.testComponentSource
8112
+ }),
8113
+ revision: Object.assign(sourceRevision, {
8114
+ compileSignatures: testCompileSignatures(status),
8115
+ componentProvenance
8116
+ }),
8117
+ nondeterministic: nondeterminism ? true : undefined,
8118
+ nondeterminism: nondeterminism || undefined,
8119
+ test: {
8120
+ taskId: status.taskId || (status.result && status.result.taskId),
8121
+ testId: status.test && status.test.id,
8122
+ testName: status.test && status.test.name,
8123
+ selectedCases: selectedNames,
8124
+ passed: status.result && status.result.passed,
8125
+ failed: status.result && status.result.failed,
8126
+ tests: status.result && status.result.tests,
8127
+ dataModeSource: status.dataModeSource || dataModeSource,
8128
+ unmatchedTestNames: unmatched
8129
+ },
8130
+ evidenceCategories: testEvidenceCategories(status, selectedNames),
8131
+ // What the run spent and whether that live AI was declared (withAiBudget), so an envelope can tell a
8132
+ // measurement run from a suite with missing mocks without re-reading every case.
8133
+ dependencies: testRunDependencies(status),
8134
+ summary: status.result && status.result.summary,
8135
+ limitations: []
8136
+ .concat(unmatched.length ? ['One or more requested test case selectors matched no case.'] : [])
8137
+ .concat(nondeterminism ? [nondeterminism.message] : []),
8138
+ rawRefs: { testStatusKey: status.taskId, durableTestRun: status.result && status.result.durableRecord ? status.taskId : undefined },
8139
+ status
8140
+ }, { quiet });
8141
+
8142
+ return { unmatched, nondeterminism };
8143
+ }
8144
+
8145
+ /**
8146
+ * One failed case per distinct failure root, largest root first, capped.
8147
+ *
8148
+ * A pivot block per failed case is one problem stated N times: the thirty-three failures of a real suite
8149
+ * are twenty-one roots, and the twelve repeats carry the same trace shape, the same components and the
8150
+ * same error. The grouped roll-up above already names every case; this picks the ones worth a full block.
8151
+ */
8152
+ /** Truncate, and SAY that it was truncated — a silent cut reads as the whole message. */
8153
+ function truncateForPivot(text, max) {
8154
+ if (text.length <= max) return text;
8155
+ return text.slice(0, max) + '… (truncated; --json for the full text)';
8156
+ }
8157
+
8158
+ function selectPivotCases(tests, limit = 6) {
8159
+ const failed = (tests || []).filter((test) => test && test.passed === false);
8160
+ const seen = new Set();
8161
+ const representatives = [];
8162
+ failed.forEach((test) => {
8163
+ const root = testFailureRoot(test);
8164
+ if (seen.has(root)) return;
8165
+ seen.add(root);
8166
+ representatives.push(test);
8167
+ });
8168
+ return { shown: representatives.slice(0, limit), failedCount: failed.length, rootCount: representatives.length };
8169
+ }
8170
+
7842
8171
  function printTestRunPivots(status = {}) {
7843
8172
  const tests = status.result && Array.isArray(status.result.tests) ? status.result.tests : [];
7844
8173
  if (!tests.length) return;
@@ -7846,7 +8175,8 @@ function printTestRunPivots(status = {}) {
7846
8175
  if (slowest && slowest.duration != null) {
7847
8176
  console.log('Slowest case:', (slowest.name || '(unnamed)') + ' in ' + formatDurationMs(slowest.duration));
7848
8177
  }
7849
- tests.filter((test) => test && test.passed === false).forEach((test) => {
8178
+ const selection = selectPivotCases(tests);
8179
+ selection.shown.forEach((test) => {
7850
8180
  console.log('Pivots for failed case:', test.name || '(unnamed)');
7851
8181
  if (test.outcome) console.log(' Outcome:', test.outcome + (test.outcomeReason && test.outcomeReason !== test.error ? ' - ' + String(test.outcomeReason).slice(0, 300) : ''));
7852
8182
  if (test.duration != null) console.log(' Duration:', formatDurationMs(test.duration));
@@ -7857,7 +8187,9 @@ function printTestRunPivots(status = {}) {
7857
8187
  if (test.threadGroupingId || test.threadGroupId || test.traceId) {
7858
8188
  console.log(' Trace:', test.traceId || test.threadGroupingId || test.threadGroupId);
7859
8189
  }
7860
- if (test.error) console.log(' Error:', String(test.error).slice(0, 500));
8190
+ // 500 chars cut the replay diagnosis mid-sentence, losing the half that says what to DO. A pivot is the
8191
+ // block a reader acts on; truncating its one actionable sentence to save four lines is a bad trade.
8192
+ if (test.error) console.log(' Error:', truncateForPivot(String(test.error), 1200));
7861
8193
  const diagnostics = test.diagnostics && typeof test.diagnostics === 'object' ? test.diagnostics : null;
7862
8194
  if (diagnostics && Object.keys(diagnostics).length) {
7863
8195
  console.log(' Diagnostics:', JSON.stringify(diagnostics).slice(0, 1200));
@@ -7873,6 +8205,12 @@ function printTestRunPivots(status = {}) {
7873
8205
  console.log(' Live HTTP calls:', test.liveHttpCalls.length + (intentional ? ' (' + intentional + ' intentional, inside withAiBudget)' : ''));
7874
8206
  }
7875
8207
  });
8208
+ // Named, not silently dropped: a reader must be able to tell "this is everything" from "this is a sample".
8209
+ if (selection.rootCount > selection.shown.length) {
8210
+ console.log('Pivots shown for ' + selection.shown.length + ' of ' + selection.rootCount +
8211
+ ' failure root(s) (' + selection.failedCount + ' failed case(s)). Every root is listed above; ' +
8212
+ 'use --names "<case>" for one, or --json for all.');
8213
+ }
7876
8214
  }
7877
8215
 
7878
8216
  function formatDurationMs(value) {
@@ -8086,6 +8424,114 @@ function stagedMaskedByVariantLines(masked, accountId) {
8086
8424
  return lines;
8087
8425
  }
8088
8426
 
8427
+ /**
8428
+ * The assertion ROOT of one failed case: what broke, with the particulars of this case removed.
8429
+ *
8430
+ * Deliberately built only from properties of the JVM/Groovy failure format - a power-assert's
8431
+ * `Expression:`/`Values:` decoration, an `assert <expr>` head, an exception class prefix - and never from
8432
+ * any vocabulary a suite happens to use. Identifiers and numbers are replaced because two cases failing
8433
+ * the same way differ exactly in those.
8434
+ */
8435
+ function testFailureRoot(test = {}) {
8436
+ let text = String(test.error || test.outcomeReason || test.outcome || 'failed').trim();
8437
+ // Groovy's power assert appends the rendered expression and every intermediate value.
8438
+ text = text.split(/\.\s+(?:Expression|Values):/)[0];
8439
+ const assertion = text.match(/assert\s+(.+)$/);
8440
+ if (assertion) text = 'assert ' + assertion[1];
8441
+ text = text
8442
+ .replace(/['"][0-9a-fA-F]{8}-[0-9a-fA-F-]{4,}['"]/g, "'<id>'")
8443
+ .replace(/\b[0-9a-fA-F]{8}-[0-9a-fA-F-]{27,}\b/g, '<id>')
8444
+ .replace(/\b\d[\d.,]*\b/g, '<n>')
8445
+ .replace(/\s+/g, ' ')
8446
+ .trim();
8447
+ if (text.length <= FAILURE_ROOT_LABEL_CHARS) return text;
8448
+ // Cut on a word boundary: "found no usable stored provid" reads as a different error than the one it is.
8449
+ const cut = text.slice(0, FAILURE_ROOT_LABEL_CHARS);
8450
+ const lastSpace = cut.lastIndexOf(' ');
8451
+ return (lastSpace > FAILURE_ROOT_LABEL_CHARS * 0.6 ? cut.slice(0, lastSpace) : cut) + '\u2026';
8452
+ }
8453
+
8454
+ /** Long enough to tell two failures apart, short enough that a root list stays a list. */
8455
+ const FAILURE_ROOT_LABEL_CHARS = 120;
8456
+
8457
+ /**
8458
+ * Failed cases grouped by that root, largest first.
8459
+ *
8460
+ * Thirty-three failures printed in run order read as thirty-three problems; the same run grouped is four.
8461
+ * This is the line that decides whether the next move is a narrow root-cause pass or another broad rerun,
8462
+ * so it is computed from the run itself rather than left to the reader.
8463
+ */
8464
+ function testFailureGroups(status = {}) {
8465
+ const tests = status.result && Array.isArray(status.result.tests) ? status.result.tests : [];
8466
+ const groups = new Map();
8467
+ tests.filter((test) => test && test.passed === false && !test.interrupted).forEach((test) => {
8468
+ const root = testFailureRoot(test);
8469
+ if (!groups.has(root)) groups.set(root, { root, count: 0, cases: [] });
8470
+ const group = groups.get(root);
8471
+ group.count += 1;
8472
+ group.cases.push(test.name || '(unnamed)');
8473
+ });
8474
+ return Array.from(groups.values()).sort((left, right) => right.count - left.count || left.root.localeCompare(right.root));
8475
+ }
8476
+
8477
+ const FAILURE_GROUPS_SHOWN = 8;
8478
+
8479
+ function testFailureGroupLines(status = {}, options = {}) {
8480
+ const groups = testFailureGroups(status);
8481
+ if (!groups.length) return [];
8482
+ const failed = groups.reduce((sum, group) => sum + group.count, 0);
8483
+ const lines = [''];
8484
+ lines.push('Failure roots (' + failed + ' failed case(s), ' + groups.length + ' distinct root(s)):');
8485
+ groups.slice(0, FAILURE_GROUPS_SHOWN).forEach((group) => {
8486
+ lines.push(' ' + String(group.count) + 'x ' + group.root);
8487
+ // The representative case is what `--names` takes, so the next command is a copy of this line.
8488
+ lines.push(' e.g. ' + group.cases[0] + (group.count > 1 ? ' (+' + (group.count - 1) + ' more)' : ''));
8489
+ });
8490
+ if (groups.length > FAILURE_GROUPS_SHOWN) {
8491
+ lines.push(' ...' + (groups.length - FAILURE_GROUPS_SHOWN) + ' more root(s); --json for all');
8492
+ }
8493
+ const testRef = options.testRef || (status.test && (status.test.id || status.test.name)) ||
8494
+ (status.result && (status.result.testId || status.result.testName));
8495
+ if (groups[0] && groups[0].count > 1 && testRef) {
8496
+ lines.push('Largest root first: remits-cli test run --test ' + JSON.stringify(String(testRef)) +
8497
+ ' --names ' + JSON.stringify(groups[0].cases[0]));
8498
+ }
8499
+ return lines;
8500
+ }
8501
+
8502
+ /**
8503
+ * The verdict of a run in four lines, shared by `test run` and `test status`.
8504
+ *
8505
+ * `test run` used to print the whole run object as pretty JSON before its human summary. Measured on one
8506
+ * 92-case suite that is 220 KB - most of it per-case ids repeated once per case - spent by the command an
8507
+ * agent runs most often, on the turn where it has the least room left to reason. Every byte is still one
8508
+ * `--json` away.
8509
+ */
8510
+ function testRunHeadlineLines(status = {}, options = {}) {
8511
+ const result = status.result || {};
8512
+ const lines = [];
8513
+ lines.push('Test run status: ' + (status.status || 'unknown'));
8514
+ if (options.taskId || status.taskId) lines.push('Task ID: ' + (options.taskId || status.taskId));
8515
+ if (options.source) lines.push('Source: ' + options.source);
8516
+ const actualDataMode = status.dataMode || result.dataMode || options.dataMode;
8517
+ let dataModeLine = 'Data mode: ' + (actualDataMode || 'unknown');
8518
+ // A durable run answers about the lane it RAN in, which need not be the lane this command asked for.
8519
+ // Reporting only the run's own lane is correct and reads as if the request had been honoured.
8520
+ if (options.requestedDataMode && actualDataMode && options.requestedDataMode !== actualDataMode) {
8521
+ dataModeLine += ' (you asked for ' + options.requestedDataMode + '; this run was recorded in the ' +
8522
+ actualDataMode + ' lane, and that is what it proves)';
8523
+ }
8524
+ lines.push(dataModeLine);
8525
+ if (result.total != null || result.passed != null) {
8526
+ lines.push('Cases: ' + (result.passed || 0) + ' passed, ' + (result.failed || 0) + ' failed, ' +
8527
+ (result.total || 0) + ' total');
8528
+ }
8529
+ if (status.error || status.message || result.error) {
8530
+ lines.push('Message: ' + (status.error || status.message || result.error));
8531
+ }
8532
+ return lines;
8533
+ }
8534
+
8089
8535
  // The corpus-style roll-up printed after a run: outcome counts, case duration percentiles, AI usage split
8090
8536
  // live/mocked, and the durable record. Built only from the result the platform returned.
8091
8537
  function testRunSummaryLines(status = {}) {
@@ -8341,6 +8787,7 @@ async function testCommand(flags) {
8341
8787
  printLocalStateWarnings(cwd, flags);
8342
8788
  printStagingLane(branchName, workspace, workspaceSource(cwd, flags));
8343
8789
  printStagingLaneOwnerNotice(start.staging || {});
8790
+ printCliAdvisory(start);
8344
8791
  if (start.staging && Array.isArray(start.staging.accountLanes)) {
8345
8792
  printOrphanedWorkspaceWarning(start.staging);
8346
8793
  printAccountLanes({ accountLanes: start.staging.accountLanes, branchName, workspace });
@@ -8356,6 +8803,48 @@ async function testCommand(flags) {
8356
8803
  });
8357
8804
  runtimeState.currentTestTaskId = start.taskId;
8358
8805
 
8806
+ const waitForCompletion = waitForTestCompletion(flags);
8807
+ if (!waitForCompletion) {
8808
+ recordEvidenceEntry(cwd, {
8809
+ type: 'test_run',
8810
+ success: null,
8811
+ pending: true,
8812
+ claim: 'Test run ' + String(testRef),
8813
+ world: buildCommandWorld(start, {
8814
+ accountId,
8815
+ dataMode,
8816
+ dataModeSource,
8817
+ branchName,
8818
+ workspace,
8819
+ host: normalizeBaseUrl(baseUrl),
8820
+ sourceLayer: start.staging && start.staging.testComponentSource
8821
+ }),
8822
+ test: {
8823
+ taskId: start.taskId,
8824
+ testId: start.test && start.test.id,
8825
+ testName: start.test && start.test.name,
8826
+ selectedCases: names,
8827
+ dataModeSource
8828
+ },
8829
+ evidenceCategories: testEvidenceCategories(start, names),
8830
+ rawRefs: { testStatusKey: start.taskId },
8831
+ status: start
8832
+ }, { accountId, dataMode, branchName, workspace, baseUrl: normalizeBaseUrl(baseUrl), envelopeId: activeEnvelopeId });
8833
+ runtimeState.currentTestTaskId = null;
8834
+ if (jsonOutput) {
8835
+ console.log(JSON.stringify(Object.assign({}, start, {
8836
+ pending: true,
8837
+ wait: false,
8838
+ statusCommand: 'remits-cli test status --task-id ' + start.taskId
8839
+ }), null, 2));
8840
+ } else {
8841
+ console.log('Not waiting (--wait false).');
8842
+ console.log('Poll this run: remits-cli test status --task-id ' + start.taskId);
8843
+ console.log('Terminal evidence attaches when `test status` reads a completed, failed or interrupted run.');
8844
+ }
8845
+ return;
8846
+ }
8847
+
8359
8848
  let stopWs = null;
8360
8849
  if (flags.watch !== 'false') {
8361
8850
  const topic = session.websocketTopic || start.websocketTopic || (session.user && String(session.user.uuid || '').replace(/-/g, ''));
@@ -8376,10 +8865,16 @@ async function testCommand(flags) {
8376
8865
  if (jsonOutput) {
8377
8866
  console.log(JSON.stringify(status, null, 2));
8378
8867
  } else {
8379
- console.log('Final status:', JSON.stringify(status, null, 2));
8380
8868
  printStagingLaneOwnerNotice(status.staging || {});
8869
+ testRunHeadlineLines(status, { taskId: start.taskId, dataMode, requestedDataMode: dataMode })
8870
+ .forEach((line) => console.log(line));
8381
8871
  testRunSummaryLines(status).forEach((line) => console.log(line));
8872
+ testFailureGroupLines(status, { testRef }).forEach((line) => console.log(line));
8382
8873
  printTestRunPivots(status);
8874
+ // The whole run object used to be printed here as pretty JSON. One 92-case suite measured 220 KB,
8875
+ // most of it identifiers repeated once per case, spent on the turn with the least room left. It is
8876
+ // still one flag away, and the durable record keeps it after the live status expires.
8877
+ console.log('Full run payload: remits-cli test status --task-id ' + start.taskId + ' --json');
8383
8878
  }
8384
8879
 
8385
8880
  // A selector that matched no case is a mis-specified run, not a passing one. Say so in the terminal
@@ -8404,17 +8899,24 @@ async function testCommand(flags) {
8404
8899
  process.exitCode = 1;
8405
8900
  }
8406
8901
 
8407
- const componentProvenance = testComponentProvenance(status);
8408
- const sourceRevision = collectVerificationSource(cwd, flags);
8409
- const priorPackets = await readVerificationPacketsForDiagnostics(api, cwd, session, accountId, activeEnvelopeId, {
8410
- packetType: 'test_run',
8411
- suite: status.test && status.test.name,
8412
- max: 50
8413
- });
8414
- const nondeterminism = detectNondeterministicTestRun(priorPackets, status, componentProvenance, {
8415
- laneContentHash: status.staging && status.staging.laneSummary && status.staging.laneSummary.contentHash,
8416
- gitHead: sourceRevision.gitHead
8902
+ const evidence = await appendTerminalTestRunEvidence({
8903
+ api,
8904
+ cwd,
8905
+ session,
8906
+ accountId,
8907
+ flags: verification.evidenceFlags,
8908
+ status,
8909
+ testRef,
8910
+ selectedNames: names,
8911
+ dataMode,
8912
+ dataModeSource,
8913
+ branchName,
8914
+ workspace,
8915
+ baseUrl,
8916
+ activeEnvelopeId,
8917
+ quiet: jsonOutput
8417
8918
  });
8919
+ const nondeterminism = evidence.nondeterminism;
8418
8920
  if (nondeterminism && !jsonOutput) {
8419
8921
  console.log('Nondeterministic signal:', nondeterminism.message);
8420
8922
  nondeterminism.flips.slice(0, 6).forEach((flip) => {
@@ -8422,40 +8924,6 @@ async function testCommand(flags) {
8422
8924
  ' (previous packet ' + (flip.previousPacketId || 'unknown') + ')');
8423
8925
  });
8424
8926
  }
8425
-
8426
- await appendVerificationPacket(api, cwd, session, accountId, verification.evidenceFlags, {
8427
- type: 'test_run',
8428
- success: status.status === 'completed' && !(status.result && status.result.failed > 0) && !unmatched.length,
8429
- claim: 'Test run ' + String(testRef),
8430
- world: buildCommandWorld(status, { accountId, dataMode: status.dataMode || dataMode, dataModeSource, branchName, workspace, host: normalizeBaseUrl(baseUrl), sourceLayer: status.staging && status.staging.testComponentSource }),
8431
- revision: Object.assign(sourceRevision, {
8432
- compileSignatures: testCompileSignatures(status),
8433
- componentProvenance
8434
- }),
8435
- nondeterministic: nondeterminism ? true : undefined,
8436
- nondeterminism: nondeterminism || undefined,
8437
- test: {
8438
- taskId: start.taskId,
8439
- testId: status.test && status.test.id,
8440
- testName: status.test && status.test.name,
8441
- selectedCases: names,
8442
- passed: status.result && status.result.passed,
8443
- failed: status.result && status.result.failed,
8444
- tests: status.result && status.result.tests,
8445
- dataModeSource: status.dataModeSource || dataModeSource,
8446
- unmatchedTestNames: unmatched
8447
- },
8448
- evidenceCategories: testEvidenceCategories(status, names),
8449
- // What the run spent and whether that live AI was declared (withAiBudget), so an envelope can tell a
8450
- // measurement run from a suite with missing mocks without re-reading every case.
8451
- dependencies: testRunDependencies(status),
8452
- summary: status.result && status.result.summary,
8453
- limitations: []
8454
- .concat(unmatched.length ? ['One or more requested test case selectors matched no case.'] : [])
8455
- .concat(nondeterminism ? [nondeterminism.message] : []),
8456
- rawRefs: { testStatusKey: start.taskId, durableTestRun: status.result && status.result.durableRecord ? start.taskId : undefined },
8457
- status
8458
- }, { quiet: jsonOutput });
8459
8927
  }
8460
8928
 
8461
8929
  async function testStatusCommand(flags) {
@@ -8465,9 +8933,11 @@ async function testStatusCommand(flags) {
8465
8933
  const { session, accountId } = sessionContext;
8466
8934
  const baseUrl = flags['base-url'] || session.baseUrl || DEFAULT_BASE_URL;
8467
8935
  const branchName = flags.branch || currentBranch(cwd);
8936
+ const workspace = resolveWorkspace(cwd, flags);
8468
8937
  const dataMode = hasExplicitDataModeFlag(flags)
8469
8938
  ? resolveDataMode(flags, null)
8470
8939
  : DEFAULT_DATA_MODE;
8940
+ const dataModeSource = dataModeFlagSource(flags);
8471
8941
  const taskId = flags['task-id'] || flags.taskId || flags.id || (flags._ && flags._[2]);
8472
8942
  const jsonOutput = flagEnabled(flags.json);
8473
8943
  if (!taskId) throw new Error('Missing --task-id <taskId>');
@@ -8481,31 +8951,94 @@ async function testStatusCommand(flags) {
8481
8951
  taskId,
8482
8952
  dataMode
8483
8953
  }).then((r) => r.data);
8954
+ if (!status.taskId) status.taskId = taskId;
8955
+
8956
+ let statusVerification = { envelopeId: null, evidenceFlags: flags };
8957
+ if (testStatusIsTerminal(status)) {
8958
+ statusVerification = await verificationPreflight(api, cwd, session, accountId, flags, {
8959
+ packetType: 'test_run',
8960
+ world: buildCommandWorld(status, {
8961
+ accountId,
8962
+ dataMode: status.dataMode || dataMode,
8963
+ dataModeSource,
8964
+ branchName,
8965
+ workspace,
8966
+ host: normalizeBaseUrl(baseUrl),
8967
+ sourceLayer: status.staging && status.staging.testComponentSource
8968
+ }),
8969
+ test: status.test && status.test.id !== undefined && status.test.id !== null
8970
+ ? { testId: status.test.id, testName: status.test.name }
8971
+ : { testName: status.test && status.test.name }
8972
+ }, { command: 'test status', neverRefuse: true, quiet: jsonOutput });
8973
+ }
8484
8974
 
8485
8975
  if (jsonOutput) {
8976
+ if (testStatusIsTerminal(status)) {
8977
+ await appendTerminalTestRunEvidence({
8978
+ api,
8979
+ cwd,
8980
+ session,
8981
+ accountId,
8982
+ flags: statusVerification.evidenceFlags,
8983
+ status,
8984
+ testRef: status.test && (status.test.name || status.test.id) || taskId,
8985
+ selectedNames: status.result && Array.isArray(status.result.tests) ? status.result.tests.map((t) => t && t.name).filter(Boolean) : [],
8986
+ dataMode,
8987
+ dataModeSource,
8988
+ branchName,
8989
+ workspace,
8990
+ baseUrl,
8991
+ activeEnvelopeId: statusVerification.envelopeId,
8992
+ quiet: true
8993
+ });
8994
+ }
8486
8995
  console.log(JSON.stringify(status, null, 2));
8487
8996
  return status;
8488
8997
  }
8489
8998
  printSessionResolutionWarning(sessionContext);
8490
8999
  printResolvedBaseUrl(baseUrl);
8491
- console.log('Test run status:', status.status || 'unknown');
8492
- console.log('Task ID:', taskId);
8493
9000
  // Say where this answer came from. A durable record is a finished snapshot rebuilt from the database after the
8494
9001
  // live status was gone; without this line a reconstructed run is indistinguishable from one still being watched.
8495
- console.log('Source:', status.durable
8496
- ? 'durable run record (the live status has expired; this run is final)'
8497
- : 'live run status');
8498
- console.log('Data mode:', status.dataMode || dataMode);
8499
- if (status.result) {
8500
- console.log('Cases:', (status.result.passed || 0) + ' passed, ' + (status.result.failed || 0) + ' failed, ' + (status.result.total || 0) + ' total');
8501
- }
8502
- if (status.error || status.message) {
8503
- console.log('Message:', status.error || status.message);
8504
- }
9002
+ testRunHeadlineLines(status, {
9003
+ taskId,
9004
+ source: status.durable
9005
+ ? 'durable run record (the live status has expired; this run is final)'
9006
+ : 'live run status',
9007
+ dataMode,
9008
+ // A durable run reports the lane it RAN in. When that differs from the lane this command asked for,
9009
+ // say so: auditing prod activity and being handed a test-lane run is correct and reads as if it were not.
9010
+ requestedDataMode: hasExplicitDataModeFlag(flags) ? dataMode : null
9011
+ }).forEach((line) => console.log(line));
8505
9012
  if (status.world) runWorldLines(status.world, { host: normalizeBaseUrl(baseUrl) }).forEach((line) => console.log(line));
8506
9013
  testRunSummaryLines(status).forEach((line) => console.log(line));
9014
+ testFailureGroupLines(status).forEach((line) => console.log(line));
8507
9015
  printTestRunPivots(status);
8508
- if (status.status === 'failed' || (status.result && status.result.failed > 0)) {
9016
+ if (testStatusIsTerminal(status)) {
9017
+ const evidence = await appendTerminalTestRunEvidence({
9018
+ api,
9019
+ cwd,
9020
+ session,
9021
+ accountId,
9022
+ flags: statusVerification.evidenceFlags,
9023
+ status,
9024
+ testRef: status.test && (status.test.name || status.test.id) || taskId,
9025
+ selectedNames: status.result && Array.isArray(status.result.tests) ? status.result.tests.map((t) => t && t.name).filter(Boolean) : [],
9026
+ dataMode,
9027
+ dataModeSource,
9028
+ branchName,
9029
+ workspace,
9030
+ baseUrl,
9031
+ activeEnvelopeId: statusVerification.envelopeId
9032
+ });
9033
+ if (evidence.nondeterminism) {
9034
+ console.log('Nondeterministic signal:', evidence.nondeterminism.message);
9035
+ evidence.nondeterminism.flips.slice(0, 6).forEach((flip) => {
9036
+ console.log(' - ' + flip.caseName + ': ' + (flip.previousPassed ? 'passed' : 'failed') + ' -> ' + (flip.currentPassed ? 'passed' : 'failed') +
9037
+ ' (previous packet ' + (flip.previousPacketId || 'unknown') + ')');
9038
+ });
9039
+ }
9040
+ }
9041
+ if (status.status === 'failed' || status.status === 'interrupted' || (status.result && status.result.failed > 0)) {
8509
9042
  process.exitCode = 1;
8510
9043
  }
8511
9044
  return status;
@@ -8521,10 +9054,13 @@ async function testRunsCommand(flags) {
8521
9054
  const baseUrl = flags['base-url'] || session.baseUrl || DEFAULT_BASE_URL;
8522
9055
  const api = buildAxios(baseUrl, session.token);
8523
9056
  const testRef = flags.test || flags['test-id'] || flags.name;
9057
+ const allDataLanes = flagEnabled(flags['all-lanes']) || flagEnabled(flags['all-data-lanes']);
8524
9058
  const data = await loggedPost(api, cwd, '/cli/testRuns', {
8525
9059
  token: session.token,
8526
9060
  accountId,
8527
9061
  asAccountId: flags['as-account'] || flags['as-account-id'],
9062
+ dataMode: resolveDataMode(flags, session),
9063
+ allDataLanes: allDataLanes || undefined,
8528
9064
  testId: testRef && /^\d+$/.test(String(testRef)) ? Number(testRef) : undefined,
8529
9065
  testName: testRef && !/^\d+$/.test(String(testRef)) ? String(testRef) : undefined,
8530
9066
  max: flags.limit || flags.max || 20
@@ -8532,9 +9068,17 @@ async function testRunsCommand(flags) {
8532
9068
  if (!data.success) throw new Error(data.message || 'Could not list test runs');
8533
9069
 
8534
9070
  if (flagEnabled(flags.compare)) {
8535
- const runs = data.runs || [];
8536
- if (runs.length < 2) throw new Error('--compare needs at least two recorded runs' + (testRef ? ' of ' + testRef : '') + '; found ' + runs.length);
8537
- return testCompareCommand(Object.assign({}, flags, { base: runs[1].taskId, head: runs[0].taskId }));
9071
+ const pair = selectComparableRuns(data.runs || []);
9072
+ if (!pair) {
9073
+ throw new Error('--compare needs two COMPARABLE recorded runs' + (testRef ? ' of ' + testRef : '') +
9074
+ ' — same data lane, branch, workspace and case count. Found ' + (data.runs || []).length +
9075
+ ' run(s); pick two explicitly with: remits-cli test compare --base <taskId> --head <taskId>');
9076
+ }
9077
+ if (pair.skipped) {
9078
+ console.error('Comparing the latest two comparable runs (' + describeRunShape(pair.head) + '); ' +
9079
+ pair.skipped + ' newer run(s) differ in lane, branch, workspace or case count and were skipped.');
9080
+ }
9081
+ return testCompareCommand(Object.assign({}, flags, { base: pair.base.taskId, head: pair.head.taskId }));
8538
9082
  }
8539
9083
  if (flagEnabled(flags.json)) {
8540
9084
  console.log(JSON.stringify(data, null, 2));
@@ -8543,6 +9087,9 @@ async function testRunsCommand(flags) {
8543
9087
  printSessionResolutionWarning(sessionContext);
8544
9088
  printResolvedBaseUrl(baseUrl);
8545
9089
  console.log('Recorded test runs' + (testRef ? ' for ' + testRef : '') + ' (account ' + data.accountId + '), newest first:');
9090
+ // Which lane this list is, said once. Pass counts from two lanes are two different facts.
9091
+ console.log('Data lane: ' + (data.dataLaneScope || 'all-data-lanes') +
9092
+ (data.otherLaneCount ? ' (' + data.otherLaneCount + ' more run(s) in the other lane; --all-lanes to include)' : ''));
8546
9093
  (data.runs || []).forEach((run) => {
8547
9094
  console.log('- ' + run.taskId + ' ' + (run.status || 'unknown') + ' ' + (run.passed || 0) + '/' + (run.total || 0) + ' passed' +
8548
9095
  ' ' + corpusAiLabel(run) +
@@ -8555,6 +9102,40 @@ async function testRunsCommand(flags) {
8555
9102
  return data;
8556
9103
  }
8557
9104
 
9105
+ /** The world fields that decide whether two runs are the same experiment repeated. */
9106
+ function runShapeKey(run = {}) {
9107
+ return [run.dataMode || '?', run.branchName || '?', run.workspace || 'shared',
9108
+ run.variantBranch || 'none', run.total == null ? '?' : run.total].join('|');
9109
+ }
9110
+
9111
+ function describeRunShape(run = {}) {
9112
+ return [(run.dataMode || '?') + ' lane', run.branchName || '?',
9113
+ run.workspace ? 'ws:' + run.workspace : 'shared lane',
9114
+ (run.total == null ? '?' : run.total) + ' case(s)'].join(' · ');
9115
+ }
9116
+
9117
+ /**
9118
+ * The newest two runs that are actually comparable, and how many newer ones were passed over.
9119
+ *
9120
+ * `--compare` used to take runs[0] and runs[1] unconditionally. On a real history that pairs a 92-case
9121
+ * suite with a 1-case `--names` probe, or a prod-lane run with a test-lane one, and reports the difference
9122
+ * as regressions. A comparison across worlds is not a comparison.
9123
+ */
9124
+ function selectComparableRuns(runs) {
9125
+ const list = Array.isArray(runs) ? runs.filter(Boolean) : [];
9126
+ for (let head = 0; head < list.length; head += 1) {
9127
+ const key = runShapeKey(list[head]);
9128
+ for (let base = head + 1; base < list.length; base += 1) {
9129
+ if (runShapeKey(list[base]) === key) {
9130
+ // Every run passed over, including the ones BETWEEN head and base — a 1-case probe sitting between
9131
+ // two full suites is exactly the run whose absence from the comparison needs saying.
9132
+ return { head: list[head], base: list[base], skipped: base - 1 };
9133
+ }
9134
+ }
9135
+ }
9136
+ return null;
9137
+ }
9138
+
8558
9139
  // `test compare --base <taskId> --head <taskId>`: per-case outcome, cost and duration deltas between two runs,
8559
9140
  // read from the durable records.
8560
9141
  async function testCompareCommand(flags) {
@@ -8585,6 +9166,18 @@ async function testCompareCommand(flags) {
8585
9166
  .filter((key) => comparison.baseWorld && comparison.headWorld && String(comparison.baseWorld[key] || '') !== String(comparison.headWorld[key] || ''))
8586
9167
  .map((key) => key + ' ' + short(String(comparison.baseWorld[key] || 'none')) + ' -> ' + short(String(comparison.headWorld[key] || 'none')));
8587
9168
  if (worldDiff.length) console.log('World changed: ' + worldDiff.join('; '));
9169
+ // A lane or case-count difference is not a delta to interpret, it is two different experiments. Say so
9170
+ // rather than letting "regressed (33)" stand for "these runs never measured the same thing".
9171
+ const baseLane = comparison.baseWorld && comparison.baseWorld.dataMode;
9172
+ const headLane = comparison.headWorld && comparison.headWorld.dataMode;
9173
+ if (baseLane && headLane && baseLane !== headLane) {
9174
+ console.log('NOT COMPARABLE: these runs are in different DATA LANES (' + baseLane + ' vs ' + headLane +
9175
+ '). A record\'s dataMode is the lane it was written in; the pass counts below are two different facts.');
9176
+ }
9177
+ if (comparison.base.total && comparison.head.total && comparison.base.total !== comparison.head.total) {
9178
+ console.log('NOT COMPARABLE like-for-like: ' + comparison.base.total + ' case(s) vs ' + comparison.head.total +
9179
+ '. One of these is a narrowed --names run; "absent" below means the case did not run, not that it broke.');
9180
+ }
8588
9181
  if (comparison.improved.length) console.log('Improved (' + comparison.improved.length + '): ' + comparison.improved.join(', '));
8589
9182
  if (comparison.regressed.length) console.log('Regressed (' + comparison.regressed.length + '): ' + comparison.regressed.join(', '));
8590
9183
  comparison.changed.filter((c) => c.direction === 'changed').forEach((c) => console.log('Changed: ' + c.name + ' ' + c.base + ' -> ' + c.head));
@@ -9565,10 +10158,16 @@ async function verifyCommand(flags, subcommand) {
9565
10158
  if (flags.manifest || flags.file) {
9566
10159
  manifest = parseManifestFile(flags.manifest || flags.file);
9567
10160
  }
9568
- if (dataMode === 'prod' && !manifestRequiredEvidence(manifest).length && !flagEnabled(flags['no-contract'])) {
10161
+ // PROD only, and framed as risk rather than as an unmet obligation. An envelope with no contract is an
10162
+ // evidence log, which is the normal and complete mode; warning about it on every lane taught agents that
10163
+ // their own record-keeping was an exam they were failing, and they responded by starting it over.
10164
+ if (dataMode === 'prod' && !manifestRequiredEvidence(manifest).length && !manifestClaims(manifest).length &&
10165
+ !flagEnabled(flags['no-contract'])) {
9569
10166
  console.error('');
9570
- console.error('Warning: starting a prod-data verification envelope with no required evidence contract.');
9571
- console.error('Attach a manifest with requiredEvidence before destructive work, or pass --no-contract when this is intentionally evidence-only.');
10167
+ console.error('Note: this envelope records PRODUCTION-data evidence with no acceptance contract.');
10168
+ console.error('That is fine for investigation. Before destructive or irreversible production work, name what');
10169
+ console.error('must be true so the result is checkable by someone else:');
10170
+ console.error(' remits-cli verify claim <id> --text "what must be true" [--test "<suite>"]');
9572
10171
  console.error('');
9573
10172
  }
9574
10173
  let statusResponse = null;
@@ -9598,12 +10197,17 @@ async function verifyCommand(flags, subcommand) {
9598
10197
  ticketId: flags.ticket,
9599
10198
  account: collectVerificationAccount(cwd, accountId, baseUrl, dataMode),
9600
10199
  source,
9601
- manifest
10200
+ manifest,
10201
+ noContract: flagEnabled(flags['no-contract']) || undefined
9602
10202
  });
9603
10203
  const envelope = response.envelope;
9604
10204
  writeLocalVerificationEnvelope(cwd, envelope);
9605
10205
  writeActiveVerificationEnvelope(cwd, envelope.envelopeId, activeContext);
9606
10206
  console.log('Verification envelope started:', envelope.envelopeId);
10207
+ if (!manifestRequiredEvidence(manifest).length && !manifestClaims(manifest).length) {
10208
+ console.log('Mode: evidence log (no acceptance contract asked for). Commands attach world-stamped evidence;');
10209
+ console.log(' nothing is outstanding. Add `verify claim <id> --text "..."` only if someone needs a verdict.');
10210
+ }
9607
10211
  printLocalCommandContext(cwd, flags, { accountId, branchName, workspace, dataMode });
9608
10212
  printLocalStateWarnings(cwd, flags);
9609
10213
  console.log('Target:', checkoutWorldLine + ', dataMode=' + dataMode +
@@ -9683,6 +10287,40 @@ async function verifyCommand(flags, subcommand) {
9683
10287
  return envelope;
9684
10288
  }
9685
10289
 
10290
+ if (sub === 'claim' || sub === 'claims') {
10291
+ const claimId = flags.id || (flags._ && flags._[2]);
10292
+ const text = flags.text || flags.claim || flags.summary || flags.note;
10293
+ if (!claimId) {
10294
+ throw new Error('Usage: remits-cli verify claim <id> --text "what must be true" [--test "<suite>"] [--packet-type <type>]');
10295
+ }
10296
+ const claim = { id: String(claimId) };
10297
+ if (text) claim.text = String(text);
10298
+ // A claim may name the shape of evidence that proves it, so one line of contract can bind to a suite
10299
+ // instead of needing a second parallel requiredEvidence list kept in sync by hand.
10300
+ const suite = flags.test || flags.suite;
10301
+ if (suite) {
10302
+ claim.packetType = 'test_run';
10303
+ claim.suite = String(suite);
10304
+ if (flags.case || flags.cases) {
10305
+ claim.cases = [].concat(flags.case || []).concat(flags.cases || [])
10306
+ .flatMap((value) => String(value).split(',')).map((value) => value.trim()).filter(Boolean);
10307
+ }
10308
+ } else if (flags['packet-type'] || flags.packetType) {
10309
+ claim.packetType = String(flags['packet-type'] || flags.packetType);
10310
+ }
10311
+ const response = await postVerificationCommand(api, cwd, session, accountId, 'claim', { envelopeId, claim });
10312
+ writeLocalVerificationEnvelope(cwd, response.envelope);
10313
+ const acceptance = (response.envelope || {}).acceptance || {};
10314
+ console.log('Claim recorded on envelope ' + envelopeId + ': ' + claim.id);
10315
+ console.log('Contract now requires ' + (acceptance.requiredEvidence || []).length + ' item(s).');
10316
+ console.log('Prove it by adding --claim ' + claim.id + ' to the command that demonstrates it, e.g.');
10317
+ console.log(' remits-cli verify test --test "<suite>" --claim ' + claim.id);
10318
+ console.log(' remits-cli verify attach --claim ' + claim.id + ' --note "what you observed"');
10319
+ printAcceptanceWarnings(response.envelope);
10320
+ printEnvelopeWarnings(response.envelope);
10321
+ return response.envelope;
10322
+ }
10323
+
9686
10324
  if (sub === 'attach') {
9687
10325
  const artifacts = [];
9688
10326
  let type = 'manual_observation';
@@ -9719,7 +10357,7 @@ async function verifyCommand(flags, subcommand) {
9719
10357
  artifacts,
9720
10358
  evidenceCategories
9721
10359
  };
9722
- const response = await appendVerificationPacket(api, cwd, session, accountId, { 'verify-envelope': envelopeId }, packet);
10360
+ const response = await appendVerificationPacket(api, cwd, session, accountId, Object.assign({}, flags, { 'verify-envelope': envelopeId }), packet);
9723
10361
  console.log('Evidence attached:', (response.packet || packet).packetId);
9724
10362
  return response;
9725
10363
  }
@@ -9730,7 +10368,9 @@ async function verifyCommand(flags, subcommand) {
9730
10368
  const packet = {
9731
10369
  type: sub === 'browser-start' ? 'browser_session' : (sub === 'browser-snapshot' ? 'browser_snapshot' : 'browser_step'),
9732
10370
  success: !flagEnabled(flags.failed),
9733
- claim: flags.claim || flags.label || action,
10371
+ // `--claim` now names a declared acceptance claim id, so the free-text label for a browser step is
10372
+ // `--label`. Reading both from one flag made the id double as prose and the prose double as an id.
10373
+ claim: flags.label || flags.text || action,
9734
10374
  world: buildCommandWorld(null, { accountId, dataMode, branchName, workspace, host: normalizeBaseUrl(baseUrl) }),
9735
10375
  browser: {
9736
10376
  action,
@@ -9743,7 +10383,7 @@ async function verifyCommand(flags, subcommand) {
9743
10383
  evidenceCategories: [sub, action === 'upload' ? 'browser_session.actual_upload' : null, action === 'click' ? 'browser_session.actual_clicks' : null]
9744
10384
  .filter(Boolean)
9745
10385
  };
9746
- const response = await appendVerificationPacket(api, cwd, session, accountId, { 'verify-envelope': envelopeId }, packet);
10386
+ const response = await appendVerificationPacket(api, cwd, session, accountId, Object.assign({}, flags, { 'verify-envelope': envelopeId }), packet);
9747
10387
  console.log('Browser evidence attached:', (response.packet || packet).packetId);
9748
10388
  return response;
9749
10389
  }
@@ -10701,6 +11341,7 @@ function buildRepoSnapshot(entry) {
10701
11341
  const repoFiles = [
10702
11342
  { label: 'account-info.json', path: entry.accountInfoPath || path.join(directory, 'account-info.json'), mode: 'json' },
10703
11343
  { label: 'account-hierarchy.json', path: path.join(directory, 'account-hierarchy.json'), mode: 'json' },
11344
+ { label: 'account-analytics.json', path: path.join(directory, 'account-analytics.json'), mode: 'json' },
10704
11345
  { label: '.remits-cli/active-actor/current-session.txt', path: localPaths.currentSessionFile, mode: 'text' },
10705
11346
  { label: '.remits-cli/shared/tools/tools.json', path: path.join(localPaths.toolsDir, 'tools.json'), mode: 'json' },
10706
11347
  { label: '.remits-cli/active-actor/session log (tail)', path: currentSessionLog, mode: 'tail' },
@@ -14708,23 +15349,57 @@ function slimActivitySmell(smell) {
14708
15349
  testName: s.testName || null,
14709
15350
  branchName: s.branchName || null,
14710
15351
  workspace: s.workspace || null,
15352
+ packetCount: s.packetCount == null ? null : s.packetCount,
15353
+ requiredCount: s.requiredCount == null ? null : s.requiredCount,
15354
+ openCount: s.openCount == null ? null : s.openCount,
15355
+ command: s.command || null,
14711
15356
  updatedAt: formatTime(s.updatedAtMs || s.updatedAt || s.at || s.createdAt)
14712
15357
  };
14713
15358
  }
14714
15359
 
15360
+ /**
15361
+ * The next COMMAND, not the smell restated. This list previously mapped each signal to its own summary
15362
+ * sentence, so "Next actions" contained no action; and it filtered on severity <= 2, which permanently
15363
+ * excluded `no_required_evidence` - the one signal that explains why an envelope never closes.
15364
+ * A signal that knows its own fix is actionable at any severity.
15365
+ */
15366
+ function buildActivityNextActions(smells) {
15367
+ const seen = new Set();
15368
+ return (smells || [])
15369
+ .filter((smell) => smell && (smell.command || smellSeverityRank(smell) <= 2))
15370
+ .sort((left, right) => {
15371
+ const command = (right.command ? 1 : 0) - (left.command ? 1 : 0);
15372
+ if (command) return command;
15373
+ return smellSeverityRank(left) - smellSeverityRank(right);
15374
+ })
15375
+ .map((smell) => {
15376
+ const title = smell.title || smell.envelopeSummary || smell.testName ||
15377
+ (Array.isArray(smell.workspaces) ? smell.workspaces.join(', ') : null);
15378
+ return {
15379
+ type: smell.type || 'signal',
15380
+ severity: smell.severity || 'info',
15381
+ why: (title ? title + ' — ' : '') + (smell.summary || 'inspect this signal'),
15382
+ command: smell.command || null
15383
+ };
15384
+ })
15385
+ .filter((action) => {
15386
+ // Dedupe on the COMMAND when there is one: two different signals about one envelope that resolve to
15387
+ // the same next move are one next move.
15388
+ const key = action.command ? 'cmd|' + action.command : action.type + '|' + action.why;
15389
+ if (seen.has(key)) return false;
15390
+ seen.add(key);
15391
+ return true;
15392
+ })
15393
+ .slice(0, 6);
15394
+ }
15395
+
14715
15396
  function compactActivityInspect(response) {
14716
15397
  const counts = response.counts || {};
14717
15398
  const smells = sortedActivitySmells(response.smells);
14718
15399
  const lanes = Array.isArray(response.lanes) ? response.lanes : [];
14719
15400
  const envelopes = Array.isArray(response.envelopes) ? response.envelopes : [];
14720
15401
  const testRuns = Array.isArray(response.testRuns) ? response.testRuns : [];
14721
- const nextActions = smells
14722
- .filter((smell) => smellSeverityRank(smell) <= 2)
14723
- .slice(0, 5)
14724
- .map((smell) => {
14725
- const title = smell.title || smell.envelopeSummary || smell.testName;
14726
- return (smell.type || 'signal') + (title ? ' (' + title + ')' : '') + ': ' + (smell.summary || 'inspect this signal');
14727
- });
15402
+ const nextActions = buildActivityNextActions(smells);
14728
15403
 
14729
15404
  return {
14730
15405
  success: response.success,
@@ -14809,7 +15484,14 @@ function printActivityInspectSummary(response) {
14809
15484
  (counts.lanes || 0) + ' lane(s), ' +
14810
15485
  (counts.verificationEnvelopes || 0) + ' envelope(s), ' +
14811
15486
  (counts.testRuns || 0) + ' test run(s), ' +
14812
- (counts.corpusResults || 0) + ' corpus result(s)');
15487
+ (counts.corpusResults || 0) + ' corpus result(s)' +
15488
+ (counts.distinctStagedEntries != null && counts.distinctStagedEntries !== counts.stagedEntries
15489
+ ? ', ' + counts.stagedEntries + ' staged entries across lanes (' + counts.distinctStagedEntries + ' distinct)'
15490
+ : (counts.stagedEntries ? ', ' + counts.stagedEntries + ' staged entries' : '')));
15491
+ if (response.envelopeScope === 'all-data-lanes') {
15492
+ console.log('Note: test runs and corpus results are narrowed to the ' + response.dataMode +
15493
+ ' lane; envelopes are listed across BOTH lanes (stage/sync evidence is lane-independent).');
15494
+ }
14813
15495
 
14814
15496
  const smells = sortedActivitySmells(response.smells);
14815
15497
  console.log('');
@@ -14819,6 +15501,18 @@ function printActivityInspectSummary(response) {
14819
15501
  console.log('Top smells (' + smells.length + '):');
14820
15502
  smells.slice(0, 20).forEach((smell) => {
14821
15503
  console.log(' [' + (smell.severity || 'info') + '] ' + (smell.type || 'signal') + ': ' + smell.summary);
15504
+ // The numbers were in the JSON and not on the line, so twenty signals read as three sentences repeated.
15505
+ const facts = [
15506
+ smell.packetCount != null ? smell.packetCount + ' packet(s)' : null,
15507
+ smell.currentFailureCount != null ? smell.currentFailureCount + ' current failure(s)' : null,
15508
+ smell.requiredCount != null ? smell.requiredCount + ' requirement(s)' : null,
15509
+ smell.openCount != null ? smell.openCount + ' open' : null,
15510
+ smell.recentRuns != null ? smell.recentRuns + ' recent run(s)' : null,
15511
+ smell.failedRuns != null ? smell.failedRuns + ' failed' : null,
15512
+ smell.stagedCount != null ? smell.stagedCount + ' staged' : null,
15513
+ Array.isArray(smell.workspaces) ? smell.workspaces.join(', ') : null
15514
+ ].filter(Boolean);
15515
+ if (facts.length) console.log(' ' + facts.join(' · '));
14822
15516
  const pivots = [
14823
15517
  smell.accountId ? 'account ' + smell.accountId : null,
14824
15518
  smell.laneId ? 'lane ' + smell.laneId : null,
@@ -14831,12 +15525,15 @@ function printActivityInspectSummary(response) {
14831
15525
  if (pivots.length) console.log(' ' + pivots.join(' · '));
14832
15526
  });
14833
15527
  if (smells.length > 20) console.log(' ...' + (smells.length - 20) + ' more; rerun with --json');
14834
- const actions = compactActivityInspect(response).nextActions;
14835
- if (actions.length) {
14836
- console.log('');
14837
- console.log('Next actions:');
14838
- actions.forEach((action) => console.log(' - ' + action));
14839
- }
15528
+ }
15529
+ const actions = buildActivityNextActions(smells);
15530
+ if (actions.length) {
15531
+ console.log('');
15532
+ console.log('Next actions:');
15533
+ actions.forEach((action) => {
15534
+ console.log(' - ' + action.why);
15535
+ if (action.command) console.log(' ' + action.command);
15536
+ });
14840
15537
  }
14841
15538
 
14842
15539
  // "2 agent(s)" with no names is the shape of the question, not the answer: an agent asking who else is in
@@ -14882,9 +15579,20 @@ function printActivityInspectSummary(response) {
14882
15579
  console.log('');
14883
15580
  console.log('Recent verification envelopes:');
14884
15581
  envelopes.slice(0, 8).forEach((env) => {
15582
+ const world = [env.branchName || null, env.workspace ? 'ws:' + env.workspace : null, env.dataMode || null]
15583
+ .filter(Boolean).join('/');
14885
15584
  console.log(' - ' + (env.status || 'unknown') + ' · ' + (env.packetCount || 0) + ' packet(s) · ' +
15585
+ (world ? '[' + world + '] · ' : '') +
14886
15586
  (env.summary || env.envelopeId) + ' · ' + formatTime(env.updatedAtMs || env.lastPacketAtMs || env.createdAt));
14887
- if (env.currentFailureCount) console.log(' current failures: ' + env.currentFailureCount);
15587
+ if (env.noRequiredEvidence) {
15588
+ console.log(' evidence log — no verdict requested; add a claim only if someone needs one: remits-cli verify claim <id> --text "..." --envelope ' + env.envelopeId);
15589
+ }
15590
+ // A failure recorded in an envelope that promised nothing is HISTORY, not an outstanding task — the
15591
+ // same rule the smells already apply (`promises` in activitySmells). Printing it as "current failures"
15592
+ // re-created, one line lower, exactly the unfinishable-looking work the evidence/verdict split removed.
15593
+ if (env.currentFailureCount) {
15594
+ console.log(' ' + (env.noRequiredEvidence ? 'failed evidence recorded: ' : 'current failures: ') + env.currentFailureCount);
15595
+ }
14888
15596
  });
14889
15597
  }
14890
15598
 
@@ -14946,6 +15654,62 @@ function listWorkstreamIndexes(cwd) {
14946
15654
  .sort((a, b) => String(b.updatedAt || '').localeCompare(String(a.updatedAt || '')));
14947
15655
  }
14948
15656
 
15657
+ async function evidenceCommand(flags) {
15658
+ const cwd = process.cwd();
15659
+ ensureLocalState(cwd);
15660
+ const limit = Math.max(1, Math.min(parseInt(flags.limit || flags.max || '40', 10) || 40, 500));
15661
+ const entries = readEvidenceTrail(cwd, limit);
15662
+ if (flagEnabled(flags.json)) {
15663
+ console.log(JSON.stringify({ success: true, actor: resolveLocalActor().id, count: entries.length, entries }, null, 2));
15664
+ return entries;
15665
+ }
15666
+ const paths = localStatePaths(cwd);
15667
+ console.log('Evidence trail for actor ' + paths.actor.id + ' (' + paths.evidenceFile + ')');
15668
+ if (!entries.length) {
15669
+ console.log('Nothing recorded yet. Every stage, test, token, tool and sync appends one world-stamped line here,');
15670
+ console.log('with no envelope and nothing to start.');
15671
+ return entries;
15672
+ }
15673
+ // Grouped by WORLD, because "what have I proven" is only answerable per world. An agent that ran the same
15674
+ // suite in two lanes has two different facts, and a flat list hides exactly that.
15675
+ const groups = new Map();
15676
+ entries.forEach((entry) => {
15677
+ const key = describeEvidenceWorld(entry.world);
15678
+ if (!groups.has(key)) groups.set(key, []);
15679
+ groups.get(key).push(entry);
15680
+ });
15681
+ groups.forEach((rows, world) => {
15682
+ console.log('');
15683
+ console.log(world);
15684
+ const byClaim = new Map();
15685
+ rows.forEach((row) => {
15686
+ const key = (row.type || 'unknown') + '|' + (row.claim || row.type || 'evidence') + '|' + evidenceOutcome(row);
15687
+ if (!byClaim.has(key)) byClaim.set(key, { row, count: 0 });
15688
+ const bucket = byClaim.get(key);
15689
+ bucket.count += 1;
15690
+ bucket.row = row;
15691
+ });
15692
+ Array.from(byClaim.values())
15693
+ .sort((left, right) => right.count - left.count)
15694
+ .slice(0, 15)
15695
+ .forEach((bucket) => {
15696
+ const row = bucket.row;
15697
+ const layer = row.world && row.world.sourceLayer ? ' source ' + row.world.sourceLayer : '';
15698
+ console.log(' ' + (bucket.count > 1 ? bucket.count + ' x ' : '') +
15699
+ evidenceOutcomeLabel(row) + (row.type || 'unknown') + ': ' + (row.claim || '(no claim)') +
15700
+ ' ' + formatTime(row.ts) + layer +
15701
+ (Array.isArray(row.claims) && row.claims.length ? ' [claims ' + row.claims.join(', ') + ']' : '') +
15702
+ (row.envelopeId ? ' [envelope ' + row.envelopeId.slice(0, 8) + ']' : ''));
15703
+ });
15704
+ if (byClaim.size > 15) console.log(' ...' + (byClaim.size - 15) + ' more distinct entries; --json for all');
15705
+ });
15706
+ console.log('');
15707
+ console.log(entries.length + ' entry(s) shown. This is a log, not a verdict: nothing here is outstanding.');
15708
+ console.log('If someone needs a checkable verdict, start an envelope and name what must be true:');
15709
+ console.log(' remits-cli verify start --summary "..." && remits-cli verify claim <id> --text "..."');
15710
+ return entries;
15711
+ }
15712
+
14949
15713
  async function workstreamCommand(flags, subcommand) {
14950
15714
  const cwd = process.cwd();
14951
15715
  ensureLocalState(cwd);
@@ -15653,10 +16417,18 @@ function releaseAutoUpdateLock() {
15653
16417
  } catch (_) {}
15654
16418
  }
15655
16419
 
16420
+ /**
16421
+ * May this invocation check for, and install, a newer CLI?
16422
+ *
16423
+ * `--json` used to disable it outright, to keep update chatter out of a document a program is parsing.
16424
+ * The effect was that the callers who pass `--json` on every command - which is every AI agent - never
16425
+ * upgraded at all, so each release of better diagnostics reached the population that needed it least.
16426
+ * The chatter now goes to stderr (see `autoUpdateIfNeeded`), which is where a program's stdout contract
16427
+ * says it belongs, and the check stays on.
16428
+ */
15656
16429
  function shouldAutoUpdate(command, flags) {
15657
16430
  if (!command || command === 'help' || command === '--help') return false;
15658
16431
  if (flags && flags['no-auto-update']) return false;
15659
- if (flags && flagEnabled(flags.json)) return false;
15660
16432
  if (process.env.REMITS_CLI_AUTO_UPDATE && AUTO_UPDATE_DISABLED_VALUES.has(String(process.env.REMITS_CLI_AUTO_UPDATE).toLowerCase())) {
15661
16433
  return false;
15662
16434
  }
@@ -15664,10 +16436,31 @@ function shouldAutoUpdate(command, flags) {
15664
16436
  return true;
15665
16437
  }
15666
16438
 
16439
+ /** True when the last registry check is recent enough that another one would only cost latency. */
16440
+ function autoUpdateCheckedRecently() {
16441
+ try {
16442
+ const stamp = JSON.parse(fs.readFileSync(AUTO_UPDATE_STAMP_FILE, 'utf8'));
16443
+ return Number(stamp.checkedAtMs) > Date.now() - AUTO_UPDATE_CHECK_INTERVAL_MS;
16444
+ } catch (_) {
16445
+ return false;
16446
+ }
16447
+ }
16448
+
16449
+ function recordAutoUpdateCheck(latest) {
16450
+ try {
16451
+ ensureSessionDir();
16452
+ fs.writeFileSync(AUTO_UPDATE_STAMP_FILE,
16453
+ JSON.stringify({ checkedAtMs: Date.now(), latest: latest || null }, null, 2));
16454
+ } catch (_) {}
16455
+ }
16456
+
15667
16457
  function autoUpdateIfNeeded(originalArgv, options = {}) {
15668
16458
  const requireSuccess = Boolean(options.requireSuccess);
16459
+ // Update chatter is diagnostics, never part of a `--json` document. stderr keeps both promises at once.
16460
+ const note = (line) => console.error(line);
15669
16461
  let lockAcquired = false;
15670
16462
  try {
16463
+ if (!requireSuccess && autoUpdateCheckedRecently()) return false;
15671
16464
  lockAcquired = acquireAutoUpdateLock();
15672
16465
  if (!lockAcquired) {
15673
16466
  if (requireSuccess) {
@@ -15681,20 +16474,22 @@ function autoUpdateIfNeeded(originalArgv, options = {}) {
15681
16474
  encoding: 'utf8',
15682
16475
  stdio: ['ignore', 'pipe', 'ignore']
15683
16476
  }).trim();
16477
+ recordAutoUpdateCheck(latest);
15684
16478
  if (latest && compareSemver(latest, currentVersion) > 0) {
15685
- console.log('[update] New version available: ' + currentVersion + ' -> ' + latest + '. Installing...');
16479
+ note('[update] New version available: ' + currentVersion + ' -> ' + latest + '. Installing...');
16480
+ // npm's own progress output goes to stderr too: a `--json` caller's stdout must stay one document.
15686
16481
  const install = spawnSync(npmCommand(), ['install', '-g', REMITS_CLI_PACKAGE_NAME + '@latest'], {
15687
- stdio: 'inherit',
16482
+ stdio: ['ignore', process.stderr, process.stderr],
15688
16483
  env: process.env
15689
16484
  });
15690
16485
  if (install.error || install.status !== 0) {
15691
16486
  if (requireSuccess) {
15692
16487
  throw new Error('Auto-update failed; remits-cli start requires the latest published version before continuing.');
15693
16488
  }
15694
- console.log('[update] Update failed; continuing with version ' + currentVersion + '.');
16489
+ note('[update] Update failed; continuing with version ' + currentVersion + '.');
15695
16490
  return false;
15696
16491
  }
15697
- console.log('[update] Updated to ' + latest + '. Re-running command...');
16492
+ note('[update] Updated to ' + latest + '. Re-running command...');
15698
16493
  const rerun = spawnSync(remitsCliCommand(), originalArgv, {
15699
16494
  stdio: 'inherit',
15700
16495
  env: Object.assign({}, process.env, { REMITS_CLI_AUTO_UPDATED: '1' })
@@ -15703,7 +16498,7 @@ function autoUpdateIfNeeded(originalArgv, options = {}) {
15703
16498
  if (requireSuccess) {
15704
16499
  throw new Error('Auto-update succeeded but re-running the updated remits-cli command failed.');
15705
16500
  }
15706
- console.log('[update] Re-run failed; continuing with version ' + currentVersion + '.');
16501
+ note('[update] Re-run failed; continuing with version ' + currentVersion + '.');
15707
16502
  return false;
15708
16503
  }
15709
16504
  process.exitCode = rerun.status === null ? 1 : rerun.status;
@@ -15862,13 +16657,14 @@ function printComponentsHelp(subcommand) {
15862
16657
  console.log(' remits-cli components branch <name> [--json] # overridden/added/removed + drift');
15863
16658
  console.log(' remits-cli components branch <name> --diff <componentId> --component-type <kind> [--json]');
15864
16659
  console.log(' remits-cli components branch <name> --subscribers [--json]');
16660
+ console.log(' remits-cli components branch <name> --copy-to <newBranch> [--dry-run] [--force] [--json]');
15865
16661
  console.log(' remits-cli components branch <name> --subscribe <accountId> [--parent-account <id>] [--domain <host>] [--dry-run] [--confirm-primary-edge]');
15866
16662
  console.log(' remits-cli components branch <name> --unsubscribe <accountId>');
15867
16663
  console.log(' remits-cli components branch <name> --retire [--force]');
15868
16664
  }
15869
16665
 
15870
16666
  function printTestHelp() {
15871
- console.log('Usage: remits-cli test run --test <id|name> [--base-url URL] [--branch stagingScope] [--names "a|b"] [--watch true|false] [--data-mode test|prod] [--as-account ID] [--variant-branch NAME|none] [--json]');
16667
+ console.log('Usage: remits-cli test run --test <id|name> [--base-url URL] [--branch stagingScope] [--names "a|b"] [--watch true|false] [--wait true|false] [--data-mode test|prod] [--as-account ID] [--variant-branch NAME|none] [--json]');
15872
16668
  console.log(' remits-cli test status --task-id <taskId> [--base-url URL] [--account-id ID] [--branch stagingScope] [--data-mode test|prod] [--json]');
15873
16669
  console.log(' remits-cli test runs [--test <id|name>] [--limit 20] [--compare] [--json] # durable run history');
15874
16670
  console.log(' remits-cli test compare --base <taskId> --head <taskId> [--json] # per-case outcome/cost deltas');
@@ -15887,6 +16683,8 @@ function printTestHelp() {
15887
16683
  console.log(' to force production/subscription semantics from a variant checkout.');
15888
16684
  console.log(' --branch changes only the CLI staging namespace for test execution. Pair an unused');
15889
16685
  console.log(' value with --variant-branch none when existing staged entries would shadow DB rows.');
16686
+ console.log(' --watch false disables websocket progress streaming; the CLI still waits for final status.');
16687
+ console.log(' --wait false returns after launch with a task id; test status attaches terminal evidence.');
15890
16688
  console.log(' --json prints only the final status JSON to stdout; banners and progress go to stderr.');
15891
16689
  console.log(' --data-mode prod intentionally targets live production data.');
15892
16690
  console.log(' Every finished run is recorded durably: test status keeps working after the live status');
@@ -16021,6 +16819,9 @@ function printVerifyHelp() {
16021
16819
  console.log(' remits-cli verify use <envelopeId>');
16022
16820
  console.log(' remits-cli verify current');
16023
16821
  console.log(' remits-cli verify clear [--all]');
16822
+ console.log(' remits-cli verify claim <id> --text "what must be true" [--test "<suite>"] [--packet-type TYPE]');
16823
+ console.log(' # declares a contract in ONE command when someone needs a verdict; no-contract envelopes are evidence logs');
16824
+ console.log(' # prove it by adding --claim <id> to any verify test/token/tool/stage/attach command');
16024
16825
  console.log(' remits-cli verify show [--envelope ID] [--summary|--compact] [--json]');
16025
16826
  console.log(' remits-cli verify status [--envelope ID]');
16026
16827
  console.log(' remits-cli verify report [--envelope ID]');
@@ -16142,8 +16943,9 @@ async function main() {
16142
16943
  console.log(' remits-cli tool --name <toolName> [--base-url URL] [--account-id ID] [--as-account ID] [--target-account ID] [--branch BRANCH] [--input \"{...}\"|--input-file file.json] [--data-mode test|prod] [--scope self|children|hierarchy] [--account-ids 1,2,3] [--variant-branch NAME|none] [--timeout-ms 60000] [--async true --wait true]');
16143
16944
  console.log(' remits-cli tool status --call-id <callId> [--base-url URL] [--account-id ID] [--data-mode test|prod]');
16144
16945
  console.log(' remits-cli workspace [show|use <name>|use --auto|clear] # isolate staging when several agents share a repo');
16145
- console.log(' remits-cli verify start --summary "..." [--manifest file.json] # start a verification envelope');
16146
- console.log(' remits-cli verify [current|list|use|clear|show|status|report|attach|abandon|supersede|test|token|stage|sync|tool]');
16946
+ console.log(' remits-cli evidence [--limit N] [--json] # what you have run and in which world (always recorded)');
16947
+ console.log(' remits-cli verify start --summary "..." [--manifest file.json] # OPTIONAL: ask for a checkable verdict');
16948
+ console.log(' remits-cli verify [current|list|use|clear|show|status|report|claim|attach|abandon|supersede|test|token|stage|sync|tool]');
16147
16949
  console.log(' remits-cli components stage [--workset|--changed-only] [--base-url URL] [--account-id ID] [--branch BRANCH] [--workspace NAME] [--data-mode test|prod] [--json|--verbose]');
16148
16950
  console.log(' remits-cli components status [--base-url URL] [--account-id ID] [--branch BRANCH] [--workspace NAME] [--component-type TYPE --component-id ID] [--json|--verbose]');
16149
16951
  console.log(' remits-cli components lanes [--base-url URL] [--account-id ID] [--json]');
@@ -16154,10 +16956,11 @@ async function main() {
16154
16956
  console.log(' remits-cli components promotion [<branch>] [--json] [--no-fail] # promotion readiness + ordered next steps');
16155
16957
  console.log(' remits-cli components branches [--json] # committed branch variants for this account');
16156
16958
  console.log(' remits-cli components branch <name> [--diff <componentId> --component-type <kind>] [--subscribers] [--json]');
16959
+ console.log(' remits-cli components branch <name> --copy-to <newBranch> [--dry-run] [--force] [--json]');
16157
16960
  console.log(' remits-cli components branch <name> --subscribe <accountId> [--dry-run] [--confirm-primary-edge]');
16158
16961
  console.log(' remits-cli components branch <name> --unsubscribe <accountId> # return that account to trunk');
16159
16962
  console.log(' remits-cli components branch <name> --retire [--force] # delete the branch\'s overlays');
16160
- console.log(' remits-cli test run --test <id|name> [--base-url URL] [--branch stagingScope] [--names "a|b"] [--watch true|false] [--data-mode test|prod] [--as-account ID] [--variant-branch NAME|none] [--json]');
16963
+ console.log(' remits-cli test run --test <id|name> [--base-url URL] [--branch stagingScope] [--names "a|b"] [--watch true|false] [--wait true|false] [--data-mode test|prod] [--as-account ID] [--variant-branch NAME|none] [--json]');
16161
16964
  console.log(' remits-cli test runs [--test <id|name>] [--compare] # durable run history; compare latest two');
16162
16965
  console.log(' remits-cli test compare --base <taskId> --head <taskId>');
16163
16966
  console.log(' remits-cli corpus import --manifest corpus-manifest.json # seed a test corpus: cases + artifacts');
@@ -16383,6 +17186,22 @@ async function main() {
16383
17186
  throw new Error('Unknown activity subcommand: ' + subcommand);
16384
17187
  }
16385
17188
 
17189
+ if (command === 'evidence') {
17190
+ if (wantsHelp) {
17191
+ console.log('Usage: remits-cli evidence [--limit N] [--json]');
17192
+ console.log('');
17193
+ console.log('What you have actually run, and in which world. Every stage, test, token, tool and sync');
17194
+ console.log('appends one world-stamped line automatically - there is nothing to start and nothing to');
17195
+ console.log('satisfy. Grouped by world, because the same suite run in two lanes is two different facts.');
17196
+ console.log('');
17197
+ console.log('This is per-actor, so agents sharing a checkout never read each other\'s trail.');
17198
+ console.log('Use `remits-cli verify` only when someone needs a checkable VERDICT on top of this.');
17199
+ return;
17200
+ }
17201
+ await evidenceCommand(args);
17202
+ return;
17203
+ }
17204
+
16386
17205
  if (command === 'workstream') {
16387
17206
  if (wantsHelp) {
16388
17207
  console.log('Usage: remits-cli workstream status [--workstream ID|--workspace NAME] [--json]');