@remits/remits-cli 0.1.136 → 0.1.137
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/index.js +569 -85
- package/package.json +1 -1
- package/skills/remits-cli/SKILL.md +1 -0
- package/skills/remits-cli/references/branch-variants.md +25 -0
- package/skills/remits-cli/references/command-reference.md +53 -2
- package/skills/remits-cli/references/component-resolution.md +13 -0
- package/skills/remits-cli/references/development-loop.md +43 -2
- package/skills/remits-cli/references/tool-reference.md +19 -2
package/index.js
CHANGED
|
@@ -4,18 +4,18 @@
|
|
|
4
4
|
## Table of Contents
|
|
5
5
|
|
|
6
6
|
- L22 Runtime Bootstrap And Shared State
|
|
7
|
-
-
|
|
8
|
-
-
|
|
9
|
-
-
|
|
10
|
-
-
|
|
11
|
-
-
|
|
12
|
-
-
|
|
13
|
-
-
|
|
14
|
-
-
|
|
15
|
-
-
|
|
16
|
-
-
|
|
17
|
-
-
|
|
18
|
-
-
|
|
7
|
+
- L123 Sessions, Account Resolution, And Production Guards
|
|
8
|
+
- L615 Local State, Workspaces, And Verification Evidence
|
|
9
|
+
- L2652 Account Repos, Guide Sync, And Platform Repo
|
|
10
|
+
- L3460 Component Discovery And HTTP Logging
|
|
11
|
+
- L4107 Skill Delivery And TOC Resolution
|
|
12
|
+
- L4417 Auth And Component Staging
|
|
13
|
+
- L5032 Component Summaries, Status, And Sync Gates
|
|
14
|
+
- L7104 Branches, Promotion, Commit, And Test Runs
|
|
15
|
+
- L9193 Tokens, Tools, Verification, And Config
|
|
16
|
+
- L10620 Service Dashboard And WebSocket Listener
|
|
17
|
+
- L12742 Agent And Ticket Workflows
|
|
18
|
+
- L16337 Help, Auto Update, And Command Dispatch
|
|
19
19
|
*/
|
|
20
20
|
|
|
21
21
|
/*
|
|
@@ -58,6 +58,10 @@ const SERVICE_STATE_FILE = path.join(SESSION_DIR, 'service-state.json');
|
|
|
58
58
|
const ISSUES_DIR = path.join(SESSION_DIR, 'issues');
|
|
59
59
|
const WEBSOCKET_STATE_FILE = path.join(SESSION_DIR, 'websocket-state.json');
|
|
60
60
|
const AUTO_UPDATE_LOCK_DIR = path.join(SESSION_DIR, 'auto-update.lock');
|
|
61
|
+
const AUTO_UPDATE_STAMP_FILE = path.join(SESSION_DIR, 'auto-update-check.json');
|
|
62
|
+
// `npm view` is a network round trip. It used to run on EVERY command, which is a per-command latency tax
|
|
63
|
+
// on the agents that issue the most commands.
|
|
64
|
+
const AUTO_UPDATE_CHECK_INTERVAL_MS = 6 * 60 * 60 * 1000;
|
|
61
65
|
const AUTO_UPDATE_DISABLED_VALUES = new Set(['0', 'false', 'no', 'off']);
|
|
62
66
|
const REMITS_CLI_PACKAGE_NAME = '@remits/remits-cli';
|
|
63
67
|
const DEFAULT_DATA_MODE = 'test';
|
|
@@ -154,6 +158,14 @@ function flagEnabled(value) {
|
|
|
154
158
|
return value === true || value === 'true' || value === '1' || value === 'yes';
|
|
155
159
|
}
|
|
156
160
|
|
|
161
|
+
function flagDisabled(value) {
|
|
162
|
+
return value === false || value === 'false' || value === '0' || value === 'no';
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
function waitForTestCompletion(flags = {}) {
|
|
166
|
+
return !(flagDisabled(flags.wait) || flagEnabled(flags['no-wait']) || flagEnabled(flags.noWait));
|
|
167
|
+
}
|
|
168
|
+
|
|
157
169
|
function withStdoutRoutedToStderr(enabled, fn) {
|
|
158
170
|
if (!enabled) return fn();
|
|
159
171
|
const originalLog = console.log;
|
|
@@ -898,6 +910,22 @@ function printLocalStateWarnings(cwd, flags = {}) {
|
|
|
898
910
|
warnings.forEach((warning) => console.log(' - ' + warning));
|
|
899
911
|
}
|
|
900
912
|
|
|
913
|
+
/**
|
|
914
|
+
* "Your remits-cli is older than this platform's" - printed once per process, from ONE place.
|
|
915
|
+
*
|
|
916
|
+
* The platform attaches `cliAdvisory` to the two responses every working loop passes through (stage and
|
|
917
|
+
* test-run start) and returns nothing at all when the caller is current, so this renders only when there
|
|
918
|
+
* is something to act on. stderr, because a `--json` caller's stdout is a document.
|
|
919
|
+
*/
|
|
920
|
+
let cliAdvisoryPrinted = false;
|
|
921
|
+
function printCliAdvisory(response) {
|
|
922
|
+
const advisory = response && response.cliAdvisory;
|
|
923
|
+
if (!advisory || cliAdvisoryPrinted) return;
|
|
924
|
+
cliAdvisoryPrinted = true;
|
|
925
|
+
console.error('');
|
|
926
|
+
console.error('[remits-cli ' + (advisory.current || '?') + ' -> ' + (advisory.expected || '?') + '] ' + advisory.message);
|
|
927
|
+
}
|
|
928
|
+
|
|
901
929
|
function printLocalCommandContext(cwd, flags = {}, context = {}) {
|
|
902
930
|
const paths = localStatePaths(cwd);
|
|
903
931
|
const branchName = context.branchName || flags.branch || safeGitValue(cwd, 'git rev-parse --abbrev-ref HEAD') || 'unknown';
|
|
@@ -3834,6 +3862,12 @@ function stageTimeoutMs(components, flags = {}) {
|
|
|
3834
3862
|
|
|
3835
3863
|
function buildAxios(baseUrl, token, timeoutMs = 60000) {
|
|
3836
3864
|
const headers = token ? { Authorization: 'Bearer ' + token } : {};
|
|
3865
|
+
// The CLI version on EVERY request, from the one place every request is built. It used to travel only on
|
|
3866
|
+
// `agent register`, so the platform knew the version of the sessions that registered - and the agents that
|
|
3867
|
+
// loop hardest are exactly the ones that never do. Without it the platform cannot tell an agent that the
|
|
3868
|
+
// diagnostics it is missing exist, and every output improvement lands invisibly.
|
|
3869
|
+
const cliVersion = readCliVersion();
|
|
3870
|
+
if (cliVersion) headers['X-Remits-Cli-Version'] = cliVersion;
|
|
3837
3871
|
return axios.create({ baseURL: baseUrl, timeout: parsePositiveInt(timeoutMs, 60000), headers });
|
|
3838
3872
|
}
|
|
3839
3873
|
|
|
@@ -5232,6 +5266,9 @@ function printStageSummary(response, flags) {
|
|
|
5232
5266
|
if (typeof printAccountLanes === 'function') {
|
|
5233
5267
|
printAccountLanes(response);
|
|
5234
5268
|
}
|
|
5269
|
+
if (typeof printCliAdvisory === 'function') {
|
|
5270
|
+
printCliAdvisory(response);
|
|
5271
|
+
}
|
|
5235
5272
|
printComponentCommandResponse('Components stage', response, flags);
|
|
5236
5273
|
}
|
|
5237
5274
|
|
|
@@ -7083,6 +7120,10 @@ async function branchesComponentsCommand(flags) {
|
|
|
7083
7120
|
// after the subcommand is the branch name when present.
|
|
7084
7121
|
const positional = flags._ && flags._[2];
|
|
7085
7122
|
const branchName = flags.branch || positional;
|
|
7123
|
+
const copyToBranch = flags['copy-to'] || flags.copyTo;
|
|
7124
|
+
if ((flags['copy-to'] !== undefined || flags.copyTo !== undefined) && !branchName) {
|
|
7125
|
+
throw new Error('components branch <source> --copy-to <target> requires a source branch name');
|
|
7126
|
+
}
|
|
7086
7127
|
|
|
7087
7128
|
// Without a branch there is nothing to detail, so always list. With one, the mutating flags
|
|
7088
7129
|
// (--subscribe/--unsubscribe/--retire) win, then the narrower read views, and the default is the
|
|
@@ -7092,10 +7133,14 @@ async function branchesComponentsCommand(flags) {
|
|
|
7092
7133
|
if (flags.subscribe !== undefined) mode = 'subscribe';
|
|
7093
7134
|
else if (flags.unsubscribe !== undefined) mode = 'unsubscribe';
|
|
7094
7135
|
else if (flagEnabled(flags.retire)) mode = 'retire';
|
|
7136
|
+
else if (flags['copy-to'] !== undefined || flags.copyTo !== undefined) mode = 'copy';
|
|
7095
7137
|
else if (flags.diff !== undefined) mode = 'diff';
|
|
7096
7138
|
else if (flagEnabled(flags.subscribers)) mode = 'subscribers';
|
|
7097
7139
|
else mode = 'status';
|
|
7098
7140
|
}
|
|
7141
|
+
if (mode === 'copy' && (copyToBranch === true || !String(copyToBranch || '').trim())) {
|
|
7142
|
+
throw new Error('components branch <source> --copy-to <target> requires a target branch name');
|
|
7143
|
+
}
|
|
7099
7144
|
|
|
7100
7145
|
// `--diff 42` carries the component id inline; a bare `--diff` falls back to --component-id/--name.
|
|
7101
7146
|
const componentId = mode === 'diff'
|
|
@@ -7118,6 +7163,7 @@ async function branchesComponentsCommand(flags) {
|
|
|
7118
7163
|
componentType: flags['component-type'] || flags.type,
|
|
7119
7164
|
componentId,
|
|
7120
7165
|
componentName: flags['component-name'] || flags.name,
|
|
7166
|
+
targetBranchName: copyToBranch,
|
|
7121
7167
|
subscribeAccountId,
|
|
7122
7168
|
parentAccountId: flags['parent-account'],
|
|
7123
7169
|
// Optional branch-scoped custom host set on the same edge as the subscription.
|
|
@@ -7241,6 +7287,33 @@ function printBranchesSummary(response) {
|
|
|
7241
7287
|
return;
|
|
7242
7288
|
}
|
|
7243
7289
|
|
|
7290
|
+
if (response.mode === 'copy') {
|
|
7291
|
+
console.log(response.message || ('Copied branch overlays to ' + response.targetBranch));
|
|
7292
|
+
console.log('Source branch:', response.branch);
|
|
7293
|
+
console.log('Target branch:', response.targetBranch);
|
|
7294
|
+
// A dry run writes nothing, so `copied` is 0 by contract — printing it as "Copied overlays: 0"
|
|
7295
|
+
// under a "Would copy 2" message reads as a failed copy. Name the plan instead.
|
|
7296
|
+
if (response.dryRun) {
|
|
7297
|
+
console.log('Plan only (dry run) — overlays that would be copied:', response.plannedCopies || 0);
|
|
7298
|
+
if (response.existingCount) console.log('Existing overlays that would be replaced:', response.existingCount);
|
|
7299
|
+
} else {
|
|
7300
|
+
console.log('Copied overlays:', response.copied || 0);
|
|
7301
|
+
if (response.overwritten) console.log('Replaced existing overlays:', response.overwritten);
|
|
7302
|
+
}
|
|
7303
|
+
// A source overlay that is identical to trunk in both content and metadata is sparse and is never
|
|
7304
|
+
// stored, so it is named rather than silently missing from the count.
|
|
7305
|
+
const skippedSparse = response.skippedIdenticalToTrunk || [];
|
|
7306
|
+
if (skippedSparse.length) {
|
|
7307
|
+
console.log('Skipped as identical to trunk:', skippedSparse.join(', '));
|
|
7308
|
+
}
|
|
7309
|
+
// The number an operator needs BEFORE deciding to pass --force: a target branch with live
|
|
7310
|
+
// subscribers is code those accounts are running right now.
|
|
7311
|
+
if (response.targetSubscriberCount) {
|
|
7312
|
+
console.log('Live subscribers on the target branch:', response.targetSubscriberCount);
|
|
7313
|
+
}
|
|
7314
|
+
return;
|
|
7315
|
+
}
|
|
7316
|
+
|
|
7244
7317
|
if (response.mode === 'subscribe' || response.mode === 'unsubscribe' || response.mode === 'retire') {
|
|
7245
7318
|
console.log(response.message);
|
|
7246
7319
|
if (response.requiresConfirmation) {
|
|
@@ -7852,7 +7925,7 @@ async function waitForStatus(api, cwd, accountId, branchName, taskId, token, dat
|
|
|
7852
7925
|
dataMode
|
|
7853
7926
|
}).then((r) => r.data);
|
|
7854
7927
|
|
|
7855
|
-
if (status
|
|
7928
|
+
if (testStatusIsTerminal(status)) {
|
|
7856
7929
|
return status;
|
|
7857
7930
|
}
|
|
7858
7931
|
await new Promise((r) => setTimeout(r, pollDelayMs));
|
|
@@ -7860,6 +7933,10 @@ async function waitForStatus(api, cwd, accountId, branchName, taskId, token, dat
|
|
|
7860
7933
|
}
|
|
7861
7934
|
}
|
|
7862
7935
|
|
|
7936
|
+
function testStatusIsTerminal(status = {}) {
|
|
7937
|
+
return ['completed', 'failed', 'interrupted'].includes(String(status.status || '').toLowerCase());
|
|
7938
|
+
}
|
|
7939
|
+
|
|
7863
7940
|
function testEvidenceCategories(status = {}, selectedNames = []) {
|
|
7864
7941
|
const categories = new Set(['test_run']);
|
|
7865
7942
|
const test = status.test || {};
|
|
@@ -7989,6 +8066,108 @@ function detectNondeterministicTestRun(priorPackets, currentStatus, currentProve
|
|
|
7989
8066
|
return null;
|
|
7990
8067
|
}
|
|
7991
8068
|
|
|
8069
|
+
async function appendTerminalTestRunEvidence(options = {}) {
|
|
8070
|
+
const {
|
|
8071
|
+
api,
|
|
8072
|
+
cwd,
|
|
8073
|
+
session,
|
|
8074
|
+
accountId,
|
|
8075
|
+
flags,
|
|
8076
|
+
status,
|
|
8077
|
+
testRef,
|
|
8078
|
+
selectedNames = [],
|
|
8079
|
+
dataMode,
|
|
8080
|
+
dataModeSource,
|
|
8081
|
+
branchName,
|
|
8082
|
+
workspace,
|
|
8083
|
+
baseUrl,
|
|
8084
|
+
activeEnvelopeId,
|
|
8085
|
+
quiet
|
|
8086
|
+
} = options;
|
|
8087
|
+
const unmatched = (status.result && status.result.unmatchedTestNames) || [];
|
|
8088
|
+
const componentProvenance = testComponentProvenance(status);
|
|
8089
|
+
const sourceRevision = collectVerificationSource(cwd, flags);
|
|
8090
|
+
const priorPackets = await readVerificationPacketsForDiagnostics(api, cwd, session, accountId, activeEnvelopeId, {
|
|
8091
|
+
packetType: 'test_run',
|
|
8092
|
+
suite: status.test && status.test.name,
|
|
8093
|
+
max: 50
|
|
8094
|
+
});
|
|
8095
|
+
const nondeterminism = detectNondeterministicTestRun(priorPackets, status, componentProvenance, {
|
|
8096
|
+
laneContentHash: status.staging && status.staging.laneSummary && status.staging.laneSummary.contentHash,
|
|
8097
|
+
gitHead: sourceRevision.gitHead
|
|
8098
|
+
});
|
|
8099
|
+
|
|
8100
|
+
await appendVerificationPacket(api, cwd, session, accountId, flags, {
|
|
8101
|
+
type: 'test_run',
|
|
8102
|
+
success: status.status === 'completed' && !(status.result && status.result.failed > 0) && !unmatched.length,
|
|
8103
|
+
claim: 'Test run ' + String(testRef || (status.test && (status.test.name || status.test.id)) || status.taskId || 'unknown'),
|
|
8104
|
+
world: buildCommandWorld(status, {
|
|
8105
|
+
accountId,
|
|
8106
|
+
dataMode: status.dataMode || dataMode,
|
|
8107
|
+
dataModeSource,
|
|
8108
|
+
branchName,
|
|
8109
|
+
workspace,
|
|
8110
|
+
host: normalizeBaseUrl(baseUrl),
|
|
8111
|
+
sourceLayer: status.staging && status.staging.testComponentSource
|
|
8112
|
+
}),
|
|
8113
|
+
revision: Object.assign(sourceRevision, {
|
|
8114
|
+
compileSignatures: testCompileSignatures(status),
|
|
8115
|
+
componentProvenance
|
|
8116
|
+
}),
|
|
8117
|
+
nondeterministic: nondeterminism ? true : undefined,
|
|
8118
|
+
nondeterminism: nondeterminism || undefined,
|
|
8119
|
+
test: {
|
|
8120
|
+
taskId: status.taskId || (status.result && status.result.taskId),
|
|
8121
|
+
testId: status.test && status.test.id,
|
|
8122
|
+
testName: status.test && status.test.name,
|
|
8123
|
+
selectedCases: selectedNames,
|
|
8124
|
+
passed: status.result && status.result.passed,
|
|
8125
|
+
failed: status.result && status.result.failed,
|
|
8126
|
+
tests: status.result && status.result.tests,
|
|
8127
|
+
dataModeSource: status.dataModeSource || dataModeSource,
|
|
8128
|
+
unmatchedTestNames: unmatched
|
|
8129
|
+
},
|
|
8130
|
+
evidenceCategories: testEvidenceCategories(status, selectedNames),
|
|
8131
|
+
// What the run spent and whether that live AI was declared (withAiBudget), so an envelope can tell a
|
|
8132
|
+
// measurement run from a suite with missing mocks without re-reading every case.
|
|
8133
|
+
dependencies: testRunDependencies(status),
|
|
8134
|
+
summary: status.result && status.result.summary,
|
|
8135
|
+
limitations: []
|
|
8136
|
+
.concat(unmatched.length ? ['One or more requested test case selectors matched no case.'] : [])
|
|
8137
|
+
.concat(nondeterminism ? [nondeterminism.message] : []),
|
|
8138
|
+
rawRefs: { testStatusKey: status.taskId, durableTestRun: status.result && status.result.durableRecord ? status.taskId : undefined },
|
|
8139
|
+
status
|
|
8140
|
+
}, { quiet });
|
|
8141
|
+
|
|
8142
|
+
return { unmatched, nondeterminism };
|
|
8143
|
+
}
|
|
8144
|
+
|
|
8145
|
+
/**
|
|
8146
|
+
* One failed case per distinct failure root, largest root first, capped.
|
|
8147
|
+
*
|
|
8148
|
+
* A pivot block per failed case is one problem stated N times: the thirty-three failures of a real suite
|
|
8149
|
+
* are twenty-one roots, and the twelve repeats carry the same trace shape, the same components and the
|
|
8150
|
+
* same error. The grouped roll-up above already names every case; this picks the ones worth a full block.
|
|
8151
|
+
*/
|
|
8152
|
+
/** Truncate, and SAY that it was truncated — a silent cut reads as the whole message. */
|
|
8153
|
+
function truncateForPivot(text, max) {
|
|
8154
|
+
if (text.length <= max) return text;
|
|
8155
|
+
return text.slice(0, max) + '… (truncated; --json for the full text)';
|
|
8156
|
+
}
|
|
8157
|
+
|
|
8158
|
+
function selectPivotCases(tests, limit = 6) {
|
|
8159
|
+
const failed = (tests || []).filter((test) => test && test.passed === false);
|
|
8160
|
+
const seen = new Set();
|
|
8161
|
+
const representatives = [];
|
|
8162
|
+
failed.forEach((test) => {
|
|
8163
|
+
const root = testFailureRoot(test);
|
|
8164
|
+
if (seen.has(root)) return;
|
|
8165
|
+
seen.add(root);
|
|
8166
|
+
representatives.push(test);
|
|
8167
|
+
});
|
|
8168
|
+
return { shown: representatives.slice(0, limit), failedCount: failed.length, rootCount: representatives.length };
|
|
8169
|
+
}
|
|
8170
|
+
|
|
7992
8171
|
function printTestRunPivots(status = {}) {
|
|
7993
8172
|
const tests = status.result && Array.isArray(status.result.tests) ? status.result.tests : [];
|
|
7994
8173
|
if (!tests.length) return;
|
|
@@ -7996,7 +8175,8 @@ function printTestRunPivots(status = {}) {
|
|
|
7996
8175
|
if (slowest && slowest.duration != null) {
|
|
7997
8176
|
console.log('Slowest case:', (slowest.name || '(unnamed)') + ' in ' + formatDurationMs(slowest.duration));
|
|
7998
8177
|
}
|
|
7999
|
-
|
|
8178
|
+
const selection = selectPivotCases(tests);
|
|
8179
|
+
selection.shown.forEach((test) => {
|
|
8000
8180
|
console.log('Pivots for failed case:', test.name || '(unnamed)');
|
|
8001
8181
|
if (test.outcome) console.log(' Outcome:', test.outcome + (test.outcomeReason && test.outcomeReason !== test.error ? ' - ' + String(test.outcomeReason).slice(0, 300) : ''));
|
|
8002
8182
|
if (test.duration != null) console.log(' Duration:', formatDurationMs(test.duration));
|
|
@@ -8007,7 +8187,9 @@ function printTestRunPivots(status = {}) {
|
|
|
8007
8187
|
if (test.threadGroupingId || test.threadGroupId || test.traceId) {
|
|
8008
8188
|
console.log(' Trace:', test.traceId || test.threadGroupingId || test.threadGroupId);
|
|
8009
8189
|
}
|
|
8010
|
-
|
|
8190
|
+
// 500 chars cut the replay diagnosis mid-sentence, losing the half that says what to DO. A pivot is the
|
|
8191
|
+
// block a reader acts on; truncating its one actionable sentence to save four lines is a bad trade.
|
|
8192
|
+
if (test.error) console.log(' Error:', truncateForPivot(String(test.error), 1200));
|
|
8011
8193
|
const diagnostics = test.diagnostics && typeof test.diagnostics === 'object' ? test.diagnostics : null;
|
|
8012
8194
|
if (diagnostics && Object.keys(diagnostics).length) {
|
|
8013
8195
|
console.log(' Diagnostics:', JSON.stringify(diagnostics).slice(0, 1200));
|
|
@@ -8023,6 +8205,12 @@ function printTestRunPivots(status = {}) {
|
|
|
8023
8205
|
console.log(' Live HTTP calls:', test.liveHttpCalls.length + (intentional ? ' (' + intentional + ' intentional, inside withAiBudget)' : ''));
|
|
8024
8206
|
}
|
|
8025
8207
|
});
|
|
8208
|
+
// Named, not silently dropped: a reader must be able to tell "this is everything" from "this is a sample".
|
|
8209
|
+
if (selection.rootCount > selection.shown.length) {
|
|
8210
|
+
console.log('Pivots shown for ' + selection.shown.length + ' of ' + selection.rootCount +
|
|
8211
|
+
' failure root(s) (' + selection.failedCount + ' failed case(s)). Every root is listed above; ' +
|
|
8212
|
+
'use --names "<case>" for one, or --json for all.');
|
|
8213
|
+
}
|
|
8026
8214
|
}
|
|
8027
8215
|
|
|
8028
8216
|
function formatDurationMs(value) {
|
|
@@ -8236,6 +8424,114 @@ function stagedMaskedByVariantLines(masked, accountId) {
|
|
|
8236
8424
|
return lines;
|
|
8237
8425
|
}
|
|
8238
8426
|
|
|
8427
|
+
/**
|
|
8428
|
+
* The assertion ROOT of one failed case: what broke, with the particulars of this case removed.
|
|
8429
|
+
*
|
|
8430
|
+
* Deliberately built only from properties of the JVM/Groovy failure format - a power-assert's
|
|
8431
|
+
* `Expression:`/`Values:` decoration, an `assert <expr>` head, an exception class prefix - and never from
|
|
8432
|
+
* any vocabulary a suite happens to use. Identifiers and numbers are replaced because two cases failing
|
|
8433
|
+
* the same way differ exactly in those.
|
|
8434
|
+
*/
|
|
8435
|
+
function testFailureRoot(test = {}) {
|
|
8436
|
+
let text = String(test.error || test.outcomeReason || test.outcome || 'failed').trim();
|
|
8437
|
+
// Groovy's power assert appends the rendered expression and every intermediate value.
|
|
8438
|
+
text = text.split(/\.\s+(?:Expression|Values):/)[0];
|
|
8439
|
+
const assertion = text.match(/assert\s+(.+)$/);
|
|
8440
|
+
if (assertion) text = 'assert ' + assertion[1];
|
|
8441
|
+
text = text
|
|
8442
|
+
.replace(/['"][0-9a-fA-F]{8}-[0-9a-fA-F-]{4,}['"]/g, "'<id>'")
|
|
8443
|
+
.replace(/\b[0-9a-fA-F]{8}-[0-9a-fA-F-]{27,}\b/g, '<id>')
|
|
8444
|
+
.replace(/\b\d[\d.,]*\b/g, '<n>')
|
|
8445
|
+
.replace(/\s+/g, ' ')
|
|
8446
|
+
.trim();
|
|
8447
|
+
if (text.length <= FAILURE_ROOT_LABEL_CHARS) return text;
|
|
8448
|
+
// Cut on a word boundary: "found no usable stored provid" reads as a different error than the one it is.
|
|
8449
|
+
const cut = text.slice(0, FAILURE_ROOT_LABEL_CHARS);
|
|
8450
|
+
const lastSpace = cut.lastIndexOf(' ');
|
|
8451
|
+
return (lastSpace > FAILURE_ROOT_LABEL_CHARS * 0.6 ? cut.slice(0, lastSpace) : cut) + '\u2026';
|
|
8452
|
+
}
|
|
8453
|
+
|
|
8454
|
+
/** Long enough to tell two failures apart, short enough that a root list stays a list. */
|
|
8455
|
+
const FAILURE_ROOT_LABEL_CHARS = 120;
|
|
8456
|
+
|
|
8457
|
+
/**
|
|
8458
|
+
* Failed cases grouped by that root, largest first.
|
|
8459
|
+
*
|
|
8460
|
+
* Thirty-three failures printed in run order read as thirty-three problems; the same run grouped is four.
|
|
8461
|
+
* This is the line that decides whether the next move is a narrow root-cause pass or another broad rerun,
|
|
8462
|
+
* so it is computed from the run itself rather than left to the reader.
|
|
8463
|
+
*/
|
|
8464
|
+
function testFailureGroups(status = {}) {
|
|
8465
|
+
const tests = status.result && Array.isArray(status.result.tests) ? status.result.tests : [];
|
|
8466
|
+
const groups = new Map();
|
|
8467
|
+
tests.filter((test) => test && test.passed === false && !test.interrupted).forEach((test) => {
|
|
8468
|
+
const root = testFailureRoot(test);
|
|
8469
|
+
if (!groups.has(root)) groups.set(root, { root, count: 0, cases: [] });
|
|
8470
|
+
const group = groups.get(root);
|
|
8471
|
+
group.count += 1;
|
|
8472
|
+
group.cases.push(test.name || '(unnamed)');
|
|
8473
|
+
});
|
|
8474
|
+
return Array.from(groups.values()).sort((left, right) => right.count - left.count || left.root.localeCompare(right.root));
|
|
8475
|
+
}
|
|
8476
|
+
|
|
8477
|
+
const FAILURE_GROUPS_SHOWN = 8;
|
|
8478
|
+
|
|
8479
|
+
function testFailureGroupLines(status = {}, options = {}) {
|
|
8480
|
+
const groups = testFailureGroups(status);
|
|
8481
|
+
if (!groups.length) return [];
|
|
8482
|
+
const failed = groups.reduce((sum, group) => sum + group.count, 0);
|
|
8483
|
+
const lines = [''];
|
|
8484
|
+
lines.push('Failure roots (' + failed + ' failed case(s), ' + groups.length + ' distinct root(s)):');
|
|
8485
|
+
groups.slice(0, FAILURE_GROUPS_SHOWN).forEach((group) => {
|
|
8486
|
+
lines.push(' ' + String(group.count) + 'x ' + group.root);
|
|
8487
|
+
// The representative case is what `--names` takes, so the next command is a copy of this line.
|
|
8488
|
+
lines.push(' e.g. ' + group.cases[0] + (group.count > 1 ? ' (+' + (group.count - 1) + ' more)' : ''));
|
|
8489
|
+
});
|
|
8490
|
+
if (groups.length > FAILURE_GROUPS_SHOWN) {
|
|
8491
|
+
lines.push(' ...' + (groups.length - FAILURE_GROUPS_SHOWN) + ' more root(s); --json for all');
|
|
8492
|
+
}
|
|
8493
|
+
const testRef = options.testRef || (status.test && (status.test.id || status.test.name)) ||
|
|
8494
|
+
(status.result && (status.result.testId || status.result.testName));
|
|
8495
|
+
if (groups[0] && groups[0].count > 1 && testRef) {
|
|
8496
|
+
lines.push('Largest root first: remits-cli test run --test ' + JSON.stringify(String(testRef)) +
|
|
8497
|
+
' --names ' + JSON.stringify(groups[0].cases[0]));
|
|
8498
|
+
}
|
|
8499
|
+
return lines;
|
|
8500
|
+
}
|
|
8501
|
+
|
|
8502
|
+
/**
|
|
8503
|
+
* The verdict of a run in four lines, shared by `test run` and `test status`.
|
|
8504
|
+
*
|
|
8505
|
+
* `test run` used to print the whole run object as pretty JSON before its human summary. Measured on one
|
|
8506
|
+
* 92-case suite that is 220 KB - most of it per-case ids repeated once per case - spent by the command an
|
|
8507
|
+
* agent runs most often, on the turn where it has the least room left to reason. Every byte is still one
|
|
8508
|
+
* `--json` away.
|
|
8509
|
+
*/
|
|
8510
|
+
function testRunHeadlineLines(status = {}, options = {}) {
|
|
8511
|
+
const result = status.result || {};
|
|
8512
|
+
const lines = [];
|
|
8513
|
+
lines.push('Test run status: ' + (status.status || 'unknown'));
|
|
8514
|
+
if (options.taskId || status.taskId) lines.push('Task ID: ' + (options.taskId || status.taskId));
|
|
8515
|
+
if (options.source) lines.push('Source: ' + options.source);
|
|
8516
|
+
const actualDataMode = status.dataMode || result.dataMode || options.dataMode;
|
|
8517
|
+
let dataModeLine = 'Data mode: ' + (actualDataMode || 'unknown');
|
|
8518
|
+
// A durable run answers about the lane it RAN in, which need not be the lane this command asked for.
|
|
8519
|
+
// Reporting only the run's own lane is correct and reads as if the request had been honoured.
|
|
8520
|
+
if (options.requestedDataMode && actualDataMode && options.requestedDataMode !== actualDataMode) {
|
|
8521
|
+
dataModeLine += ' (you asked for ' + options.requestedDataMode + '; this run was recorded in the ' +
|
|
8522
|
+
actualDataMode + ' lane, and that is what it proves)';
|
|
8523
|
+
}
|
|
8524
|
+
lines.push(dataModeLine);
|
|
8525
|
+
if (result.total != null || result.passed != null) {
|
|
8526
|
+
lines.push('Cases: ' + (result.passed || 0) + ' passed, ' + (result.failed || 0) + ' failed, ' +
|
|
8527
|
+
(result.total || 0) + ' total');
|
|
8528
|
+
}
|
|
8529
|
+
if (status.error || status.message || result.error) {
|
|
8530
|
+
lines.push('Message: ' + (status.error || status.message || result.error));
|
|
8531
|
+
}
|
|
8532
|
+
return lines;
|
|
8533
|
+
}
|
|
8534
|
+
|
|
8239
8535
|
// The corpus-style roll-up printed after a run: outcome counts, case duration percentiles, AI usage split
|
|
8240
8536
|
// live/mocked, and the durable record. Built only from the result the platform returned.
|
|
8241
8537
|
function testRunSummaryLines(status = {}) {
|
|
@@ -8491,6 +8787,7 @@ async function testCommand(flags) {
|
|
|
8491
8787
|
printLocalStateWarnings(cwd, flags);
|
|
8492
8788
|
printStagingLane(branchName, workspace, workspaceSource(cwd, flags));
|
|
8493
8789
|
printStagingLaneOwnerNotice(start.staging || {});
|
|
8790
|
+
printCliAdvisory(start);
|
|
8494
8791
|
if (start.staging && Array.isArray(start.staging.accountLanes)) {
|
|
8495
8792
|
printOrphanedWorkspaceWarning(start.staging);
|
|
8496
8793
|
printAccountLanes({ accountLanes: start.staging.accountLanes, branchName, workspace });
|
|
@@ -8506,6 +8803,48 @@ async function testCommand(flags) {
|
|
|
8506
8803
|
});
|
|
8507
8804
|
runtimeState.currentTestTaskId = start.taskId;
|
|
8508
8805
|
|
|
8806
|
+
const waitForCompletion = waitForTestCompletion(flags);
|
|
8807
|
+
if (!waitForCompletion) {
|
|
8808
|
+
recordEvidenceEntry(cwd, {
|
|
8809
|
+
type: 'test_run',
|
|
8810
|
+
success: null,
|
|
8811
|
+
pending: true,
|
|
8812
|
+
claim: 'Test run ' + String(testRef),
|
|
8813
|
+
world: buildCommandWorld(start, {
|
|
8814
|
+
accountId,
|
|
8815
|
+
dataMode,
|
|
8816
|
+
dataModeSource,
|
|
8817
|
+
branchName,
|
|
8818
|
+
workspace,
|
|
8819
|
+
host: normalizeBaseUrl(baseUrl),
|
|
8820
|
+
sourceLayer: start.staging && start.staging.testComponentSource
|
|
8821
|
+
}),
|
|
8822
|
+
test: {
|
|
8823
|
+
taskId: start.taskId,
|
|
8824
|
+
testId: start.test && start.test.id,
|
|
8825
|
+
testName: start.test && start.test.name,
|
|
8826
|
+
selectedCases: names,
|
|
8827
|
+
dataModeSource
|
|
8828
|
+
},
|
|
8829
|
+
evidenceCategories: testEvidenceCategories(start, names),
|
|
8830
|
+
rawRefs: { testStatusKey: start.taskId },
|
|
8831
|
+
status: start
|
|
8832
|
+
}, { accountId, dataMode, branchName, workspace, baseUrl: normalizeBaseUrl(baseUrl), envelopeId: activeEnvelopeId });
|
|
8833
|
+
runtimeState.currentTestTaskId = null;
|
|
8834
|
+
if (jsonOutput) {
|
|
8835
|
+
console.log(JSON.stringify(Object.assign({}, start, {
|
|
8836
|
+
pending: true,
|
|
8837
|
+
wait: false,
|
|
8838
|
+
statusCommand: 'remits-cli test status --task-id ' + start.taskId
|
|
8839
|
+
}), null, 2));
|
|
8840
|
+
} else {
|
|
8841
|
+
console.log('Not waiting (--wait false).');
|
|
8842
|
+
console.log('Poll this run: remits-cli test status --task-id ' + start.taskId);
|
|
8843
|
+
console.log('Terminal evidence attaches when `test status` reads a completed, failed or interrupted run.');
|
|
8844
|
+
}
|
|
8845
|
+
return;
|
|
8846
|
+
}
|
|
8847
|
+
|
|
8509
8848
|
let stopWs = null;
|
|
8510
8849
|
if (flags.watch !== 'false') {
|
|
8511
8850
|
const topic = session.websocketTopic || start.websocketTopic || (session.user && String(session.user.uuid || '').replace(/-/g, ''));
|
|
@@ -8526,10 +8865,16 @@ async function testCommand(flags) {
|
|
|
8526
8865
|
if (jsonOutput) {
|
|
8527
8866
|
console.log(JSON.stringify(status, null, 2));
|
|
8528
8867
|
} else {
|
|
8529
|
-
console.log('Final status:', JSON.stringify(status, null, 2));
|
|
8530
8868
|
printStagingLaneOwnerNotice(status.staging || {});
|
|
8869
|
+
testRunHeadlineLines(status, { taskId: start.taskId, dataMode, requestedDataMode: dataMode })
|
|
8870
|
+
.forEach((line) => console.log(line));
|
|
8531
8871
|
testRunSummaryLines(status).forEach((line) => console.log(line));
|
|
8872
|
+
testFailureGroupLines(status, { testRef }).forEach((line) => console.log(line));
|
|
8532
8873
|
printTestRunPivots(status);
|
|
8874
|
+
// The whole run object used to be printed here as pretty JSON. One 92-case suite measured 220 KB,
|
|
8875
|
+
// most of it identifiers repeated once per case, spent on the turn with the least room left. It is
|
|
8876
|
+
// still one flag away, and the durable record keeps it after the live status expires.
|
|
8877
|
+
console.log('Full run payload: remits-cli test status --task-id ' + start.taskId + ' --json');
|
|
8533
8878
|
}
|
|
8534
8879
|
|
|
8535
8880
|
// A selector that matched no case is a mis-specified run, not a passing one. Say so in the terminal
|
|
@@ -8554,17 +8899,24 @@ async function testCommand(flags) {
|
|
|
8554
8899
|
process.exitCode = 1;
|
|
8555
8900
|
}
|
|
8556
8901
|
|
|
8557
|
-
const
|
|
8558
|
-
|
|
8559
|
-
|
|
8560
|
-
|
|
8561
|
-
|
|
8562
|
-
|
|
8563
|
-
|
|
8564
|
-
|
|
8565
|
-
|
|
8566
|
-
|
|
8902
|
+
const evidence = await appendTerminalTestRunEvidence({
|
|
8903
|
+
api,
|
|
8904
|
+
cwd,
|
|
8905
|
+
session,
|
|
8906
|
+
accountId,
|
|
8907
|
+
flags: verification.evidenceFlags,
|
|
8908
|
+
status,
|
|
8909
|
+
testRef,
|
|
8910
|
+
selectedNames: names,
|
|
8911
|
+
dataMode,
|
|
8912
|
+
dataModeSource,
|
|
8913
|
+
branchName,
|
|
8914
|
+
workspace,
|
|
8915
|
+
baseUrl,
|
|
8916
|
+
activeEnvelopeId,
|
|
8917
|
+
quiet: jsonOutput
|
|
8567
8918
|
});
|
|
8919
|
+
const nondeterminism = evidence.nondeterminism;
|
|
8568
8920
|
if (nondeterminism && !jsonOutput) {
|
|
8569
8921
|
console.log('Nondeterministic signal:', nondeterminism.message);
|
|
8570
8922
|
nondeterminism.flips.slice(0, 6).forEach((flip) => {
|
|
@@ -8572,40 +8924,6 @@ async function testCommand(flags) {
|
|
|
8572
8924
|
' (previous packet ' + (flip.previousPacketId || 'unknown') + ')');
|
|
8573
8925
|
});
|
|
8574
8926
|
}
|
|
8575
|
-
|
|
8576
|
-
await appendVerificationPacket(api, cwd, session, accountId, verification.evidenceFlags, {
|
|
8577
|
-
type: 'test_run',
|
|
8578
|
-
success: status.status === 'completed' && !(status.result && status.result.failed > 0) && !unmatched.length,
|
|
8579
|
-
claim: 'Test run ' + String(testRef),
|
|
8580
|
-
world: buildCommandWorld(status, { accountId, dataMode: status.dataMode || dataMode, dataModeSource, branchName, workspace, host: normalizeBaseUrl(baseUrl), sourceLayer: status.staging && status.staging.testComponentSource }),
|
|
8581
|
-
revision: Object.assign(sourceRevision, {
|
|
8582
|
-
compileSignatures: testCompileSignatures(status),
|
|
8583
|
-
componentProvenance
|
|
8584
|
-
}),
|
|
8585
|
-
nondeterministic: nondeterminism ? true : undefined,
|
|
8586
|
-
nondeterminism: nondeterminism || undefined,
|
|
8587
|
-
test: {
|
|
8588
|
-
taskId: start.taskId,
|
|
8589
|
-
testId: status.test && status.test.id,
|
|
8590
|
-
testName: status.test && status.test.name,
|
|
8591
|
-
selectedCases: names,
|
|
8592
|
-
passed: status.result && status.result.passed,
|
|
8593
|
-
failed: status.result && status.result.failed,
|
|
8594
|
-
tests: status.result && status.result.tests,
|
|
8595
|
-
dataModeSource: status.dataModeSource || dataModeSource,
|
|
8596
|
-
unmatchedTestNames: unmatched
|
|
8597
|
-
},
|
|
8598
|
-
evidenceCategories: testEvidenceCategories(status, names),
|
|
8599
|
-
// What the run spent and whether that live AI was declared (withAiBudget), so an envelope can tell a
|
|
8600
|
-
// measurement run from a suite with missing mocks without re-reading every case.
|
|
8601
|
-
dependencies: testRunDependencies(status),
|
|
8602
|
-
summary: status.result && status.result.summary,
|
|
8603
|
-
limitations: []
|
|
8604
|
-
.concat(unmatched.length ? ['One or more requested test case selectors matched no case.'] : [])
|
|
8605
|
-
.concat(nondeterminism ? [nondeterminism.message] : []),
|
|
8606
|
-
rawRefs: { testStatusKey: start.taskId, durableTestRun: status.result && status.result.durableRecord ? start.taskId : undefined },
|
|
8607
|
-
status
|
|
8608
|
-
}, { quiet: jsonOutput });
|
|
8609
8927
|
}
|
|
8610
8928
|
|
|
8611
8929
|
async function testStatusCommand(flags) {
|
|
@@ -8615,9 +8933,11 @@ async function testStatusCommand(flags) {
|
|
|
8615
8933
|
const { session, accountId } = sessionContext;
|
|
8616
8934
|
const baseUrl = flags['base-url'] || session.baseUrl || DEFAULT_BASE_URL;
|
|
8617
8935
|
const branchName = flags.branch || currentBranch(cwd);
|
|
8936
|
+
const workspace = resolveWorkspace(cwd, flags);
|
|
8618
8937
|
const dataMode = hasExplicitDataModeFlag(flags)
|
|
8619
8938
|
? resolveDataMode(flags, null)
|
|
8620
8939
|
: DEFAULT_DATA_MODE;
|
|
8940
|
+
const dataModeSource = dataModeFlagSource(flags);
|
|
8621
8941
|
const taskId = flags['task-id'] || flags.taskId || flags.id || (flags._ && flags._[2]);
|
|
8622
8942
|
const jsonOutput = flagEnabled(flags.json);
|
|
8623
8943
|
if (!taskId) throw new Error('Missing --task-id <taskId>');
|
|
@@ -8631,31 +8951,94 @@ async function testStatusCommand(flags) {
|
|
|
8631
8951
|
taskId,
|
|
8632
8952
|
dataMode
|
|
8633
8953
|
}).then((r) => r.data);
|
|
8954
|
+
if (!status.taskId) status.taskId = taskId;
|
|
8955
|
+
|
|
8956
|
+
let statusVerification = { envelopeId: null, evidenceFlags: flags };
|
|
8957
|
+
if (testStatusIsTerminal(status)) {
|
|
8958
|
+
statusVerification = await verificationPreflight(api, cwd, session, accountId, flags, {
|
|
8959
|
+
packetType: 'test_run',
|
|
8960
|
+
world: buildCommandWorld(status, {
|
|
8961
|
+
accountId,
|
|
8962
|
+
dataMode: status.dataMode || dataMode,
|
|
8963
|
+
dataModeSource,
|
|
8964
|
+
branchName,
|
|
8965
|
+
workspace,
|
|
8966
|
+
host: normalizeBaseUrl(baseUrl),
|
|
8967
|
+
sourceLayer: status.staging && status.staging.testComponentSource
|
|
8968
|
+
}),
|
|
8969
|
+
test: status.test && status.test.id !== undefined && status.test.id !== null
|
|
8970
|
+
? { testId: status.test.id, testName: status.test.name }
|
|
8971
|
+
: { testName: status.test && status.test.name }
|
|
8972
|
+
}, { command: 'test status', neverRefuse: true, quiet: jsonOutput });
|
|
8973
|
+
}
|
|
8634
8974
|
|
|
8635
8975
|
if (jsonOutput) {
|
|
8976
|
+
if (testStatusIsTerminal(status)) {
|
|
8977
|
+
await appendTerminalTestRunEvidence({
|
|
8978
|
+
api,
|
|
8979
|
+
cwd,
|
|
8980
|
+
session,
|
|
8981
|
+
accountId,
|
|
8982
|
+
flags: statusVerification.evidenceFlags,
|
|
8983
|
+
status,
|
|
8984
|
+
testRef: status.test && (status.test.name || status.test.id) || taskId,
|
|
8985
|
+
selectedNames: status.result && Array.isArray(status.result.tests) ? status.result.tests.map((t) => t && t.name).filter(Boolean) : [],
|
|
8986
|
+
dataMode,
|
|
8987
|
+
dataModeSource,
|
|
8988
|
+
branchName,
|
|
8989
|
+
workspace,
|
|
8990
|
+
baseUrl,
|
|
8991
|
+
activeEnvelopeId: statusVerification.envelopeId,
|
|
8992
|
+
quiet: true
|
|
8993
|
+
});
|
|
8994
|
+
}
|
|
8636
8995
|
console.log(JSON.stringify(status, null, 2));
|
|
8637
8996
|
return status;
|
|
8638
8997
|
}
|
|
8639
8998
|
printSessionResolutionWarning(sessionContext);
|
|
8640
8999
|
printResolvedBaseUrl(baseUrl);
|
|
8641
|
-
console.log('Test run status:', status.status || 'unknown');
|
|
8642
|
-
console.log('Task ID:', taskId);
|
|
8643
9000
|
// Say where this answer came from. A durable record is a finished snapshot rebuilt from the database after the
|
|
8644
9001
|
// live status was gone; without this line a reconstructed run is indistinguishable from one still being watched.
|
|
8645
|
-
|
|
8646
|
-
|
|
8647
|
-
:
|
|
8648
|
-
|
|
8649
|
-
|
|
8650
|
-
|
|
8651
|
-
|
|
8652
|
-
|
|
8653
|
-
|
|
8654
|
-
}
|
|
9002
|
+
testRunHeadlineLines(status, {
|
|
9003
|
+
taskId,
|
|
9004
|
+
source: status.durable
|
|
9005
|
+
? 'durable run record (the live status has expired; this run is final)'
|
|
9006
|
+
: 'live run status',
|
|
9007
|
+
dataMode,
|
|
9008
|
+
// A durable run reports the lane it RAN in. When that differs from the lane this command asked for,
|
|
9009
|
+
// say so: auditing prod activity and being handed a test-lane run is correct and reads as if it were not.
|
|
9010
|
+
requestedDataMode: hasExplicitDataModeFlag(flags) ? dataMode : null
|
|
9011
|
+
}).forEach((line) => console.log(line));
|
|
8655
9012
|
if (status.world) runWorldLines(status.world, { host: normalizeBaseUrl(baseUrl) }).forEach((line) => console.log(line));
|
|
8656
9013
|
testRunSummaryLines(status).forEach((line) => console.log(line));
|
|
9014
|
+
testFailureGroupLines(status).forEach((line) => console.log(line));
|
|
8657
9015
|
printTestRunPivots(status);
|
|
8658
|
-
if (
|
|
9016
|
+
if (testStatusIsTerminal(status)) {
|
|
9017
|
+
const evidence = await appendTerminalTestRunEvidence({
|
|
9018
|
+
api,
|
|
9019
|
+
cwd,
|
|
9020
|
+
session,
|
|
9021
|
+
accountId,
|
|
9022
|
+
flags: statusVerification.evidenceFlags,
|
|
9023
|
+
status,
|
|
9024
|
+
testRef: status.test && (status.test.name || status.test.id) || taskId,
|
|
9025
|
+
selectedNames: status.result && Array.isArray(status.result.tests) ? status.result.tests.map((t) => t && t.name).filter(Boolean) : [],
|
|
9026
|
+
dataMode,
|
|
9027
|
+
dataModeSource,
|
|
9028
|
+
branchName,
|
|
9029
|
+
workspace,
|
|
9030
|
+
baseUrl,
|
|
9031
|
+
activeEnvelopeId: statusVerification.envelopeId
|
|
9032
|
+
});
|
|
9033
|
+
if (evidence.nondeterminism) {
|
|
9034
|
+
console.log('Nondeterministic signal:', evidence.nondeterminism.message);
|
|
9035
|
+
evidence.nondeterminism.flips.slice(0, 6).forEach((flip) => {
|
|
9036
|
+
console.log(' - ' + flip.caseName + ': ' + (flip.previousPassed ? 'passed' : 'failed') + ' -> ' + (flip.currentPassed ? 'passed' : 'failed') +
|
|
9037
|
+
' (previous packet ' + (flip.previousPacketId || 'unknown') + ')');
|
|
9038
|
+
});
|
|
9039
|
+
}
|
|
9040
|
+
}
|
|
9041
|
+
if (status.status === 'failed' || status.status === 'interrupted' || (status.result && status.result.failed > 0)) {
|
|
8659
9042
|
process.exitCode = 1;
|
|
8660
9043
|
}
|
|
8661
9044
|
return status;
|
|
@@ -8671,10 +9054,13 @@ async function testRunsCommand(flags) {
|
|
|
8671
9054
|
const baseUrl = flags['base-url'] || session.baseUrl || DEFAULT_BASE_URL;
|
|
8672
9055
|
const api = buildAxios(baseUrl, session.token);
|
|
8673
9056
|
const testRef = flags.test || flags['test-id'] || flags.name;
|
|
9057
|
+
const allDataLanes = flagEnabled(flags['all-lanes']) || flagEnabled(flags['all-data-lanes']);
|
|
8674
9058
|
const data = await loggedPost(api, cwd, '/cli/testRuns', {
|
|
8675
9059
|
token: session.token,
|
|
8676
9060
|
accountId,
|
|
8677
9061
|
asAccountId: flags['as-account'] || flags['as-account-id'],
|
|
9062
|
+
dataMode: resolveDataMode(flags, session),
|
|
9063
|
+
allDataLanes: allDataLanes || undefined,
|
|
8678
9064
|
testId: testRef && /^\d+$/.test(String(testRef)) ? Number(testRef) : undefined,
|
|
8679
9065
|
testName: testRef && !/^\d+$/.test(String(testRef)) ? String(testRef) : undefined,
|
|
8680
9066
|
max: flags.limit || flags.max || 20
|
|
@@ -8682,9 +9068,17 @@ async function testRunsCommand(flags) {
|
|
|
8682
9068
|
if (!data.success) throw new Error(data.message || 'Could not list test runs');
|
|
8683
9069
|
|
|
8684
9070
|
if (flagEnabled(flags.compare)) {
|
|
8685
|
-
const
|
|
8686
|
-
if (
|
|
8687
|
-
|
|
9071
|
+
const pair = selectComparableRuns(data.runs || []);
|
|
9072
|
+
if (!pair) {
|
|
9073
|
+
throw new Error('--compare needs two COMPARABLE recorded runs' + (testRef ? ' of ' + testRef : '') +
|
|
9074
|
+
' — same data lane, branch, workspace and case count. Found ' + (data.runs || []).length +
|
|
9075
|
+
' run(s); pick two explicitly with: remits-cli test compare --base <taskId> --head <taskId>');
|
|
9076
|
+
}
|
|
9077
|
+
if (pair.skipped) {
|
|
9078
|
+
console.error('Comparing the latest two comparable runs (' + describeRunShape(pair.head) + '); ' +
|
|
9079
|
+
pair.skipped + ' newer run(s) differ in lane, branch, workspace or case count and were skipped.');
|
|
9080
|
+
}
|
|
9081
|
+
return testCompareCommand(Object.assign({}, flags, { base: pair.base.taskId, head: pair.head.taskId }));
|
|
8688
9082
|
}
|
|
8689
9083
|
if (flagEnabled(flags.json)) {
|
|
8690
9084
|
console.log(JSON.stringify(data, null, 2));
|
|
@@ -8693,6 +9087,9 @@ async function testRunsCommand(flags) {
|
|
|
8693
9087
|
printSessionResolutionWarning(sessionContext);
|
|
8694
9088
|
printResolvedBaseUrl(baseUrl);
|
|
8695
9089
|
console.log('Recorded test runs' + (testRef ? ' for ' + testRef : '') + ' (account ' + data.accountId + '), newest first:');
|
|
9090
|
+
// Which lane this list is, said once. Pass counts from two lanes are two different facts.
|
|
9091
|
+
console.log('Data lane: ' + (data.dataLaneScope || 'all-data-lanes') +
|
|
9092
|
+
(data.otherLaneCount ? ' (' + data.otherLaneCount + ' more run(s) in the other lane; --all-lanes to include)' : ''));
|
|
8696
9093
|
(data.runs || []).forEach((run) => {
|
|
8697
9094
|
console.log('- ' + run.taskId + ' ' + (run.status || 'unknown') + ' ' + (run.passed || 0) + '/' + (run.total || 0) + ' passed' +
|
|
8698
9095
|
' ' + corpusAiLabel(run) +
|
|
@@ -8705,6 +9102,40 @@ async function testRunsCommand(flags) {
|
|
|
8705
9102
|
return data;
|
|
8706
9103
|
}
|
|
8707
9104
|
|
|
9105
|
+
/** The world fields that decide whether two runs are the same experiment repeated. */
|
|
9106
|
+
function runShapeKey(run = {}) {
|
|
9107
|
+
return [run.dataMode || '?', run.branchName || '?', run.workspace || 'shared',
|
|
9108
|
+
run.variantBranch || 'none', run.total == null ? '?' : run.total].join('|');
|
|
9109
|
+
}
|
|
9110
|
+
|
|
9111
|
+
function describeRunShape(run = {}) {
|
|
9112
|
+
return [(run.dataMode || '?') + ' lane', run.branchName || '?',
|
|
9113
|
+
run.workspace ? 'ws:' + run.workspace : 'shared lane',
|
|
9114
|
+
(run.total == null ? '?' : run.total) + ' case(s)'].join(' · ');
|
|
9115
|
+
}
|
|
9116
|
+
|
|
9117
|
+
/**
|
|
9118
|
+
* The newest two runs that are actually comparable, and how many newer ones were passed over.
|
|
9119
|
+
*
|
|
9120
|
+
* `--compare` used to take runs[0] and runs[1] unconditionally. On a real history that pairs a 92-case
|
|
9121
|
+
* suite with a 1-case `--names` probe, or a prod-lane run with a test-lane one, and reports the difference
|
|
9122
|
+
* as regressions. A comparison across worlds is not a comparison.
|
|
9123
|
+
*/
|
|
9124
|
+
function selectComparableRuns(runs) {
|
|
9125
|
+
const list = Array.isArray(runs) ? runs.filter(Boolean) : [];
|
|
9126
|
+
for (let head = 0; head < list.length; head += 1) {
|
|
9127
|
+
const key = runShapeKey(list[head]);
|
|
9128
|
+
for (let base = head + 1; base < list.length; base += 1) {
|
|
9129
|
+
if (runShapeKey(list[base]) === key) {
|
|
9130
|
+
// Every run passed over, including the ones BETWEEN head and base — a 1-case probe sitting between
|
|
9131
|
+
// two full suites is exactly the run whose absence from the comparison needs saying.
|
|
9132
|
+
return { head: list[head], base: list[base], skipped: base - 1 };
|
|
9133
|
+
}
|
|
9134
|
+
}
|
|
9135
|
+
}
|
|
9136
|
+
return null;
|
|
9137
|
+
}
|
|
9138
|
+
|
|
8708
9139
|
// `test compare --base <taskId> --head <taskId>`: per-case outcome, cost and duration deltas between two runs,
|
|
8709
9140
|
// read from the durable records.
|
|
8710
9141
|
async function testCompareCommand(flags) {
|
|
@@ -8735,6 +9166,18 @@ async function testCompareCommand(flags) {
|
|
|
8735
9166
|
.filter((key) => comparison.baseWorld && comparison.headWorld && String(comparison.baseWorld[key] || '') !== String(comparison.headWorld[key] || ''))
|
|
8736
9167
|
.map((key) => key + ' ' + short(String(comparison.baseWorld[key] || 'none')) + ' -> ' + short(String(comparison.headWorld[key] || 'none')));
|
|
8737
9168
|
if (worldDiff.length) console.log('World changed: ' + worldDiff.join('; '));
|
|
9169
|
+
// A lane or case-count difference is not a delta to interpret, it is two different experiments. Say so
|
|
9170
|
+
// rather than letting "regressed (33)" stand for "these runs never measured the same thing".
|
|
9171
|
+
const baseLane = comparison.baseWorld && comparison.baseWorld.dataMode;
|
|
9172
|
+
const headLane = comparison.headWorld && comparison.headWorld.dataMode;
|
|
9173
|
+
if (baseLane && headLane && baseLane !== headLane) {
|
|
9174
|
+
console.log('NOT COMPARABLE: these runs are in different DATA LANES (' + baseLane + ' vs ' + headLane +
|
|
9175
|
+
'). A record\'s dataMode is the lane it was written in; the pass counts below are two different facts.');
|
|
9176
|
+
}
|
|
9177
|
+
if (comparison.base.total && comparison.head.total && comparison.base.total !== comparison.head.total) {
|
|
9178
|
+
console.log('NOT COMPARABLE like-for-like: ' + comparison.base.total + ' case(s) vs ' + comparison.head.total +
|
|
9179
|
+
'. One of these is a narrowed --names run; "absent" below means the case did not run, not that it broke.');
|
|
9180
|
+
}
|
|
8738
9181
|
if (comparison.improved.length) console.log('Improved (' + comparison.improved.length + '): ' + comparison.improved.join(', '));
|
|
8739
9182
|
if (comparison.regressed.length) console.log('Regressed (' + comparison.regressed.length + '): ' + comparison.regressed.join(', '));
|
|
8740
9183
|
comparison.changed.filter((c) => c.direction === 'changed').forEach((c) => console.log('Changed: ' + c.name + ' ' + c.base + ' -> ' + c.head));
|
|
@@ -10898,6 +11341,7 @@ function buildRepoSnapshot(entry) {
|
|
|
10898
11341
|
const repoFiles = [
|
|
10899
11342
|
{ label: 'account-info.json', path: entry.accountInfoPath || path.join(directory, 'account-info.json'), mode: 'json' },
|
|
10900
11343
|
{ label: 'account-hierarchy.json', path: path.join(directory, 'account-hierarchy.json'), mode: 'json' },
|
|
11344
|
+
{ label: 'account-analytics.json', path: path.join(directory, 'account-analytics.json'), mode: 'json' },
|
|
10901
11345
|
{ label: '.remits-cli/active-actor/current-session.txt', path: localPaths.currentSessionFile, mode: 'text' },
|
|
10902
11346
|
{ label: '.remits-cli/shared/tools/tools.json', path: path.join(localPaths.toolsDir, 'tools.json'), mode: 'json' },
|
|
10903
11347
|
{ label: '.remits-cli/active-actor/session log (tail)', path: currentSessionLog, mode: 'tail' },
|
|
@@ -15143,7 +15587,12 @@ function printActivityInspectSummary(response) {
|
|
|
15143
15587
|
if (env.noRequiredEvidence) {
|
|
15144
15588
|
console.log(' evidence log — no verdict requested; add a claim only if someone needs one: remits-cli verify claim <id> --text "..." --envelope ' + env.envelopeId);
|
|
15145
15589
|
}
|
|
15146
|
-
|
|
15590
|
+
// A failure recorded in an envelope that promised nothing is HISTORY, not an outstanding task — the
|
|
15591
|
+
// same rule the smells already apply (`promises` in activitySmells). Printing it as "current failures"
|
|
15592
|
+
// re-created, one line lower, exactly the unfinishable-looking work the evidence/verdict split removed.
|
|
15593
|
+
if (env.currentFailureCount) {
|
|
15594
|
+
console.log(' ' + (env.noRequiredEvidence ? 'failed evidence recorded: ' : 'current failures: ') + env.currentFailureCount);
|
|
15595
|
+
}
|
|
15147
15596
|
});
|
|
15148
15597
|
}
|
|
15149
15598
|
|
|
@@ -15968,10 +16417,18 @@ function releaseAutoUpdateLock() {
|
|
|
15968
16417
|
} catch (_) {}
|
|
15969
16418
|
}
|
|
15970
16419
|
|
|
16420
|
+
/**
|
|
16421
|
+
* May this invocation check for, and install, a newer CLI?
|
|
16422
|
+
*
|
|
16423
|
+
* `--json` used to disable it outright, to keep update chatter out of a document a program is parsing.
|
|
16424
|
+
* The effect was that the callers who pass `--json` on every command - which is every AI agent - never
|
|
16425
|
+
* upgraded at all, so each release of better diagnostics reached the population that needed it least.
|
|
16426
|
+
* The chatter now goes to stderr (see `autoUpdateIfNeeded`), which is where a program's stdout contract
|
|
16427
|
+
* says it belongs, and the check stays on.
|
|
16428
|
+
*/
|
|
15971
16429
|
function shouldAutoUpdate(command, flags) {
|
|
15972
16430
|
if (!command || command === 'help' || command === '--help') return false;
|
|
15973
16431
|
if (flags && flags['no-auto-update']) return false;
|
|
15974
|
-
if (flags && flagEnabled(flags.json)) return false;
|
|
15975
16432
|
if (process.env.REMITS_CLI_AUTO_UPDATE && AUTO_UPDATE_DISABLED_VALUES.has(String(process.env.REMITS_CLI_AUTO_UPDATE).toLowerCase())) {
|
|
15976
16433
|
return false;
|
|
15977
16434
|
}
|
|
@@ -15979,10 +16436,31 @@ function shouldAutoUpdate(command, flags) {
|
|
|
15979
16436
|
return true;
|
|
15980
16437
|
}
|
|
15981
16438
|
|
|
16439
|
+
/** True when the last registry check is recent enough that another one would only cost latency. */
|
|
16440
|
+
function autoUpdateCheckedRecently() {
|
|
16441
|
+
try {
|
|
16442
|
+
const stamp = JSON.parse(fs.readFileSync(AUTO_UPDATE_STAMP_FILE, 'utf8'));
|
|
16443
|
+
return Number(stamp.checkedAtMs) > Date.now() - AUTO_UPDATE_CHECK_INTERVAL_MS;
|
|
16444
|
+
} catch (_) {
|
|
16445
|
+
return false;
|
|
16446
|
+
}
|
|
16447
|
+
}
|
|
16448
|
+
|
|
16449
|
+
function recordAutoUpdateCheck(latest) {
|
|
16450
|
+
try {
|
|
16451
|
+
ensureSessionDir();
|
|
16452
|
+
fs.writeFileSync(AUTO_UPDATE_STAMP_FILE,
|
|
16453
|
+
JSON.stringify({ checkedAtMs: Date.now(), latest: latest || null }, null, 2));
|
|
16454
|
+
} catch (_) {}
|
|
16455
|
+
}
|
|
16456
|
+
|
|
15982
16457
|
function autoUpdateIfNeeded(originalArgv, options = {}) {
|
|
15983
16458
|
const requireSuccess = Boolean(options.requireSuccess);
|
|
16459
|
+
// Update chatter is diagnostics, never part of a `--json` document. stderr keeps both promises at once.
|
|
16460
|
+
const note = (line) => console.error(line);
|
|
15984
16461
|
let lockAcquired = false;
|
|
15985
16462
|
try {
|
|
16463
|
+
if (!requireSuccess && autoUpdateCheckedRecently()) return false;
|
|
15986
16464
|
lockAcquired = acquireAutoUpdateLock();
|
|
15987
16465
|
if (!lockAcquired) {
|
|
15988
16466
|
if (requireSuccess) {
|
|
@@ -15996,20 +16474,22 @@ function autoUpdateIfNeeded(originalArgv, options = {}) {
|
|
|
15996
16474
|
encoding: 'utf8',
|
|
15997
16475
|
stdio: ['ignore', 'pipe', 'ignore']
|
|
15998
16476
|
}).trim();
|
|
16477
|
+
recordAutoUpdateCheck(latest);
|
|
15999
16478
|
if (latest && compareSemver(latest, currentVersion) > 0) {
|
|
16000
|
-
|
|
16479
|
+
note('[update] New version available: ' + currentVersion + ' -> ' + latest + '. Installing...');
|
|
16480
|
+
// npm's own progress output goes to stderr too: a `--json` caller's stdout must stay one document.
|
|
16001
16481
|
const install = spawnSync(npmCommand(), ['install', '-g', REMITS_CLI_PACKAGE_NAME + '@latest'], {
|
|
16002
|
-
stdio: '
|
|
16482
|
+
stdio: ['ignore', process.stderr, process.stderr],
|
|
16003
16483
|
env: process.env
|
|
16004
16484
|
});
|
|
16005
16485
|
if (install.error || install.status !== 0) {
|
|
16006
16486
|
if (requireSuccess) {
|
|
16007
16487
|
throw new Error('Auto-update failed; remits-cli start requires the latest published version before continuing.');
|
|
16008
16488
|
}
|
|
16009
|
-
|
|
16489
|
+
note('[update] Update failed; continuing with version ' + currentVersion + '.');
|
|
16010
16490
|
return false;
|
|
16011
16491
|
}
|
|
16012
|
-
|
|
16492
|
+
note('[update] Updated to ' + latest + '. Re-running command...');
|
|
16013
16493
|
const rerun = spawnSync(remitsCliCommand(), originalArgv, {
|
|
16014
16494
|
stdio: 'inherit',
|
|
16015
16495
|
env: Object.assign({}, process.env, { REMITS_CLI_AUTO_UPDATED: '1' })
|
|
@@ -16018,7 +16498,7 @@ function autoUpdateIfNeeded(originalArgv, options = {}) {
|
|
|
16018
16498
|
if (requireSuccess) {
|
|
16019
16499
|
throw new Error('Auto-update succeeded but re-running the updated remits-cli command failed.');
|
|
16020
16500
|
}
|
|
16021
|
-
|
|
16501
|
+
note('[update] Re-run failed; continuing with version ' + currentVersion + '.');
|
|
16022
16502
|
return false;
|
|
16023
16503
|
}
|
|
16024
16504
|
process.exitCode = rerun.status === null ? 1 : rerun.status;
|
|
@@ -16177,13 +16657,14 @@ function printComponentsHelp(subcommand) {
|
|
|
16177
16657
|
console.log(' remits-cli components branch <name> [--json] # overridden/added/removed + drift');
|
|
16178
16658
|
console.log(' remits-cli components branch <name> --diff <componentId> --component-type <kind> [--json]');
|
|
16179
16659
|
console.log(' remits-cli components branch <name> --subscribers [--json]');
|
|
16660
|
+
console.log(' remits-cli components branch <name> --copy-to <newBranch> [--dry-run] [--force] [--json]');
|
|
16180
16661
|
console.log(' remits-cli components branch <name> --subscribe <accountId> [--parent-account <id>] [--domain <host>] [--dry-run] [--confirm-primary-edge]');
|
|
16181
16662
|
console.log(' remits-cli components branch <name> --unsubscribe <accountId>');
|
|
16182
16663
|
console.log(' remits-cli components branch <name> --retire [--force]');
|
|
16183
16664
|
}
|
|
16184
16665
|
|
|
16185
16666
|
function printTestHelp() {
|
|
16186
|
-
console.log('Usage: remits-cli test run --test <id|name> [--base-url URL] [--branch stagingScope] [--names "a|b"] [--watch true|false] [--data-mode test|prod] [--as-account ID] [--variant-branch NAME|none] [--json]');
|
|
16667
|
+
console.log('Usage: remits-cli test run --test <id|name> [--base-url URL] [--branch stagingScope] [--names "a|b"] [--watch true|false] [--wait true|false] [--data-mode test|prod] [--as-account ID] [--variant-branch NAME|none] [--json]');
|
|
16187
16668
|
console.log(' remits-cli test status --task-id <taskId> [--base-url URL] [--account-id ID] [--branch stagingScope] [--data-mode test|prod] [--json]');
|
|
16188
16669
|
console.log(' remits-cli test runs [--test <id|name>] [--limit 20] [--compare] [--json] # durable run history');
|
|
16189
16670
|
console.log(' remits-cli test compare --base <taskId> --head <taskId> [--json] # per-case outcome/cost deltas');
|
|
@@ -16202,6 +16683,8 @@ function printTestHelp() {
|
|
|
16202
16683
|
console.log(' to force production/subscription semantics from a variant checkout.');
|
|
16203
16684
|
console.log(' --branch changes only the CLI staging namespace for test execution. Pair an unused');
|
|
16204
16685
|
console.log(' value with --variant-branch none when existing staged entries would shadow DB rows.');
|
|
16686
|
+
console.log(' --watch false disables websocket progress streaming; the CLI still waits for final status.');
|
|
16687
|
+
console.log(' --wait false returns after launch with a task id; test status attaches terminal evidence.');
|
|
16205
16688
|
console.log(' --json prints only the final status JSON to stdout; banners and progress go to stderr.');
|
|
16206
16689
|
console.log(' --data-mode prod intentionally targets live production data.');
|
|
16207
16690
|
console.log(' Every finished run is recorded durably: test status keeps working after the live status');
|
|
@@ -16473,10 +16956,11 @@ async function main() {
|
|
|
16473
16956
|
console.log(' remits-cli components promotion [<branch>] [--json] [--no-fail] # promotion readiness + ordered next steps');
|
|
16474
16957
|
console.log(' remits-cli components branches [--json] # committed branch variants for this account');
|
|
16475
16958
|
console.log(' remits-cli components branch <name> [--diff <componentId> --component-type <kind>] [--subscribers] [--json]');
|
|
16959
|
+
console.log(' remits-cli components branch <name> --copy-to <newBranch> [--dry-run] [--force] [--json]');
|
|
16476
16960
|
console.log(' remits-cli components branch <name> --subscribe <accountId> [--dry-run] [--confirm-primary-edge]');
|
|
16477
16961
|
console.log(' remits-cli components branch <name> --unsubscribe <accountId> # return that account to trunk');
|
|
16478
16962
|
console.log(' remits-cli components branch <name> --retire [--force] # delete the branch\'s overlays');
|
|
16479
|
-
console.log(' remits-cli test run --test <id|name> [--base-url URL] [--branch stagingScope] [--names "a|b"] [--watch true|false] [--data-mode test|prod] [--as-account ID] [--variant-branch NAME|none] [--json]');
|
|
16963
|
+
console.log(' remits-cli test run --test <id|name> [--base-url URL] [--branch stagingScope] [--names "a|b"] [--watch true|false] [--wait true|false] [--data-mode test|prod] [--as-account ID] [--variant-branch NAME|none] [--json]');
|
|
16480
16964
|
console.log(' remits-cli test runs [--test <id|name>] [--compare] # durable run history; compare latest two');
|
|
16481
16965
|
console.log(' remits-cli test compare --base <taskId> --head <taskId>');
|
|
16482
16966
|
console.log(' remits-cli corpus import --manifest corpus-manifest.json # seed a test corpus: cases + artifacts');
|
package/package.json
CHANGED
|
@@ -48,6 +48,7 @@ reference below, load it before you act, not after something surprises you.
|
|
|
48
48
|
| register this session as an agent, or ask why a routed ticket never started | `agent-sessions.md` | `serve` vs `register`, what registration does, worker spawning per agent kind, routing order, the control center, multi-session auth |
|
|
49
49
|
| investigate live behavior | `investigation.md` | which tool reads which record, correlation keys, reading a record's `content`, HTTP audits, AI activity, node/`localMode`, the production support flows |
|
|
50
50
|
| call any `mcp_*` tool | `tool-reference.md` | every tool's parameters, semantics, and traps — read the tool's entry before building its input |
|
|
51
|
+
| run, describe, poll, or interrupt an Action from the CLI | `tool-reference.md` → `mcp_run_action` | Action execution through `remits-cli tool --name mcp_run_action`; there is no separate top-level `remits-cli action` / `run action` wrapper |
|
|
51
52
|
| reason about repo discovery, auth state, or what a command actually sent | `cli-state.md` | the global `~/.remits-cli/` control plane vs per-repo `./.remits-cli/`, and which file answers which question |
|
|
52
53
|
| need an exact flag, or the auth / host / async surface | `command-reference.md` | authentication, host vs data mode, the two async mechanisms, hierarchy-scoped reads, the full command list, prod banners |
|
|
53
54
|
| give up on something that misbehaved | `troubleshooting.md` | the symptom→fix table, the two kinds of escalation, and the escalation bundle |
|
|
@@ -408,6 +408,31 @@ from that checkout). Its first real sync (or commit) needs `--create-variant-bra
|
|
|
408
408
|
variants or a subscriber the platform treats it as a feature branch and refuses to land it, so creating a
|
|
409
409
|
new variant branch is always a stated decision, never a side effect of a feature branch's name.
|
|
410
410
|
|
|
411
|
+
For a nested branch cut from an existing variant branch, seed the platform overlays explicitly after the git
|
|
412
|
+
branch is created and pushed:
|
|
413
|
+
|
|
414
|
+
```bash
|
|
415
|
+
remits-cli components branch forked --copy-to sandbox --dry-run
|
|
416
|
+
remits-cli components branch forked --copy-to sandbox
|
|
417
|
+
remits-cli components sync --safe --branch sandbox
|
|
418
|
+
```
|
|
419
|
+
|
|
420
|
+
The copy is a DB overlay bootstrap, not a git write. It makes inherited overlays from `forked` already
|
|
421
|
+
`storedCurrent` on the first `sandbox` sync, so the safe diff gate only has to account for the new branch's
|
|
422
|
+
real file changes.
|
|
423
|
+
|
|
424
|
+
**Copy before you subscribe, or the copy needs `--force`.** The copy refuses a target that already has
|
|
425
|
+
stored overlays *or* live subscribers, and that second guard fires even when the target has no overlays at
|
|
426
|
+
all — a branch somebody is already resolving is code those accounts are running right now, and replacing it
|
|
427
|
+
has to be a stated decision. Seeding first and subscribing after keeps the plain form working. The copy is
|
|
428
|
+
also all-or-nothing: the delete of any replaced overlays and every inserted row share one transaction, so a
|
|
429
|
+
failure leaves the target exactly as it was rather than half-populated.
|
|
430
|
+
|
|
431
|
+
Run it as `--dry-run` first. The plan reports `plannedCopies` — the number the real run will write — which
|
|
432
|
+
is not always the source branch's row count: a source overlay that has become identical to trunk in both
|
|
433
|
+
content and metadata is sparse, is not stored, and is listed as skipped instead of being silently dropped
|
|
434
|
+
from the total.
|
|
435
|
+
|
|
411
436
|
**Trunk moving also invalidates a branch.** Variant sparseness compares branch content against *current*
|
|
412
437
|
trunk, so a trunk change can make an overlay obsolete without the branch changing at all. A trunk sync
|
|
413
438
|
drops the affected branches' cached sync verdicts, so the next `components sync` on the branch really
|
|
@@ -15,6 +15,8 @@
|
|
|
15
15
|
- [Hierarchy-scoped tool reads](#hierarchy-scoped-tool-reads)
|
|
16
16
|
- [Data Mode](#data-mode)
|
|
17
17
|
- [Command Reference](#command-reference)
|
|
18
|
+
- [Reading a failing run](#reading-a-failing-run)
|
|
19
|
+
- [A pass count only means something within one data lane](#a-pass-count-only-means-something-within-one-data-lane)
|
|
18
20
|
- [Verification envelopes](#verification-envelopes)
|
|
19
21
|
- [Staging modes: workset vs full snapshot](#staging-modes-workset-vs-full-snapshot)
|
|
20
22
|
- [Prod banners and retryable failures](#prod-banners-and-retryable-failures)
|
|
@@ -236,12 +238,13 @@ remits-cli components branches [--json] # branche
|
|
|
236
238
|
remits-cli components branch <name> [--json] # one branch: owner account, overridden / added / removed, drift flags, subscribers
|
|
237
239
|
remits-cli components branch <name> --diff <componentId> --component-type <kind> [--json]
|
|
238
240
|
remits-cli components branch <name> --subscribers [--json]
|
|
241
|
+
remits-cli components branch <name> --copy-to <newBranch> [--dry-run] [--force] [--json] # seed a new variant branch with this branch's stored overlays; force if target has overlays/subscribers
|
|
239
242
|
remits-cli components branch <name> --subscribe <accountId> [--parent-account <id>] [--domain <host>] [--dry-run] [--confirm-primary-edge] # make an account resolve this branch
|
|
240
243
|
remits-cli components branch <name> --unsubscribe <accountId> # return that account to trunk
|
|
241
244
|
remits-cli components branch <name> --retire [--force] # delete the branch's overlays
|
|
242
|
-
remits-cli test run --test <id|name> [--branch <stagingScope>] [--names "a|b"] [--watch true|false] [--data-mode test|prod] [--as-account <ID>] [--variant-branch <name|none>] [--json]
|
|
245
|
+
remits-cli test run --test <id|name> [--branch <stagingScope>] [--names "a|b"] [--watch true|false] [--wait true|false] [--data-mode test|prod] [--as-account <ID>] [--variant-branch <name|none>] [--json]
|
|
243
246
|
remits-cli test status --task-id <taskId> [--branch <stagingScope>] [--data-mode test|prod] [--json] # falls back to the DURABLE run record once the live status expires
|
|
244
|
-
remits-cli test runs [--test <id|name>] [--limit 20] [--compare] [--as-account <ID>] [--json]
|
|
247
|
+
remits-cli test runs [--test <id|name>] [--limit 20] [--compare] [--all-lanes] [--as-account <ID>] [--json] # durable run history: pass counts, live AI cost, lane, content hash
|
|
245
248
|
remits-cli test compare --base <taskId> --head <taskId> [--json] # per-case improved / regressed / changed, cost and world deltas
|
|
246
249
|
remits-cli corpus import --manifest corpus-manifest.json [--corpus <name>] [--as-account <ID>] [--data-mode test|prod --confirm-prod] [--json] # seed an evaluation corpus: cases + immutable artifacts; idempotent by caseKey
|
|
247
250
|
remits-cli corpus cases --corpus <name> [--split S] [--tag T|--tags T,U] [--key K|--keys K,L] [--include-values] [--include-retired] [--limit N] [--json]
|
|
@@ -252,6 +255,54 @@ remits-cli corpus consistency --corpus <name> --case <caseKey> [--json]
|
|
|
252
255
|
remits-cli corpus retire --corpus <name> --case <caseKey> [--case ...] [--restore] [--json] # drop a case from future runs; past measurements stay readable
|
|
253
256
|
```
|
|
254
257
|
|
|
258
|
+
`test run` starts a server-side task immediately. By default the CLI process polls until the task is
|
|
259
|
+
terminal; `--watch false` disables websocket progress streaming but still waits. Pass `--wait false` to
|
|
260
|
+
return after launch with the task id. A single run with `--names "a|b|c"` selects cases into one suite task,
|
|
261
|
+
and those cases execute sequentially in declaration order. To overlap independent slow cases, launch
|
|
262
|
+
separate `remits-cli test run --names "<case>" --wait false` commands from the same staged lane and keep the
|
|
263
|
+
printed task ids; complete each proof with `test status --task-id <id>`. Avoid parallel runs for cases that
|
|
264
|
+
share mutable fixtures, suite-level side effects, or undeclared live AI/provider calls.
|
|
265
|
+
|
|
266
|
+
### Reading a failing run
|
|
267
|
+
|
|
268
|
+
`test run` and `test status` print a **verdict**, not the run. The whole run object used to be dumped as
|
|
269
|
+
pretty JSON in human mode - 220 KB for one 92-case suite, most of it identifiers repeated once per case - so
|
|
270
|
+
the command an agent runs most often was the one that spent its remaining room to think. Every byte is still
|
|
271
|
+
there behind `--json`, and behind `test status --task-id <id> --json` once the live status expires.
|
|
272
|
+
|
|
273
|
+
What you get instead, and what to do with it:
|
|
274
|
+
|
|
275
|
+
```text
|
|
276
|
+
Cases: 59 passed, 33 failed, 92 total
|
|
277
|
+
|
|
278
|
+
Failure roots (33 failed case(s), 21 distinct root(s)):
|
|
279
|
+
10x assert statement.data.processingStatus == 'Analyzed'
|
|
280
|
+
e.g. bundled UK statements (+9 more)
|
|
281
|
+
2x assert statement.data.feeBreakdownChecked == true
|
|
282
|
+
e.g. 7851 Flat Rate (+1 more)
|
|
283
|
+
Largest root first: remits-cli test run --test "Statement Reader Calculations" --names "bundled UK statements"
|
|
284
|
+
```
|
|
285
|
+
|
|
286
|
+
**Thirty-three failures are not thirty-three problems.** Cases are grouped by their assertion root - the
|
|
287
|
+
failure with the particulars (ids, numbers, Groovy's `Expression:`/`Values:` decoration) removed - largest
|
|
288
|
+
group first, and one pivot block is printed per root rather than per case. Fix the largest root against one
|
|
289
|
+
named case, then re-run the whole suite. Re-running the suite before you have a root is how a session spends
|
|
290
|
+
an afternoon at the same pass count.
|
|
291
|
+
|
|
292
|
+
### A pass count only means something within one data lane
|
|
293
|
+
|
|
294
|
+
`test runs` is scoped to the data lane of the command by default, and says so:
|
|
295
|
+
|
|
296
|
+
```text
|
|
297
|
+
Data lane: test (6 more run(s) in the other lane; --all-lanes to include)
|
|
298
|
+
```
|
|
299
|
+
|
|
300
|
+
A run's `dataMode` **is** the lane it was written in, so `64/92` in the prod lane and `59/92` in the test lane
|
|
301
|
+
are two different facts, not a regression. `--compare` (and `test runs --compare`) picks the newest two runs
|
|
302
|
+
of the same **shape** - same data lane, branch, workspace and case count - and says how many newer runs it
|
|
303
|
+
skipped; it refuses rather than comparing across worlds. `test status --data-mode prod` on a run recorded in
|
|
304
|
+
the test lane returns the run's real lane and now says that is what happened.
|
|
305
|
+
|
|
255
306
|
A run labelled **`AI MOCKED`** replayed every AI turn from `aiMock`: it produces the same outcomes, scores and
|
|
256
307
|
metrics a measured run would, so read that label before treating green as evidence. `compare` warns when base and
|
|
257
308
|
head used different AI modes; `consistency` warns before calling a case unstable when its runs mixed modes.
|
|
@@ -16,6 +16,7 @@
|
|
|
16
16
|
- [When staged overrides apply](#when-staged-overrides-apply)
|
|
17
17
|
- [Diagnosing which version is in play](#diagnosing-which-version-is-in-play)
|
|
18
18
|
- [Working alongside other agents: the staging WORKSPACE](#working-alongside-other-agents-the-staging-workspace)
|
|
19
|
+
- [Keeping your remits-cli current](#keeping-your-remits-cli-current)
|
|
19
20
|
- [A lane holds an OVERLAY; your workset is a different number](#a-lane-holds-an-overlay-your-workset-is-a-different-number)
|
|
20
21
|
- [Stage / sync / clear with remits-cli](#stage--sync--clear-with-remits-cli)
|
|
21
22
|
- [Stale after sync / commit (the in-memory compile cache)](#stale-after-sync--commit-the-in-memory-compile-cache)
|
|
@@ -250,6 +251,18 @@ index with `remits-cli workstream status`.
|
|
|
250
251
|
`null` means unknown — a lane staged by an older CLI, or a branch whose last sync is not recorded —
|
|
251
252
|
never "fresh".
|
|
252
253
|
|
|
254
|
+
### Keeping your remits-cli current
|
|
255
|
+
|
|
256
|
+
The diagnostics in these references only exist in the CLI that prints them. An older install does not print a
|
|
257
|
+
worse version of them — it prints nothing, with no error, so nothing tells you what you are not being shown.
|
|
258
|
+
`components stage` and `test run` warn when yours is behind:
|
|
259
|
+
|
|
260
|
+
```text
|
|
261
|
+
[remits-cli 0.1.120 -> 0.1.136] ... npm install -g @remits/remits-cli@latest
|
|
262
|
+
```
|
|
263
|
+
|
|
264
|
+
Upgrade when you see it. Auto-update handles it for you on most commands, including `--json` ones.
|
|
265
|
+
|
|
253
266
|
### A lane holds an OVERLAY; your workset is a different number
|
|
254
267
|
|
|
255
268
|
This is the distinction that decides whether a lane is legible to anyone but you.
|
|
@@ -20,6 +20,7 @@
|
|
|
20
20
|
- [Stage your workset, not the whole repo](#stage-your-workset-not-the-whole-repo)
|
|
21
21
|
- [Three numbers, three questions](#three-numbers-three-questions)
|
|
22
22
|
- [Step 4: Verify the Change](#step-4-verify-the-change)
|
|
23
|
+
- [Many failures are usually few causes](#many-failures-are-usually-few-causes)
|
|
23
24
|
- [Step 5: Iterate If Needed](#step-5-iterate-if-needed)
|
|
24
25
|
- [Step 6: Update Documentation](#step-6-update-documentation)
|
|
25
26
|
- [Temporary Experiment Workflow](#temporary-experiment-workflow)
|
|
@@ -411,14 +412,54 @@ Tests run on the platform against your staged snapshot. They stream results in r
|
|
|
411
412
|
id + content hash, platform sync vs local HEAD. If it is not the world you meant, stop: the result will be
|
|
412
413
|
about a different world.
|
|
413
414
|
|
|
415
|
+
The platform launch is asynchronous. By default `remits-cli test run` waits for its task to finish, and any
|
|
416
|
+
cases selected by `--names "a|b"` execute sequentially inside that suite task. If several slow cases are
|
|
417
|
+
independent, stage once, then start separate `test run --names "<case>" --wait false` commands from the same
|
|
418
|
+
lane and keep the task ids they print. Complete each proof with
|
|
419
|
+
`remits-cli test status --task-id <id>`; that terminal status read records and attaches the final `test_run`
|
|
420
|
+
evidence. Do not parallelize cases that mutate the same fixture, rely on shared suite setup state, or make
|
|
421
|
+
undeclared live AI/provider calls.
|
|
422
|
+
|
|
414
423
|
If cases fail, read the printed summary and pivots first: each case has an `outcome`
|
|
415
424
|
(`failed`, `error`, `budget_exceeded`, `provider_unavailable`, …) with its reason, timing, trace id, AI usage
|
|
416
425
|
(live vs mocked, cost), bounded `report(...)` diagnostics, live HTTP signals, and resolved component
|
|
417
426
|
provenance. Fix the code, re-stage, and re-run only after those pivots explain the failure.
|
|
418
427
|
|
|
428
|
+
##### Many failures are usually few causes
|
|
429
|
+
|
|
430
|
+
When several cases fail, the run leads with **failure roots** — the failures grouped by their assertion with
|
|
431
|
+
the particulars (ids, numbers, Groovy's `Expression:`/`Values:` decoration) stripped out, largest group first,
|
|
432
|
+
one pivot block per root rather than per case:
|
|
433
|
+
|
|
434
|
+
```text
|
|
435
|
+
Failure roots (33 failed case(s), 21 distinct root(s)):
|
|
436
|
+
10x assert statement.data.processingStatus == 'Analyzed'
|
|
437
|
+
e.g. bundled UK statements (+9 more)
|
|
438
|
+
Largest root first: remits-cli test run --test "Statement Reader Calculations" --names "bundled UK statements"
|
|
439
|
+
```
|
|
440
|
+
|
|
441
|
+
**Work the largest root against ONE named case, then re-run the suite.** A full re-run after every edit is the
|
|
442
|
+
most expensive way to learn nothing: a real account spent two days and twenty-five runs holding a suite at
|
|
443
|
+
56–59 of 92 while ten of its thirty-three failures were one cause. The narrow run is seconds, tells you
|
|
444
|
+
whether the cause moved, and leaves your context for the actual reasoning.
|
|
445
|
+
|
|
446
|
+
Two failure roots mean "stop and look elsewhere", not "iterate harder":
|
|
447
|
+
|
|
448
|
+
- **`aiMock ctx.replay(...) found no usable stored provider response`** — the message says which of three
|
|
449
|
+
things happened. *No rows at all* for that sessionId means the stored session this fixture borrows is not
|
|
450
|
+
on this platform and will not come back; the case cannot pass until the mock builds its own response with
|
|
451
|
+
`ctx.toolCall(...)` / `ctx.content(...)`. Do not re-run it. (Replaying a session that IS there renews it,
|
|
452
|
+
so a suite that runs regularly keeps its fixtures.)
|
|
453
|
+
- **`Method too large` / `Class too large` / a synthetic `_closureNN`** — a JVM limit on one method body, not
|
|
454
|
+
a bug in the line it names. Staging now warns *before* the refusal and names the closure's source line span.
|
|
455
|
+
Split that body; see `development-guide.md` → *Keep Component Bodies Split*.
|
|
456
|
+
|
|
419
457
|
Every finished run is recorded durably: `remits-cli test status --task-id <id>` answers after the live status
|
|
420
|
-
expires, `remits-cli test runs --test <name> --compare` compares the latest run with the
|
|
421
|
-
`remits-cli test compare --base <id> --head <id>` compares any two.
|
|
458
|
+
expires, `remits-cli test runs --test <name> --compare` compares the latest run with the newest **comparable**
|
|
459
|
+
one, and `remits-cli test compare --base <id> --head <id>` compares any two. Comparable means the same data
|
|
460
|
+
lane, branch, workspace and case count: a run's `dataMode` is the lane it was written in, so `64/92` in prod
|
|
461
|
+
and `59/92` in test are two facts and not a trend. `test runs` lists one lane at a time and names it
|
|
462
|
+
(`--all-lanes` to see both). For evaluation suites (a `corpus(name)` of
|
|
422
463
|
cases seeded with `remits-cli corpus import`, one case per corpus case, intentional live AI inside
|
|
423
464
|
`withAiBudget(...)`, measurements compared with `remits-cli corpus compare` / `corpus consistency`), read
|
|
424
465
|
`guides/test-components.md` → *Evaluation Suites And Corpora*.
|
|
@@ -76,6 +76,9 @@ remits-cli tool --account-id 21 --as-account 37 --target-account 37 --name mcp_r
|
|
|
76
76
|
# poll by run id: --account-id 21 --as-account 37 --target-account 37 --input '{"controlAction":"status","actionRunId":"my-stable-run-id"}'
|
|
77
77
|
```
|
|
78
78
|
|
|
79
|
+
That is the CLI Action-run surface. The command is still `remits-cli tool`; this CLI does not have a
|
|
80
|
+
separate top-level `remits-cli action`, `remits-cli actions`, or `remits-cli run action` wrapper.
|
|
81
|
+
|
|
79
82
|
For long tools that lack their own async mode, use the CLI transport async (`--async true`), optionally with
|
|
80
83
|
`--wait true` to poll locally, and `remits-cli tool status --call-id <callId>`. Do not stack both mechanisms
|
|
81
84
|
(see `command-reference.md` → *Tool Execution Lifecycle*). Use `--timeout-ms <ms>` only to adjust the per-request client timeout; it is
|
|
@@ -403,12 +406,18 @@ Front-stage references:
|
|
|
403
406
|
| `page` | no | 1-based page number. Default: `1` |
|
|
404
407
|
| `pageSize` | no | Results per page. Default: `25`, max: `100` |
|
|
405
408
|
| `summaryOnly` | no | Detail mode: return the MAP (record index + stats + timeline) with no payloads. Same as `parts:["index"]` |
|
|
406
|
-
| `
|
|
409
|
+
| `compact` | no | Detail mode: after reading the MAP, open selected records without repeating the `records` index and `timeline`; keeps `groupingSummary`, `representativeSession`, `stats`, selected ids, and requested payload sections |
|
|
410
|
+
| `payloadOnly` | no | Alias for `compact` |
|
|
411
|
+
| `parts` | no | Detail parts: `index`, `map`, `records`, `stats`, `transcript`, `tool_calls`, and record sections `full`, `conversation_messages`, `system`, `response`, `tools` |
|
|
407
412
|
| `sections` | no | Alias for `parts` |
|
|
408
413
|
| `recordIds` | no | Detail mode: open only these request/response record IDs |
|
|
409
414
|
| `toolCallIds` | no | Detail mode: return the FULL exact input/result from `ai_tool_call` for these tool-call ids (what the tool PRODUCED — see the lens caveat) |
|
|
410
415
|
| `consolidateContext` | no | When `true`, collapses repeated XML-like prompt context into a consolidated section |
|
|
411
416
|
|
|
417
|
+
Detail responses expose both lenses: `groupingSummary` is the authoritative grouping-wide summary
|
|
418
|
+
(counts, lanes, cost, status), while `representativeSession` names the concrete session used for
|
|
419
|
+
owner/runtime metadata. `session` remains only as a compatibility alias for older callers.
|
|
420
|
+
|
|
412
421
|
Spend: each search row carries `liveRequestCount`, `mockedRequestCount`, `lanes`, `testTaskId` and
|
|
413
422
|
`estimatedCost`/`estimatedCostUsd` = **live spend only** (a mocked turn is never priced, even when it replays a
|
|
414
423
|
recording with a cost). `pageTotals` sums the page. Turns recorded before the platform stored the mock flag are
|
|
@@ -416,12 +425,16 @@ recording with a cost). `pageTotals` sums the page. Turns recorded before the pl
|
|
|
416
425
|
is refused rather than silently widened; the applied bounds are echoed as `filters.rangeStart`/`rangeEnd`.
|
|
417
426
|
To audit one Test run: `{"action":"search","testTaskId":"<taskId>","pageSize":100}`.
|
|
418
427
|
|
|
419
|
-
Audit flow: `action:"search"` to find the grouping → `action:"detail"` + `summaryOnly:true` for the MAP → re-call detail with `recordIds`/`toolCallIds` + `parts` to open exactly what you need. Prefer the map → open flow over a full-detail dump. Remember the two-lens rule: `tool_calls`/`toolCallIds` is what the tool PRODUCED; `conversation_messages`/`transcript` is what the AI CONSUMED (after any `_offload`/`_hideResult`/`_message`/supersede/evict transform).
|
|
428
|
+
Audit flow: `action:"search"` to find the grouping → `action:"detail"` + `summaryOnly:true` for the MAP → re-call detail with `recordIds`/`toolCallIds` + `parts` to open exactly what you need, usually with `compact:true` once the map has chosen the row. Prefer the map → open flow over a full-detail dump. Remember the two-lens rule: `tool_calls`/`toolCallIds` is what the tool PRODUCED; `conversation_messages`/`transcript` is what the AI CONSUMED (after any `_offload`/`_hideResult`/`_message`/supersede/evict transform).
|
|
420
429
|
|
|
421
430
|
### `mcp_run_action`
|
|
422
431
|
Run an Action on a target account, with explicit prod/test data mode, optional staged branch resolution, and
|
|
423
432
|
staged-vs-DB provenance in the result.
|
|
424
433
|
|
|
434
|
+
Invoke it with `remits-cli tool --name mcp_run_action`. Despite the natural shorthand "run an Action", there
|
|
435
|
+
is no separate top-level `remits-cli action` / `actions` command and no `remits-cli run action` wrapper in
|
|
436
|
+
this CLI build.
|
|
437
|
+
|
|
425
438
|
Describe the Action first when the input shape is not obvious. This does not execute the Action:
|
|
426
439
|
|
|
427
440
|
```bash
|
|
@@ -832,6 +845,10 @@ Common uses:
|
|
|
832
845
|
- `remits-cli components branch <name> --diff <id> --component-type <kind>` — compare one variant against
|
|
833
846
|
current trunk.
|
|
834
847
|
- `remits-cli components branch <name> --subscribers` — list accounts resolving that branch.
|
|
848
|
+
- `remits-cli components branch <name> --copy-to <newBranch> [--dry-run] [--force]` — seed a new variant
|
|
849
|
+
branch with the source branch's stored overlays before the first safe sync of the new branch. Dry-run
|
|
850
|
+
reports the copy plan without writes; force is required when the target already has overlays or live
|
|
851
|
+
subscribers.
|
|
835
852
|
|
|
836
853
|
### `mcp_cache`
|
|
837
854
|
Bounded read-only investigation of the platform Redis keyspace — the way to see exactly what a staged
|