@remits/remits-cli 0.1.136 → 0.1.138
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/index.js +570 -85
- package/package.json +1 -1
- package/skills/remits-cli/SKILL.md +12 -3
- package/skills/remits-cli/references/account-targeting.md +7 -5
- package/skills/remits-cli/references/branch-variants.md +25 -0
- package/skills/remits-cli/references/command-reference.md +67 -3
- package/skills/remits-cli/references/component-resolution.md +13 -0
- package/skills/remits-cli/references/development-loop.md +48 -7
- package/skills/remits-cli/references/tool-reference.md +19 -2
package/index.js
CHANGED
|
@@ -4,18 +4,18 @@
|
|
|
4
4
|
## Table of Contents
|
|
5
5
|
|
|
6
6
|
- L22 Runtime Bootstrap And Shared State
|
|
7
|
-
-
|
|
8
|
-
-
|
|
9
|
-
-
|
|
10
|
-
-
|
|
11
|
-
-
|
|
12
|
-
-
|
|
13
|
-
-
|
|
14
|
-
-
|
|
15
|
-
-
|
|
16
|
-
-
|
|
17
|
-
-
|
|
18
|
-
-
|
|
7
|
+
- L123 Sessions, Account Resolution, And Production Guards
|
|
8
|
+
- L615 Local State, Workspaces, And Verification Evidence
|
|
9
|
+
- L2652 Account Repos, Guide Sync, And Platform Repo
|
|
10
|
+
- L3460 Component Discovery And HTTP Logging
|
|
11
|
+
- L4107 Skill Delivery And TOC Resolution
|
|
12
|
+
- L4417 Auth And Component Staging
|
|
13
|
+
- L5032 Component Summaries, Status, And Sync Gates
|
|
14
|
+
- L7104 Branches, Promotion, Commit, And Test Runs
|
|
15
|
+
- L9193 Tokens, Tools, Verification, And Config
|
|
16
|
+
- L10620 Service Dashboard And WebSocket Listener
|
|
17
|
+
- L12743 Agent And Ticket Workflows
|
|
18
|
+
- L16338 Help, Auto Update, And Command Dispatch
|
|
19
19
|
*/
|
|
20
20
|
|
|
21
21
|
/*
|
|
@@ -58,6 +58,10 @@ const SERVICE_STATE_FILE = path.join(SESSION_DIR, 'service-state.json');
|
|
|
58
58
|
const ISSUES_DIR = path.join(SESSION_DIR, 'issues');
|
|
59
59
|
const WEBSOCKET_STATE_FILE = path.join(SESSION_DIR, 'websocket-state.json');
|
|
60
60
|
const AUTO_UPDATE_LOCK_DIR = path.join(SESSION_DIR, 'auto-update.lock');
|
|
61
|
+
const AUTO_UPDATE_STAMP_FILE = path.join(SESSION_DIR, 'auto-update-check.json');
|
|
62
|
+
// `npm view` is a network round trip. It used to run on EVERY command, which is a per-command latency tax
|
|
63
|
+
// on the agents that issue the most commands.
|
|
64
|
+
const AUTO_UPDATE_CHECK_INTERVAL_MS = 6 * 60 * 60 * 1000;
|
|
61
65
|
const AUTO_UPDATE_DISABLED_VALUES = new Set(['0', 'false', 'no', 'off']);
|
|
62
66
|
const REMITS_CLI_PACKAGE_NAME = '@remits/remits-cli';
|
|
63
67
|
const DEFAULT_DATA_MODE = 'test';
|
|
@@ -154,6 +158,14 @@ function flagEnabled(value) {
|
|
|
154
158
|
return value === true || value === 'true' || value === '1' || value === 'yes';
|
|
155
159
|
}
|
|
156
160
|
|
|
161
|
+
function flagDisabled(value) {
|
|
162
|
+
return value === false || value === 'false' || value === '0' || value === 'no';
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
function waitForTestCompletion(flags = {}) {
|
|
166
|
+
return !(flagDisabled(flags.wait) || flagEnabled(flags['no-wait']) || flagEnabled(flags.noWait));
|
|
167
|
+
}
|
|
168
|
+
|
|
157
169
|
function withStdoutRoutedToStderr(enabled, fn) {
|
|
158
170
|
if (!enabled) return fn();
|
|
159
171
|
const originalLog = console.log;
|
|
@@ -898,6 +910,22 @@ function printLocalStateWarnings(cwd, flags = {}) {
|
|
|
898
910
|
warnings.forEach((warning) => console.log(' - ' + warning));
|
|
899
911
|
}
|
|
900
912
|
|
|
913
|
+
/**
|
|
914
|
+
* "Your remits-cli is older than this platform's" - printed once per process, from ONE place.
|
|
915
|
+
*
|
|
916
|
+
* The platform attaches `cliAdvisory` to the two responses every working loop passes through (stage and
|
|
917
|
+
* test-run start) and returns nothing at all when the caller is current, so this renders only when there
|
|
918
|
+
* is something to act on. stderr, because a `--json` caller's stdout is a document.
|
|
919
|
+
*/
|
|
920
|
+
let cliAdvisoryPrinted = false;
|
|
921
|
+
function printCliAdvisory(response) {
|
|
922
|
+
const advisory = response && response.cliAdvisory;
|
|
923
|
+
if (!advisory || cliAdvisoryPrinted) return;
|
|
924
|
+
cliAdvisoryPrinted = true;
|
|
925
|
+
console.error('');
|
|
926
|
+
console.error('[remits-cli ' + (advisory.current || '?') + ' -> ' + (advisory.expected || '?') + '] ' + advisory.message);
|
|
927
|
+
}
|
|
928
|
+
|
|
901
929
|
function printLocalCommandContext(cwd, flags = {}, context = {}) {
|
|
902
930
|
const paths = localStatePaths(cwd);
|
|
903
931
|
const branchName = context.branchName || flags.branch || safeGitValue(cwd, 'git rev-parse --abbrev-ref HEAD') || 'unknown';
|
|
@@ -3834,6 +3862,12 @@ function stageTimeoutMs(components, flags = {}) {
|
|
|
3834
3862
|
|
|
3835
3863
|
function buildAxios(baseUrl, token, timeoutMs = 60000) {
|
|
3836
3864
|
const headers = token ? { Authorization: 'Bearer ' + token } : {};
|
|
3865
|
+
// The CLI version on EVERY request, from the one place every request is built. It used to travel only on
|
|
3866
|
+
// `agent register`, so the platform knew the version of the sessions that registered - and the agents that
|
|
3867
|
+
// loop hardest are exactly the ones that never do. Without it the platform cannot tell an agent that the
|
|
3868
|
+
// diagnostics it is missing exist, and every output improvement lands invisibly.
|
|
3869
|
+
const cliVersion = readCliVersion();
|
|
3870
|
+
if (cliVersion) headers['X-Remits-Cli-Version'] = cliVersion;
|
|
3837
3871
|
return axios.create({ baseURL: baseUrl, timeout: parsePositiveInt(timeoutMs, 60000), headers });
|
|
3838
3872
|
}
|
|
3839
3873
|
|
|
@@ -5232,6 +5266,9 @@ function printStageSummary(response, flags) {
|
|
|
5232
5266
|
if (typeof printAccountLanes === 'function') {
|
|
5233
5267
|
printAccountLanes(response);
|
|
5234
5268
|
}
|
|
5269
|
+
if (typeof printCliAdvisory === 'function') {
|
|
5270
|
+
printCliAdvisory(response);
|
|
5271
|
+
}
|
|
5235
5272
|
printComponentCommandResponse('Components stage', response, flags);
|
|
5236
5273
|
}
|
|
5237
5274
|
|
|
@@ -7083,6 +7120,10 @@ async function branchesComponentsCommand(flags) {
|
|
|
7083
7120
|
// after the subcommand is the branch name when present.
|
|
7084
7121
|
const positional = flags._ && flags._[2];
|
|
7085
7122
|
const branchName = flags.branch || positional;
|
|
7123
|
+
const copyToBranch = flags['copy-to'] || flags.copyTo;
|
|
7124
|
+
if ((flags['copy-to'] !== undefined || flags.copyTo !== undefined) && !branchName) {
|
|
7125
|
+
throw new Error('components branch <source> --copy-to <target> requires a source branch name');
|
|
7126
|
+
}
|
|
7086
7127
|
|
|
7087
7128
|
// Without a branch there is nothing to detail, so always list. With one, the mutating flags
|
|
7088
7129
|
// (--subscribe/--unsubscribe/--retire) win, then the narrower read views, and the default is the
|
|
@@ -7092,10 +7133,14 @@ async function branchesComponentsCommand(flags) {
|
|
|
7092
7133
|
if (flags.subscribe !== undefined) mode = 'subscribe';
|
|
7093
7134
|
else if (flags.unsubscribe !== undefined) mode = 'unsubscribe';
|
|
7094
7135
|
else if (flagEnabled(flags.retire)) mode = 'retire';
|
|
7136
|
+
else if (flags['copy-to'] !== undefined || flags.copyTo !== undefined) mode = 'copy';
|
|
7095
7137
|
else if (flags.diff !== undefined) mode = 'diff';
|
|
7096
7138
|
else if (flagEnabled(flags.subscribers)) mode = 'subscribers';
|
|
7097
7139
|
else mode = 'status';
|
|
7098
7140
|
}
|
|
7141
|
+
if (mode === 'copy' && (copyToBranch === true || !String(copyToBranch || '').trim())) {
|
|
7142
|
+
throw new Error('components branch <source> --copy-to <target> requires a target branch name');
|
|
7143
|
+
}
|
|
7099
7144
|
|
|
7100
7145
|
// `--diff 42` carries the component id inline; a bare `--diff` falls back to --component-id/--name.
|
|
7101
7146
|
const componentId = mode === 'diff'
|
|
@@ -7118,6 +7163,7 @@ async function branchesComponentsCommand(flags) {
|
|
|
7118
7163
|
componentType: flags['component-type'] || flags.type,
|
|
7119
7164
|
componentId,
|
|
7120
7165
|
componentName: flags['component-name'] || flags.name,
|
|
7166
|
+
targetBranchName: copyToBranch,
|
|
7121
7167
|
subscribeAccountId,
|
|
7122
7168
|
parentAccountId: flags['parent-account'],
|
|
7123
7169
|
// Optional branch-scoped custom host set on the same edge as the subscription.
|
|
@@ -7241,6 +7287,33 @@ function printBranchesSummary(response) {
|
|
|
7241
7287
|
return;
|
|
7242
7288
|
}
|
|
7243
7289
|
|
|
7290
|
+
if (response.mode === 'copy') {
|
|
7291
|
+
console.log(response.message || ('Copied branch overlays to ' + response.targetBranch));
|
|
7292
|
+
console.log('Source branch:', response.branch);
|
|
7293
|
+
console.log('Target branch:', response.targetBranch);
|
|
7294
|
+
// A dry run writes nothing, so `copied` is 0 by contract — printing it as "Copied overlays: 0"
|
|
7295
|
+
// under a "Would copy 2" message reads as a failed copy. Name the plan instead.
|
|
7296
|
+
if (response.dryRun) {
|
|
7297
|
+
console.log('Plan only (dry run) — overlays that would be copied:', response.plannedCopies || 0);
|
|
7298
|
+
if (response.existingCount) console.log('Existing overlays that would be replaced:', response.existingCount);
|
|
7299
|
+
} else {
|
|
7300
|
+
console.log('Copied overlays:', response.copied || 0);
|
|
7301
|
+
if (response.overwritten) console.log('Replaced existing overlays:', response.overwritten);
|
|
7302
|
+
}
|
|
7303
|
+
// A source overlay that is identical to trunk in both content and metadata is sparse and is never
|
|
7304
|
+
// stored, so it is named rather than silently missing from the count.
|
|
7305
|
+
const skippedSparse = response.skippedIdenticalToTrunk || [];
|
|
7306
|
+
if (skippedSparse.length) {
|
|
7307
|
+
console.log('Skipped as identical to trunk:', skippedSparse.join(', '));
|
|
7308
|
+
}
|
|
7309
|
+
// The number an operator needs BEFORE deciding to pass --force: a target branch with live
|
|
7310
|
+
// subscribers is code those accounts are running right now.
|
|
7311
|
+
if (response.targetSubscriberCount) {
|
|
7312
|
+
console.log('Live subscribers on the target branch:', response.targetSubscriberCount);
|
|
7313
|
+
}
|
|
7314
|
+
return;
|
|
7315
|
+
}
|
|
7316
|
+
|
|
7244
7317
|
if (response.mode === 'subscribe' || response.mode === 'unsubscribe' || response.mode === 'retire') {
|
|
7245
7318
|
console.log(response.message);
|
|
7246
7319
|
if (response.requiresConfirmation) {
|
|
@@ -7852,7 +7925,7 @@ async function waitForStatus(api, cwd, accountId, branchName, taskId, token, dat
|
|
|
7852
7925
|
dataMode
|
|
7853
7926
|
}).then((r) => r.data);
|
|
7854
7927
|
|
|
7855
|
-
if (status
|
|
7928
|
+
if (testStatusIsTerminal(status)) {
|
|
7856
7929
|
return status;
|
|
7857
7930
|
}
|
|
7858
7931
|
await new Promise((r) => setTimeout(r, pollDelayMs));
|
|
@@ -7860,6 +7933,10 @@ async function waitForStatus(api, cwd, accountId, branchName, taskId, token, dat
|
|
|
7860
7933
|
}
|
|
7861
7934
|
}
|
|
7862
7935
|
|
|
7936
|
+
function testStatusIsTerminal(status = {}) {
|
|
7937
|
+
return ['completed', 'failed', 'interrupted'].includes(String(status.status || '').toLowerCase());
|
|
7938
|
+
}
|
|
7939
|
+
|
|
7863
7940
|
function testEvidenceCategories(status = {}, selectedNames = []) {
|
|
7864
7941
|
const categories = new Set(['test_run']);
|
|
7865
7942
|
const test = status.test || {};
|
|
@@ -7989,6 +8066,108 @@ function detectNondeterministicTestRun(priorPackets, currentStatus, currentProve
|
|
|
7989
8066
|
return null;
|
|
7990
8067
|
}
|
|
7991
8068
|
|
|
8069
|
+
async function appendTerminalTestRunEvidence(options = {}) {
|
|
8070
|
+
const {
|
|
8071
|
+
api,
|
|
8072
|
+
cwd,
|
|
8073
|
+
session,
|
|
8074
|
+
accountId,
|
|
8075
|
+
flags,
|
|
8076
|
+
status,
|
|
8077
|
+
testRef,
|
|
8078
|
+
selectedNames = [],
|
|
8079
|
+
dataMode,
|
|
8080
|
+
dataModeSource,
|
|
8081
|
+
branchName,
|
|
8082
|
+
workspace,
|
|
8083
|
+
baseUrl,
|
|
8084
|
+
activeEnvelopeId,
|
|
8085
|
+
quiet
|
|
8086
|
+
} = options;
|
|
8087
|
+
const unmatched = (status.result && status.result.unmatchedTestNames) || [];
|
|
8088
|
+
const componentProvenance = testComponentProvenance(status);
|
|
8089
|
+
const sourceRevision = collectVerificationSource(cwd, flags);
|
|
8090
|
+
const priorPackets = await readVerificationPacketsForDiagnostics(api, cwd, session, accountId, activeEnvelopeId, {
|
|
8091
|
+
packetType: 'test_run',
|
|
8092
|
+
suite: status.test && status.test.name,
|
|
8093
|
+
max: 50
|
|
8094
|
+
});
|
|
8095
|
+
const nondeterminism = detectNondeterministicTestRun(priorPackets, status, componentProvenance, {
|
|
8096
|
+
laneContentHash: status.staging && status.staging.laneSummary && status.staging.laneSummary.contentHash,
|
|
8097
|
+
gitHead: sourceRevision.gitHead
|
|
8098
|
+
});
|
|
8099
|
+
|
|
8100
|
+
await appendVerificationPacket(api, cwd, session, accountId, flags, {
|
|
8101
|
+
type: 'test_run',
|
|
8102
|
+
success: status.status === 'completed' && !(status.result && status.result.failed > 0) && !unmatched.length,
|
|
8103
|
+
claim: 'Test run ' + String(testRef || (status.test && (status.test.name || status.test.id)) || status.taskId || 'unknown'),
|
|
8104
|
+
world: buildCommandWorld(status, {
|
|
8105
|
+
accountId,
|
|
8106
|
+
dataMode: status.dataMode || dataMode,
|
|
8107
|
+
dataModeSource,
|
|
8108
|
+
branchName,
|
|
8109
|
+
workspace,
|
|
8110
|
+
host: normalizeBaseUrl(baseUrl),
|
|
8111
|
+
sourceLayer: status.staging && status.staging.testComponentSource
|
|
8112
|
+
}),
|
|
8113
|
+
revision: Object.assign(sourceRevision, {
|
|
8114
|
+
compileSignatures: testCompileSignatures(status),
|
|
8115
|
+
componentProvenance
|
|
8116
|
+
}),
|
|
8117
|
+
nondeterministic: nondeterminism ? true : undefined,
|
|
8118
|
+
nondeterminism: nondeterminism || undefined,
|
|
8119
|
+
test: {
|
|
8120
|
+
taskId: status.taskId || (status.result && status.result.taskId),
|
|
8121
|
+
testId: status.test && status.test.id,
|
|
8122
|
+
testName: status.test && status.test.name,
|
|
8123
|
+
selectedCases: selectedNames,
|
|
8124
|
+
passed: status.result && status.result.passed,
|
|
8125
|
+
failed: status.result && status.result.failed,
|
|
8126
|
+
tests: status.result && status.result.tests,
|
|
8127
|
+
dataModeSource: status.dataModeSource || dataModeSource,
|
|
8128
|
+
unmatchedTestNames: unmatched
|
|
8129
|
+
},
|
|
8130
|
+
evidenceCategories: testEvidenceCategories(status, selectedNames),
|
|
8131
|
+
// What the run spent and whether that live AI was declared (withAiBudget), so an envelope can tell a
|
|
8132
|
+
// measurement run from a suite with missing mocks without re-reading every case.
|
|
8133
|
+
dependencies: testRunDependencies(status),
|
|
8134
|
+
summary: status.result && status.result.summary,
|
|
8135
|
+
limitations: []
|
|
8136
|
+
.concat(unmatched.length ? ['One or more requested test case selectors matched no case.'] : [])
|
|
8137
|
+
.concat(nondeterminism ? [nondeterminism.message] : []),
|
|
8138
|
+
rawRefs: { testStatusKey: status.taskId, durableTestRun: status.result && status.result.durableRecord ? status.taskId : undefined },
|
|
8139
|
+
status
|
|
8140
|
+
}, { quiet });
|
|
8141
|
+
|
|
8142
|
+
return { unmatched, nondeterminism };
|
|
8143
|
+
}
|
|
8144
|
+
|
|
8145
|
+
/**
|
|
8146
|
+
* One failed case per distinct failure root, largest root first, capped.
|
|
8147
|
+
*
|
|
8148
|
+
* A pivot block per failed case is one problem stated N times: the thirty-three failures of a real suite
|
|
8149
|
+
* are twenty-one roots, and the twelve repeats carry the same trace shape, the same components and the
|
|
8150
|
+
* same error. The grouped roll-up above already names every case; this picks the ones worth a full block.
|
|
8151
|
+
*/
|
|
8152
|
+
/** Truncate, and SAY that it was truncated — a silent cut reads as the whole message. */
|
|
8153
|
+
function truncateForPivot(text, max) {
|
|
8154
|
+
if (text.length <= max) return text;
|
|
8155
|
+
return text.slice(0, max) + '… (truncated; --json for the full text)';
|
|
8156
|
+
}
|
|
8157
|
+
|
|
8158
|
+
function selectPivotCases(tests, limit = 6) {
|
|
8159
|
+
const failed = (tests || []).filter((test) => test && test.passed === false);
|
|
8160
|
+
const seen = new Set();
|
|
8161
|
+
const representatives = [];
|
|
8162
|
+
failed.forEach((test) => {
|
|
8163
|
+
const root = testFailureRoot(test);
|
|
8164
|
+
if (seen.has(root)) return;
|
|
8165
|
+
seen.add(root);
|
|
8166
|
+
representatives.push(test);
|
|
8167
|
+
});
|
|
8168
|
+
return { shown: representatives.slice(0, limit), failedCount: failed.length, rootCount: representatives.length };
|
|
8169
|
+
}
|
|
8170
|
+
|
|
7992
8171
|
function printTestRunPivots(status = {}) {
|
|
7993
8172
|
const tests = status.result && Array.isArray(status.result.tests) ? status.result.tests : [];
|
|
7994
8173
|
if (!tests.length) return;
|
|
@@ -7996,7 +8175,8 @@ function printTestRunPivots(status = {}) {
|
|
|
7996
8175
|
if (slowest && slowest.duration != null) {
|
|
7997
8176
|
console.log('Slowest case:', (slowest.name || '(unnamed)') + ' in ' + formatDurationMs(slowest.duration));
|
|
7998
8177
|
}
|
|
7999
|
-
|
|
8178
|
+
const selection = selectPivotCases(tests);
|
|
8179
|
+
selection.shown.forEach((test) => {
|
|
8000
8180
|
console.log('Pivots for failed case:', test.name || '(unnamed)');
|
|
8001
8181
|
if (test.outcome) console.log(' Outcome:', test.outcome + (test.outcomeReason && test.outcomeReason !== test.error ? ' - ' + String(test.outcomeReason).slice(0, 300) : ''));
|
|
8002
8182
|
if (test.duration != null) console.log(' Duration:', formatDurationMs(test.duration));
|
|
@@ -8007,7 +8187,9 @@ function printTestRunPivots(status = {}) {
|
|
|
8007
8187
|
if (test.threadGroupingId || test.threadGroupId || test.traceId) {
|
|
8008
8188
|
console.log(' Trace:', test.traceId || test.threadGroupingId || test.threadGroupId);
|
|
8009
8189
|
}
|
|
8010
|
-
|
|
8190
|
+
// 500 chars cut the replay diagnosis mid-sentence, losing the half that says what to DO. A pivot is the
|
|
8191
|
+
// block a reader acts on; truncating its one actionable sentence to save four lines is a bad trade.
|
|
8192
|
+
if (test.error) console.log(' Error:', truncateForPivot(String(test.error), 1200));
|
|
8011
8193
|
const diagnostics = test.diagnostics && typeof test.diagnostics === 'object' ? test.diagnostics : null;
|
|
8012
8194
|
if (diagnostics && Object.keys(diagnostics).length) {
|
|
8013
8195
|
console.log(' Diagnostics:', JSON.stringify(diagnostics).slice(0, 1200));
|
|
@@ -8023,6 +8205,12 @@ function printTestRunPivots(status = {}) {
|
|
|
8023
8205
|
console.log(' Live HTTP calls:', test.liveHttpCalls.length + (intentional ? ' (' + intentional + ' intentional, inside withAiBudget)' : ''));
|
|
8024
8206
|
}
|
|
8025
8207
|
});
|
|
8208
|
+
// Named, not silently dropped: a reader must be able to tell "this is everything" from "this is a sample".
|
|
8209
|
+
if (selection.rootCount > selection.shown.length) {
|
|
8210
|
+
console.log('Pivots shown for ' + selection.shown.length + ' of ' + selection.rootCount +
|
|
8211
|
+
' failure root(s) (' + selection.failedCount + ' failed case(s)). Every root is listed above; ' +
|
|
8212
|
+
'use --names "<case>" for one, or --json for all.');
|
|
8213
|
+
}
|
|
8026
8214
|
}
|
|
8027
8215
|
|
|
8028
8216
|
function formatDurationMs(value) {
|
|
@@ -8236,6 +8424,114 @@ function stagedMaskedByVariantLines(masked, accountId) {
|
|
|
8236
8424
|
return lines;
|
|
8237
8425
|
}
|
|
8238
8426
|
|
|
8427
|
+
/**
|
|
8428
|
+
* The assertion ROOT of one failed case: what broke, with the particulars of this case removed.
|
|
8429
|
+
*
|
|
8430
|
+
* Deliberately built only from properties of the JVM/Groovy failure format - a power-assert's
|
|
8431
|
+
* `Expression:`/`Values:` decoration, an `assert <expr>` head, an exception class prefix - and never from
|
|
8432
|
+
* any vocabulary a suite happens to use. Identifiers and numbers are replaced because two cases failing
|
|
8433
|
+
* the same way differ exactly in those.
|
|
8434
|
+
*/
|
|
8435
|
+
function testFailureRoot(test = {}) {
|
|
8436
|
+
let text = String(test.error || test.outcomeReason || test.outcome || 'failed').trim();
|
|
8437
|
+
// Groovy's power assert appends the rendered expression and every intermediate value.
|
|
8438
|
+
text = text.split(/\.\s+(?:Expression|Values):/)[0];
|
|
8439
|
+
const assertion = text.match(/assert\s+(.+)$/);
|
|
8440
|
+
if (assertion) text = 'assert ' + assertion[1];
|
|
8441
|
+
text = text
|
|
8442
|
+
.replace(/['"][0-9a-fA-F]{8}-[0-9a-fA-F-]{4,}['"]/g, "'<id>'")
|
|
8443
|
+
.replace(/\b[0-9a-fA-F]{8}-[0-9a-fA-F-]{27,}\b/g, '<id>')
|
|
8444
|
+
.replace(/\b\d[\d.,]*\b/g, '<n>')
|
|
8445
|
+
.replace(/\s+/g, ' ')
|
|
8446
|
+
.trim();
|
|
8447
|
+
if (text.length <= FAILURE_ROOT_LABEL_CHARS) return text;
|
|
8448
|
+
// Cut on a word boundary: "found no usable stored provid" reads as a different error than the one it is.
|
|
8449
|
+
const cut = text.slice(0, FAILURE_ROOT_LABEL_CHARS);
|
|
8450
|
+
const lastSpace = cut.lastIndexOf(' ');
|
|
8451
|
+
return (lastSpace > FAILURE_ROOT_LABEL_CHARS * 0.6 ? cut.slice(0, lastSpace) : cut) + '\u2026';
|
|
8452
|
+
}
|
|
8453
|
+
|
|
8454
|
+
/** Long enough to tell two failures apart, short enough that a root list stays a list. */
|
|
8455
|
+
const FAILURE_ROOT_LABEL_CHARS = 120;
|
|
8456
|
+
|
|
8457
|
+
/**
|
|
8458
|
+
* Failed cases grouped by that root, largest first.
|
|
8459
|
+
*
|
|
8460
|
+
* Thirty-three failures printed in run order read as thirty-three problems; the same run grouped is four.
|
|
8461
|
+
* This is the line that decides whether the next move is a narrow root-cause pass or another broad rerun,
|
|
8462
|
+
* so it is computed from the run itself rather than left to the reader.
|
|
8463
|
+
*/
|
|
8464
|
+
function testFailureGroups(status = {}) {
|
|
8465
|
+
const tests = status.result && Array.isArray(status.result.tests) ? status.result.tests : [];
|
|
8466
|
+
const groups = new Map();
|
|
8467
|
+
tests.filter((test) => test && test.passed === false && !test.interrupted).forEach((test) => {
|
|
8468
|
+
const root = testFailureRoot(test);
|
|
8469
|
+
if (!groups.has(root)) groups.set(root, { root, count: 0, cases: [] });
|
|
8470
|
+
const group = groups.get(root);
|
|
8471
|
+
group.count += 1;
|
|
8472
|
+
group.cases.push(test.name || '(unnamed)');
|
|
8473
|
+
});
|
|
8474
|
+
return Array.from(groups.values()).sort((left, right) => right.count - left.count || left.root.localeCompare(right.root));
|
|
8475
|
+
}
|
|
8476
|
+
|
|
8477
|
+
const FAILURE_GROUPS_SHOWN = 8;
|
|
8478
|
+
|
|
8479
|
+
function testFailureGroupLines(status = {}, options = {}) {
|
|
8480
|
+
const groups = testFailureGroups(status);
|
|
8481
|
+
if (!groups.length) return [];
|
|
8482
|
+
const failed = groups.reduce((sum, group) => sum + group.count, 0);
|
|
8483
|
+
const lines = [''];
|
|
8484
|
+
lines.push('Failure roots (' + failed + ' failed case(s), ' + groups.length + ' distinct root(s)):');
|
|
8485
|
+
groups.slice(0, FAILURE_GROUPS_SHOWN).forEach((group) => {
|
|
8486
|
+
lines.push(' ' + String(group.count) + 'x ' + group.root);
|
|
8487
|
+
// The representative case is what `--names` takes, so the next command is a copy of this line.
|
|
8488
|
+
lines.push(' e.g. ' + group.cases[0] + (group.count > 1 ? ' (+' + (group.count - 1) + ' more)' : ''));
|
|
8489
|
+
});
|
|
8490
|
+
if (groups.length > FAILURE_GROUPS_SHOWN) {
|
|
8491
|
+
lines.push(' ...' + (groups.length - FAILURE_GROUPS_SHOWN) + ' more root(s); --json for all');
|
|
8492
|
+
}
|
|
8493
|
+
const testRef = options.testRef || (status.test && (status.test.id || status.test.name)) ||
|
|
8494
|
+
(status.result && (status.result.testId || status.result.testName));
|
|
8495
|
+
if (groups[0] && groups[0].count > 1 && testRef) {
|
|
8496
|
+
lines.push('Largest root first: remits-cli test run --test ' + JSON.stringify(String(testRef)) +
|
|
8497
|
+
' --names ' + JSON.stringify(groups[0].cases[0]));
|
|
8498
|
+
}
|
|
8499
|
+
return lines;
|
|
8500
|
+
}
|
|
8501
|
+
|
|
8502
|
+
/**
|
|
8503
|
+
* The verdict of a run in four lines, shared by `test run` and `test status`.
|
|
8504
|
+
*
|
|
8505
|
+
* `test run` used to print the whole run object as pretty JSON before its human summary. Measured on one
|
|
8506
|
+
* 92-case suite that is 220 KB - most of it per-case ids repeated once per case - spent by the command an
|
|
8507
|
+
* agent runs most often, on the turn where it has the least room left to reason. Every byte is still one
|
|
8508
|
+
* `--json` away.
|
|
8509
|
+
*/
|
|
8510
|
+
function testRunHeadlineLines(status = {}, options = {}) {
|
|
8511
|
+
const result = status.result || {};
|
|
8512
|
+
const lines = [];
|
|
8513
|
+
lines.push('Test run status: ' + (status.status || 'unknown'));
|
|
8514
|
+
if (options.taskId || status.taskId) lines.push('Task ID: ' + (options.taskId || status.taskId));
|
|
8515
|
+
if (options.source) lines.push('Source: ' + options.source);
|
|
8516
|
+
const actualDataMode = status.dataMode || result.dataMode || options.dataMode;
|
|
8517
|
+
let dataModeLine = 'Data mode: ' + (actualDataMode || 'unknown');
|
|
8518
|
+
// A durable run answers about the lane it RAN in, which need not be the lane this command asked for.
|
|
8519
|
+
// Reporting only the run's own lane is correct and reads as if the request had been honoured.
|
|
8520
|
+
if (options.requestedDataMode && actualDataMode && options.requestedDataMode !== actualDataMode) {
|
|
8521
|
+
dataModeLine += ' (you asked for ' + options.requestedDataMode + '; this run was recorded in the ' +
|
|
8522
|
+
actualDataMode + ' lane, and that is what it proves)';
|
|
8523
|
+
}
|
|
8524
|
+
lines.push(dataModeLine);
|
|
8525
|
+
if (result.total != null || result.passed != null) {
|
|
8526
|
+
lines.push('Cases: ' + (result.passed || 0) + ' passed, ' + (result.failed || 0) + ' failed, ' +
|
|
8527
|
+
(result.total || 0) + ' total');
|
|
8528
|
+
}
|
|
8529
|
+
if (status.error || status.message || result.error) {
|
|
8530
|
+
lines.push('Message: ' + (status.error || status.message || result.error));
|
|
8531
|
+
}
|
|
8532
|
+
return lines;
|
|
8533
|
+
}
|
|
8534
|
+
|
|
8239
8535
|
// The corpus-style roll-up printed after a run: outcome counts, case duration percentiles, AI usage split
|
|
8240
8536
|
// live/mocked, and the durable record. Built only from the result the platform returned.
|
|
8241
8537
|
function testRunSummaryLines(status = {}) {
|
|
@@ -8491,6 +8787,7 @@ async function testCommand(flags) {
|
|
|
8491
8787
|
printLocalStateWarnings(cwd, flags);
|
|
8492
8788
|
printStagingLane(branchName, workspace, workspaceSource(cwd, flags));
|
|
8493
8789
|
printStagingLaneOwnerNotice(start.staging || {});
|
|
8790
|
+
printCliAdvisory(start);
|
|
8494
8791
|
if (start.staging && Array.isArray(start.staging.accountLanes)) {
|
|
8495
8792
|
printOrphanedWorkspaceWarning(start.staging);
|
|
8496
8793
|
printAccountLanes({ accountLanes: start.staging.accountLanes, branchName, workspace });
|
|
@@ -8506,6 +8803,48 @@ async function testCommand(flags) {
|
|
|
8506
8803
|
});
|
|
8507
8804
|
runtimeState.currentTestTaskId = start.taskId;
|
|
8508
8805
|
|
|
8806
|
+
const waitForCompletion = waitForTestCompletion(flags);
|
|
8807
|
+
if (!waitForCompletion) {
|
|
8808
|
+
recordEvidenceEntry(cwd, {
|
|
8809
|
+
type: 'test_run',
|
|
8810
|
+
success: null,
|
|
8811
|
+
pending: true,
|
|
8812
|
+
claim: 'Test run ' + String(testRef),
|
|
8813
|
+
world: buildCommandWorld(start, {
|
|
8814
|
+
accountId,
|
|
8815
|
+
dataMode,
|
|
8816
|
+
dataModeSource,
|
|
8817
|
+
branchName,
|
|
8818
|
+
workspace,
|
|
8819
|
+
host: normalizeBaseUrl(baseUrl),
|
|
8820
|
+
sourceLayer: start.staging && start.staging.testComponentSource
|
|
8821
|
+
}),
|
|
8822
|
+
test: {
|
|
8823
|
+
taskId: start.taskId,
|
|
8824
|
+
testId: start.test && start.test.id,
|
|
8825
|
+
testName: start.test && start.test.name,
|
|
8826
|
+
selectedCases: names,
|
|
8827
|
+
dataModeSource
|
|
8828
|
+
},
|
|
8829
|
+
evidenceCategories: testEvidenceCategories(start, names),
|
|
8830
|
+
rawRefs: { testStatusKey: start.taskId },
|
|
8831
|
+
status: start
|
|
8832
|
+
}, { accountId, dataMode, branchName, workspace, baseUrl: normalizeBaseUrl(baseUrl), envelopeId: activeEnvelopeId });
|
|
8833
|
+
runtimeState.currentTestTaskId = null;
|
|
8834
|
+
if (jsonOutput) {
|
|
8835
|
+
console.log(JSON.stringify(Object.assign({}, start, {
|
|
8836
|
+
pending: true,
|
|
8837
|
+
wait: false,
|
|
8838
|
+
statusCommand: 'remits-cli test status --task-id ' + start.taskId
|
|
8839
|
+
}), null, 2));
|
|
8840
|
+
} else {
|
|
8841
|
+
console.log('Not waiting (--wait false).');
|
|
8842
|
+
console.log('Poll this run: remits-cli test status --task-id ' + start.taskId);
|
|
8843
|
+
console.log('Terminal evidence attaches when `test status` reads a completed, failed or interrupted run.');
|
|
8844
|
+
}
|
|
8845
|
+
return;
|
|
8846
|
+
}
|
|
8847
|
+
|
|
8509
8848
|
let stopWs = null;
|
|
8510
8849
|
if (flags.watch !== 'false') {
|
|
8511
8850
|
const topic = session.websocketTopic || start.websocketTopic || (session.user && String(session.user.uuid || '').replace(/-/g, ''));
|
|
@@ -8526,10 +8865,16 @@ async function testCommand(flags) {
|
|
|
8526
8865
|
if (jsonOutput) {
|
|
8527
8866
|
console.log(JSON.stringify(status, null, 2));
|
|
8528
8867
|
} else {
|
|
8529
|
-
console.log('Final status:', JSON.stringify(status, null, 2));
|
|
8530
8868
|
printStagingLaneOwnerNotice(status.staging || {});
|
|
8869
|
+
testRunHeadlineLines(status, { taskId: start.taskId, dataMode, requestedDataMode: dataMode })
|
|
8870
|
+
.forEach((line) => console.log(line));
|
|
8531
8871
|
testRunSummaryLines(status).forEach((line) => console.log(line));
|
|
8872
|
+
testFailureGroupLines(status, { testRef }).forEach((line) => console.log(line));
|
|
8532
8873
|
printTestRunPivots(status);
|
|
8874
|
+
// The whole run object used to be printed here as pretty JSON. One 92-case suite measured 220 KB,
|
|
8875
|
+
// most of it identifiers repeated once per case, spent on the turn with the least room left. It is
|
|
8876
|
+
// still one flag away, and the durable record keeps it after the live status expires.
|
|
8877
|
+
console.log('Full run payload: remits-cli test status --task-id ' + start.taskId + ' --json');
|
|
8533
8878
|
}
|
|
8534
8879
|
|
|
8535
8880
|
// A selector that matched no case is a mis-specified run, not a passing one. Say so in the terminal
|
|
@@ -8554,17 +8899,24 @@ async function testCommand(flags) {
|
|
|
8554
8899
|
process.exitCode = 1;
|
|
8555
8900
|
}
|
|
8556
8901
|
|
|
8557
|
-
const
|
|
8558
|
-
|
|
8559
|
-
|
|
8560
|
-
|
|
8561
|
-
|
|
8562
|
-
|
|
8563
|
-
|
|
8564
|
-
|
|
8565
|
-
|
|
8566
|
-
|
|
8902
|
+
const evidence = await appendTerminalTestRunEvidence({
|
|
8903
|
+
api,
|
|
8904
|
+
cwd,
|
|
8905
|
+
session,
|
|
8906
|
+
accountId,
|
|
8907
|
+
flags: verification.evidenceFlags,
|
|
8908
|
+
status,
|
|
8909
|
+
testRef,
|
|
8910
|
+
selectedNames: names,
|
|
8911
|
+
dataMode,
|
|
8912
|
+
dataModeSource,
|
|
8913
|
+
branchName,
|
|
8914
|
+
workspace,
|
|
8915
|
+
baseUrl,
|
|
8916
|
+
activeEnvelopeId,
|
|
8917
|
+
quiet: jsonOutput
|
|
8567
8918
|
});
|
|
8919
|
+
const nondeterminism = evidence.nondeterminism;
|
|
8568
8920
|
if (nondeterminism && !jsonOutput) {
|
|
8569
8921
|
console.log('Nondeterministic signal:', nondeterminism.message);
|
|
8570
8922
|
nondeterminism.flips.slice(0, 6).forEach((flip) => {
|
|
@@ -8572,40 +8924,6 @@ async function testCommand(flags) {
|
|
|
8572
8924
|
' (previous packet ' + (flip.previousPacketId || 'unknown') + ')');
|
|
8573
8925
|
});
|
|
8574
8926
|
}
|
|
8575
|
-
|
|
8576
|
-
await appendVerificationPacket(api, cwd, session, accountId, verification.evidenceFlags, {
|
|
8577
|
-
type: 'test_run',
|
|
8578
|
-
success: status.status === 'completed' && !(status.result && status.result.failed > 0) && !unmatched.length,
|
|
8579
|
-
claim: 'Test run ' + String(testRef),
|
|
8580
|
-
world: buildCommandWorld(status, { accountId, dataMode: status.dataMode || dataMode, dataModeSource, branchName, workspace, host: normalizeBaseUrl(baseUrl), sourceLayer: status.staging && status.staging.testComponentSource }),
|
|
8581
|
-
revision: Object.assign(sourceRevision, {
|
|
8582
|
-
compileSignatures: testCompileSignatures(status),
|
|
8583
|
-
componentProvenance
|
|
8584
|
-
}),
|
|
8585
|
-
nondeterministic: nondeterminism ? true : undefined,
|
|
8586
|
-
nondeterminism: nondeterminism || undefined,
|
|
8587
|
-
test: {
|
|
8588
|
-
taskId: start.taskId,
|
|
8589
|
-
testId: status.test && status.test.id,
|
|
8590
|
-
testName: status.test && status.test.name,
|
|
8591
|
-
selectedCases: names,
|
|
8592
|
-
passed: status.result && status.result.passed,
|
|
8593
|
-
failed: status.result && status.result.failed,
|
|
8594
|
-
tests: status.result && status.result.tests,
|
|
8595
|
-
dataModeSource: status.dataModeSource || dataModeSource,
|
|
8596
|
-
unmatchedTestNames: unmatched
|
|
8597
|
-
},
|
|
8598
|
-
evidenceCategories: testEvidenceCategories(status, names),
|
|
8599
|
-
// What the run spent and whether that live AI was declared (withAiBudget), so an envelope can tell a
|
|
8600
|
-
// measurement run from a suite with missing mocks without re-reading every case.
|
|
8601
|
-
dependencies: testRunDependencies(status),
|
|
8602
|
-
summary: status.result && status.result.summary,
|
|
8603
|
-
limitations: []
|
|
8604
|
-
.concat(unmatched.length ? ['One or more requested test case selectors matched no case.'] : [])
|
|
8605
|
-
.concat(nondeterminism ? [nondeterminism.message] : []),
|
|
8606
|
-
rawRefs: { testStatusKey: start.taskId, durableTestRun: status.result && status.result.durableRecord ? start.taskId : undefined },
|
|
8607
|
-
status
|
|
8608
|
-
}, { quiet: jsonOutput });
|
|
8609
8927
|
}
|
|
8610
8928
|
|
|
8611
8929
|
async function testStatusCommand(flags) {
|
|
@@ -8615,9 +8933,11 @@ async function testStatusCommand(flags) {
|
|
|
8615
8933
|
const { session, accountId } = sessionContext;
|
|
8616
8934
|
const baseUrl = flags['base-url'] || session.baseUrl || DEFAULT_BASE_URL;
|
|
8617
8935
|
const branchName = flags.branch || currentBranch(cwd);
|
|
8936
|
+
const workspace = resolveWorkspace(cwd, flags);
|
|
8618
8937
|
const dataMode = hasExplicitDataModeFlag(flags)
|
|
8619
8938
|
? resolveDataMode(flags, null)
|
|
8620
8939
|
: DEFAULT_DATA_MODE;
|
|
8940
|
+
const dataModeSource = dataModeFlagSource(flags);
|
|
8621
8941
|
const taskId = flags['task-id'] || flags.taskId || flags.id || (flags._ && flags._[2]);
|
|
8622
8942
|
const jsonOutput = flagEnabled(flags.json);
|
|
8623
8943
|
if (!taskId) throw new Error('Missing --task-id <taskId>');
|
|
@@ -8631,31 +8951,94 @@ async function testStatusCommand(flags) {
|
|
|
8631
8951
|
taskId,
|
|
8632
8952
|
dataMode
|
|
8633
8953
|
}).then((r) => r.data);
|
|
8954
|
+
if (!status.taskId) status.taskId = taskId;
|
|
8955
|
+
|
|
8956
|
+
let statusVerification = { envelopeId: null, evidenceFlags: flags };
|
|
8957
|
+
if (testStatusIsTerminal(status)) {
|
|
8958
|
+
statusVerification = await verificationPreflight(api, cwd, session, accountId, flags, {
|
|
8959
|
+
packetType: 'test_run',
|
|
8960
|
+
world: buildCommandWorld(status, {
|
|
8961
|
+
accountId,
|
|
8962
|
+
dataMode: status.dataMode || dataMode,
|
|
8963
|
+
dataModeSource,
|
|
8964
|
+
branchName,
|
|
8965
|
+
workspace,
|
|
8966
|
+
host: normalizeBaseUrl(baseUrl),
|
|
8967
|
+
sourceLayer: status.staging && status.staging.testComponentSource
|
|
8968
|
+
}),
|
|
8969
|
+
test: status.test && status.test.id !== undefined && status.test.id !== null
|
|
8970
|
+
? { testId: status.test.id, testName: status.test.name }
|
|
8971
|
+
: { testName: status.test && status.test.name }
|
|
8972
|
+
}, { command: 'test status', neverRefuse: true, quiet: jsonOutput });
|
|
8973
|
+
}
|
|
8634
8974
|
|
|
8635
8975
|
if (jsonOutput) {
|
|
8976
|
+
if (testStatusIsTerminal(status)) {
|
|
8977
|
+
await appendTerminalTestRunEvidence({
|
|
8978
|
+
api,
|
|
8979
|
+
cwd,
|
|
8980
|
+
session,
|
|
8981
|
+
accountId,
|
|
8982
|
+
flags: statusVerification.evidenceFlags,
|
|
8983
|
+
status,
|
|
8984
|
+
testRef: status.test && (status.test.name || status.test.id) || taskId,
|
|
8985
|
+
selectedNames: status.result && Array.isArray(status.result.tests) ? status.result.tests.map((t) => t && t.name).filter(Boolean) : [],
|
|
8986
|
+
dataMode,
|
|
8987
|
+
dataModeSource,
|
|
8988
|
+
branchName,
|
|
8989
|
+
workspace,
|
|
8990
|
+
baseUrl,
|
|
8991
|
+
activeEnvelopeId: statusVerification.envelopeId,
|
|
8992
|
+
quiet: true
|
|
8993
|
+
});
|
|
8994
|
+
}
|
|
8636
8995
|
console.log(JSON.stringify(status, null, 2));
|
|
8637
8996
|
return status;
|
|
8638
8997
|
}
|
|
8639
8998
|
printSessionResolutionWarning(sessionContext);
|
|
8640
8999
|
printResolvedBaseUrl(baseUrl);
|
|
8641
|
-
console.log('Test run status:', status.status || 'unknown');
|
|
8642
|
-
console.log('Task ID:', taskId);
|
|
8643
9000
|
// Say where this answer came from. A durable record is a finished snapshot rebuilt from the database after the
|
|
8644
9001
|
// live status was gone; without this line a reconstructed run is indistinguishable from one still being watched.
|
|
8645
|
-
|
|
8646
|
-
|
|
8647
|
-
:
|
|
8648
|
-
|
|
8649
|
-
|
|
8650
|
-
|
|
8651
|
-
|
|
8652
|
-
|
|
8653
|
-
|
|
8654
|
-
}
|
|
9002
|
+
testRunHeadlineLines(status, {
|
|
9003
|
+
taskId,
|
|
9004
|
+
source: status.durable
|
|
9005
|
+
? 'durable run record (the live status has expired; this run is final)'
|
|
9006
|
+
: 'live run status',
|
|
9007
|
+
dataMode,
|
|
9008
|
+
// A durable run reports the lane it RAN in. When that differs from the lane this command asked for,
|
|
9009
|
+
// say so: auditing prod activity and being handed a test-lane run is correct and reads as if it were not.
|
|
9010
|
+
requestedDataMode: hasExplicitDataModeFlag(flags) ? dataMode : null
|
|
9011
|
+
}).forEach((line) => console.log(line));
|
|
8655
9012
|
if (status.world) runWorldLines(status.world, { host: normalizeBaseUrl(baseUrl) }).forEach((line) => console.log(line));
|
|
8656
9013
|
testRunSummaryLines(status).forEach((line) => console.log(line));
|
|
9014
|
+
testFailureGroupLines(status).forEach((line) => console.log(line));
|
|
8657
9015
|
printTestRunPivots(status);
|
|
8658
|
-
if (
|
|
9016
|
+
if (testStatusIsTerminal(status)) {
|
|
9017
|
+
const evidence = await appendTerminalTestRunEvidence({
|
|
9018
|
+
api,
|
|
9019
|
+
cwd,
|
|
9020
|
+
session,
|
|
9021
|
+
accountId,
|
|
9022
|
+
flags: statusVerification.evidenceFlags,
|
|
9023
|
+
status,
|
|
9024
|
+
testRef: status.test && (status.test.name || status.test.id) || taskId,
|
|
9025
|
+
selectedNames: status.result && Array.isArray(status.result.tests) ? status.result.tests.map((t) => t && t.name).filter(Boolean) : [],
|
|
9026
|
+
dataMode,
|
|
9027
|
+
dataModeSource,
|
|
9028
|
+
branchName,
|
|
9029
|
+
workspace,
|
|
9030
|
+
baseUrl,
|
|
9031
|
+
activeEnvelopeId: statusVerification.envelopeId
|
|
9032
|
+
});
|
|
9033
|
+
if (evidence.nondeterminism) {
|
|
9034
|
+
console.log('Nondeterministic signal:', evidence.nondeterminism.message);
|
|
9035
|
+
evidence.nondeterminism.flips.slice(0, 6).forEach((flip) => {
|
|
9036
|
+
console.log(' - ' + flip.caseName + ': ' + (flip.previousPassed ? 'passed' : 'failed') + ' -> ' + (flip.currentPassed ? 'passed' : 'failed') +
|
|
9037
|
+
' (previous packet ' + (flip.previousPacketId || 'unknown') + ')');
|
|
9038
|
+
});
|
|
9039
|
+
}
|
|
9040
|
+
}
|
|
9041
|
+
if (status.status === 'failed' || status.status === 'interrupted' || (status.result && status.result.failed > 0)) {
|
|
8659
9042
|
process.exitCode = 1;
|
|
8660
9043
|
}
|
|
8661
9044
|
return status;
|
|
@@ -8671,10 +9054,13 @@ async function testRunsCommand(flags) {
|
|
|
8671
9054
|
const baseUrl = flags['base-url'] || session.baseUrl || DEFAULT_BASE_URL;
|
|
8672
9055
|
const api = buildAxios(baseUrl, session.token);
|
|
8673
9056
|
const testRef = flags.test || flags['test-id'] || flags.name;
|
|
9057
|
+
const allDataLanes = flagEnabled(flags['all-lanes']) || flagEnabled(flags['all-data-lanes']);
|
|
8674
9058
|
const data = await loggedPost(api, cwd, '/cli/testRuns', {
|
|
8675
9059
|
token: session.token,
|
|
8676
9060
|
accountId,
|
|
8677
9061
|
asAccountId: flags['as-account'] || flags['as-account-id'],
|
|
9062
|
+
dataMode: resolveDataMode(flags, session),
|
|
9063
|
+
allDataLanes: allDataLanes || undefined,
|
|
8678
9064
|
testId: testRef && /^\d+$/.test(String(testRef)) ? Number(testRef) : undefined,
|
|
8679
9065
|
testName: testRef && !/^\d+$/.test(String(testRef)) ? String(testRef) : undefined,
|
|
8680
9066
|
max: flags.limit || flags.max || 20
|
|
@@ -8682,9 +9068,17 @@ async function testRunsCommand(flags) {
|
|
|
8682
9068
|
if (!data.success) throw new Error(data.message || 'Could not list test runs');
|
|
8683
9069
|
|
|
8684
9070
|
if (flagEnabled(flags.compare)) {
|
|
8685
|
-
const
|
|
8686
|
-
if (
|
|
8687
|
-
|
|
9071
|
+
const pair = selectComparableRuns(data.runs || []);
|
|
9072
|
+
if (!pair) {
|
|
9073
|
+
throw new Error('--compare needs two COMPARABLE recorded runs' + (testRef ? ' of ' + testRef : '') +
|
|
9074
|
+
' — same data lane, branch, workspace and case count. Found ' + (data.runs || []).length +
|
|
9075
|
+
' run(s); pick two explicitly with: remits-cli test compare --base <taskId> --head <taskId>');
|
|
9076
|
+
}
|
|
9077
|
+
if (pair.skipped) {
|
|
9078
|
+
console.error('Comparing the latest two comparable runs (' + describeRunShape(pair.head) + '); ' +
|
|
9079
|
+
pair.skipped + ' newer run(s) differ in lane, branch, workspace or case count and were skipped.');
|
|
9080
|
+
}
|
|
9081
|
+
return testCompareCommand(Object.assign({}, flags, { base: pair.base.taskId, head: pair.head.taskId }));
|
|
8688
9082
|
}
|
|
8689
9083
|
if (flagEnabled(flags.json)) {
|
|
8690
9084
|
console.log(JSON.stringify(data, null, 2));
|
|
@@ -8693,6 +9087,9 @@ async function testRunsCommand(flags) {
|
|
|
8693
9087
|
printSessionResolutionWarning(sessionContext);
|
|
8694
9088
|
printResolvedBaseUrl(baseUrl);
|
|
8695
9089
|
console.log('Recorded test runs' + (testRef ? ' for ' + testRef : '') + ' (account ' + data.accountId + '), newest first:');
|
|
9090
|
+
// Which lane this list is, said once. Pass counts from two lanes are two different facts.
|
|
9091
|
+
console.log('Data lane: ' + (data.dataLaneScope || 'all-data-lanes') +
|
|
9092
|
+
(data.otherLaneCount ? ' (' + data.otherLaneCount + ' more run(s) in the other lane; --all-lanes to include)' : ''));
|
|
8696
9093
|
(data.runs || []).forEach((run) => {
|
|
8697
9094
|
console.log('- ' + run.taskId + ' ' + (run.status || 'unknown') + ' ' + (run.passed || 0) + '/' + (run.total || 0) + ' passed' +
|
|
8698
9095
|
' ' + corpusAiLabel(run) +
|
|
@@ -8705,6 +9102,40 @@ async function testRunsCommand(flags) {
|
|
|
8705
9102
|
return data;
|
|
8706
9103
|
}
|
|
8707
9104
|
|
|
9105
|
+
/** The world fields that decide whether two runs are the same experiment repeated. */
|
|
9106
|
+
function runShapeKey(run = {}) {
|
|
9107
|
+
return [run.dataMode || '?', run.branchName || '?', run.workspace || 'shared',
|
|
9108
|
+
run.variantBranch || 'none', run.total == null ? '?' : run.total].join('|');
|
|
9109
|
+
}
|
|
9110
|
+
|
|
9111
|
+
function describeRunShape(run = {}) {
|
|
9112
|
+
return [(run.dataMode || '?') + ' lane', run.branchName || '?',
|
|
9113
|
+
run.workspace ? 'ws:' + run.workspace : 'shared lane',
|
|
9114
|
+
(run.total == null ? '?' : run.total) + ' case(s)'].join(' · ');
|
|
9115
|
+
}
|
|
9116
|
+
|
|
9117
|
+
/**
|
|
9118
|
+
* The newest two runs that are actually comparable, and how many newer ones were passed over.
|
|
9119
|
+
*
|
|
9120
|
+
* `--compare` used to take runs[0] and runs[1] unconditionally. On a real history that pairs a 92-case
|
|
9121
|
+
* suite with a 1-case `--names` probe, or a prod-lane run with a test-lane one, and reports the difference
|
|
9122
|
+
* as regressions. A comparison across worlds is not a comparison.
|
|
9123
|
+
*/
|
|
9124
|
+
function selectComparableRuns(runs) {
|
|
9125
|
+
const list = Array.isArray(runs) ? runs.filter(Boolean) : [];
|
|
9126
|
+
for (let head = 0; head < list.length; head += 1) {
|
|
9127
|
+
const key = runShapeKey(list[head]);
|
|
9128
|
+
for (let base = head + 1; base < list.length; base += 1) {
|
|
9129
|
+
if (runShapeKey(list[base]) === key) {
|
|
9130
|
+
// Every run passed over, including the ones BETWEEN head and base — a 1-case probe sitting between
|
|
9131
|
+
// two full suites is exactly the run whose absence from the comparison needs saying.
|
|
9132
|
+
return { head: list[head], base: list[base], skipped: base - 1 };
|
|
9133
|
+
}
|
|
9134
|
+
}
|
|
9135
|
+
}
|
|
9136
|
+
return null;
|
|
9137
|
+
}
|
|
9138
|
+
|
|
8708
9139
|
// `test compare --base <taskId> --head <taskId>`: per-case outcome, cost and duration deltas between two runs,
|
|
8709
9140
|
// read from the durable records.
|
|
8710
9141
|
async function testCompareCommand(flags) {
|
|
@@ -8735,6 +9166,18 @@ async function testCompareCommand(flags) {
|
|
|
8735
9166
|
.filter((key) => comparison.baseWorld && comparison.headWorld && String(comparison.baseWorld[key] || '') !== String(comparison.headWorld[key] || ''))
|
|
8736
9167
|
.map((key) => key + ' ' + short(String(comparison.baseWorld[key] || 'none')) + ' -> ' + short(String(comparison.headWorld[key] || 'none')));
|
|
8737
9168
|
if (worldDiff.length) console.log('World changed: ' + worldDiff.join('; '));
|
|
9169
|
+
// A lane or case-count difference is not a delta to interpret, it is two different experiments. Say so
|
|
9170
|
+
// rather than letting "regressed (33)" stand for "these runs never measured the same thing".
|
|
9171
|
+
const baseLane = comparison.baseWorld && comparison.baseWorld.dataMode;
|
|
9172
|
+
const headLane = comparison.headWorld && comparison.headWorld.dataMode;
|
|
9173
|
+
if (baseLane && headLane && baseLane !== headLane) {
|
|
9174
|
+
console.log('NOT COMPARABLE: these runs are in different DATA LANES (' + baseLane + ' vs ' + headLane +
|
|
9175
|
+
'). A record\'s dataMode is the lane it was written in; the pass counts below are two different facts.');
|
|
9176
|
+
}
|
|
9177
|
+
if (comparison.base.total && comparison.head.total && comparison.base.total !== comparison.head.total) {
|
|
9178
|
+
console.log('NOT COMPARABLE like-for-like: ' + comparison.base.total + ' case(s) vs ' + comparison.head.total +
|
|
9179
|
+
'. One of these is a narrowed --names run; "absent" below means the case did not run, not that it broke.');
|
|
9180
|
+
}
|
|
8738
9181
|
if (comparison.improved.length) console.log('Improved (' + comparison.improved.length + '): ' + comparison.improved.join(', '));
|
|
8739
9182
|
if (comparison.regressed.length) console.log('Regressed (' + comparison.regressed.length + '): ' + comparison.regressed.join(', '));
|
|
8740
9183
|
comparison.changed.filter((c) => c.direction === 'changed').forEach((c) => console.log('Changed: ' + c.name + ' ' + c.base + ' -> ' + c.head));
|
|
@@ -10896,8 +11339,10 @@ function buildRepoSnapshot(entry) {
|
|
|
10896
11339
|
? path.join(localPaths.legacy.sessionsDir, legacySessionName + '.jsonl')
|
|
10897
11340
|
: null;
|
|
10898
11341
|
const repoFiles = [
|
|
11342
|
+
{ label: 'account-boot.json', path: path.join(directory, 'account-boot.json'), mode: 'json' },
|
|
10899
11343
|
{ label: 'account-info.json', path: entry.accountInfoPath || path.join(directory, 'account-info.json'), mode: 'json' },
|
|
10900
11344
|
{ label: 'account-hierarchy.json', path: path.join(directory, 'account-hierarchy.json'), mode: 'json' },
|
|
11345
|
+
{ label: 'account-analytics.json', path: path.join(directory, 'account-analytics.json'), mode: 'json' },
|
|
10901
11346
|
{ label: '.remits-cli/active-actor/current-session.txt', path: localPaths.currentSessionFile, mode: 'text' },
|
|
10902
11347
|
{ label: '.remits-cli/shared/tools/tools.json', path: path.join(localPaths.toolsDir, 'tools.json'), mode: 'json' },
|
|
10903
11348
|
{ label: '.remits-cli/active-actor/session log (tail)', path: currentSessionLog, mode: 'tail' },
|
|
@@ -15143,7 +15588,12 @@ function printActivityInspectSummary(response) {
|
|
|
15143
15588
|
if (env.noRequiredEvidence) {
|
|
15144
15589
|
console.log(' evidence log — no verdict requested; add a claim only if someone needs one: remits-cli verify claim <id> --text "..." --envelope ' + env.envelopeId);
|
|
15145
15590
|
}
|
|
15146
|
-
|
|
15591
|
+
// A failure recorded in an envelope that promised nothing is HISTORY, not an outstanding task — the
|
|
15592
|
+
// same rule the smells already apply (`promises` in activitySmells). Printing it as "current failures"
|
|
15593
|
+
// re-created, one line lower, exactly the unfinishable-looking work the evidence/verdict split removed.
|
|
15594
|
+
if (env.currentFailureCount) {
|
|
15595
|
+
console.log(' ' + (env.noRequiredEvidence ? 'failed evidence recorded: ' : 'current failures: ') + env.currentFailureCount);
|
|
15596
|
+
}
|
|
15147
15597
|
});
|
|
15148
15598
|
}
|
|
15149
15599
|
|
|
@@ -15968,10 +16418,18 @@ function releaseAutoUpdateLock() {
|
|
|
15968
16418
|
} catch (_) {}
|
|
15969
16419
|
}
|
|
15970
16420
|
|
|
16421
|
+
/**
|
|
16422
|
+
* May this invocation check for, and install, a newer CLI?
|
|
16423
|
+
*
|
|
16424
|
+
* `--json` used to disable it outright, to keep update chatter out of a document a program is parsing.
|
|
16425
|
+
* The effect was that the callers who pass `--json` on every command - which is every AI agent - never
|
|
16426
|
+
* upgraded at all, so each release of better diagnostics reached the population that needed it least.
|
|
16427
|
+
* The chatter now goes to stderr (see `autoUpdateIfNeeded`), which is where a program's stdout contract
|
|
16428
|
+
* says it belongs, and the check stays on.
|
|
16429
|
+
*/
|
|
15971
16430
|
function shouldAutoUpdate(command, flags) {
|
|
15972
16431
|
if (!command || command === 'help' || command === '--help') return false;
|
|
15973
16432
|
if (flags && flags['no-auto-update']) return false;
|
|
15974
|
-
if (flags && flagEnabled(flags.json)) return false;
|
|
15975
16433
|
if (process.env.REMITS_CLI_AUTO_UPDATE && AUTO_UPDATE_DISABLED_VALUES.has(String(process.env.REMITS_CLI_AUTO_UPDATE).toLowerCase())) {
|
|
15976
16434
|
return false;
|
|
15977
16435
|
}
|
|
@@ -15979,10 +16437,31 @@ function shouldAutoUpdate(command, flags) {
|
|
|
15979
16437
|
return true;
|
|
15980
16438
|
}
|
|
15981
16439
|
|
|
16440
|
+
/** True when the last registry check is recent enough that another one would only cost latency. */
|
|
16441
|
+
function autoUpdateCheckedRecently() {
|
|
16442
|
+
try {
|
|
16443
|
+
const stamp = JSON.parse(fs.readFileSync(AUTO_UPDATE_STAMP_FILE, 'utf8'));
|
|
16444
|
+
return Number(stamp.checkedAtMs) > Date.now() - AUTO_UPDATE_CHECK_INTERVAL_MS;
|
|
16445
|
+
} catch (_) {
|
|
16446
|
+
return false;
|
|
16447
|
+
}
|
|
16448
|
+
}
|
|
16449
|
+
|
|
16450
|
+
function recordAutoUpdateCheck(latest) {
|
|
16451
|
+
try {
|
|
16452
|
+
ensureSessionDir();
|
|
16453
|
+
fs.writeFileSync(AUTO_UPDATE_STAMP_FILE,
|
|
16454
|
+
JSON.stringify({ checkedAtMs: Date.now(), latest: latest || null }, null, 2));
|
|
16455
|
+
} catch (_) {}
|
|
16456
|
+
}
|
|
16457
|
+
|
|
15982
16458
|
function autoUpdateIfNeeded(originalArgv, options = {}) {
|
|
15983
16459
|
const requireSuccess = Boolean(options.requireSuccess);
|
|
16460
|
+
// Update chatter is diagnostics, never part of a `--json` document. stderr keeps both promises at once.
|
|
16461
|
+
const note = (line) => console.error(line);
|
|
15984
16462
|
let lockAcquired = false;
|
|
15985
16463
|
try {
|
|
16464
|
+
if (!requireSuccess && autoUpdateCheckedRecently()) return false;
|
|
15986
16465
|
lockAcquired = acquireAutoUpdateLock();
|
|
15987
16466
|
if (!lockAcquired) {
|
|
15988
16467
|
if (requireSuccess) {
|
|
@@ -15996,20 +16475,22 @@ function autoUpdateIfNeeded(originalArgv, options = {}) {
|
|
|
15996
16475
|
encoding: 'utf8',
|
|
15997
16476
|
stdio: ['ignore', 'pipe', 'ignore']
|
|
15998
16477
|
}).trim();
|
|
16478
|
+
recordAutoUpdateCheck(latest);
|
|
15999
16479
|
if (latest && compareSemver(latest, currentVersion) > 0) {
|
|
16000
|
-
|
|
16480
|
+
note('[update] New version available: ' + currentVersion + ' -> ' + latest + '. Installing...');
|
|
16481
|
+
// npm's own progress output goes to stderr too: a `--json` caller's stdout must stay one document.
|
|
16001
16482
|
const install = spawnSync(npmCommand(), ['install', '-g', REMITS_CLI_PACKAGE_NAME + '@latest'], {
|
|
16002
|
-
stdio: '
|
|
16483
|
+
stdio: ['ignore', process.stderr, process.stderr],
|
|
16003
16484
|
env: process.env
|
|
16004
16485
|
});
|
|
16005
16486
|
if (install.error || install.status !== 0) {
|
|
16006
16487
|
if (requireSuccess) {
|
|
16007
16488
|
throw new Error('Auto-update failed; remits-cli start requires the latest published version before continuing.');
|
|
16008
16489
|
}
|
|
16009
|
-
|
|
16490
|
+
note('[update] Update failed; continuing with version ' + currentVersion + '.');
|
|
16010
16491
|
return false;
|
|
16011
16492
|
}
|
|
16012
|
-
|
|
16493
|
+
note('[update] Updated to ' + latest + '. Re-running command...');
|
|
16013
16494
|
const rerun = spawnSync(remitsCliCommand(), originalArgv, {
|
|
16014
16495
|
stdio: 'inherit',
|
|
16015
16496
|
env: Object.assign({}, process.env, { REMITS_CLI_AUTO_UPDATED: '1' })
|
|
@@ -16018,7 +16499,7 @@ function autoUpdateIfNeeded(originalArgv, options = {}) {
|
|
|
16018
16499
|
if (requireSuccess) {
|
|
16019
16500
|
throw new Error('Auto-update succeeded but re-running the updated remits-cli command failed.');
|
|
16020
16501
|
}
|
|
16021
|
-
|
|
16502
|
+
note('[update] Re-run failed; continuing with version ' + currentVersion + '.');
|
|
16022
16503
|
return false;
|
|
16023
16504
|
}
|
|
16024
16505
|
process.exitCode = rerun.status === null ? 1 : rerun.status;
|
|
@@ -16177,13 +16658,14 @@ function printComponentsHelp(subcommand) {
|
|
|
16177
16658
|
console.log(' remits-cli components branch <name> [--json] # overridden/added/removed + drift');
|
|
16178
16659
|
console.log(' remits-cli components branch <name> --diff <componentId> --component-type <kind> [--json]');
|
|
16179
16660
|
console.log(' remits-cli components branch <name> --subscribers [--json]');
|
|
16661
|
+
console.log(' remits-cli components branch <name> --copy-to <newBranch> [--dry-run] [--force] [--json]');
|
|
16180
16662
|
console.log(' remits-cli components branch <name> --subscribe <accountId> [--parent-account <id>] [--domain <host>] [--dry-run] [--confirm-primary-edge]');
|
|
16181
16663
|
console.log(' remits-cli components branch <name> --unsubscribe <accountId>');
|
|
16182
16664
|
console.log(' remits-cli components branch <name> --retire [--force]');
|
|
16183
16665
|
}
|
|
16184
16666
|
|
|
16185
16667
|
function printTestHelp() {
|
|
16186
|
-
console.log('Usage: remits-cli test run --test <id|name> [--base-url URL] [--branch stagingScope] [--names "a|b"] [--watch true|false] [--data-mode test|prod] [--as-account ID] [--variant-branch NAME|none] [--json]');
|
|
16668
|
+
console.log('Usage: remits-cli test run --test <id|name> [--base-url URL] [--branch stagingScope] [--names "a|b"] [--watch true|false] [--wait true|false] [--data-mode test|prod] [--as-account ID] [--variant-branch NAME|none] [--json]');
|
|
16187
16669
|
console.log(' remits-cli test status --task-id <taskId> [--base-url URL] [--account-id ID] [--branch stagingScope] [--data-mode test|prod] [--json]');
|
|
16188
16670
|
console.log(' remits-cli test runs [--test <id|name>] [--limit 20] [--compare] [--json] # durable run history');
|
|
16189
16671
|
console.log(' remits-cli test compare --base <taskId> --head <taskId> [--json] # per-case outcome/cost deltas');
|
|
@@ -16202,6 +16684,8 @@ function printTestHelp() {
|
|
|
16202
16684
|
console.log(' to force production/subscription semantics from a variant checkout.');
|
|
16203
16685
|
console.log(' --branch changes only the CLI staging namespace for test execution. Pair an unused');
|
|
16204
16686
|
console.log(' value with --variant-branch none when existing staged entries would shadow DB rows.');
|
|
16687
|
+
console.log(' --watch false disables websocket progress streaming; the CLI still waits for final status.');
|
|
16688
|
+
console.log(' --wait false returns after launch with a task id; test status attaches terminal evidence.');
|
|
16205
16689
|
console.log(' --json prints only the final status JSON to stdout; banners and progress go to stderr.');
|
|
16206
16690
|
console.log(' --data-mode prod intentionally targets live production data.');
|
|
16207
16691
|
console.log(' Every finished run is recorded durably: test status keeps working after the live status');
|
|
@@ -16473,10 +16957,11 @@ async function main() {
|
|
|
16473
16957
|
console.log(' remits-cli components promotion [<branch>] [--json] [--no-fail] # promotion readiness + ordered next steps');
|
|
16474
16958
|
console.log(' remits-cli components branches [--json] # committed branch variants for this account');
|
|
16475
16959
|
console.log(' remits-cli components branch <name> [--diff <componentId> --component-type <kind>] [--subscribers] [--json]');
|
|
16960
|
+
console.log(' remits-cli components branch <name> --copy-to <newBranch> [--dry-run] [--force] [--json]');
|
|
16476
16961
|
console.log(' remits-cli components branch <name> --subscribe <accountId> [--dry-run] [--confirm-primary-edge]');
|
|
16477
16962
|
console.log(' remits-cli components branch <name> --unsubscribe <accountId> # return that account to trunk');
|
|
16478
16963
|
console.log(' remits-cli components branch <name> --retire [--force] # delete the branch\'s overlays');
|
|
16479
|
-
console.log(' remits-cli test run --test <id|name> [--base-url URL] [--branch stagingScope] [--names "a|b"] [--watch true|false] [--data-mode test|prod] [--as-account ID] [--variant-branch NAME|none] [--json]');
|
|
16964
|
+
console.log(' remits-cli test run --test <id|name> [--base-url URL] [--branch stagingScope] [--names "a|b"] [--watch true|false] [--wait true|false] [--data-mode test|prod] [--as-account ID] [--variant-branch NAME|none] [--json]');
|
|
16480
16965
|
console.log(' remits-cli test runs [--test <id|name>] [--compare] # durable run history; compare latest two');
|
|
16481
16966
|
console.log(' remits-cli test compare --base <taskId> --head <taskId>');
|
|
16482
16967
|
console.log(' remits-cli corpus import --manifest corpus-manifest.json # seed a test corpus: cases + artifacts');
|