bmad-method-test-architecture-enterprise 1.21.0 → 1.21.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/cli/lib/agent-adapters.js +2 -2
- package/cli/lib/parse-report.js +7 -6
- package/cli/lib/run-agent.js +3 -3
- package/cli/test-review.js +59 -5
- package/docs/reference/tea-test-review-cli.md +5 -3
- package/package.json +1 -1
- package/test/fixtures/test-review-cli/stub-agent.js +2 -0
- package/test/test-test-review-cli.js +56 -12
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
"name": "bmad-method-test-architecture-enterprise",
|
|
13
13
|
"source": "./",
|
|
14
14
|
"description": "Master Test Architect module for quality strategy, test automation, CI/CD quality gates, and structured testing education. Part of the BMad Method ecosystem.",
|
|
15
|
-
"version": "1.21.
|
|
15
|
+
"version": "1.21.2",
|
|
16
16
|
"author": {
|
|
17
17
|
"name": "Murat K Ozcan (TEA Creator) & Brian (BMad) Madison"
|
|
18
18
|
},
|
|
@@ -109,7 +109,7 @@ function modelFromArgs(flags, extra = []) {
|
|
|
109
109
|
}
|
|
110
110
|
|
|
111
111
|
/**
|
|
112
|
-
* Model argv for an adapter, suppressed when the --
|
|
112
|
+
* Model argv for an adapter, suppressed when the --agent-arg passthrough
|
|
113
113
|
* already sets the model itself.
|
|
114
114
|
*
|
|
115
115
|
* The suppression is not politeness, it is required for codex: clap rejects a
|
|
@@ -171,7 +171,7 @@ const AGENT_ADAPTERS = {
|
|
|
171
171
|
// on a one-word prompt, measured 2026-08-03), but it is codex-only, so
|
|
172
172
|
// pinning it in this vendor-agnostic table would give the flag a meaning
|
|
173
173
|
// no other adapter can honor. Set it per run with
|
|
174
|
-
// --
|
|
174
|
+
// --agent-arg -c --agent-arg model_reasoning_effort=low.
|
|
175
175
|
buildArgv: (extra = [], model) => [
|
|
176
176
|
'exec',
|
|
177
177
|
'--skip-git-repo-check',
|
package/cli/lib/parse-report.js
CHANGED
|
@@ -4,9 +4,10 @@
|
|
|
4
4
|
* Strict, section-aware schema (every element is mandatory):
|
|
5
5
|
* - YAML frontmatter declaring workflowType: testarch-test-review and a
|
|
6
6
|
* non-empty stepsCompleted list.
|
|
7
|
-
* - A "
|
|
8
|
-
* "## Decision" section; the two must agree
|
|
9
|
-
* value must be one of the legal enum, to which
|
|
7
|
+
* - A "Recommendation:" line, with or without Markdown bolding, in BOTH the
|
|
8
|
+
* "## Executive Summary" and the "## Decision" section; the two must agree
|
|
9
|
+
* (case-insensitively) and the value must be one of the legal enum, to which
|
|
10
|
+
* it is normalized.
|
|
10
11
|
* - "**Quality Score**: N/100" with N an integer in 0-100.
|
|
11
12
|
* - A "**Total Violations**:" line with all four severity counts.
|
|
12
13
|
* - A "## Quality Score Breakdown" ledger whose arithmetic reproduces the
|
|
@@ -41,7 +42,7 @@
|
|
|
41
42
|
const path = require('node:path');
|
|
42
43
|
|
|
43
44
|
const RECOMMENDATION_ENUM = ['Approve', 'Approve with Comments', 'Request Changes', 'Block'];
|
|
44
|
-
const RECOMMENDATION_LINE =
|
|
45
|
+
const RECOMMENDATION_LINE = /^[ \t]*(?:\*\*Recommendation\*\*:|\*\*Recommendation:\*\*|Recommendation:)[ \t]*([^\r\n]+?)[ \t]*$/m;
|
|
45
46
|
const CONTEXT_BASIS_ENUM = ['none', 'pr_diff', 'pr_diff_truncated'];
|
|
46
47
|
const CONTEXT_BASIS_LINE_SOURCE = String.raw`^[ \t]*\*\*Context Basis:?\*\*:?[ \t]*([^\r\n]+)[ \t]*$`;
|
|
47
48
|
const CONTEXT_WAIVERS_LINE_SOURCE = String.raw`^[ \t]*\*\*Context Waivers Applied:?\*\*:?[ \t]*([^\r\n]+)[ \t]*$`;
|
|
@@ -148,7 +149,7 @@ function recommendationFromSection(text, heading) {
|
|
|
148
149
|
}
|
|
149
150
|
const match = section.match(RECOMMENDATION_LINE);
|
|
150
151
|
if (!match) {
|
|
151
|
-
unparseable(`Report is missing the "
|
|
152
|
+
unparseable(`Report is missing the "Recommendation:" line in the "## ${heading}" section`);
|
|
152
153
|
}
|
|
153
154
|
return normalizeRecommendation(match[1], `"## ${heading}"`);
|
|
154
155
|
}
|
|
@@ -514,7 +515,7 @@ function parseReport(reportText, runContract = {}) {
|
|
|
514
515
|
const executive = recommendationFromSection(text, 'Executive Summary');
|
|
515
516
|
const decision = recommendationFromSection(text, 'Decision');
|
|
516
517
|
if (executive !== decision) {
|
|
517
|
-
unparseable(`Report has conflicting "
|
|
518
|
+
unparseable(`Report has conflicting "Recommendation:" lines (${executive} in Executive Summary vs ${decision} in Decision)`);
|
|
518
519
|
}
|
|
519
520
|
|
|
520
521
|
const scoreMatch = text.match(SCORE_PATTERN);
|
package/cli/lib/run-agent.js
CHANGED
|
@@ -57,13 +57,13 @@ function buildMinimalEnv(envPass = [], sourceEnv = process.env, adapterEnvNames
|
|
|
57
57
|
* @param {string} [options.agent] - Adapter key from agent-adapters.js (default claude).
|
|
58
58
|
* @param {string} [options.agentCommand] - Executable override (--agent-cmd); replaces only the
|
|
59
59
|
* adapter's default command, not its argv/env.
|
|
60
|
-
* @param {string[]} [options.agentArgs] - Extra args appended after the adapter's own argv (--
|
|
60
|
+
* @param {string[]} [options.agentArgs] - Extra args appended after the adapter's own argv (--agent-arg passthrough).
|
|
61
61
|
* @param {string} [options.model] - Model to pin for this run; defaults to the adapter's defaultModel.
|
|
62
62
|
* @param {number} [options.timeout] - Wall-clock timeout in ms (default 1800000); SIGTERM on expiry.
|
|
63
63
|
* @param {string} [options.cwd] - Working directory for the agent.
|
|
64
64
|
* @param {string[]} [options.envPass] - Extra env var names allowed through to the child.
|
|
65
65
|
* @param {string[]} [options.spawnPrefix] - Isolation wrapper (e.g. sandbox-exec -f profile).
|
|
66
|
-
* @returns {string}
|
|
66
|
+
* @returns {{ stdout: string, stderr: string }} Captured output from a successful agent run.
|
|
67
67
|
* @throws {Error} AGENT_NOT_FOUND when the executable is missing, AGENT_FAILED
|
|
68
68
|
* on spawn error, timeout, or non-zero exit.
|
|
69
69
|
*/
|
|
@@ -128,7 +128,7 @@ function runAgent(
|
|
|
128
128
|
throw error;
|
|
129
129
|
}
|
|
130
130
|
|
|
131
|
-
return result.stdout;
|
|
131
|
+
return { stdout: result.stdout || '', stderr: result.stderr || '' };
|
|
132
132
|
}
|
|
133
133
|
|
|
134
134
|
module.exports = { runAgent, buildMinimalEnv };
|
package/cli/test-review.js
CHANGED
|
@@ -59,6 +59,8 @@ const SCOPES = new Set(['single', 'directory', 'suite']);
|
|
|
59
59
|
const FAIL_ON_LEVELS = new Set(['request-changes', 'block']);
|
|
60
60
|
const DEFAULT_TIMEOUT_MS = 1_800_000; // 30 minutes
|
|
61
61
|
const ENV_PASS_NAME = /^[A-Za-z_][A-Za-z0-9_]*$/;
|
|
62
|
+
const AGENT_OUTPUT_TAIL_LINES = 20;
|
|
63
|
+
const AGENT_OUTPUT_TAIL_CHARS = 8000;
|
|
62
64
|
|
|
63
65
|
function fail(exitCode, message) {
|
|
64
66
|
console.error(`tea-test-review: ${message}`);
|
|
@@ -69,6 +71,52 @@ function collect(value, previous) {
|
|
|
69
71
|
return [...previous, value];
|
|
70
72
|
}
|
|
71
73
|
|
|
74
|
+
/**
|
|
75
|
+
* Keep the pre-multi-vendor passthrough spelling working while exposing the
|
|
76
|
+
* generic name as the only documented interface. Normalizing argv before
|
|
77
|
+
* Commander parses it preserves the exact order when old and new spellings
|
|
78
|
+
* are mixed, which matters for paired vendor arguments such as `-c value`.
|
|
79
|
+
*
|
|
80
|
+
* @param {string[]} argv - Process argv.
|
|
81
|
+
* @returns {{ argv: string[], usedDeprecatedAlias: boolean }}
|
|
82
|
+
*/
|
|
83
|
+
function normalizeAgentArgAliases(argv) {
|
|
84
|
+
let usedDeprecatedAlias = false;
|
|
85
|
+
const normalized = argv.map((arg) => {
|
|
86
|
+
if (arg === '--claude-arg') {
|
|
87
|
+
usedDeprecatedAlias = true;
|
|
88
|
+
return '--agent-arg';
|
|
89
|
+
}
|
|
90
|
+
if (arg.startsWith('--claude-arg=')) {
|
|
91
|
+
usedDeprecatedAlias = true;
|
|
92
|
+
return `--agent-arg=${arg.slice('--claude-arg='.length)}`;
|
|
93
|
+
}
|
|
94
|
+
return arg;
|
|
95
|
+
});
|
|
96
|
+
return { argv: normalized, usedDeprecatedAlias };
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
function boundedAgentOutputTail(value) {
|
|
100
|
+
const byLines = String(value || '')
|
|
101
|
+
.trim()
|
|
102
|
+
.split(/\r?\n/)
|
|
103
|
+
.slice(-AGENT_OUTPUT_TAIL_LINES)
|
|
104
|
+
.join('\n');
|
|
105
|
+
return byLines.length > AGENT_OUTPUT_TAIL_CHARS ? byLines.slice(-AGENT_OUTPUT_TAIL_CHARS) : byLines;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
function printMissingReportDiagnostics(agentResult) {
|
|
109
|
+
for (const [label, value] of [
|
|
110
|
+
['stdout', agentResult && agentResult.stdout],
|
|
111
|
+
['stderr', agentResult && agentResult.stderr],
|
|
112
|
+
]) {
|
|
113
|
+
const tail = boundedAgentOutputTail(value);
|
|
114
|
+
if (tail) {
|
|
115
|
+
console.error(`Agent ${label} before missing report (bounded tail):\n${tail}`);
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
|
|
72
120
|
/**
|
|
73
121
|
* Validate a --waive-until value: it must be a real calendar date in YYYY-MM-DD
|
|
74
122
|
* form, strictly after the local today (day granularity, local timezone).
|
|
@@ -150,7 +198,7 @@ function main() {
|
|
|
150
198
|
.join(', ')})`,
|
|
151
199
|
)
|
|
152
200
|
.option('--agent-cmd <path>', 'override the agent executable (advanced; used by tests with a stub agent)')
|
|
153
|
-
.option('--
|
|
201
|
+
.option('--agent-arg <arg>', 'extra argument appended to the selected agent CLI argv (repeatable)', collect, [])
|
|
154
202
|
.option(
|
|
155
203
|
'--env-pass <NAME>',
|
|
156
204
|
'environment variable name allowed through to the agent beyond the minimal default set (repeatable)',
|
|
@@ -183,8 +231,12 @@ function main() {
|
|
|
183
231
|
.option('--pact-mcp <mode>', `force tea_pact_mcp, overriding _bmad/tea/config.yaml (${PACT_MCP_VALUES.join('|')}; default: none)`);
|
|
184
232
|
|
|
185
233
|
program.exitOverride();
|
|
234
|
+
const normalizedArgv = normalizeAgentArgAliases(process.argv);
|
|
235
|
+
if (normalizedArgv.usedDeprecatedAlias) {
|
|
236
|
+
console.error('tea-test-review: --claude-arg is deprecated; use --agent-arg.');
|
|
237
|
+
}
|
|
186
238
|
try {
|
|
187
|
-
program.parse(
|
|
239
|
+
program.parse(normalizedArgv.argv);
|
|
188
240
|
} catch (error) {
|
|
189
241
|
// --help/--version print their output and throw with exitCode 0.
|
|
190
242
|
if (error.exitCode === 0) {
|
|
@@ -215,7 +267,7 @@ function main() {
|
|
|
215
267
|
let resolvedModel = null;
|
|
216
268
|
if (options.agent !== 'none') {
|
|
217
269
|
try {
|
|
218
|
-
resolvedModel = resolveModel(options.agent, options.model, options.
|
|
270
|
+
resolvedModel = resolveModel(options.agent, options.model, options.agentArg);
|
|
219
271
|
} catch (error) {
|
|
220
272
|
if (error.code === 'MODEL_ARG_INVALID' || error.code === 'MODEL_ARG_CONFLICT') {
|
|
221
273
|
fail(EXIT.ENV_ERROR, error.message);
|
|
@@ -552,11 +604,12 @@ function main() {
|
|
|
552
604
|
});
|
|
553
605
|
|
|
554
606
|
const stopHeartbeat = startHeartbeat();
|
|
607
|
+
let agentResult;
|
|
555
608
|
try {
|
|
556
|
-
runAgent(prompt, {
|
|
609
|
+
agentResult = runAgent(prompt, {
|
|
557
610
|
agent: options.agent,
|
|
558
611
|
agentCommand: options.agentCmd,
|
|
559
|
-
agentArgs: options.
|
|
612
|
+
agentArgs: options.agentArg,
|
|
560
613
|
model: options.model,
|
|
561
614
|
timeout: timeoutMs,
|
|
562
615
|
cwd: agentCwd,
|
|
@@ -568,6 +621,7 @@ function main() {
|
|
|
568
621
|
}
|
|
569
622
|
|
|
570
623
|
if (!fs.existsSync(agentOutputPath) || fs.statSync(agentOutputPath).mtimeMs <= runStart) {
|
|
624
|
+
printMissingReportDiagnostics(agentResult);
|
|
571
625
|
// Throw rather than fail()/process.exit() here: this runs inside
|
|
572
626
|
// withIsolation's callback, and exiting the process skips its finally
|
|
573
627
|
// block (restoreModes(), the chmod-fallback permission restore) along
|
|
@@ -87,7 +87,7 @@ The job needs `contents: read` and `pull-requests: write`, and forks receive no
|
|
|
87
87
|
| `--agent <agent>` | `claude` | `claude` or `codex` spawn the matching CLI via its adapter (`cli/lib/agent-adapters.js`); `none` prints the prompt only. |
|
|
88
88
|
| `--model <model>` | `claude`: `sonnet`, `codex`: `gpt-5.6-sol` | Model the agent runs on. Overrides whatever the vendor CLI would pick from its own config. Rejected with `--agent none`. |
|
|
89
89
|
| `--agent-cmd <path>` | the selected adapter's command | Override the agent executable; the selected `--agent` adapter's argv/env still apply (advanced). |
|
|
90
|
-
| `--
|
|
90
|
+
| `--agent-arg <arg>` | - | Extra argument appended to the selected agent's argv (repeatable). |
|
|
91
91
|
| `--env-pass <NAME>` | - | Env var allowed through beyond the default set (repeatable). |
|
|
92
92
|
| `--timeout-ms <n>` | `1800000` (30 min) | Agent wall-clock timeout (SIGTERM on expiry). |
|
|
93
93
|
| `--min-score <n>` | - | Fail when the quality score is below `n` (0-100). |
|
|
@@ -132,13 +132,15 @@ Left unstated, the model is whatever the vendor CLI resolves for itself: `~/.cod
|
|
|
132
132
|
| `claude` | `sonnet` | `--model <model>` |
|
|
133
133
|
| `codex` | `gpt-5.6-sol` | `--model <model>` |
|
|
134
134
|
|
|
135
|
-
The resolved model travels in the verdict JSON as `model`, alongside `agent`, so a stored verdict says what produced it. A model supplied through `--
|
|
135
|
+
The resolved model travels in the verdict JSON as `model`, alongside `agent`, so a stored verdict says what produced it. A model supplied through `--agent-arg` becomes the resolved value too. Combining `--model` with a passthrough model, or declaring multiple passthrough models, is rejected before spawn. Two scores are only comparable when those two fields match.
|
|
136
|
+
|
|
137
|
+
`--claude-arg` remains accepted as a deprecated alias for compatibility with workflows created before multi-vendor support. It emits a migration warning and preserves argument order. New workflows should use `--agent-arg`.
|
|
136
138
|
|
|
137
139
|
The pinned values are aliases: they hold the tier steady, not the exact weights. Pass a fully-qualified slug to `--model` when a run has to be reproducible across model generations.
|
|
138
140
|
|
|
139
141
|
`--model` is rejected with `--agent none`, which runs no agent. Honoring it there would be a lie, and quietly dropping it is the failure mode `--model` exists to remove.
|
|
140
142
|
|
|
141
|
-
**Codex reasoning effort is a second unstated input, and it is not pinned here.** A local `model_reasoning_effort = "max"` costs about ten extra seconds even on a one-word prompt, and far more on a real review. It is codex-only, so it gets no vendor-agnostic flag; set it per run with `--
|
|
143
|
+
**Codex reasoning effort is a second unstated input, and it is not pinned here.** A local `model_reasoning_effort = "max"` costs about ten extra seconds even on a one-word prompt, and far more on a real review. It is codex-only, so it gets no vendor-agnostic flag; set it per run with `--agent-arg -c --agent-arg model_reasoning_effort=low`.
|
|
142
144
|
|
|
143
145
|
## TEA config resolution
|
|
144
146
|
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"$schema": "https://json.schemastore.org/package.json",
|
|
3
3
|
"name": "bmad-method-test-architecture-enterprise",
|
|
4
|
-
"version": "1.21.
|
|
4
|
+
"version": "1.21.2",
|
|
5
5
|
"description": "Master Test Architect for quality strategy, test automation, and release gates",
|
|
6
6
|
"keywords": [
|
|
7
7
|
"bmad",
|
|
@@ -592,6 +592,28 @@ async function runTests() {
|
|
|
592
592
|
assert(false, 'colon-in-bold fixture parses', error.message);
|
|
593
593
|
}
|
|
594
594
|
|
|
595
|
+
// Regression: a live Codex CI run produced the correct Decision value but
|
|
596
|
+
// omitted Markdown bolding from that one label. Styling cannot invalidate
|
|
597
|
+
// an otherwise complete verdict whose two Recommendation values agree.
|
|
598
|
+
try {
|
|
599
|
+
const source = readFixture('reports', 'request-changes.md');
|
|
600
|
+
const liveCodexShape = source.replace(
|
|
601
|
+
'## Decision\n\n**Recommendation**: Request Changes',
|
|
602
|
+
'## Decision\n\nRecommendation: Request Changes',
|
|
603
|
+
);
|
|
604
|
+
if (liveCodexShape === source) {
|
|
605
|
+
throw new Error('plain Recommendation regression setup did not modify the fixture');
|
|
606
|
+
}
|
|
607
|
+
const plainDecision = parseReport(liveCodexShape);
|
|
608
|
+
assert(
|
|
609
|
+
plainDecision.recommendation === 'Request Changes' && plainDecision.qualityScore === 63,
|
|
610
|
+
'plain Decision Recommendation from live Codex output parses',
|
|
611
|
+
JSON.stringify(plainDecision),
|
|
612
|
+
);
|
|
613
|
+
} catch (error) {
|
|
614
|
+
assert(false, 'plain Decision Recommendation from live Codex output parses', error.message);
|
|
615
|
+
}
|
|
616
|
+
|
|
595
617
|
try {
|
|
596
618
|
const lowercase = parseReport(readFixture('reports', 'lowercase.md'));
|
|
597
619
|
assert(lowercase.recommendation === 'Approve', 'lowercase fixture: "approve" normalizes to Approve', JSON.stringify(lowercase));
|
|
@@ -1352,7 +1374,7 @@ async function runTests() {
|
|
|
1352
1374
|
const argv = adapter.buildArgv(['--extra-marker']);
|
|
1353
1375
|
assert(
|
|
1354
1376
|
Array.isArray(argv) && argv.includes('--extra-marker') && argv.at(-1) === '--extra-marker',
|
|
1355
|
-
`${name} adapter buildArgv appends extra args (--
|
|
1377
|
+
`${name} adapter buildArgv appends extra args (--agent-arg passthrough) last`,
|
|
1356
1378
|
JSON.stringify(argv),
|
|
1357
1379
|
);
|
|
1358
1380
|
assert(typeof adapter.command === 'string' && adapter.command.length > 0, `${name} adapter declares a default command`);
|
|
@@ -1651,7 +1673,7 @@ async function runTests() {
|
|
|
1651
1673
|
help.status === 0 &&
|
|
1652
1674
|
[
|
|
1653
1675
|
'--agent-cmd',
|
|
1654
|
-
'--
|
|
1676
|
+
'--agent-arg',
|
|
1655
1677
|
'--timeout-ms',
|
|
1656
1678
|
'--test-glob',
|
|
1657
1679
|
'--env-pass',
|
|
@@ -1815,17 +1837,17 @@ async function runTests() {
|
|
|
1815
1837
|
assert(false, 'verdict JSON records the pinned default when --model is absent', error.message);
|
|
1816
1838
|
}
|
|
1817
1839
|
|
|
1818
|
-
// The
|
|
1819
|
-
//
|
|
1840
|
+
// The generic passthrough can name a model. A second --model would be a
|
|
1841
|
+
// clap usage error on codex.
|
|
1820
1842
|
const passthroughRun = modelRun(
|
|
1821
1843
|
'model-passthrough',
|
|
1822
|
-
['--
|
|
1844
|
+
['--agent-arg', '-m', '--agent-arg', 'passthrough-model'],
|
|
1823
1845
|
'passthrough-model',
|
|
1824
1846
|
'codex',
|
|
1825
1847
|
);
|
|
1826
1848
|
assert(
|
|
1827
1849
|
passthroughRun.status === 0 && passthroughRun.stderr.includes('model passthrough-model'),
|
|
1828
|
-
'
|
|
1850
|
+
'an --agent-arg model passthrough becomes the resolved model and suppresses the pinned default',
|
|
1829
1851
|
`status=${passthroughRun.status} stderr=${passthroughRun.stderr}`,
|
|
1830
1852
|
);
|
|
1831
1853
|
try {
|
|
@@ -1839,9 +1861,23 @@ async function runTests() {
|
|
|
1839
1861
|
assert(false, 'verdict JSON records the passthrough model that actually produced the score', error.message);
|
|
1840
1862
|
}
|
|
1841
1863
|
|
|
1864
|
+
const legacyPassthroughRun = modelRun(
|
|
1865
|
+
'model-passthrough-legacy-alias',
|
|
1866
|
+
['--claude-arg', '-m', '--claude-arg', 'legacy-passthrough-model'],
|
|
1867
|
+
'legacy-passthrough-model',
|
|
1868
|
+
'codex',
|
|
1869
|
+
);
|
|
1870
|
+
assert(
|
|
1871
|
+
legacyPassthroughRun.status === 0 &&
|
|
1872
|
+
legacyPassthroughRun.stderr.includes('--claude-arg is deprecated; use --agent-arg') &&
|
|
1873
|
+
legacyPassthroughRun.stderr.includes('model legacy-passthrough-model'),
|
|
1874
|
+
'the legacy --claude-arg alias preserves passthrough order and emits a migration warning',
|
|
1875
|
+
`status=${legacyPassthroughRun.status} stderr=${legacyPassthroughRun.stderr}`,
|
|
1876
|
+
);
|
|
1877
|
+
|
|
1842
1878
|
for (const [agent, passthroughArg, expected] of [
|
|
1843
|
-
['claude', '--
|
|
1844
|
-
['codex', '--
|
|
1879
|
+
['claude', '--agent-arg=--model=claude-equals', 'claude-equals'],
|
|
1880
|
+
['codex', '--agent-arg=-m=codex-equals', 'codex-equals'],
|
|
1845
1881
|
]) {
|
|
1846
1882
|
const equalsRun = modelRun(`model-passthrough-equals-${agent}`, [passthroughArg], expected, agent);
|
|
1847
1883
|
assert(
|
|
@@ -1860,7 +1896,7 @@ async function runTests() {
|
|
|
1860
1896
|
fixtureProject,
|
|
1861
1897
|
'--model',
|
|
1862
1898
|
'explicit-model',
|
|
1863
|
-
'--
|
|
1899
|
+
'--agent-arg=--model=passthrough-model',
|
|
1864
1900
|
]);
|
|
1865
1901
|
assert(
|
|
1866
1902
|
modelConflict.status === 2 && modelConflict.stderr.includes('both --model'),
|
|
@@ -1875,8 +1911,8 @@ async function runTests() {
|
|
|
1875
1911
|
'tests/checkout.spec.ts',
|
|
1876
1912
|
'--project-root',
|
|
1877
1913
|
fixtureProject,
|
|
1878
|
-
'--
|
|
1879
|
-
'--
|
|
1914
|
+
'--agent-arg=-m=first-model',
|
|
1915
|
+
'--agent-arg=--model=second-model',
|
|
1880
1916
|
]);
|
|
1881
1917
|
assert(
|
|
1882
1918
|
duplicatePassthroughModel.status === 2 && duplicatePassthroughModel.stderr.includes('declares the model 2 times'),
|
|
@@ -1891,7 +1927,7 @@ async function runTests() {
|
|
|
1891
1927
|
'tests/checkout.spec.ts',
|
|
1892
1928
|
'--project-root',
|
|
1893
1929
|
fixtureProject,
|
|
1894
|
-
'--
|
|
1930
|
+
'--agent-arg=--model',
|
|
1895
1931
|
]);
|
|
1896
1932
|
assert(
|
|
1897
1933
|
missingPassthroughModel.status === 2 && missingPassthroughModel.stderr.includes('passthrough value'),
|
|
@@ -2135,6 +2171,14 @@ async function runTests() {
|
|
|
2135
2171
|
'stub writing nothing exits 3 despite a stale pre-placed report',
|
|
2136
2172
|
`status=${staleRun.status} stderr=${staleRun.stderr}`,
|
|
2137
2173
|
);
|
|
2174
|
+
assert(
|
|
2175
|
+
staleRun.stderr.includes('Agent stdout before missing report (bounded tail):') &&
|
|
2176
|
+
staleRun.stderr.includes('exited successfully without writing the requested report') &&
|
|
2177
|
+
staleRun.stderr.includes('Agent stderr before missing report (bounded tail):') &&
|
|
2178
|
+
staleRun.stderr.includes('success-path stderr before missing report'),
|
|
2179
|
+
'an exit-0 missing-report failure surfaces bounded stdout and stderr diagnostics',
|
|
2180
|
+
staleRun.stderr,
|
|
2181
|
+
);
|
|
2138
2182
|
assert(!staleRun.stdout.includes('"recommendation": "Block"'), 'stale pre-placed report is never parsed', staleRun.stdout);
|
|
2139
2183
|
assert(!fs.existsSync(staleOut), 'stale pre-placed report is deleted before the agent runs', staleOut);
|
|
2140
2184
|
|