bmad-method-test-architecture-enterprise 1.21.0 → 1.21.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -12,7 +12,7 @@
12
12
  "name": "bmad-method-test-architecture-enterprise",
13
13
  "source": "./",
14
14
  "description": "Master Test Architect module for quality strategy, test automation, CI/CD quality gates, and structured testing education. Part of the BMad Method ecosystem.",
15
- "version": "1.21.0",
15
+ "version": "1.21.2",
16
16
  "author": {
17
17
  "name": "Murat K Ozcan (TEA Creator) & Brian (BMad) Madison"
18
18
  },
@@ -109,7 +109,7 @@ function modelFromArgs(flags, extra = []) {
109
109
  }
110
110
 
111
111
  /**
112
- * Model argv for an adapter, suppressed when the --claude-arg passthrough
112
+ * Model argv for an adapter, suppressed when the --agent-arg passthrough
113
113
  * already sets the model itself.
114
114
  *
115
115
  * The suppression is not politeness, it is required for codex: clap rejects a
@@ -171,7 +171,7 @@ const AGENT_ADAPTERS = {
171
171
  // on a one-word prompt, measured 2026-08-03), but it is codex-only, so
172
172
  // pinning it in this vendor-agnostic table would give the flag a meaning
173
173
  // no other adapter can honor. Set it per run with
174
- // --claude-arg -c --claude-arg model_reasoning_effort=low.
174
+ // --agent-arg -c --agent-arg model_reasoning_effort=low.
175
175
  buildArgv: (extra = [], model) => [
176
176
  'exec',
177
177
  '--skip-git-repo-check',
@@ -4,9 +4,10 @@
4
4
  * Strict, section-aware schema (every element is mandatory):
5
5
  * - YAML frontmatter declaring workflowType: testarch-test-review and a
6
6
  * non-empty stepsCompleted list.
7
- * - A "**Recommendation**:" line in BOTH the "## Executive Summary" and the
8
- * "## Decision" section; the two must agree (case-insensitively) and the
9
- * value must be one of the legal enum, to which it is normalized.
7
+ * - A "Recommendation:" line, with or without Markdown bolding, in BOTH the
8
+ * "## Executive Summary" and the "## Decision" section; the two must agree
9
+ * (case-insensitively) and the value must be one of the legal enum, to which
10
+ * it is normalized.
10
11
  * - "**Quality Score**: N/100" with N an integer in 0-100.
11
12
  * - A "**Total Violations**:" line with all four severity counts.
12
13
  * - A "## Quality Score Breakdown" ledger whose arithmetic reproduces the
@@ -41,7 +42,7 @@
41
42
  const path = require('node:path');
42
43
 
43
44
  const RECOMMENDATION_ENUM = ['Approve', 'Approve with Comments', 'Request Changes', 'Block'];
44
- const RECOMMENDATION_LINE = /\*\*Recommendation:?\*\*:?[ \t]*([^\n]+)/;
45
+ const RECOMMENDATION_LINE = /^[ \t]*(?:\*\*Recommendation\*\*:|\*\*Recommendation:\*\*|Recommendation:)[ \t]*([^\r\n]+?)[ \t]*$/m;
45
46
  const CONTEXT_BASIS_ENUM = ['none', 'pr_diff', 'pr_diff_truncated'];
46
47
  const CONTEXT_BASIS_LINE_SOURCE = String.raw`^[ \t]*\*\*Context Basis:?\*\*:?[ \t]*([^\r\n]+)[ \t]*$`;
47
48
  const CONTEXT_WAIVERS_LINE_SOURCE = String.raw`^[ \t]*\*\*Context Waivers Applied:?\*\*:?[ \t]*([^\r\n]+)[ \t]*$`;
@@ -148,7 +149,7 @@ function recommendationFromSection(text, heading) {
148
149
  }
149
150
  const match = section.match(RECOMMENDATION_LINE);
150
151
  if (!match) {
151
- unparseable(`Report is missing the "**Recommendation**:" line in the "## ${heading}" section`);
152
+ unparseable(`Report is missing the "Recommendation:" line in the "## ${heading}" section`);
152
153
  }
153
154
  return normalizeRecommendation(match[1], `"## ${heading}"`);
154
155
  }
@@ -514,7 +515,7 @@ function parseReport(reportText, runContract = {}) {
514
515
  const executive = recommendationFromSection(text, 'Executive Summary');
515
516
  const decision = recommendationFromSection(text, 'Decision');
516
517
  if (executive !== decision) {
517
- unparseable(`Report has conflicting "**Recommendation**:" lines (${executive} in Executive Summary vs ${decision} in Decision)`);
518
+ unparseable(`Report has conflicting "Recommendation:" lines (${executive} in Executive Summary vs ${decision} in Decision)`);
518
519
  }
519
520
 
520
521
  const scoreMatch = text.match(SCORE_PATTERN);
@@ -57,13 +57,13 @@ function buildMinimalEnv(envPass = [], sourceEnv = process.env, adapterEnvNames
57
57
  * @param {string} [options.agent] - Adapter key from agent-adapters.js (default claude).
58
58
  * @param {string} [options.agentCommand] - Executable override (--agent-cmd); replaces only the
59
59
  * adapter's default command, not its argv/env.
60
- * @param {string[]} [options.agentArgs] - Extra args appended after the adapter's own argv (--claude-arg passthrough).
60
+ * @param {string[]} [options.agentArgs] - Extra args appended after the adapter's own argv (--agent-arg passthrough).
61
61
  * @param {string} [options.model] - Model to pin for this run; defaults to the adapter's defaultModel.
62
62
  * @param {number} [options.timeout] - Wall-clock timeout in ms (default 1800000); SIGTERM on expiry.
63
63
  * @param {string} [options.cwd] - Working directory for the agent.
64
64
  * @param {string[]} [options.envPass] - Extra env var names allowed through to the child.
65
65
  * @param {string[]} [options.spawnPrefix] - Isolation wrapper (e.g. sandbox-exec -f profile).
66
- * @returns {string} Agent stdout.
66
+ * @returns {{ stdout: string, stderr: string }} Captured output from a successful agent run.
67
67
  * @throws {Error} AGENT_NOT_FOUND when the executable is missing, AGENT_FAILED
68
68
  * on spawn error, timeout, or non-zero exit.
69
69
  */
@@ -128,7 +128,7 @@ function runAgent(
128
128
  throw error;
129
129
  }
130
130
 
131
- return result.stdout;
131
+ return { stdout: result.stdout || '', stderr: result.stderr || '' };
132
132
  }
133
133
 
134
134
  module.exports = { runAgent, buildMinimalEnv };
@@ -59,6 +59,8 @@ const SCOPES = new Set(['single', 'directory', 'suite']);
59
59
  const FAIL_ON_LEVELS = new Set(['request-changes', 'block']);
60
60
  const DEFAULT_TIMEOUT_MS = 1_800_000; // 30 minutes
61
61
  const ENV_PASS_NAME = /^[A-Za-z_][A-Za-z0-9_]*$/;
62
+ const AGENT_OUTPUT_TAIL_LINES = 20;
63
+ const AGENT_OUTPUT_TAIL_CHARS = 8000;
62
64
 
63
65
  function fail(exitCode, message) {
64
66
  console.error(`tea-test-review: ${message}`);
@@ -69,6 +71,52 @@ function collect(value, previous) {
69
71
  return [...previous, value];
70
72
  }
71
73
 
74
+ /**
75
+ * Keep the pre-multi-vendor passthrough spelling working while exposing the
76
+ * generic name as the only documented interface. Normalizing argv before
77
+ * Commander parses it preserves the exact order when old and new spellings
78
+ * are mixed, which matters for paired vendor arguments such as `-c value`.
79
+ *
80
+ * @param {string[]} argv - Process argv.
81
+ * @returns {{ argv: string[], usedDeprecatedAlias: boolean }}
82
+ */
83
+ function normalizeAgentArgAliases(argv) {
84
+ let usedDeprecatedAlias = false;
85
+ const normalized = argv.map((arg) => {
86
+ if (arg === '--claude-arg') {
87
+ usedDeprecatedAlias = true;
88
+ return '--agent-arg';
89
+ }
90
+ if (arg.startsWith('--claude-arg=')) {
91
+ usedDeprecatedAlias = true;
92
+ return `--agent-arg=${arg.slice('--claude-arg='.length)}`;
93
+ }
94
+ return arg;
95
+ });
96
+ return { argv: normalized, usedDeprecatedAlias };
97
+ }
98
+
99
+ function boundedAgentOutputTail(value) {
100
+ const byLines = String(value || '')
101
+ .trim()
102
+ .split(/\r?\n/)
103
+ .slice(-AGENT_OUTPUT_TAIL_LINES)
104
+ .join('\n');
105
+ return byLines.length > AGENT_OUTPUT_TAIL_CHARS ? byLines.slice(-AGENT_OUTPUT_TAIL_CHARS) : byLines;
106
+ }
107
+
108
+ function printMissingReportDiagnostics(agentResult) {
109
+ for (const [label, value] of [
110
+ ['stdout', agentResult && agentResult.stdout],
111
+ ['stderr', agentResult && agentResult.stderr],
112
+ ]) {
113
+ const tail = boundedAgentOutputTail(value);
114
+ if (tail) {
115
+ console.error(`Agent ${label} before missing report (bounded tail):\n${tail}`);
116
+ }
117
+ }
118
+ }
119
+
72
120
  /**
73
121
  * Validate a --waive-until value: it must be a real calendar date in YYYY-MM-DD
74
122
  * form, strictly after the local today (day granularity, local timezone).
@@ -150,7 +198,7 @@ function main() {
150
198
  .join(', ')})`,
151
199
  )
152
200
  .option('--agent-cmd <path>', 'override the agent executable (advanced; used by tests with a stub agent)')
153
- .option('--claude-arg <arg>', 'extra argument passed through to the claude CLI (repeatable)', collect, [])
201
+ .option('--agent-arg <arg>', 'extra argument appended to the selected agent CLI argv (repeatable)', collect, [])
154
202
  .option(
155
203
  '--env-pass <NAME>',
156
204
  'environment variable name allowed through to the agent beyond the minimal default set (repeatable)',
@@ -183,8 +231,12 @@ function main() {
183
231
  .option('--pact-mcp <mode>', `force tea_pact_mcp, overriding _bmad/tea/config.yaml (${PACT_MCP_VALUES.join('|')}; default: none)`);
184
232
 
185
233
  program.exitOverride();
234
+ const normalizedArgv = normalizeAgentArgAliases(process.argv);
235
+ if (normalizedArgv.usedDeprecatedAlias) {
236
+ console.error('tea-test-review: --claude-arg is deprecated; use --agent-arg.');
237
+ }
186
238
  try {
187
- program.parse(process.argv);
239
+ program.parse(normalizedArgv.argv);
188
240
  } catch (error) {
189
241
  // --help/--version print their output and throw with exitCode 0.
190
242
  if (error.exitCode === 0) {
@@ -215,7 +267,7 @@ function main() {
215
267
  let resolvedModel = null;
216
268
  if (options.agent !== 'none') {
217
269
  try {
218
- resolvedModel = resolveModel(options.agent, options.model, options.claudeArg);
270
+ resolvedModel = resolveModel(options.agent, options.model, options.agentArg);
219
271
  } catch (error) {
220
272
  if (error.code === 'MODEL_ARG_INVALID' || error.code === 'MODEL_ARG_CONFLICT') {
221
273
  fail(EXIT.ENV_ERROR, error.message);
@@ -552,11 +604,12 @@ function main() {
552
604
  });
553
605
 
554
606
  const stopHeartbeat = startHeartbeat();
607
+ let agentResult;
555
608
  try {
556
- runAgent(prompt, {
609
+ agentResult = runAgent(prompt, {
557
610
  agent: options.agent,
558
611
  agentCommand: options.agentCmd,
559
- agentArgs: options.claudeArg,
612
+ agentArgs: options.agentArg,
560
613
  model: options.model,
561
614
  timeout: timeoutMs,
562
615
  cwd: agentCwd,
@@ -568,6 +621,7 @@ function main() {
568
621
  }
569
622
 
570
623
  if (!fs.existsSync(agentOutputPath) || fs.statSync(agentOutputPath).mtimeMs <= runStart) {
624
+ printMissingReportDiagnostics(agentResult);
571
625
  // Throw rather than fail()/process.exit() here: this runs inside
572
626
  // withIsolation's callback, and exiting the process skips its finally
573
627
  // block (restoreModes(), the chmod-fallback permission restore) along
@@ -87,7 +87,7 @@ The job needs `contents: read` and `pull-requests: write`, and forks receive no
87
87
  | `--agent <agent>` | `claude` | `claude` or `codex` spawn the matching CLI via its adapter (`cli/lib/agent-adapters.js`); `none` prints the prompt only. |
88
88
  | `--model <model>` | `claude`: `sonnet`, `codex`: `gpt-5.6-sol` | Model the agent runs on. Overrides whatever the vendor CLI would pick from its own config. Rejected with `--agent none`. |
89
89
  | `--agent-cmd <path>` | the selected adapter's command | Override the agent executable; the selected `--agent` adapter's argv/env still apply (advanced). |
90
- | `--claude-arg <arg>` | - | Extra argument appended to the selected agent's argv (repeatable; name predates multi-vendor support). |
90
+ | `--agent-arg <arg>` | - | Extra argument appended to the selected agent's argv (repeatable). |
91
91
  | `--env-pass <NAME>` | - | Env var allowed through beyond the default set (repeatable). |
92
92
  | `--timeout-ms <n>` | `1800000` (30 min) | Agent wall-clock timeout (SIGTERM on expiry). |
93
93
  | `--min-score <n>` | - | Fail when the quality score is below `n` (0-100). |
@@ -132,13 +132,15 @@ Left unstated, the model is whatever the vendor CLI resolves for itself: `~/.cod
132
132
  | `claude` | `sonnet` | `--model <model>` |
133
133
  | `codex` | `gpt-5.6-sol` | `--model <model>` |
134
134
 
135
- The resolved model travels in the verdict JSON as `model`, alongside `agent`, so a stored verdict says what produced it. A model supplied through `--claude-arg` becomes the resolved value too. Combining `--model` with a passthrough model, or declaring multiple passthrough models, is rejected before spawn. Two scores are only comparable when those two fields match.
135
+ The resolved model travels in the verdict JSON as `model`, alongside `agent`, so a stored verdict says what produced it. A model supplied through `--agent-arg` becomes the resolved value too. Combining `--model` with a passthrough model, or declaring multiple passthrough models, is rejected before spawn. Two scores are only comparable when those two fields match.
136
+
137
+ `--claude-arg` remains accepted as a deprecated alias for compatibility with workflows created before multi-vendor support. It emits a migration warning and preserves argument order. New workflows should use `--agent-arg`.
136
138
 
137
139
  The pinned values are aliases: they hold the tier steady, not the exact weights. Pass a fully-qualified slug to `--model` when a run has to be reproducible across model generations.
138
140
 
139
141
  `--model` is rejected with `--agent none`, which runs no agent. Honoring it there would be a lie, and quietly dropping it is the failure mode `--model` exists to remove.
140
142
 
141
- **Codex reasoning effort is a second unstated input, and it is not pinned here.** A local `model_reasoning_effort = "max"` costs about ten extra seconds even on a one-word prompt, and far more on a real review. It is codex-only, so it gets no vendor-agnostic flag; set it per run with `--claude-arg -c --claude-arg model_reasoning_effort=low`.
143
+ **Codex reasoning effort is a second unstated input, and it is not pinned here.** A local `model_reasoning_effort = "max"` costs about ten extra seconds even on a one-word prompt, and far more on a real review. It is codex-only, so it gets no vendor-agnostic flag; set it per run with `--agent-arg -c --agent-arg model_reasoning_effort=low`.
142
144
 
143
145
  ## TEA config resolution
144
146
 
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "$schema": "https://json.schemastore.org/package.json",
3
3
  "name": "bmad-method-test-architecture-enterprise",
4
- "version": "1.21.0",
4
+ "version": "1.21.2",
5
5
  "description": "Master Test Architect for quality strategy, test automation, and release gates",
6
6
  "keywords": [
7
7
  "bmad",
@@ -94,6 +94,8 @@ if (mode === 'fail') {
94
94
  }
95
95
 
96
96
  if (mode === 'nothing') {
97
+ console.log('stub-agent: exited successfully without writing the requested report');
98
+ console.error('stub-agent: success-path stderr before missing report');
97
99
  process.exit(0);
98
100
  }
99
101
 
@@ -592,6 +592,28 @@ async function runTests() {
592
592
  assert(false, 'colon-in-bold fixture parses', error.message);
593
593
  }
594
594
 
595
+ // Regression: a live Codex CI run produced the correct Decision value but
596
+ // omitted Markdown bolding from that one label. Styling cannot invalidate
597
+ // an otherwise complete verdict whose two Recommendation values agree.
598
+ try {
599
+ const source = readFixture('reports', 'request-changes.md');
600
+ const liveCodexShape = source.replace(
601
+ '## Decision\n\n**Recommendation**: Request Changes',
602
+ '## Decision\n\nRecommendation: Request Changes',
603
+ );
604
+ if (liveCodexShape === source) {
605
+ throw new Error('plain Recommendation regression setup did not modify the fixture');
606
+ }
607
+ const plainDecision = parseReport(liveCodexShape);
608
+ assert(
609
+ plainDecision.recommendation === 'Request Changes' && plainDecision.qualityScore === 63,
610
+ 'plain Decision Recommendation from live Codex output parses',
611
+ JSON.stringify(plainDecision),
612
+ );
613
+ } catch (error) {
614
+ assert(false, 'plain Decision Recommendation from live Codex output parses', error.message);
615
+ }
616
+
595
617
  try {
596
618
  const lowercase = parseReport(readFixture('reports', 'lowercase.md'));
597
619
  assert(lowercase.recommendation === 'Approve', 'lowercase fixture: "approve" normalizes to Approve', JSON.stringify(lowercase));
@@ -1352,7 +1374,7 @@ async function runTests() {
1352
1374
  const argv = adapter.buildArgv(['--extra-marker']);
1353
1375
  assert(
1354
1376
  Array.isArray(argv) && argv.includes('--extra-marker') && argv.at(-1) === '--extra-marker',
1355
- `${name} adapter buildArgv appends extra args (--claude-arg passthrough) last`,
1377
+ `${name} adapter buildArgv appends extra args (--agent-arg passthrough) last`,
1356
1378
  JSON.stringify(argv),
1357
1379
  );
1358
1380
  assert(typeof adapter.command === 'string' && adapter.command.length > 0, `${name} adapter declares a default command`);
@@ -1651,7 +1673,7 @@ async function runTests() {
1651
1673
  help.status === 0 &&
1652
1674
  [
1653
1675
  '--agent-cmd',
1654
- '--claude-arg',
1676
+ '--agent-arg',
1655
1677
  '--timeout-ms',
1656
1678
  '--test-glob',
1657
1679
  '--env-pass',
@@ -1815,17 +1837,17 @@ async function runTests() {
1815
1837
  assert(false, 'verdict JSON records the pinned default when --model is absent', error.message);
1816
1838
  }
1817
1839
 
1818
- // The --claude-arg escape hatch predates --model and still has to work: a
1819
- // second --model would be a clap usage error on codex.
1840
+ // The generic passthrough can name a model. A second --model would be a
1841
+ // clap usage error on codex.
1820
1842
  const passthroughRun = modelRun(
1821
1843
  'model-passthrough',
1822
- ['--claude-arg', '-m', '--claude-arg', 'passthrough-model'],
1844
+ ['--agent-arg', '-m', '--agent-arg', 'passthrough-model'],
1823
1845
  'passthrough-model',
1824
1846
  'codex',
1825
1847
  );
1826
1848
  assert(
1827
1849
  passthroughRun.status === 0 && passthroughRun.stderr.includes('model passthrough-model'),
1828
- 'a --claude-arg model passthrough becomes the resolved model and suppresses the pinned default',
1850
+ 'an --agent-arg model passthrough becomes the resolved model and suppresses the pinned default',
1829
1851
  `status=${passthroughRun.status} stderr=${passthroughRun.stderr}`,
1830
1852
  );
1831
1853
  try {
@@ -1839,9 +1861,23 @@ async function runTests() {
1839
1861
  assert(false, 'verdict JSON records the passthrough model that actually produced the score', error.message);
1840
1862
  }
1841
1863
 
1864
+ const legacyPassthroughRun = modelRun(
1865
+ 'model-passthrough-legacy-alias',
1866
+ ['--claude-arg', '-m', '--claude-arg', 'legacy-passthrough-model'],
1867
+ 'legacy-passthrough-model',
1868
+ 'codex',
1869
+ );
1870
+ assert(
1871
+ legacyPassthroughRun.status === 0 &&
1872
+ legacyPassthroughRun.stderr.includes('--claude-arg is deprecated; use --agent-arg') &&
1873
+ legacyPassthroughRun.stderr.includes('model legacy-passthrough-model'),
1874
+ 'the legacy --claude-arg alias preserves passthrough order and emits a migration warning',
1875
+ `status=${legacyPassthroughRun.status} stderr=${legacyPassthroughRun.stderr}`,
1876
+ );
1877
+
1842
1878
  for (const [agent, passthroughArg, expected] of [
1843
- ['claude', '--claude-arg=--model=claude-equals', 'claude-equals'],
1844
- ['codex', '--claude-arg=-m=codex-equals', 'codex-equals'],
1879
+ ['claude', '--agent-arg=--model=claude-equals', 'claude-equals'],
1880
+ ['codex', '--agent-arg=-m=codex-equals', 'codex-equals'],
1845
1881
  ]) {
1846
1882
  const equalsRun = modelRun(`model-passthrough-equals-${agent}`, [passthroughArg], expected, agent);
1847
1883
  assert(
@@ -1860,7 +1896,7 @@ async function runTests() {
1860
1896
  fixtureProject,
1861
1897
  '--model',
1862
1898
  'explicit-model',
1863
- '--claude-arg=--model=passthrough-model',
1899
+ '--agent-arg=--model=passthrough-model',
1864
1900
  ]);
1865
1901
  assert(
1866
1902
  modelConflict.status === 2 && modelConflict.stderr.includes('both --model'),
@@ -1875,8 +1911,8 @@ async function runTests() {
1875
1911
  'tests/checkout.spec.ts',
1876
1912
  '--project-root',
1877
1913
  fixtureProject,
1878
- '--claude-arg=-m=first-model',
1879
- '--claude-arg=--model=second-model',
1914
+ '--agent-arg=-m=first-model',
1915
+ '--agent-arg=--model=second-model',
1880
1916
  ]);
1881
1917
  assert(
1882
1918
  duplicatePassthroughModel.status === 2 && duplicatePassthroughModel.stderr.includes('declares the model 2 times'),
@@ -1891,7 +1927,7 @@ async function runTests() {
1891
1927
  'tests/checkout.spec.ts',
1892
1928
  '--project-root',
1893
1929
  fixtureProject,
1894
- '--claude-arg=--model',
1930
+ '--agent-arg=--model',
1895
1931
  ]);
1896
1932
  assert(
1897
1933
  missingPassthroughModel.status === 2 && missingPassthroughModel.stderr.includes('passthrough value'),
@@ -2135,6 +2171,14 @@ async function runTests() {
2135
2171
  'stub writing nothing exits 3 despite a stale pre-placed report',
2136
2172
  `status=${staleRun.status} stderr=${staleRun.stderr}`,
2137
2173
  );
2174
+ assert(
2175
+ staleRun.stderr.includes('Agent stdout before missing report (bounded tail):') &&
2176
+ staleRun.stderr.includes('exited successfully without writing the requested report') &&
2177
+ staleRun.stderr.includes('Agent stderr before missing report (bounded tail):') &&
2178
+ staleRun.stderr.includes('success-path stderr before missing report'),
2179
+ 'an exit-0 missing-report failure surfaces bounded stdout and stderr diagnostics',
2180
+ staleRun.stderr,
2181
+ );
2138
2182
  assert(!staleRun.stdout.includes('"recommendation": "Block"'), 'stale pre-placed report is never parsed', staleRun.stdout);
2139
2183
  assert(!fs.existsSync(staleOut), 'stale pre-placed report is deleted before the agent runs', staleOut);
2140
2184