bmad-method-test-architecture-enterprise 1.21.2 → 1.21.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -12,7 +12,7 @@
12
12
  "name": "bmad-method-test-architecture-enterprise",
13
13
  "source": "./",
14
14
  "description": "Master Test Architect module for quality strategy, test automation, CI/CD quality gates, and structured testing education. Part of the BMad Method ecosystem.",
15
- "version": "1.21.2",
15
+ "version": "1.21.3",
16
16
  "author": {
17
17
  "name": "Murat K Ozcan (TEA Creator) & Brian (BMad) Madison"
18
18
  },
@@ -136,11 +136,12 @@ function buildPrompt({
136
136
  '- **Recommendation** must be exactly one of: Approve | Approve with Comments | Request Changes | Block',
137
137
  '- A "## Decision" section is required, spelled exactly that, and its **Recommendation** must match the',
138
138
  " Executive Summary's. Do not rename the heading after the sentence that describes it.",
139
- '- **Quality Score**: N/100 is required and must be an integer from 0 to 100.',
139
+ '- **Quality Score**: N/100 is required and must be an integer from 0 to 100. The CLI treats this agent-written',
140
+ ' number as provisional and replaces it with the deterministic ledger result before gating.',
140
141
  '- The **Total Violations**: line is required, with Critical, High, Medium, and Low counts.',
141
- '- The "## Quality Score Breakdown" section is required and its ledger must reproduce the score. The CLI',
142
- ' recomputes 100 - (Critical×10 + High×5 + Medium×2 + Low×1) + Total Bonus and rejects any disagreement,',
143
- ' so the deduction ledger is the only scoring model: never a weighted average and never a judgment adjustment.',
142
+ '- The "## Quality Score Breakdown" section is required. The CLI computes the authoritative score as',
143
+ ' 100 - (Critical×10 + High×5 + Medium×2 + Low×1) + Total Bonus, then normalizes the report score and grade',
144
+ ' before gating. The deduction ledger is the only scoring model: never a weighted average or judgment adjustment.',
144
145
  '- Each of the six bonus categories is worth 0 or 5, so "Total Bonus" is a multiple of 5 from 0 to 30.',
145
146
  '- Grade is exactly one of A, B, C, D, F, with no modifier such as A+.',
146
147
  `- The Executive Summary must carry exactly one "**Context Basis**: ${contextBasis}" line, exactly that value.`,
@@ -10,9 +10,9 @@
10
10
  * it is normalized.
11
11
  * - "**Quality Score**: N/100" with N an integer in 0-100.
12
12
  * - A "**Total Violations**:" line with all four severity counts.
13
- * - A "## Quality Score Breakdown" ledger whose arithmetic reproduces the
14
- * published score; the skill's deduction model is the only scoring model, so
15
- * a score that contradicts its own breakdown is rejected rather than gated on.
13
+ * - A "## Quality Score Breakdown" ledger from which the CLI computes the
14
+ * authoritative score; the skill's deduction model is the only scoring
15
+ * model, so agent arithmetic never controls the gate.
16
16
  * - A "## Reviewed Files" section listing every reviewed file.
17
17
  * - Exactly one "**Context Basis**:" line inside the Executive Summary, plus a
18
18
  * "## Review Context" manifest whenever that basis is not `none`.
@@ -46,7 +46,7 @@ const RECOMMENDATION_LINE = /^[ \t]*(?:\*\*Recommendation\*\*:|\*\*Recommendatio
46
46
  const CONTEXT_BASIS_ENUM = ['none', 'pr_diff', 'pr_diff_truncated'];
47
47
  const CONTEXT_BASIS_LINE_SOURCE = String.raw`^[ \t]*\*\*Context Basis:?\*\*:?[ \t]*([^\r\n]+)[ \t]*$`;
48
48
  const CONTEXT_WAIVERS_LINE_SOURCE = String.raw`^[ \t]*\*\*Context Waivers Applied:?\*\*:?[ \t]*([^\r\n]+)[ \t]*$`;
49
- const SCORE_PATTERN = /\*\*Quality Score\*\*:\s*(\d+)\s*\/\s*100/;
49
+ const SCORE_PATTERN = /\*\*Quality Score\*\*:\s*(\d+)\s*\/\s*100(?:[ \t]*\([ \t]*([A-F])(?=[ \t)-]))?/;
50
50
  const VIOLATIONS_LINE = /\*\*Total Violations:?\*\*:?[ \t]*([^\n]+)/;
51
51
  const VIOLATION_LEVELS = ['Critical', 'High', 'Medium', 'Low'];
52
52
  // The template always prints the bonus with a leading "+" (every fixture in
@@ -444,20 +444,19 @@ function verifyRunContract({ reviewedFiles, contextBasis, contextFiles }, runCon
444
444
  }
445
445
 
446
446
  /**
447
- * Recompute the template's deduction ledger and reject a report whose
448
- * arithmetic disagrees with the score it published.
447
+ * Compute the authoritative quality score from the template's deduction
448
+ * ledger. The agent's published score is presentation data only.
449
449
  *
450
- * Two live runs over an identical four-file set returned 83/100 and 92/100, and
451
- * the lower one printed a breakdown that summed to 92, so a published score
452
- * cannot be trusted on its face. The ledger in `test-review-template.md` is the
453
- * workflow's only scoring model, which makes it recomputable here from the
454
- * violation counts the report already declares.
450
+ * Live runs have repeatedly published arithmetic that contradicts their own
451
+ * ledgers. The ledger in `test-review-template.md` is the workflow's only
452
+ * scoring model, so the CLI derives the score from the violation counts and
453
+ * bonus instead of asking a probabilistic producer to perform gate arithmetic.
455
454
  *
456
455
  * The breakdown sits inside a fenced block, which the verdict scan strips, so
457
456
  * this reads the raw report instead and anchors on the section heading: only
458
457
  * the ledger under "## Quality Score Breakdown" is ever consulted.
459
458
  */
460
- function verifyScoreLedger(rawText, qualityScore, violations) {
459
+ function deriveQualityScore(rawText, violations) {
461
460
  // extractSection's regex takes the first match; on raw (fence-intact) text
462
461
  // that is exploitable if the reviewed file's own quoted content contains a
463
462
  // second "## Quality Score Breakdown" heading earlier in the report than
@@ -488,13 +487,64 @@ function verifyScoreLedger(rawText, qualityScore, violations) {
488
487
  const key = level.toLowerCase();
489
488
  return sum + violations[key] * SEVERITY_DEDUCTIONS[key];
490
489
  }, 0);
491
- const expected = Math.max(0, Math.min(100, 100 - deductions + bonus));
492
- if (expected !== qualityScore) {
493
- unparseable(
494
- `Report Quality Score ${qualityScore} contradicts its own breakdown: ` +
495
- `100 - ${deductions} deductions + ${bonus} bonus = ${expected}`,
496
- );
497
- }
490
+ return Math.max(0, Math.min(100, 100 - deductions + bonus));
491
+ }
492
+
493
+ function gradeForScore(score) {
494
+ if (score >= 90) return 'A';
495
+ if (score >= 80) return 'B';
496
+ if (score >= 70) return 'C';
497
+ if (score >= 60) return 'D';
498
+ return 'F';
499
+ }
500
+
501
+ /** Normalize the report's schema-owned score and grade fields to CLI arithmetic. */
502
+ function normalizeReportScore(reportText, qualityScore) {
503
+ const grade = gradeForScore(qualityScore);
504
+ let inFence = false;
505
+ let section = null;
506
+ let summaryNormalized = false;
507
+ let finalScoreNormalized = false;
508
+ let finalGradeNormalized = false;
509
+
510
+ return reportText
511
+ .split('\n')
512
+ .map((originalLine) => {
513
+ let line = originalLine;
514
+ if (/^\s*```/.test(line)) {
515
+ inFence = !inFence;
516
+ return line;
517
+ }
518
+
519
+ if (!inFence) {
520
+ const heading = /^##[ \t]+([^\r\n]+?)[ \t]*\r?$/.exec(line);
521
+ if (heading) {
522
+ section = heading[1];
523
+ }
524
+ if (!summaryNormalized && /^[ \t]*\*\*Quality Score\*\*:/.test(line)) {
525
+ line = line
526
+ .replace(/^([ \t]*\*\*Quality Score\*\*:[ \t]*)\d+([ \t]*\/[ \t]*100)/, `$1${qualityScore}$2`)
527
+ .replace(/^([ \t]*\*\*Quality Score\*\*:[^\r\n]*\([ \t]*)[A-F](?=[ \t)-])/, `$1${grade}`);
528
+ summaryNormalized = true;
529
+ }
530
+ }
531
+
532
+ // The score ledger is itself a fenced block. Its fields remain active
533
+ // only inside the unique Quality Score Breakdown section validated by
534
+ // deriveQualityScore; unrelated fenced examples stay untouched.
535
+ if (section === 'Quality Score Breakdown') {
536
+ if (!finalScoreNormalized && /^[ \t]*Final Score[ \t]*:/.test(line)) {
537
+ line = line.replace(/^([ \t]*Final Score[ \t]*:[ \t]*)\d+([ \t]*\/[ \t]*100[ \t]*\r?)$/, `$1${qualityScore}$2`);
538
+ finalScoreNormalized = true;
539
+ }
540
+ if (!finalGradeNormalized && /^[ \t]*Grade[ \t]*:/.test(line)) {
541
+ line = line.replace(/^([ \t]*Grade[ \t]*:[ \t]*)[A-F]([ \t]*\r?)$/, `$1${grade}$2`);
542
+ finalGradeNormalized = true;
543
+ }
544
+ }
545
+ return line;
546
+ })
547
+ .join('\n');
498
548
  }
499
549
 
500
550
  /**
@@ -502,7 +552,7 @@ function verifyScoreLedger(rawText, qualityScore, violations) {
502
552
  *
503
553
  * @param {string} reportText - Full test-review.md contents.
504
554
  * @param {object} [runContract] - Exact reviewed/context evidence supplied by the runner.
505
- * @returns {{recommendation: string, qualityScore: number, violations: object,
555
+ * @returns {{recommendation: string, qualityScore: number, reportedQualityScore?: number, violations: object,
506
556
  * reviewedFiles: string[], contextBasis: string, contextFiles: string[],
507
557
  * contextWaiversApplied: number}}
508
558
  * @throws {Error} With code REPORT_UNPARSEABLE on any missing/invalid element.
@@ -522,9 +572,10 @@ function parseReport(reportText, runContract = {}) {
522
572
  if (!scoreMatch) {
523
573
  unparseable('Report is missing the "**Quality Score**: N/100" line');
524
574
  }
525
- const qualityScore = Number.parseInt(scoreMatch[1], 10);
526
- if (qualityScore < 0 || qualityScore > 100) {
527
- unparseable(`Report Quality Score ${qualityScore} is outside the required 0-100 range`);
575
+ const reportedQualityScore = Number.parseInt(scoreMatch[1], 10);
576
+ const reportedQualityGrade = scoreMatch[2];
577
+ if (reportedQualityScore < 0 || reportedQualityScore > 100) {
578
+ unparseable(`Report Quality Score ${reportedQualityScore} is outside the required 0-100 range`);
528
579
  }
529
580
 
530
581
  const violations = parseViolations(text);
@@ -535,7 +586,7 @@ function parseReport(reportText, runContract = {}) {
535
586
  );
536
587
  }
537
588
 
538
- verifyScoreLedger(reportText, qualityScore, violations);
589
+ const qualityScore = deriveQualityScore(reportText, violations);
539
590
 
540
591
  const reviewedFiles = parseReviewedFiles(text);
541
592
  const contextBasis = parseContextBasis(text);
@@ -560,7 +611,7 @@ function parseReport(reportText, runContract = {}) {
560
611
  const keyStrengths = extractBullets(extractSubsection(executiveSection, 'Key Strengths'), '✅');
561
612
  const keyWeaknesses = extractBullets(extractSubsection(executiveSection, 'Key Weaknesses'), '❌');
562
613
 
563
- return {
614
+ const parsed = {
564
615
  recommendation: executive,
565
616
  qualityScore,
566
617
  violations,
@@ -571,6 +622,13 @@ function parseReport(reportText, runContract = {}) {
571
622
  keyStrengths,
572
623
  keyWeaknesses,
573
624
  };
625
+ if (
626
+ reportedQualityScore !== qualityScore ||
627
+ (reportedQualityGrade !== undefined && reportedQualityGrade !== gradeForScore(qualityScore))
628
+ ) {
629
+ parsed.reportedQualityScore = reportedQualityScore;
630
+ }
631
+ return parsed;
574
632
  }
575
633
 
576
634
  /**
@@ -599,4 +657,4 @@ function scoreFails(score, minScore) {
599
657
  return score < minScore;
600
658
  }
601
659
 
602
- module.exports = { parseReport, verdictFor, scoreFails, CONTEXT_BASIS_ENUM };
660
+ module.exports = { parseReport, normalizeReportScore, verdictFor, scoreFails, CONTEXT_BASIS_ENUM };
@@ -15,8 +15,9 @@
15
15
  *
16
16
  * Exit codes: 0 pass/skip (or a verdict failure waived with --waive), 1 review
17
17
  * verdict fail (or a skip with --fail-on-skip / a deletions-only diff),
18
- * 2 environment/config error, 3 agent or parse failure. A waiver never applies
19
- * to exit 2 or 3: environment, agent, and parse failures are never waivable.
18
+ * 2 environment/config error, 3 agent, parse, or report-artifact failure. A
19
+ * waiver never applies to exit 2 or 3: infrastructure failures are never
20
+ * waivable.
20
21
  *
21
22
  * Usage:
22
23
  * tea-test-review --base origin/main --agent claude --json test-review.json
@@ -24,6 +25,7 @@
24
25
  */
25
26
 
26
27
  const fs = require('node:fs');
28
+ const { randomUUID } = require('node:crypto');
27
29
  const os = require('node:os');
28
30
  const path = require('node:path');
29
31
  const { spawn } = require('node:child_process');
@@ -41,7 +43,7 @@ const {
41
43
  registerExtraTestPattern,
42
44
  } = require('./lib/changed-tests');
43
45
  const { buildPrompt } = require('./lib/build-prompt');
44
- const { parseReport, verdictFor, scoreFails } = require('./lib/parse-report');
46
+ const { parseReport, normalizeReportScore, verdictFor, scoreFails } = require('./lib/parse-report');
45
47
  const { runAgent } = require('./lib/run-agent');
46
48
  const { AGENT_ADAPTERS, resolveModel } = require('./lib/agent-adapters');
47
49
  const { withIsolation, selectBackend } = require('./lib/isolate');
@@ -151,6 +153,45 @@ function writeJsonFile(jsonPath, payload) {
151
153
  }
152
154
  }
153
155
 
156
+ function reportArtifactFailure(action, artifactPath, cause) {
157
+ const error = new Error(`Failed to ${action} report artifact ${artifactPath}: ${cause.message}`);
158
+ error.code = 'REPORT_ARTIFACT';
159
+ return error;
160
+ }
161
+
162
+ function reportTemporaryPath(artifactPath) {
163
+ return path.join(path.dirname(artifactPath), `.${path.basename(artifactPath)}.${randomUUID()}.tmp`);
164
+ }
165
+
166
+ function replaceReportArtifact(artifactPath, temporaryPath, writeTemporary) {
167
+ let temporaryReady = false;
168
+ try {
169
+ fs.mkdirSync(path.dirname(artifactPath), { recursive: true });
170
+ writeTemporary(temporaryPath);
171
+ temporaryReady = true;
172
+ fs.renameSync(temporaryPath, artifactPath);
173
+ } catch (error) {
174
+ try {
175
+ fs.rmSync(temporaryPath, { force: true });
176
+ } catch {
177
+ // Preserve the artifact error that caused the failure.
178
+ }
179
+ throw reportArtifactFailure(temporaryReady ? 'replace' : 'prepare', artifactPath, error);
180
+ }
181
+ }
182
+
183
+ function copyReportArtifact(sourcePath, artifactPath, temporaryPath) {
184
+ replaceReportArtifact(artifactPath, temporaryPath, (temporary) => {
185
+ fs.copyFileSync(sourcePath, temporary);
186
+ });
187
+ }
188
+
189
+ function writeReportArtifact(artifactPath, temporaryPath, content) {
190
+ replaceReportArtifact(artifactPath, temporaryPath, (temporary) => {
191
+ fs.writeFileSync(temporary, content, 'utf8');
192
+ });
193
+ }
194
+
154
195
  function main() {
155
196
  const program = new Command();
156
197
 
@@ -554,7 +595,8 @@ function main() {
554
595
  }
555
596
  const runStart = Date.now();
556
597
 
557
- const writablePaths = [outputPath, ...(jsonPath ? [jsonPath] : [])];
598
+ const copiedReportTemporaryPath = reportTemporaryPath(outputPath);
599
+ const normalizedReportTemporaryPath = reportTemporaryPath(outputPath);
558
600
  let redirectDir = null;
559
601
  let gateFailures = [];
560
602
 
@@ -581,28 +623,25 @@ function main() {
581
623
  return () => heartbeat.kill();
582
624
  };
583
625
 
584
- const executeRun = ({ agentCwd, spawnPrefix }) => {
585
- // Under isolation the agent writes into a fresh tmpdir; the CLI copies the
586
- // artifacts back to the requested paths after a successful run.
587
- let agentOutputPath = outputPath;
588
- let agentJsonPath = jsonPath;
589
- if (isolationActive) {
590
- redirectDir = fs.mkdtempSync(path.join(os.tmpdir(), 'tea-test-review-'));
591
- agentOutputPath = path.join(redirectDir, 'test-review.md');
592
- agentJsonPath = jsonPath ? path.join(redirectDir, 'verdict.json') : null;
593
- }
594
-
595
- const prompt = buildPrompt({
596
- skillRoot,
597
- files: changedTestFiles,
598
- outputPath: agentOutputPath,
599
- scope: options.scope,
600
- testDir: options.testDir,
601
- teaConfig,
602
- contextFiles,
603
- contextBasis,
604
- });
626
+ // Under isolation the agent writes into a fresh tmpdir. Artifact processing
627
+ // happens after withIsolation restores the project, so an atomic rename never
628
+ // needs a project directory to be writable while the agent is running.
629
+ if (isolationActive) {
630
+ redirectDir = fs.mkdtempSync(path.join(os.tmpdir(), 'tea-test-review-'));
631
+ }
632
+ const agentOutputPath = redirectDir ? path.join(redirectDir, 'test-review.md') : outputPath;
633
+ const prompt = buildPrompt({
634
+ skillRoot,
635
+ files: changedTestFiles,
636
+ outputPath: agentOutputPath,
637
+ scope: options.scope,
638
+ testDir: options.testDir,
639
+ teaConfig,
640
+ contextFiles,
641
+ contextBasis,
642
+ });
605
643
 
644
+ const executeAgent = ({ agentCwd, spawnPrefix }) => {
606
645
  const stopHeartbeat = startHeartbeat();
607
646
  let agentResult;
608
647
  try {
@@ -622,41 +661,45 @@ function main() {
622
661
 
623
662
  if (!fs.existsSync(agentOutputPath) || fs.statSync(agentOutputPath).mtimeMs <= runStart) {
624
663
  printMissingReportDiagnostics(agentResult);
625
- // Throw rather than fail()/process.exit() here: this runs inside
626
- // withIsolation's callback, and exiting the process skips its finally
627
- // block (restoreModes(), the chmod-fallback permission restore) along
628
- // with the outer redirectDir cleanup below. Thrown errors propagate
629
- // through both finally blocks before the outer catch calls fail().
630
664
  const error = new Error(
631
665
  `Agent finished but no fresh report was written to ${agentOutputPath}; refusing to parse a stale or missing report.`,
632
666
  );
633
667
  error.code = 'REPORT_MISSING';
634
668
  throw error;
635
669
  }
670
+ };
636
671
 
672
+ const processReport = () => {
637
673
  // The report is copied back even when the verdict fails or parsing fails;
638
674
  // on agent failure nothing was produced and nothing is copied.
639
675
  if (redirectDir) {
640
- fs.copyFileSync(agentOutputPath, outputPath);
676
+ copyReportArtifact(agentOutputPath, outputPath, copiedReportTemporaryPath);
641
677
  }
642
678
 
679
+ const rawReport = fs.readFileSync(agentOutputPath, 'utf8');
643
680
  let parsed;
644
681
  try {
645
- parsed = parseReport(fs.readFileSync(agentOutputPath, 'utf8'), {
682
+ parsed = parseReport(rawReport, {
646
683
  reviewedFiles: changedTestFiles,
647
684
  contextFiles,
648
685
  contextBasis,
649
686
  });
650
687
  } catch (error) {
651
688
  if (error.code === 'REPORT_UNPARSEABLE') {
652
- // Same reasoning as the freshness check above: throw, don't exit, so
653
- // isolation cleanup runs before the process actually terminates.
654
689
  const wrapped = new Error(`${error.message} (report: ${outputPath})`);
655
690
  wrapped.code = 'REPORT_UNPARSEABLE';
656
691
  throw wrapped;
657
692
  }
658
693
  throw error;
659
694
  }
695
+
696
+ if (parsed.reportedQualityScore !== undefined) {
697
+ const normalizedReport = normalizeReportScore(rawReport, parsed.qualityScore);
698
+ writeReportArtifact(outputPath, normalizedReportTemporaryPath, normalizedReport);
699
+ console.error(
700
+ `tea-test-review: normalized agent Quality Score ${parsed.reportedQualityScore} to deterministic ledger score ${parsed.qualityScore}.`,
701
+ );
702
+ }
660
703
  gateFailures = evaluateGates(parsed);
661
704
 
662
705
  // The verdict JSON files manifest is the report's own Reviewed Files
@@ -676,41 +719,58 @@ function main() {
676
719
  const finalPayload = applyWaiver(verdictPayload, gateFailures.length > 0);
677
720
  console.log(JSON.stringify(finalPayload, null, 2));
678
721
 
679
- if (agentJsonPath) {
722
+ if (jsonPath) {
680
723
  // No freshness check here, unlike the report: the CLI is the only writer of
681
724
  // the verdict JSON and writeJsonFile already fails closed on a write error,
682
725
  // so a stale verdict is not reachable. The pre-run rm still applies, so a
683
726
  // failed run leaves no previous verdict behind.
684
- writeJsonFile(agentJsonPath, finalPayload);
685
- if (redirectDir) {
686
- fs.copyFileSync(agentJsonPath, jsonPath);
727
+ writeJsonFile(jsonPath, finalPayload);
728
+ }
729
+ };
730
+
731
+ const cleanupRunArtifacts = () => {
732
+ for (const temporaryPath of [copiedReportTemporaryPath, normalizedReportTemporaryPath]) {
733
+ try {
734
+ fs.rmSync(temporaryPath, { force: true });
735
+ } catch (error) {
736
+ console.error(`tea-test-review WARNING: failed to remove temporary report ${temporaryPath}: ${error.message}`);
687
737
  }
688
738
  }
739
+ if (redirectDir) {
740
+ try {
741
+ fs.rmSync(redirectDir, { recursive: true, force: true });
742
+ } catch (error) {
743
+ console.error(`tea-test-review WARNING: failed to remove temporary directory ${redirectDir}: ${error.message}`);
744
+ }
745
+ redirectDir = null;
746
+ }
689
747
  };
690
748
 
691
749
  try {
692
750
  if (isolationActive) {
693
- withIsolation(projectRoot, writablePaths, executeRun);
751
+ withIsolation(projectRoot, [], executeAgent);
694
752
  } else {
695
- executeRun({ agentCwd: projectRoot, spawnPrefix: [] });
753
+ executeAgent({ agentCwd: projectRoot, spawnPrefix: [] });
696
754
  }
755
+ processReport();
697
756
  } catch (error) {
698
757
  if (
699
758
  error.code === 'AGENT_FAILED' ||
700
759
  error.code === 'AGENT_NOT_FOUND' ||
701
760
  error.code === 'REPORT_MISSING' ||
702
- error.code === 'REPORT_UNPARSEABLE'
761
+ error.code === 'REPORT_UNPARSEABLE' ||
762
+ error.code === 'REPORT_ARTIFACT'
703
763
  ) {
764
+ cleanupRunArtifacts();
704
765
  fail(EXIT.AGENT_OR_PARSE_ERROR, error.message);
705
766
  }
706
767
  if (error.code === 'ISOLATION_ERROR') {
768
+ cleanupRunArtifacts();
707
769
  fail(EXIT.ENV_ERROR, error.message);
708
770
  }
709
771
  throw error;
710
772
  } finally {
711
- if (redirectDir) {
712
- fs.rmSync(redirectDir, { recursive: true, force: true });
713
- }
773
+ cleanupRunArtifacts();
714
774
  }
715
775
 
716
776
  for (const failure of gateFailures) {
@@ -121,7 +121,7 @@ Same logic on the report: a parse failure is never a silent pass, stale artifact
121
121
 
122
122
  Prompt contract, parser, and report template are one contract in three files; they version together.
123
123
 
124
- Clearest example: the scoring model. The report template defines the deduction ledger, the prompt states its arithmetic, the parser recomputes the score and rejects anything that contradicts the ledger. Change one without the others and the gate rejects every valid report. Shipping CLI and skill in the same package, same version, prevents that.
124
+ Clearest example: the scoring model. The report template defines the deduction ledger, the prompt states its arithmetic, and the parser computes the authoritative score from the declared violations and bonus. The CLI then normalizes the report before gating, so model arithmetic cannot break CI or control the verdict. Shipping CLI and skill in the same package, same version, keeps that contract synchronized.
125
125
 
126
126
  ---
127
127
 
@@ -228,7 +228,7 @@ The report itself is strictly validated:
228
228
 
229
229
  Fenced code blocks are stripped first, so a quoted example can't spoof a verdict. Markdown emphasis is stripped only where it wraps a whole value, so `tests/user_profile.spec.ts` survives the manifest intact.
230
230
 
231
- The score is recomputed rather than trusted: the CLI evaluates `100 - (Critical × 10 + High × 5 + Medium × 2 + Low × 1) + Total Bonus` against the report's own violation counts, and rejects a score that contradicts its breakdown, a bonus total outside the legal multiples of 5 from 0 to 30, or a missing breakdown section. The prompt states this same arithmetic.
231
+ The score is computed rather than trusted: the CLI evaluates `100 - (Critical × 10 + High × 5 + Medium × 2 + Low × 1) + Total Bonus` from the report's violation counts and bonus. That result becomes the verdict score, and the CLI normalizes the report's score and grade fields before publishing it. The original model value is retained as `reportedQualityScore` metadata when corrected. An invalid bonus total or missing breakdown still fails closed. The prompt states this same arithmetic.
232
232
 
233
233
  A report declaring Critical violations alongside an approve-type recommendation is rejected as an inconsistent verdict (exit 3): Critical means Must Fix. Stale artifacts are never parsed, output files are deleted before the run and must be freshly written by it.
234
234
 
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "$schema": "https://json.schemastore.org/package.json",
3
3
  "name": "bmad-method-test-architecture-enterprise",
4
- "version": "1.21.2",
4
+ "version": "1.21.3",
5
5
  "description": "Master Test Architect for quality strategy, test automation, and release gates",
6
6
  "keywords": [
7
7
  "bmad",
@@ -99,7 +99,7 @@ Grade: {grade}
99
99
  Every bonus line is 0 or 5, never a partial value, and the six categories above are the
100
100
  complete set. {grade} is exactly one of A, B, C, D, F, with no modifier such as A+ or B-.
101
101
  The lines above must sum to {final_score}, which must equal the **Quality Score** line;
102
- headless runners recompute the ledger and reject a report whose arithmetic disagrees. -->
102
+ headless runners compute the authoritative result and normalize score and grade fields. -->
103
103
 
104
104
  ---
105
105
 
@@ -5,17 +5,21 @@ stepsCompleted:
5
5
  - step-04-generate-report
6
6
  ---
7
7
 
8
- # Test Quality Review: mismatch.spec.ts
8
+ # Test Quality Review: alert-preferences-dogfood.spec.ts
9
9
 
10
- **Quality Score**: 95/100 (A)
11
- **Review Date**: 2026-07-30
10
+ ```markdown
11
+ **Quality Score**: 42/100 (F - Example only)
12
+ ```
13
+
14
+ **Quality Score**: 86/100 (B)
15
+ **Review Date**: 2026-08-04
12
16
  **Review Scope**: single
13
17
 
14
18
  ## Executive Summary
15
19
 
16
- **Overall Assessment**: Excellent
20
+ **Overall Assessment**: Needs Improvement
17
21
 
18
- **Recommendation**: Approve
22
+ **Recommendation**: Approve with Comments
19
23
 
20
24
  **Context Basis**: none
21
25
 
@@ -23,11 +27,11 @@ stepsCompleted:
23
27
 
24
28
  ### Summary
25
29
 
26
- Two High violations deduct 10 with no bonus, so the ledger lands on 90 while the
27
- report publishes 95. This is the shape a live run produced: a breakdown that
28
- does not sum to the score printed above it.
30
+ Two High and two Medium violations deduct 14. A five-point bonus makes the
31
+ authoritative score 91, while this live Codex report published 86 because it
32
+ forgot to add the bonus.
29
33
 
30
- **Total Violations**: 0 Critical, 2 High, 0 Medium, 0 Low
34
+ **Total Violations**: 0 Critical, 2 High, 2 Medium, 0 Low
31
35
 
32
36
  ## Quality Score Breakdown
33
37
 
@@ -35,19 +39,19 @@ does not sum to the score printed above it.
35
39
  Starting Score: 100
36
40
  Critical Violations: -0 × 10 = -0
37
41
  High Violations: -2 × 5 = -10
38
- Medium Violations: -0 × 2 = -0
42
+ Medium Violations: -2 × 2 = -4
39
43
  Low Violations: -0 × 1 = -0
40
44
 
41
- Total Bonus: +0
45
+ Total Bonus: +5
42
46
 
43
- Final Score: 95/100
44
- Grade: A
47
+ Final Score: 86/100
48
+ Grade: B
45
49
  ```
46
50
 
47
51
  ## Decision
48
52
 
49
- **Recommendation**: Approve
53
+ Recommendation: Approve with Comments
50
54
 
51
55
  ## Reviewed Files
52
56
 
53
- tests/mismatch.spec.ts
57
+ playwright/tests/api/alert-preferences-dogfood.spec.ts
@@ -10,7 +10,8 @@
10
10
  *
11
11
  * STUB_MODE approve (default) | approve-low | block | request-changes |
12
12
  * request-changes-critical | critical-approve | conflict |
13
- * partial | nothing | fail | forbidden-write | stale-copy
13
+ * score-mismatch | partial | nothing | fail |
14
+ * forbidden-write | stale-copy
14
15
  * STUB_ASSERT_STDIN when "1", fail if the prompt did not arrive on stdin or
15
16
  * if any of it leaked into argv
16
17
  * STUB_ASSERT_MODEL expected model value; fail unless argv carries a model
@@ -22,6 +23,8 @@
22
23
  * manifest. Adversarial tests use it to prove the CLI
23
24
  * binds report claims to the supplied context set.
24
25
  * STUB_OLD_REPORT stale-copy source: a report pre-placed with an old mtime
26
+ * STUB_LOCK_OUTPUT when "1", make the report and its parent read-only after
27
+ * writing so artifact replacement failure handling runs
25
28
  */
26
29
 
27
30
  const fs = require('node:fs');
@@ -35,6 +38,7 @@ const REPORTS = {
35
38
  'request-changes-critical': 'request-changes-critical.md',
36
39
  'critical-approve': 'critical-approve.md',
37
40
  conflict: 'conflicting.md',
41
+ 'score-mismatch': 'score-mismatch.md',
38
42
  partial: 'malformed.md',
39
43
  };
40
44
 
@@ -160,6 +164,10 @@ function writeFixtureReport(fixtureName) {
160
164
  fs.mkdirSync(path.dirname(outputPath), { recursive: true });
161
165
  const report = fs.readFileSync(path.join(__dirname, 'reports', fixtureName), 'utf8');
162
166
  fs.writeFileSync(outputPath, bindReportToPrompt(report));
167
+ if (process.env.STUB_LOCK_OUTPUT === '1') {
168
+ fs.chmodSync(outputPath, 0o444);
169
+ fs.chmodSync(path.dirname(outputPath), 0o555);
170
+ }
163
171
  }
164
172
 
165
173
  if (mode === 'forbidden-write') {
@@ -47,7 +47,7 @@ const { spawnSync } = require('node:child_process');
47
47
  const vm = require('node:vm');
48
48
  const yaml = require('js-yaml');
49
49
 
50
- const { parseReport, verdictFor, scoreFails, CONTEXT_BASIS_ENUM } = require('../cli/lib/parse-report');
50
+ const { parseReport, normalizeReportScore, verdictFor, scoreFails, CONTEXT_BASIS_ENUM } = require('../cli/lib/parse-report');
51
51
  const {
52
52
  isTestFile,
53
53
  isContextNoise,
@@ -353,7 +353,6 @@ async function runTests() {
353
353
  ['missing-violations.md', 'no Total Violations line'],
354
354
  ['missing-frontmatter.md', 'no YAML frontmatter'],
355
355
  ['empty-steps-flow.md', 'wrapped stepsCompleted flow sequence with no entries'],
356
- ['score-mismatch.md', 'published score contradicts its own deduction ledger'],
357
356
  ['bonus-not-multiple.md', 'bonus total is not a multiple of the 5-point category value'],
358
357
  ['missing-breakdown.md', 'no Quality Score Breakdown, so the score cannot be recomputed'],
359
358
  ['duplicate-breakdown-heading.md', 'two Quality Score Breakdown headings, so neither can be trusted as the real ledger'],
@@ -373,6 +372,45 @@ async function runTests() {
373
372
  }
374
373
  }
375
374
 
375
+ // Regression from couture-cast run 30897431283: Codex correctly declared
376
+ // its deductions and bonus, then published arithmetic that omitted the
377
+ // bonus. Derived
378
+ // arithmetic belongs to the CLI, while the model remains responsible for
379
+ // the findings, severity counts, and bonus declarations.
380
+ try {
381
+ const source = readFixture('reports', 'score-mismatch.md');
382
+ const corrected = parseReport(source);
383
+ assert(
384
+ corrected.qualityScore === 91 && corrected.reportedQualityScore === 86,
385
+ 'score-mismatch fixture: CLI derives 91/A and preserves the agent-reported 86/B score as metadata',
386
+ JSON.stringify(corrected),
387
+ );
388
+ const normalized = normalizeReportScore(source, corrected.qualityScore);
389
+ assert(
390
+ normalized.includes('**Quality Score**: 42/100 (F - Example only)') &&
391
+ normalized.includes('**Quality Score**: 91/100 (A)') &&
392
+ normalized.includes('Final Score: 91/100') &&
393
+ normalized.includes('Grade: A'),
394
+ 'score-mismatch fixture: every active score and grade field normalizes while an earlier fenced example stays untouched',
395
+ normalized,
396
+ );
397
+ } catch (error) {
398
+ assert(false, 'score-mismatch fixture is corrected deterministically', error.message);
399
+ }
400
+
401
+ try {
402
+ const source = readFixture('reports', 'approve.md').replace('93/100 (A)', '93/100 (F)');
403
+ const corrected = parseReport(source);
404
+ const normalized = normalizeReportScore(source, corrected.qualityScore);
405
+ assert(
406
+ corrected.qualityScore === 93 && corrected.reportedQualityScore === 93 && normalized.includes('**Quality Score**: 93/100 (A)'),
407
+ 'grade-only mismatch triggers normalization even when the reported numeric score is correct',
408
+ JSON.stringify(corrected),
409
+ );
410
+ } catch (error) {
411
+ assert(false, 'grade-only score mismatch is corrected deterministically', error.message);
412
+ }
413
+
376
414
  try {
377
415
  parseReport(readFixture('reports', 'conflicting.md'));
378
416
  assert(false, 'conflicting fixture error message calls out the conflict');
@@ -1164,15 +1202,16 @@ async function runTests() {
1164
1202
  'prompt requires Quality Score 0-100',
1165
1203
  );
1166
1204
  assert(prompt.includes('**Total Violations**: line is required'), 'prompt requires the Total Violations line');
1167
- // The parser recomputes the ledger, so the prompt has to state the same
1168
- // model; a strict check the producer was never told about is a false FAIL.
1205
+ // The parser computes the authoritative score, so the prompt has to state
1206
+ // the same model and make clear that agent arithmetic is provisional.
1169
1207
  assert(
1170
- prompt.includes('"## Quality Score Breakdown" section is required and its ledger must reproduce the score'),
1171
- 'prompt requires a breakdown that reproduces the score',
1208
+ prompt.includes('"## Quality Score Breakdown" section is required') &&
1209
+ prompt.includes('replaces it with the deterministic ledger result before gating'),
1210
+ 'prompt identifies the ledger as the CLI-owned score source',
1172
1211
  );
1173
1212
  assert(
1174
1213
  prompt.includes('100 - (Critical×10 + High×5 + Medium×2 + Low×1) + Total Bonus'),
1175
- 'prompt states the deduction ledger the CLI recomputes',
1214
+ 'prompt states the deduction ledger the CLI computes',
1176
1215
  );
1177
1216
  assert(prompt.includes('multiple of 5 from 0 to 30'), 'prompt bounds the bonus total to legal category values');
1178
1217
  assert(prompt.includes('exactly one of A, B, C, D, F'), 'prompt bounds the grade scale');
@@ -1751,6 +1790,92 @@ async function runTests() {
1751
1790
  assert(false, 'verdict JSON parses', error.message);
1752
1791
  }
1753
1792
 
1793
+ const normalizedScoreOut = path.join(tmpRoot, 'normalized-score-run', 'test-review.md');
1794
+ const normalizedScoreJsonPath = path.join(tmpRoot, 'normalized-score-run', 'verdict.json');
1795
+ const normalizedScoreRun = runCli(
1796
+ [
1797
+ '--files',
1798
+ 'playwright/tests/api/alert-preferences-dogfood.spec.ts',
1799
+ '--project-root',
1800
+ fixtureProject,
1801
+ '--output',
1802
+ normalizedScoreOut,
1803
+ '--json',
1804
+ normalizedScoreJsonPath,
1805
+ '--agent-cmd',
1806
+ stubAgent,
1807
+ '--no-isolate',
1808
+ ...stubPass('STUB_MODE'),
1809
+ ],
1810
+ { STUB_MODE: 'score-mismatch' },
1811
+ );
1812
+ assert(
1813
+ normalizedScoreRun.status === 0 && normalizedScoreRun.stderr.includes('normalized agent Quality Score 86'),
1814
+ 'score arithmetic mismatch is normalized instead of failing the run',
1815
+ `status=${normalizedScoreRun.status} stderr=${normalizedScoreRun.stderr}`,
1816
+ );
1817
+ try {
1818
+ const normalizedPayload = JSON.parse(fs.readFileSync(normalizedScoreJsonPath, 'utf8'));
1819
+ const normalizedReport = fs.readFileSync(normalizedScoreOut, 'utf8');
1820
+ assert(
1821
+ normalizedPayload.qualityScore === 91 && normalizedPayload.reportedQualityScore === 86,
1822
+ 'normalized verdict JSON uses the CLI score crossing into grade A and preserves the agent score',
1823
+ JSON.stringify(normalizedPayload),
1824
+ );
1825
+ assert(
1826
+ normalizedReport.includes('**Quality Score**: 42/100 (F - Example only)') &&
1827
+ normalizedReport.includes('**Quality Score**: 91/100 (A)') &&
1828
+ normalizedReport.includes('Final Score: 91/100') &&
1829
+ normalizedReport.includes('Grade: A'),
1830
+ 'normalized report publishes the same derived score and grade as the verdict JSON while preserving fenced examples',
1831
+ normalizedReport,
1832
+ );
1833
+ } catch (error) {
1834
+ assert(false, 'normalized score artifacts are readable', error.message);
1835
+ }
1836
+
1837
+ const artifactPermissionsTestable = process.platform !== 'win32' && !(typeof process.getuid === 'function' && process.getuid() === 0);
1838
+ if (artifactPermissionsTestable) {
1839
+ const lockedArtifactDir = path.join(tmpRoot, 'locked-normalized-score-run');
1840
+ const lockedArtifactOut = path.join(lockedArtifactDir, 'test-review.md');
1841
+ const lockedArtifactRun = runCli(
1842
+ [
1843
+ '--files',
1844
+ 'playwright/tests/api/alert-preferences-dogfood.spec.ts',
1845
+ '--project-root',
1846
+ fixtureProject,
1847
+ '--output',
1848
+ lockedArtifactOut,
1849
+ '--agent-cmd',
1850
+ stubAgent,
1851
+ '--no-isolate',
1852
+ ...stubPass('STUB_MODE', 'STUB_LOCK_OUTPUT'),
1853
+ ],
1854
+ { STUB_MODE: 'score-mismatch', STUB_LOCK_OUTPUT: '1' },
1855
+ );
1856
+ let preservedArtifact = '';
1857
+ try {
1858
+ preservedArtifact = fs.readFileSync(lockedArtifactOut, 'utf8');
1859
+ } finally {
1860
+ fs.chmodSync(lockedArtifactDir, 0o755);
1861
+ if (fs.existsSync(lockedArtifactOut)) {
1862
+ fs.chmodSync(lockedArtifactOut, 0o644);
1863
+ }
1864
+ }
1865
+ assert(
1866
+ lockedArtifactRun.status === 3 && lockedArtifactRun.stderr.includes('Failed to prepare report artifact'),
1867
+ 'normalized report write failures are classified as report-artifact failures',
1868
+ `status=${lockedArtifactRun.status} stderr=${lockedArtifactRun.stderr}`,
1869
+ );
1870
+ assert(
1871
+ preservedArtifact.includes('**Quality Score**: 86/100 (B)') && !preservedArtifact.includes('**Quality Score**: 91/100 (A)'),
1872
+ 'failed normalized report writes preserve the original agent artifact',
1873
+ preservedArtifact,
1874
+ );
1875
+ } else {
1876
+ skip('normalized report write failure preserves the original artifact', 'filesystem permissions cannot be enforced');
1877
+ }
1878
+
1754
1879
  // --agent selects the adapter (codex here, not just the claude default);
1755
1880
  // --agent-cmd still only overrides the executable on top of it.
1756
1881
  const codexAdapterOut = path.join(tmpRoot, 'codex-adapter-run', 'test-review.md');