bmad-method-test-architecture-enterprise 1.21.2 → 1.21.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/cli/lib/build-prompt.js +5 -4
- package/cli/lib/parse-report.js +84 -26
- package/cli/test-review.js +104 -44
- package/docs/explanation/test-review-cli-architecture.md +1 -1
- package/docs/reference/tea-test-review-cli.md +1 -1
- package/package.json +1 -1
- package/src/workflows/testarch/bmad-testarch-test-review/test-review-template.md +1 -1
- package/test/fixtures/test-review-cli/reports/score-mismatch.md +19 -15
- package/test/fixtures/test-review-cli/stub-agent.js +9 -1
- package/test/test-test-review-cli.js +132 -7
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
"name": "bmad-method-test-architecture-enterprise",
|
|
13
13
|
"source": "./",
|
|
14
14
|
"description": "Master Test Architect module for quality strategy, test automation, CI/CD quality gates, and structured testing education. Part of the BMad Method ecosystem.",
|
|
15
|
-
"version": "1.21.
|
|
15
|
+
"version": "1.21.3",
|
|
16
16
|
"author": {
|
|
17
17
|
"name": "Murat K Ozcan (TEA Creator) & Brian (BMad) Madison"
|
|
18
18
|
},
|
package/cli/lib/build-prompt.js
CHANGED
|
@@ -136,11 +136,12 @@ function buildPrompt({
|
|
|
136
136
|
'- **Recommendation** must be exactly one of: Approve | Approve with Comments | Request Changes | Block',
|
|
137
137
|
'- A "## Decision" section is required, spelled exactly that, and its **Recommendation** must match the',
|
|
138
138
|
" Executive Summary's. Do not rename the heading after the sentence that describes it.",
|
|
139
|
-
'- **Quality Score**: N/100 is required and must be an integer from 0 to 100.',
|
|
139
|
+
'- **Quality Score**: N/100 is required and must be an integer from 0 to 100. The CLI treats this agent-written',
|
|
140
|
+
' number as provisional and replaces it with the deterministic ledger result before gating.',
|
|
140
141
|
'- The **Total Violations**: line is required, with Critical, High, Medium, and Low counts.',
|
|
141
|
-
'- The "## Quality Score Breakdown" section is required
|
|
142
|
-
'
|
|
143
|
-
'
|
|
142
|
+
'- The "## Quality Score Breakdown" section is required. The CLI computes the authoritative score as',
|
|
143
|
+
' 100 - (Critical×10 + High×5 + Medium×2 + Low×1) + Total Bonus, then normalizes the report score and grade',
|
|
144
|
+
' before gating. The deduction ledger is the only scoring model: never a weighted average or judgment adjustment.',
|
|
144
145
|
'- Each of the six bonus categories is worth 0 or 5, so "Total Bonus" is a multiple of 5 from 0 to 30.',
|
|
145
146
|
'- Grade is exactly one of A, B, C, D, F, with no modifier such as A+.',
|
|
146
147
|
`- The Executive Summary must carry exactly one "**Context Basis**: ${contextBasis}" line, exactly that value.`,
|
package/cli/lib/parse-report.js
CHANGED
|
@@ -10,9 +10,9 @@
|
|
|
10
10
|
* it is normalized.
|
|
11
11
|
* - "**Quality Score**: N/100" with N an integer in 0-100.
|
|
12
12
|
* - A "**Total Violations**:" line with all four severity counts.
|
|
13
|
-
* - A "## Quality Score Breakdown" ledger
|
|
14
|
-
*
|
|
15
|
-
*
|
|
13
|
+
* - A "## Quality Score Breakdown" ledger from which the CLI computes the
|
|
14
|
+
* authoritative score; the skill's deduction model is the only scoring
|
|
15
|
+
* model, so agent arithmetic never controls the gate.
|
|
16
16
|
* - A "## Reviewed Files" section listing every reviewed file.
|
|
17
17
|
* - Exactly one "**Context Basis**:" line inside the Executive Summary, plus a
|
|
18
18
|
* "## Review Context" manifest whenever that basis is not `none`.
|
|
@@ -46,7 +46,7 @@ const RECOMMENDATION_LINE = /^[ \t]*(?:\*\*Recommendation\*\*:|\*\*Recommendatio
|
|
|
46
46
|
const CONTEXT_BASIS_ENUM = ['none', 'pr_diff', 'pr_diff_truncated'];
|
|
47
47
|
const CONTEXT_BASIS_LINE_SOURCE = String.raw`^[ \t]*\*\*Context Basis:?\*\*:?[ \t]*([^\r\n]+)[ \t]*$`;
|
|
48
48
|
const CONTEXT_WAIVERS_LINE_SOURCE = String.raw`^[ \t]*\*\*Context Waivers Applied:?\*\*:?[ \t]*([^\r\n]+)[ \t]*$`;
|
|
49
|
-
const SCORE_PATTERN = /\*\*Quality Score\*\*:\s*(\d+)\s*\/\s*100
|
|
49
|
+
const SCORE_PATTERN = /\*\*Quality Score\*\*:\s*(\d+)\s*\/\s*100(?:[ \t]*\([ \t]*([A-F])(?=[ \t)-]))?/;
|
|
50
50
|
const VIOLATIONS_LINE = /\*\*Total Violations:?\*\*:?[ \t]*([^\n]+)/;
|
|
51
51
|
const VIOLATION_LEVELS = ['Critical', 'High', 'Medium', 'Low'];
|
|
52
52
|
// The template always prints the bonus with a leading "+" (every fixture in
|
|
@@ -444,20 +444,19 @@ function verifyRunContract({ reviewedFiles, contextBasis, contextFiles }, runCon
|
|
|
444
444
|
}
|
|
445
445
|
|
|
446
446
|
/**
|
|
447
|
-
*
|
|
448
|
-
*
|
|
447
|
+
* Compute the authoritative quality score from the template's deduction
|
|
448
|
+
* ledger. The agent's published score is presentation data only.
|
|
449
449
|
*
|
|
450
|
-
*
|
|
451
|
-
*
|
|
452
|
-
*
|
|
453
|
-
*
|
|
454
|
-
* violation counts the report already declares.
|
|
450
|
+
* Live runs have repeatedly published arithmetic that contradicts their own
|
|
451
|
+
* ledgers. The ledger in `test-review-template.md` is the workflow's only
|
|
452
|
+
* scoring model, so the CLI derives the score from the violation counts and
|
|
453
|
+
* bonus instead of asking a probabilistic producer to perform gate arithmetic.
|
|
455
454
|
*
|
|
456
455
|
* The breakdown sits inside a fenced block, which the verdict scan strips, so
|
|
457
456
|
* this reads the raw report instead and anchors on the section heading: only
|
|
458
457
|
* the ledger under "## Quality Score Breakdown" is ever consulted.
|
|
459
458
|
*/
|
|
460
|
-
function
|
|
459
|
+
function deriveQualityScore(rawText, violations) {
|
|
461
460
|
// extractSection's regex takes the first match; on raw (fence-intact) text
|
|
462
461
|
// that is exploitable if the reviewed file's own quoted content contains a
|
|
463
462
|
// second "## Quality Score Breakdown" heading earlier in the report than
|
|
@@ -488,13 +487,64 @@ function verifyScoreLedger(rawText, qualityScore, violations) {
|
|
|
488
487
|
const key = level.toLowerCase();
|
|
489
488
|
return sum + violations[key] * SEVERITY_DEDUCTIONS[key];
|
|
490
489
|
}, 0);
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
490
|
+
return Math.max(0, Math.min(100, 100 - deductions + bonus));
|
|
491
|
+
}
|
|
492
|
+
|
|
493
|
+
function gradeForScore(score) {
|
|
494
|
+
if (score >= 90) return 'A';
|
|
495
|
+
if (score >= 80) return 'B';
|
|
496
|
+
if (score >= 70) return 'C';
|
|
497
|
+
if (score >= 60) return 'D';
|
|
498
|
+
return 'F';
|
|
499
|
+
}
|
|
500
|
+
|
|
501
|
+
/** Normalize the report's schema-owned score and grade fields to CLI arithmetic. */
|
|
502
|
+
function normalizeReportScore(reportText, qualityScore) {
|
|
503
|
+
const grade = gradeForScore(qualityScore);
|
|
504
|
+
let inFence = false;
|
|
505
|
+
let section = null;
|
|
506
|
+
let summaryNormalized = false;
|
|
507
|
+
let finalScoreNormalized = false;
|
|
508
|
+
let finalGradeNormalized = false;
|
|
509
|
+
|
|
510
|
+
return reportText
|
|
511
|
+
.split('\n')
|
|
512
|
+
.map((originalLine) => {
|
|
513
|
+
let line = originalLine;
|
|
514
|
+
if (/^\s*```/.test(line)) {
|
|
515
|
+
inFence = !inFence;
|
|
516
|
+
return line;
|
|
517
|
+
}
|
|
518
|
+
|
|
519
|
+
if (!inFence) {
|
|
520
|
+
const heading = /^##[ \t]+([^\r\n]+?)[ \t]*\r?$/.exec(line);
|
|
521
|
+
if (heading) {
|
|
522
|
+
section = heading[1];
|
|
523
|
+
}
|
|
524
|
+
if (!summaryNormalized && /^[ \t]*\*\*Quality Score\*\*:/.test(line)) {
|
|
525
|
+
line = line
|
|
526
|
+
.replace(/^([ \t]*\*\*Quality Score\*\*:[ \t]*)\d+([ \t]*\/[ \t]*100)/, `$1${qualityScore}$2`)
|
|
527
|
+
.replace(/^([ \t]*\*\*Quality Score\*\*:[^\r\n]*\([ \t]*)[A-F](?=[ \t)-])/, `$1${grade}`);
|
|
528
|
+
summaryNormalized = true;
|
|
529
|
+
}
|
|
530
|
+
}
|
|
531
|
+
|
|
532
|
+
// The score ledger is itself a fenced block. Its fields remain active
|
|
533
|
+
// only inside the unique Quality Score Breakdown section validated by
|
|
534
|
+
// deriveQualityScore; unrelated fenced examples stay untouched.
|
|
535
|
+
if (section === 'Quality Score Breakdown') {
|
|
536
|
+
if (!finalScoreNormalized && /^[ \t]*Final Score[ \t]*:/.test(line)) {
|
|
537
|
+
line = line.replace(/^([ \t]*Final Score[ \t]*:[ \t]*)\d+([ \t]*\/[ \t]*100[ \t]*\r?)$/, `$1${qualityScore}$2`);
|
|
538
|
+
finalScoreNormalized = true;
|
|
539
|
+
}
|
|
540
|
+
if (!finalGradeNormalized && /^[ \t]*Grade[ \t]*:/.test(line)) {
|
|
541
|
+
line = line.replace(/^([ \t]*Grade[ \t]*:[ \t]*)[A-F]([ \t]*\r?)$/, `$1${grade}$2`);
|
|
542
|
+
finalGradeNormalized = true;
|
|
543
|
+
}
|
|
544
|
+
}
|
|
545
|
+
return line;
|
|
546
|
+
})
|
|
547
|
+
.join('\n');
|
|
498
548
|
}
|
|
499
549
|
|
|
500
550
|
/**
|
|
@@ -502,7 +552,7 @@ function verifyScoreLedger(rawText, qualityScore, violations) {
|
|
|
502
552
|
*
|
|
503
553
|
* @param {string} reportText - Full test-review.md contents.
|
|
504
554
|
* @param {object} [runContract] - Exact reviewed/context evidence supplied by the runner.
|
|
505
|
-
* @returns {{recommendation: string, qualityScore: number, violations: object,
|
|
555
|
+
* @returns {{recommendation: string, qualityScore: number, reportedQualityScore?: number, violations: object,
|
|
506
556
|
* reviewedFiles: string[], contextBasis: string, contextFiles: string[],
|
|
507
557
|
* contextWaiversApplied: number}}
|
|
508
558
|
* @throws {Error} With code REPORT_UNPARSEABLE on any missing/invalid element.
|
|
@@ -522,9 +572,10 @@ function parseReport(reportText, runContract = {}) {
|
|
|
522
572
|
if (!scoreMatch) {
|
|
523
573
|
unparseable('Report is missing the "**Quality Score**: N/100" line');
|
|
524
574
|
}
|
|
525
|
-
const
|
|
526
|
-
|
|
527
|
-
|
|
575
|
+
const reportedQualityScore = Number.parseInt(scoreMatch[1], 10);
|
|
576
|
+
const reportedQualityGrade = scoreMatch[2];
|
|
577
|
+
if (reportedQualityScore < 0 || reportedQualityScore > 100) {
|
|
578
|
+
unparseable(`Report Quality Score ${reportedQualityScore} is outside the required 0-100 range`);
|
|
528
579
|
}
|
|
529
580
|
|
|
530
581
|
const violations = parseViolations(text);
|
|
@@ -535,7 +586,7 @@ function parseReport(reportText, runContract = {}) {
|
|
|
535
586
|
);
|
|
536
587
|
}
|
|
537
588
|
|
|
538
|
-
|
|
589
|
+
const qualityScore = deriveQualityScore(reportText, violations);
|
|
539
590
|
|
|
540
591
|
const reviewedFiles = parseReviewedFiles(text);
|
|
541
592
|
const contextBasis = parseContextBasis(text);
|
|
@@ -560,7 +611,7 @@ function parseReport(reportText, runContract = {}) {
|
|
|
560
611
|
const keyStrengths = extractBullets(extractSubsection(executiveSection, 'Key Strengths'), '✅');
|
|
561
612
|
const keyWeaknesses = extractBullets(extractSubsection(executiveSection, 'Key Weaknesses'), '❌');
|
|
562
613
|
|
|
563
|
-
|
|
614
|
+
const parsed = {
|
|
564
615
|
recommendation: executive,
|
|
565
616
|
qualityScore,
|
|
566
617
|
violations,
|
|
@@ -571,6 +622,13 @@ function parseReport(reportText, runContract = {}) {
|
|
|
571
622
|
keyStrengths,
|
|
572
623
|
keyWeaknesses,
|
|
573
624
|
};
|
|
625
|
+
if (
|
|
626
|
+
reportedQualityScore !== qualityScore ||
|
|
627
|
+
(reportedQualityGrade !== undefined && reportedQualityGrade !== gradeForScore(qualityScore))
|
|
628
|
+
) {
|
|
629
|
+
parsed.reportedQualityScore = reportedQualityScore;
|
|
630
|
+
}
|
|
631
|
+
return parsed;
|
|
574
632
|
}
|
|
575
633
|
|
|
576
634
|
/**
|
|
@@ -599,4 +657,4 @@ function scoreFails(score, minScore) {
|
|
|
599
657
|
return score < minScore;
|
|
600
658
|
}
|
|
601
659
|
|
|
602
|
-
module.exports = { parseReport, verdictFor, scoreFails, CONTEXT_BASIS_ENUM };
|
|
660
|
+
module.exports = { parseReport, normalizeReportScore, verdictFor, scoreFails, CONTEXT_BASIS_ENUM };
|
package/cli/test-review.js
CHANGED
|
@@ -15,8 +15,9 @@
|
|
|
15
15
|
*
|
|
16
16
|
* Exit codes: 0 pass/skip (or a verdict failure waived with --waive), 1 review
|
|
17
17
|
* verdict fail (or a skip with --fail-on-skip / a deletions-only diff),
|
|
18
|
-
* 2 environment/config error, 3 agent or
|
|
19
|
-
* to exit 2 or 3:
|
|
18
|
+
* 2 environment/config error, 3 agent, parse, or report-artifact failure. A
|
|
19
|
+
* waiver never applies to exit 2 or 3: infrastructure failures are never
|
|
20
|
+
* waivable.
|
|
20
21
|
*
|
|
21
22
|
* Usage:
|
|
22
23
|
* tea-test-review --base origin/main --agent claude --json test-review.json
|
|
@@ -24,6 +25,7 @@
|
|
|
24
25
|
*/
|
|
25
26
|
|
|
26
27
|
const fs = require('node:fs');
|
|
28
|
+
const { randomUUID } = require('node:crypto');
|
|
27
29
|
const os = require('node:os');
|
|
28
30
|
const path = require('node:path');
|
|
29
31
|
const { spawn } = require('node:child_process');
|
|
@@ -41,7 +43,7 @@ const {
|
|
|
41
43
|
registerExtraTestPattern,
|
|
42
44
|
} = require('./lib/changed-tests');
|
|
43
45
|
const { buildPrompt } = require('./lib/build-prompt');
|
|
44
|
-
const { parseReport, verdictFor, scoreFails } = require('./lib/parse-report');
|
|
46
|
+
const { parseReport, normalizeReportScore, verdictFor, scoreFails } = require('./lib/parse-report');
|
|
45
47
|
const { runAgent } = require('./lib/run-agent');
|
|
46
48
|
const { AGENT_ADAPTERS, resolveModel } = require('./lib/agent-adapters');
|
|
47
49
|
const { withIsolation, selectBackend } = require('./lib/isolate');
|
|
@@ -151,6 +153,45 @@ function writeJsonFile(jsonPath, payload) {
|
|
|
151
153
|
}
|
|
152
154
|
}
|
|
153
155
|
|
|
156
|
+
function reportArtifactFailure(action, artifactPath, cause) {
|
|
157
|
+
const error = new Error(`Failed to ${action} report artifact ${artifactPath}: ${cause.message}`);
|
|
158
|
+
error.code = 'REPORT_ARTIFACT';
|
|
159
|
+
return error;
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
function reportTemporaryPath(artifactPath) {
|
|
163
|
+
return path.join(path.dirname(artifactPath), `.${path.basename(artifactPath)}.${randomUUID()}.tmp`);
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
function replaceReportArtifact(artifactPath, temporaryPath, writeTemporary) {
|
|
167
|
+
let temporaryReady = false;
|
|
168
|
+
try {
|
|
169
|
+
fs.mkdirSync(path.dirname(artifactPath), { recursive: true });
|
|
170
|
+
writeTemporary(temporaryPath);
|
|
171
|
+
temporaryReady = true;
|
|
172
|
+
fs.renameSync(temporaryPath, artifactPath);
|
|
173
|
+
} catch (error) {
|
|
174
|
+
try {
|
|
175
|
+
fs.rmSync(temporaryPath, { force: true });
|
|
176
|
+
} catch {
|
|
177
|
+
// Preserve the artifact error that caused the failure.
|
|
178
|
+
}
|
|
179
|
+
throw reportArtifactFailure(temporaryReady ? 'replace' : 'prepare', artifactPath, error);
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
function copyReportArtifact(sourcePath, artifactPath, temporaryPath) {
|
|
184
|
+
replaceReportArtifact(artifactPath, temporaryPath, (temporary) => {
|
|
185
|
+
fs.copyFileSync(sourcePath, temporary);
|
|
186
|
+
});
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
function writeReportArtifact(artifactPath, temporaryPath, content) {
|
|
190
|
+
replaceReportArtifact(artifactPath, temporaryPath, (temporary) => {
|
|
191
|
+
fs.writeFileSync(temporary, content, 'utf8');
|
|
192
|
+
});
|
|
193
|
+
}
|
|
194
|
+
|
|
154
195
|
function main() {
|
|
155
196
|
const program = new Command();
|
|
156
197
|
|
|
@@ -554,7 +595,8 @@ function main() {
|
|
|
554
595
|
}
|
|
555
596
|
const runStart = Date.now();
|
|
556
597
|
|
|
557
|
-
const
|
|
598
|
+
const copiedReportTemporaryPath = reportTemporaryPath(outputPath);
|
|
599
|
+
const normalizedReportTemporaryPath = reportTemporaryPath(outputPath);
|
|
558
600
|
let redirectDir = null;
|
|
559
601
|
let gateFailures = [];
|
|
560
602
|
|
|
@@ -581,28 +623,25 @@ function main() {
|
|
|
581
623
|
return () => heartbeat.kill();
|
|
582
624
|
};
|
|
583
625
|
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
teaConfig,
|
|
602
|
-
contextFiles,
|
|
603
|
-
contextBasis,
|
|
604
|
-
});
|
|
626
|
+
// Under isolation the agent writes into a fresh tmpdir. Artifact processing
|
|
627
|
+
// happens after withIsolation restores the project, so an atomic rename never
|
|
628
|
+
// needs a project directory to be writable while the agent is running.
|
|
629
|
+
if (isolationActive) {
|
|
630
|
+
redirectDir = fs.mkdtempSync(path.join(os.tmpdir(), 'tea-test-review-'));
|
|
631
|
+
}
|
|
632
|
+
const agentOutputPath = redirectDir ? path.join(redirectDir, 'test-review.md') : outputPath;
|
|
633
|
+
const prompt = buildPrompt({
|
|
634
|
+
skillRoot,
|
|
635
|
+
files: changedTestFiles,
|
|
636
|
+
outputPath: agentOutputPath,
|
|
637
|
+
scope: options.scope,
|
|
638
|
+
testDir: options.testDir,
|
|
639
|
+
teaConfig,
|
|
640
|
+
contextFiles,
|
|
641
|
+
contextBasis,
|
|
642
|
+
});
|
|
605
643
|
|
|
644
|
+
const executeAgent = ({ agentCwd, spawnPrefix }) => {
|
|
606
645
|
const stopHeartbeat = startHeartbeat();
|
|
607
646
|
let agentResult;
|
|
608
647
|
try {
|
|
@@ -622,41 +661,45 @@ function main() {
|
|
|
622
661
|
|
|
623
662
|
if (!fs.existsSync(agentOutputPath) || fs.statSync(agentOutputPath).mtimeMs <= runStart) {
|
|
624
663
|
printMissingReportDiagnostics(agentResult);
|
|
625
|
-
// Throw rather than fail()/process.exit() here: this runs inside
|
|
626
|
-
// withIsolation's callback, and exiting the process skips its finally
|
|
627
|
-
// block (restoreModes(), the chmod-fallback permission restore) along
|
|
628
|
-
// with the outer redirectDir cleanup below. Thrown errors propagate
|
|
629
|
-
// through both finally blocks before the outer catch calls fail().
|
|
630
664
|
const error = new Error(
|
|
631
665
|
`Agent finished but no fresh report was written to ${agentOutputPath}; refusing to parse a stale or missing report.`,
|
|
632
666
|
);
|
|
633
667
|
error.code = 'REPORT_MISSING';
|
|
634
668
|
throw error;
|
|
635
669
|
}
|
|
670
|
+
};
|
|
636
671
|
|
|
672
|
+
const processReport = () => {
|
|
637
673
|
// The report is copied back even when the verdict fails or parsing fails;
|
|
638
674
|
// on agent failure nothing was produced and nothing is copied.
|
|
639
675
|
if (redirectDir) {
|
|
640
|
-
|
|
676
|
+
copyReportArtifact(agentOutputPath, outputPath, copiedReportTemporaryPath);
|
|
641
677
|
}
|
|
642
678
|
|
|
679
|
+
const rawReport = fs.readFileSync(agentOutputPath, 'utf8');
|
|
643
680
|
let parsed;
|
|
644
681
|
try {
|
|
645
|
-
parsed = parseReport(
|
|
682
|
+
parsed = parseReport(rawReport, {
|
|
646
683
|
reviewedFiles: changedTestFiles,
|
|
647
684
|
contextFiles,
|
|
648
685
|
contextBasis,
|
|
649
686
|
});
|
|
650
687
|
} catch (error) {
|
|
651
688
|
if (error.code === 'REPORT_UNPARSEABLE') {
|
|
652
|
-
// Same reasoning as the freshness check above: throw, don't exit, so
|
|
653
|
-
// isolation cleanup runs before the process actually terminates.
|
|
654
689
|
const wrapped = new Error(`${error.message} (report: ${outputPath})`);
|
|
655
690
|
wrapped.code = 'REPORT_UNPARSEABLE';
|
|
656
691
|
throw wrapped;
|
|
657
692
|
}
|
|
658
693
|
throw error;
|
|
659
694
|
}
|
|
695
|
+
|
|
696
|
+
if (parsed.reportedQualityScore !== undefined) {
|
|
697
|
+
const normalizedReport = normalizeReportScore(rawReport, parsed.qualityScore);
|
|
698
|
+
writeReportArtifact(outputPath, normalizedReportTemporaryPath, normalizedReport);
|
|
699
|
+
console.error(
|
|
700
|
+
`tea-test-review: normalized agent Quality Score ${parsed.reportedQualityScore} to deterministic ledger score ${parsed.qualityScore}.`,
|
|
701
|
+
);
|
|
702
|
+
}
|
|
660
703
|
gateFailures = evaluateGates(parsed);
|
|
661
704
|
|
|
662
705
|
// The verdict JSON files manifest is the report's own Reviewed Files
|
|
@@ -676,41 +719,58 @@ function main() {
|
|
|
676
719
|
const finalPayload = applyWaiver(verdictPayload, gateFailures.length > 0);
|
|
677
720
|
console.log(JSON.stringify(finalPayload, null, 2));
|
|
678
721
|
|
|
679
|
-
if (
|
|
722
|
+
if (jsonPath) {
|
|
680
723
|
// No freshness check here, unlike the report: the CLI is the only writer of
|
|
681
724
|
// the verdict JSON and writeJsonFile already fails closed on a write error,
|
|
682
725
|
// so a stale verdict is not reachable. The pre-run rm still applies, so a
|
|
683
726
|
// failed run leaves no previous verdict behind.
|
|
684
|
-
writeJsonFile(
|
|
685
|
-
|
|
686
|
-
|
|
727
|
+
writeJsonFile(jsonPath, finalPayload);
|
|
728
|
+
}
|
|
729
|
+
};
|
|
730
|
+
|
|
731
|
+
const cleanupRunArtifacts = () => {
|
|
732
|
+
for (const temporaryPath of [copiedReportTemporaryPath, normalizedReportTemporaryPath]) {
|
|
733
|
+
try {
|
|
734
|
+
fs.rmSync(temporaryPath, { force: true });
|
|
735
|
+
} catch (error) {
|
|
736
|
+
console.error(`tea-test-review WARNING: failed to remove temporary report ${temporaryPath}: ${error.message}`);
|
|
687
737
|
}
|
|
688
738
|
}
|
|
739
|
+
if (redirectDir) {
|
|
740
|
+
try {
|
|
741
|
+
fs.rmSync(redirectDir, { recursive: true, force: true });
|
|
742
|
+
} catch (error) {
|
|
743
|
+
console.error(`tea-test-review WARNING: failed to remove temporary directory ${redirectDir}: ${error.message}`);
|
|
744
|
+
}
|
|
745
|
+
redirectDir = null;
|
|
746
|
+
}
|
|
689
747
|
};
|
|
690
748
|
|
|
691
749
|
try {
|
|
692
750
|
if (isolationActive) {
|
|
693
|
-
withIsolation(projectRoot,
|
|
751
|
+
withIsolation(projectRoot, [], executeAgent);
|
|
694
752
|
} else {
|
|
695
|
-
|
|
753
|
+
executeAgent({ agentCwd: projectRoot, spawnPrefix: [] });
|
|
696
754
|
}
|
|
755
|
+
processReport();
|
|
697
756
|
} catch (error) {
|
|
698
757
|
if (
|
|
699
758
|
error.code === 'AGENT_FAILED' ||
|
|
700
759
|
error.code === 'AGENT_NOT_FOUND' ||
|
|
701
760
|
error.code === 'REPORT_MISSING' ||
|
|
702
|
-
error.code === 'REPORT_UNPARSEABLE'
|
|
761
|
+
error.code === 'REPORT_UNPARSEABLE' ||
|
|
762
|
+
error.code === 'REPORT_ARTIFACT'
|
|
703
763
|
) {
|
|
764
|
+
cleanupRunArtifacts();
|
|
704
765
|
fail(EXIT.AGENT_OR_PARSE_ERROR, error.message);
|
|
705
766
|
}
|
|
706
767
|
if (error.code === 'ISOLATION_ERROR') {
|
|
768
|
+
cleanupRunArtifacts();
|
|
707
769
|
fail(EXIT.ENV_ERROR, error.message);
|
|
708
770
|
}
|
|
709
771
|
throw error;
|
|
710
772
|
} finally {
|
|
711
|
-
|
|
712
|
-
fs.rmSync(redirectDir, { recursive: true, force: true });
|
|
713
|
-
}
|
|
773
|
+
cleanupRunArtifacts();
|
|
714
774
|
}
|
|
715
775
|
|
|
716
776
|
for (const failure of gateFailures) {
|
|
@@ -121,7 +121,7 @@ Same logic on the report: a parse failure is never a silent pass, stale artifact
|
|
|
121
121
|
|
|
122
122
|
Prompt contract, parser, and report template are one contract in three files; they version together.
|
|
123
123
|
|
|
124
|
-
Clearest example: the scoring model. The report template defines the deduction ledger, the prompt states its arithmetic, the parser
|
|
124
|
+
Clearest example: the scoring model. The report template defines the deduction ledger, the prompt states its arithmetic, and the parser computes the authoritative score from the declared violations and bonus. The CLI then normalizes the report before gating, so model arithmetic cannot break CI or control the verdict. Shipping CLI and skill in the same package, same version, keeps that contract synchronized.
|
|
125
125
|
|
|
126
126
|
---
|
|
127
127
|
|
|
@@ -228,7 +228,7 @@ The report itself is strictly validated:
|
|
|
228
228
|
|
|
229
229
|
Fenced code blocks are stripped first, so a quoted example can't spoof a verdict. Markdown emphasis is stripped only where it wraps a whole value, so `tests/user_profile.spec.ts` survives the manifest intact.
|
|
230
230
|
|
|
231
|
-
The score is
|
|
231
|
+
The score is computed rather than trusted: the CLI evaluates `100 - (Critical × 10 + High × 5 + Medium × 2 + Low × 1) + Total Bonus` from the report's violation counts and bonus. That result becomes the verdict score, and the CLI normalizes the report's score and grade fields before publishing it. The original model value is retained as `reportedQualityScore` metadata when corrected. An invalid bonus total or missing breakdown still fails closed. The prompt states this same arithmetic.
|
|
232
232
|
|
|
233
233
|
A report declaring Critical violations alongside an approve-type recommendation is rejected as an inconsistent verdict (exit 3): Critical means Must Fix. Stale artifacts are never parsed, output files are deleted before the run and must be freshly written by it.
|
|
234
234
|
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"$schema": "https://json.schemastore.org/package.json",
|
|
3
3
|
"name": "bmad-method-test-architecture-enterprise",
|
|
4
|
-
"version": "1.21.
|
|
4
|
+
"version": "1.21.3",
|
|
5
5
|
"description": "Master Test Architect for quality strategy, test automation, and release gates",
|
|
6
6
|
"keywords": [
|
|
7
7
|
"bmad",
|
|
@@ -99,7 +99,7 @@ Grade: {grade}
|
|
|
99
99
|
Every bonus line is 0 or 5, never a partial value, and the six categories above are the
|
|
100
100
|
complete set. {grade} is exactly one of A, B, C, D, F, with no modifier such as A+ or B-.
|
|
101
101
|
The lines above must sum to {final_score}, which must equal the **Quality Score** line;
|
|
102
|
-
headless runners
|
|
102
|
+
headless runners compute the authoritative result and normalize score and grade fields. -->
|
|
103
103
|
|
|
104
104
|
---
|
|
105
105
|
|
|
@@ -5,17 +5,21 @@ stepsCompleted:
|
|
|
5
5
|
- step-04-generate-report
|
|
6
6
|
---
|
|
7
7
|
|
|
8
|
-
# Test Quality Review:
|
|
8
|
+
# Test Quality Review: alert-preferences-dogfood.spec.ts
|
|
9
9
|
|
|
10
|
-
|
|
11
|
-
**
|
|
10
|
+
```markdown
|
|
11
|
+
**Quality Score**: 42/100 (F - Example only)
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
**Quality Score**: 86/100 (B)
|
|
15
|
+
**Review Date**: 2026-08-04
|
|
12
16
|
**Review Scope**: single
|
|
13
17
|
|
|
14
18
|
## Executive Summary
|
|
15
19
|
|
|
16
|
-
**Overall Assessment**:
|
|
20
|
+
**Overall Assessment**: Needs Improvement
|
|
17
21
|
|
|
18
|
-
**Recommendation**: Approve
|
|
22
|
+
**Recommendation**: Approve with Comments
|
|
19
23
|
|
|
20
24
|
**Context Basis**: none
|
|
21
25
|
|
|
@@ -23,11 +27,11 @@ stepsCompleted:
|
|
|
23
27
|
|
|
24
28
|
### Summary
|
|
25
29
|
|
|
26
|
-
Two High violations deduct
|
|
27
|
-
|
|
28
|
-
|
|
30
|
+
Two High and two Medium violations deduct 14. A five-point bonus makes the
|
|
31
|
+
authoritative score 91, while this live Codex report published 86 because it
|
|
32
|
+
forgot to add the bonus.
|
|
29
33
|
|
|
30
|
-
**Total Violations**: 0 Critical, 2 High,
|
|
34
|
+
**Total Violations**: 0 Critical, 2 High, 2 Medium, 0 Low
|
|
31
35
|
|
|
32
36
|
## Quality Score Breakdown
|
|
33
37
|
|
|
@@ -35,19 +39,19 @@ does not sum to the score printed above it.
|
|
|
35
39
|
Starting Score: 100
|
|
36
40
|
Critical Violations: -0 × 10 = -0
|
|
37
41
|
High Violations: -2 × 5 = -10
|
|
38
|
-
Medium Violations: -
|
|
42
|
+
Medium Violations: -2 × 2 = -4
|
|
39
43
|
Low Violations: -0 × 1 = -0
|
|
40
44
|
|
|
41
|
-
Total Bonus: +
|
|
45
|
+
Total Bonus: +5
|
|
42
46
|
|
|
43
|
-
Final Score:
|
|
44
|
-
Grade:
|
|
47
|
+
Final Score: 86/100
|
|
48
|
+
Grade: B
|
|
45
49
|
```
|
|
46
50
|
|
|
47
51
|
## Decision
|
|
48
52
|
|
|
49
|
-
|
|
53
|
+
Recommendation: Approve with Comments
|
|
50
54
|
|
|
51
55
|
## Reviewed Files
|
|
52
56
|
|
|
53
|
-
tests/
|
|
57
|
+
playwright/tests/api/alert-preferences-dogfood.spec.ts
|
|
@@ -10,7 +10,8 @@
|
|
|
10
10
|
*
|
|
11
11
|
* STUB_MODE approve (default) | approve-low | block | request-changes |
|
|
12
12
|
* request-changes-critical | critical-approve | conflict |
|
|
13
|
-
* partial | nothing | fail |
|
|
13
|
+
* score-mismatch | partial | nothing | fail |
|
|
14
|
+
* forbidden-write | stale-copy
|
|
14
15
|
* STUB_ASSERT_STDIN when "1", fail if the prompt did not arrive on stdin or
|
|
15
16
|
* if any of it leaked into argv
|
|
16
17
|
* STUB_ASSERT_MODEL expected model value; fail unless argv carries a model
|
|
@@ -22,6 +23,8 @@
|
|
|
22
23
|
* manifest. Adversarial tests use it to prove the CLI
|
|
23
24
|
* binds report claims to the supplied context set.
|
|
24
25
|
* STUB_OLD_REPORT stale-copy source: a report pre-placed with an old mtime
|
|
26
|
+
* STUB_LOCK_OUTPUT when "1", make the report and its parent read-only after
|
|
27
|
+
* writing so artifact replacement failure handling runs
|
|
25
28
|
*/
|
|
26
29
|
|
|
27
30
|
const fs = require('node:fs');
|
|
@@ -35,6 +38,7 @@ const REPORTS = {
|
|
|
35
38
|
'request-changes-critical': 'request-changes-critical.md',
|
|
36
39
|
'critical-approve': 'critical-approve.md',
|
|
37
40
|
conflict: 'conflicting.md',
|
|
41
|
+
'score-mismatch': 'score-mismatch.md',
|
|
38
42
|
partial: 'malformed.md',
|
|
39
43
|
};
|
|
40
44
|
|
|
@@ -160,6 +164,10 @@ function writeFixtureReport(fixtureName) {
|
|
|
160
164
|
fs.mkdirSync(path.dirname(outputPath), { recursive: true });
|
|
161
165
|
const report = fs.readFileSync(path.join(__dirname, 'reports', fixtureName), 'utf8');
|
|
162
166
|
fs.writeFileSync(outputPath, bindReportToPrompt(report));
|
|
167
|
+
if (process.env.STUB_LOCK_OUTPUT === '1') {
|
|
168
|
+
fs.chmodSync(outputPath, 0o444);
|
|
169
|
+
fs.chmodSync(path.dirname(outputPath), 0o555);
|
|
170
|
+
}
|
|
163
171
|
}
|
|
164
172
|
|
|
165
173
|
if (mode === 'forbidden-write') {
|
|
@@ -47,7 +47,7 @@ const { spawnSync } = require('node:child_process');
|
|
|
47
47
|
const vm = require('node:vm');
|
|
48
48
|
const yaml = require('js-yaml');
|
|
49
49
|
|
|
50
|
-
const { parseReport, verdictFor, scoreFails, CONTEXT_BASIS_ENUM } = require('../cli/lib/parse-report');
|
|
50
|
+
const { parseReport, normalizeReportScore, verdictFor, scoreFails, CONTEXT_BASIS_ENUM } = require('../cli/lib/parse-report');
|
|
51
51
|
const {
|
|
52
52
|
isTestFile,
|
|
53
53
|
isContextNoise,
|
|
@@ -353,7 +353,6 @@ async function runTests() {
|
|
|
353
353
|
['missing-violations.md', 'no Total Violations line'],
|
|
354
354
|
['missing-frontmatter.md', 'no YAML frontmatter'],
|
|
355
355
|
['empty-steps-flow.md', 'wrapped stepsCompleted flow sequence with no entries'],
|
|
356
|
-
['score-mismatch.md', 'published score contradicts its own deduction ledger'],
|
|
357
356
|
['bonus-not-multiple.md', 'bonus total is not a multiple of the 5-point category value'],
|
|
358
357
|
['missing-breakdown.md', 'no Quality Score Breakdown, so the score cannot be recomputed'],
|
|
359
358
|
['duplicate-breakdown-heading.md', 'two Quality Score Breakdown headings, so neither can be trusted as the real ledger'],
|
|
@@ -373,6 +372,45 @@ async function runTests() {
|
|
|
373
372
|
}
|
|
374
373
|
}
|
|
375
374
|
|
|
375
|
+
// Regression from couture-cast run 30897431283: Codex correctly declared
|
|
376
|
+
// its deductions and bonus, then published arithmetic that omitted the
|
|
377
|
+
// bonus. Derived
|
|
378
|
+
// arithmetic belongs to the CLI, while the model remains responsible for
|
|
379
|
+
// the findings, severity counts, and bonus declarations.
|
|
380
|
+
try {
|
|
381
|
+
const source = readFixture('reports', 'score-mismatch.md');
|
|
382
|
+
const corrected = parseReport(source);
|
|
383
|
+
assert(
|
|
384
|
+
corrected.qualityScore === 91 && corrected.reportedQualityScore === 86,
|
|
385
|
+
'score-mismatch fixture: CLI derives 91/A and preserves the agent-reported 86/B score as metadata',
|
|
386
|
+
JSON.stringify(corrected),
|
|
387
|
+
);
|
|
388
|
+
const normalized = normalizeReportScore(source, corrected.qualityScore);
|
|
389
|
+
assert(
|
|
390
|
+
normalized.includes('**Quality Score**: 42/100 (F - Example only)') &&
|
|
391
|
+
normalized.includes('**Quality Score**: 91/100 (A)') &&
|
|
392
|
+
normalized.includes('Final Score: 91/100') &&
|
|
393
|
+
normalized.includes('Grade: A'),
|
|
394
|
+
'score-mismatch fixture: every active score and grade field normalizes while an earlier fenced example stays untouched',
|
|
395
|
+
normalized,
|
|
396
|
+
);
|
|
397
|
+
} catch (error) {
|
|
398
|
+
assert(false, 'score-mismatch fixture is corrected deterministically', error.message);
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
try {
|
|
402
|
+
const source = readFixture('reports', 'approve.md').replace('93/100 (A)', '93/100 (F)');
|
|
403
|
+
const corrected = parseReport(source);
|
|
404
|
+
const normalized = normalizeReportScore(source, corrected.qualityScore);
|
|
405
|
+
assert(
|
|
406
|
+
corrected.qualityScore === 93 && corrected.reportedQualityScore === 93 && normalized.includes('**Quality Score**: 93/100 (A)'),
|
|
407
|
+
'grade-only mismatch triggers normalization even when the reported numeric score is correct',
|
|
408
|
+
JSON.stringify(corrected),
|
|
409
|
+
);
|
|
410
|
+
} catch (error) {
|
|
411
|
+
assert(false, 'grade-only score mismatch is corrected deterministically', error.message);
|
|
412
|
+
}
|
|
413
|
+
|
|
376
414
|
try {
|
|
377
415
|
parseReport(readFixture('reports', 'conflicting.md'));
|
|
378
416
|
assert(false, 'conflicting fixture error message calls out the conflict');
|
|
@@ -1164,15 +1202,16 @@ async function runTests() {
|
|
|
1164
1202
|
'prompt requires Quality Score 0-100',
|
|
1165
1203
|
);
|
|
1166
1204
|
assert(prompt.includes('**Total Violations**: line is required'), 'prompt requires the Total Violations line');
|
|
1167
|
-
// The parser
|
|
1168
|
-
//
|
|
1205
|
+
// The parser computes the authoritative score, so the prompt has to state
|
|
1206
|
+
// the same model and make clear that agent arithmetic is provisional.
|
|
1169
1207
|
assert(
|
|
1170
|
-
prompt.includes('"## Quality Score Breakdown" section is required
|
|
1171
|
-
|
|
1208
|
+
prompt.includes('"## Quality Score Breakdown" section is required') &&
|
|
1209
|
+
prompt.includes('replaces it with the deterministic ledger result before gating'),
|
|
1210
|
+
'prompt identifies the ledger as the CLI-owned score source',
|
|
1172
1211
|
);
|
|
1173
1212
|
assert(
|
|
1174
1213
|
prompt.includes('100 - (Critical×10 + High×5 + Medium×2 + Low×1) + Total Bonus'),
|
|
1175
|
-
'prompt states the deduction ledger the CLI
|
|
1214
|
+
'prompt states the deduction ledger the CLI computes',
|
|
1176
1215
|
);
|
|
1177
1216
|
assert(prompt.includes('multiple of 5 from 0 to 30'), 'prompt bounds the bonus total to legal category values');
|
|
1178
1217
|
assert(prompt.includes('exactly one of A, B, C, D, F'), 'prompt bounds the grade scale');
|
|
@@ -1751,6 +1790,92 @@ async function runTests() {
|
|
|
1751
1790
|
assert(false, 'verdict JSON parses', error.message);
|
|
1752
1791
|
}
|
|
1753
1792
|
|
|
1793
|
+
const normalizedScoreOut = path.join(tmpRoot, 'normalized-score-run', 'test-review.md');
|
|
1794
|
+
const normalizedScoreJsonPath = path.join(tmpRoot, 'normalized-score-run', 'verdict.json');
|
|
1795
|
+
const normalizedScoreRun = runCli(
|
|
1796
|
+
[
|
|
1797
|
+
'--files',
|
|
1798
|
+
'playwright/tests/api/alert-preferences-dogfood.spec.ts',
|
|
1799
|
+
'--project-root',
|
|
1800
|
+
fixtureProject,
|
|
1801
|
+
'--output',
|
|
1802
|
+
normalizedScoreOut,
|
|
1803
|
+
'--json',
|
|
1804
|
+
normalizedScoreJsonPath,
|
|
1805
|
+
'--agent-cmd',
|
|
1806
|
+
stubAgent,
|
|
1807
|
+
'--no-isolate',
|
|
1808
|
+
...stubPass('STUB_MODE'),
|
|
1809
|
+
],
|
|
1810
|
+
{ STUB_MODE: 'score-mismatch' },
|
|
1811
|
+
);
|
|
1812
|
+
assert(
|
|
1813
|
+
normalizedScoreRun.status === 0 && normalizedScoreRun.stderr.includes('normalized agent Quality Score 86'),
|
|
1814
|
+
'score arithmetic mismatch is normalized instead of failing the run',
|
|
1815
|
+
`status=${normalizedScoreRun.status} stderr=${normalizedScoreRun.stderr}`,
|
|
1816
|
+
);
|
|
1817
|
+
try {
|
|
1818
|
+
const normalizedPayload = JSON.parse(fs.readFileSync(normalizedScoreJsonPath, 'utf8'));
|
|
1819
|
+
const normalizedReport = fs.readFileSync(normalizedScoreOut, 'utf8');
|
|
1820
|
+
assert(
|
|
1821
|
+
normalizedPayload.qualityScore === 91 && normalizedPayload.reportedQualityScore === 86,
|
|
1822
|
+
'normalized verdict JSON uses the CLI score crossing into grade A and preserves the agent score',
|
|
1823
|
+
JSON.stringify(normalizedPayload),
|
|
1824
|
+
);
|
|
1825
|
+
assert(
|
|
1826
|
+
normalizedReport.includes('**Quality Score**: 42/100 (F - Example only)') &&
|
|
1827
|
+
normalizedReport.includes('**Quality Score**: 91/100 (A)') &&
|
|
1828
|
+
normalizedReport.includes('Final Score: 91/100') &&
|
|
1829
|
+
normalizedReport.includes('Grade: A'),
|
|
1830
|
+
'normalized report publishes the same derived score and grade as the verdict JSON while preserving fenced examples',
|
|
1831
|
+
normalizedReport,
|
|
1832
|
+
);
|
|
1833
|
+
} catch (error) {
|
|
1834
|
+
assert(false, 'normalized score artifacts are readable', error.message);
|
|
1835
|
+
}
|
|
1836
|
+
|
|
1837
|
+
const artifactPermissionsTestable = process.platform !== 'win32' && !(typeof process.getuid === 'function' && process.getuid() === 0);
|
|
1838
|
+
if (artifactPermissionsTestable) {
|
|
1839
|
+
const lockedArtifactDir = path.join(tmpRoot, 'locked-normalized-score-run');
|
|
1840
|
+
const lockedArtifactOut = path.join(lockedArtifactDir, 'test-review.md');
|
|
1841
|
+
const lockedArtifactRun = runCli(
|
|
1842
|
+
[
|
|
1843
|
+
'--files',
|
|
1844
|
+
'playwright/tests/api/alert-preferences-dogfood.spec.ts',
|
|
1845
|
+
'--project-root',
|
|
1846
|
+
fixtureProject,
|
|
1847
|
+
'--output',
|
|
1848
|
+
lockedArtifactOut,
|
|
1849
|
+
'--agent-cmd',
|
|
1850
|
+
stubAgent,
|
|
1851
|
+
'--no-isolate',
|
|
1852
|
+
...stubPass('STUB_MODE', 'STUB_LOCK_OUTPUT'),
|
|
1853
|
+
],
|
|
1854
|
+
{ STUB_MODE: 'score-mismatch', STUB_LOCK_OUTPUT: '1' },
|
|
1855
|
+
);
|
|
1856
|
+
let preservedArtifact = '';
|
|
1857
|
+
try {
|
|
1858
|
+
preservedArtifact = fs.readFileSync(lockedArtifactOut, 'utf8');
|
|
1859
|
+
} finally {
|
|
1860
|
+
fs.chmodSync(lockedArtifactDir, 0o755);
|
|
1861
|
+
if (fs.existsSync(lockedArtifactOut)) {
|
|
1862
|
+
fs.chmodSync(lockedArtifactOut, 0o644);
|
|
1863
|
+
}
|
|
1864
|
+
}
|
|
1865
|
+
assert(
|
|
1866
|
+
lockedArtifactRun.status === 3 && lockedArtifactRun.stderr.includes('Failed to prepare report artifact'),
|
|
1867
|
+
'normalized report write failures are classified as report-artifact failures',
|
|
1868
|
+
`status=${lockedArtifactRun.status} stderr=${lockedArtifactRun.stderr}`,
|
|
1869
|
+
);
|
|
1870
|
+
assert(
|
|
1871
|
+
preservedArtifact.includes('**Quality Score**: 86/100 (B)') && !preservedArtifact.includes('**Quality Score**: 91/100 (A)'),
|
|
1872
|
+
'failed normalized report writes preserve the original agent artifact',
|
|
1873
|
+
preservedArtifact,
|
|
1874
|
+
);
|
|
1875
|
+
} else {
|
|
1876
|
+
skip('normalized report write failure preserves the original artifact', 'filesystem permissions cannot be enforced');
|
|
1877
|
+
}
|
|
1878
|
+
|
|
1754
1879
|
// --agent selects the adapter (codex here, not just the claude default);
|
|
1755
1880
|
// --agent-cmd still only overrides the executable on top of it.
|
|
1756
1881
|
const codexAdapterOut = path.join(tmpRoot, 'codex-adapter-run', 'test-review.md');
|