bmad-method-test-architecture-enterprise 1.19.2-next.0 → 1.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. package/.claude-plugin/marketplace.json +2 -10
  2. package/.github/workflows/publish.yaml +1 -1
  3. package/.github/workflows/quality.yaml +3 -0
  4. package/.github/workflows/tea-test-review.yaml +228 -0
  5. package/CHANGELOG.md +26 -0
  6. package/cli/examples/README.md +53 -0
  7. package/cli/examples/pr-test-review.yml +279 -0
  8. package/cli/lib/agent-adapters.js +51 -0
  9. package/cli/lib/build-prompt.js +110 -0
  10. package/cli/lib/changed-tests.js +258 -0
  11. package/cli/lib/isolate.js +353 -0
  12. package/cli/lib/parse-report.js +380 -0
  13. package/cli/lib/resolve-skill.js +43 -0
  14. package/cli/lib/resolve-tea-config.js +171 -0
  15. package/cli/lib/run-agent.js +132 -0
  16. package/cli/test-review.js +607 -0
  17. package/docs/explanation/test-review-cli-architecture.md +120 -0
  18. package/docs/reference/tea-test-review-cli.md +147 -0
  19. package/eslint.config.mjs +12 -2
  20. package/package.json +5 -2
  21. package/src/workflows/testarch/bmad-testarch-test-review/SKILL.md +11 -0
  22. package/src/workflows/testarch/bmad-testarch-test-review/checklist.md +3 -3
  23. package/src/workflows/testarch/bmad-testarch-test-review/customize.toml +28 -0
  24. package/src/workflows/testarch/bmad-testarch-test-review/instructions.md +4 -0
  25. package/src/workflows/testarch/bmad-testarch-test-review/steps-c/step-01-load-context.md +5 -2
  26. package/src/workflows/testarch/bmad-testarch-test-review/steps-c/step-01b-resume.md +1 -1
  27. package/src/workflows/testarch/bmad-testarch-test-review/steps-c/step-02-discover-tests.md +4 -1
  28. package/src/workflows/testarch/bmad-testarch-test-review/steps-c/step-03-quality-evaluation.md +1 -1
  29. package/src/workflows/testarch/bmad-testarch-test-review/steps-c/step-03f-aggregate-scores.md +76 -47
  30. package/src/workflows/testarch/bmad-testarch-test-review/steps-c/step-04-generate-report.md +2 -1
  31. package/src/workflows/testarch/bmad-testarch-test-review/test-review-template.md +15 -0
  32. package/src/workflows/testarch/bmad-testarch-test-review/workflow.yaml +6 -0
  33. package/test/README.md +43 -9
  34. package/test/fixtures/test-review-cli/project/_bmad/tea/workflows/testarch/bmad-testarch-test-review/SKILL.md +1 -0
  35. package/test/fixtures/test-review-cli/project-claude/.claude/skills/bmad-testarch-test-review/SKILL.md +1 -0
  36. package/test/fixtures/test-review-cli/project-empty/.gitkeep +0 -0
  37. package/test/fixtures/test-review-cli/reports/approve-low-score.md +49 -0
  38. package/test/fixtures/test-review-cli/reports/approve.md +49 -0
  39. package/test/fixtures/test-review-cli/reports/bad-value.md +34 -0
  40. package/test/fixtures/test-review-cli/reports/block.md +48 -0
  41. package/test/fixtures/test-review-cli/reports/bonus-not-multiple.md +54 -0
  42. package/test/fixtures/test-review-cli/reports/colon-in-bold.md +47 -0
  43. package/test/fixtures/test-review-cli/reports/conflicting.md +34 -0
  44. package/test/fixtures/test-review-cli/reports/critical-approve.md +34 -0
  45. package/test/fixtures/test-review-cli/reports/duplicate-breakdown-heading.md +68 -0
  46. package/test/fixtures/test-review-cli/reports/empty-steps-flow.md +44 -0
  47. package/test/fixtures/test-review-cli/reports/fenced-recommendation.md +55 -0
  48. package/test/fixtures/test-review-cli/reports/key-strengths-weaknesses.md +60 -0
  49. package/test/fixtures/test-review-cli/reports/lowercase.md +49 -0
  50. package/test/fixtures/test-review-cli/reports/malformed.md +12 -0
  51. package/test/fixtures/test-review-cli/reports/missing-breakdown.md +34 -0
  52. package/test/fixtures/test-review-cli/reports/missing-decision.md +49 -0
  53. package/test/fixtures/test-review-cli/reports/missing-frontmatter.md +41 -0
  54. package/test/fixtures/test-review-cli/reports/missing-reviewed-files.md +45 -0
  55. package/test/fixtures/test-review-cli/reports/missing-score.md +48 -0
  56. package/test/fixtures/test-review-cli/reports/missing-violations.md +32 -0
  57. package/test/fixtures/test-review-cli/reports/plain-bullets-key-strengths.md +58 -0
  58. package/test/fixtures/test-review-cli/reports/request-changes-critical.md +49 -0
  59. package/test/fixtures/test-review-cli/reports/request-changes.md +49 -0
  60. package/test/fixtures/test-review-cli/reports/score-140.md +34 -0
  61. package/test/fixtures/test-review-cli/reports/score-mismatch.md +49 -0
  62. package/test/fixtures/test-review-cli/reports/wrapped-steps-flow.md +54 -0
  63. package/test/fixtures/test-review-cli/stub-agent.js +108 -0
  64. package/test/test-test-review-cli.js +2516 -0
  65. package/website/astro.config.mjs +2 -0
@@ -6,21 +6,13 @@
6
6
  "license": "MIT",
7
7
  "homepage": "https://github.com/bmad-code-org/bmad-method-test-architecture-enterprise",
8
8
  "repository": "https://github.com/bmad-code-org/bmad-method-test-architecture-enterprise",
9
- "keywords": [
10
- "bmad",
11
- "test-architect",
12
- "testing",
13
- "quality",
14
- "automation",
15
- "playwright",
16
- "test-engineering"
17
- ],
9
+ "keywords": ["bmad", "test-architect", "testing", "quality", "automation", "playwright", "test-engineering"],
18
10
  "plugins": [
19
11
  {
20
12
  "name": "bmad-method-test-architecture-enterprise",
21
13
  "source": "./",
22
14
  "description": "Master Test Architect module for quality strategy, test automation, CI/CD quality gates, and structured testing education. Part of the BMad Method ecosystem.",
23
- "version": "1.19.2-next.0",
15
+ "version": "1.20.0",
24
16
  "author": {
25
17
  "name": "Murat K Ozcan (TEA Creator) & Brian (BMad) Madison"
26
18
  },
@@ -76,7 +76,7 @@ jobs:
76
76
  run: npm ci
77
77
 
78
78
  - name: Run tests
79
- run: npm test
79
+ run: npm test && npm run test:cli
80
80
 
81
81
  - name: Derive next prerelease version
82
82
  if: github.event_name == 'push' || (github.event_name == 'workflow_dispatch' && inputs.channel == 'next')
@@ -115,3 +115,6 @@ jobs:
115
115
 
116
116
  - name: Test agent compilation components
117
117
  run: npm run test:install
118
+
119
+ - name: Test TEA test-review CLI
120
+ run: npm run test:cli
@@ -0,0 +1,228 @@
1
+ # Dogfoods tea-test-review on this repo's own test suite (test/*.js). Unlike
2
+ # cli/examples/pr-test-review.yml, this repo IS the skill source, so it points
3
+ # --skill-root at src/workflows/testarch/bmad-testarch-test-review directly
4
+ # instead of installing a pinned npm tarball.
5
+ #
6
+ # NOT YET ENABLED: no ANTHROPIC_API_KEY or CLAUDE_CODE_OAUTH_TOKEN repository
7
+ # secret exists yet, so the agent-running steps below are commented out and
8
+ # the trigger is workflow_dispatch only, not pull_request. A run without
9
+ # either secret would just fail and post a confusing "infrastructure failure"
10
+ # comment on every real PR.
11
+ #
12
+ # To enable: add one of the two secrets, uncomment the "Install the pinned
13
+ # agent CLI" and "Run headless test review" steps, and switch `on:` back to
14
+ # `pull_request: { types: [opened, synchronize, reopened] }`.
15
+ # - ANTHROPIC_API_KEY: pay-per-token via the Anthropic Console.
16
+ # - CLAUDE_CODE_OAUTH_TOKEN: a long-lived token from an existing Claude
17
+ # subscription, generated locally with `claude setup-token`. No separate
18
+ # API billing. Already supported by cli/lib/run-agent.js's env allowlist.
19
+ #
20
+ # Honest limitation: forks receive no secrets, so the review skips for fork
21
+ # PRs (the fork guard below) rather than failing. A skipped required check
22
+ # still satisfies branch protection, so this cannot be the sole gate against
23
+ # untrusted external contributions; see cli/examples/pr-test-review.yml's
24
+ # header comment for the same caveat.
25
+ name: TEA Test Review
26
+
27
+ on:
28
+ workflow_dispatch: {}
29
+
30
+ permissions: {}
31
+
32
+ jobs:
33
+ review:
34
+ name: Headless test review
35
+ runs-on: ubuntu-latest
36
+ # Self-adapting fork guard: on workflow_dispatch there's no pull_request
37
+ # context to check, so this is trivially true; on pull_request it skips
38
+ # forks (no secrets there). No change needed when switching triggers back.
39
+ if: github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository
40
+ permissions:
41
+ contents: read
42
+ outputs:
43
+ verdict: ${{ steps.verdict.outputs.verdict }}
44
+ steps:
45
+ - name: Checkout (full history for the PR diff)
46
+ uses: actions/checkout@v5
47
+ with:
48
+ fetch-depth: 0
49
+ persist-credentials: false
50
+
51
+ - name: Setup Node
52
+ uses: actions/setup-node@v6
53
+ with:
54
+ node-version-file: ".nvmrc"
55
+ cache: "npm"
56
+
57
+ - name: Install dependencies
58
+ run: npm ci
59
+
60
+ # - name: Install the pinned agent CLI
61
+ # env:
62
+ # CLAUDE_CODE_VERSION: 2.1.220
63
+ # run: npm install --global "@anthropic-ai/claude-code@${CLAUDE_CODE_VERSION}"
64
+
65
+ # - name: Run headless test review
66
+ # id: run-review
67
+ # env:
68
+ # ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
69
+ # # or: CLAUDE_CODE_OAUTH_TOKEN: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
70
+ # BASE_REF: origin/${{ github.base_ref }}
71
+ # run: node cli/test-review.js --base "$BASE_REF" --agent claude --skill-root src/workflows/testarch/bmad-testarch-test-review --test-dir test --output test-review.md --json test-review.json
72
+
73
+ - name: Upload review artifacts
74
+ if: always()
75
+ uses: actions/upload-artifact@v4
76
+ with:
77
+ name: tea-test-review
78
+ path: |
79
+ test-review.md
80
+ test-review.json
81
+
82
+ - name: Export the verdict as a job output
83
+ id: verdict
84
+ if: always()
85
+ env:
86
+ REVIEW_OUTCOME: ${{ steps.run-review.outcome }}
87
+ run: |
88
+ verdict="failed"
89
+ if [ "$REVIEW_OUTCOME" = "success" ]; then
90
+ verdict="passed"
91
+ fi
92
+ echo "verdict=$verdict" >> "$GITHUB_OUTPUT"
93
+
94
+ comment:
95
+ name: Comment the outcome
96
+ needs: review
97
+ # Only meaningful with a PR to comment on; skips cleanly on workflow_dispatch.
98
+ if: always() && github.event_name == 'pull_request'
99
+ runs-on: ubuntu-latest
100
+ permissions:
101
+ pull-requests: write
102
+ steps:
103
+ - name: Download review artifacts
104
+ id: download
105
+ uses: actions/download-artifact@v4
106
+ continue-on-error: true
107
+ with:
108
+ name: tea-test-review
109
+
110
+ - name: Find-and-update the review comment
111
+ uses: actions/github-script@v7
112
+ env:
113
+ DOWNLOAD_OUTCOME: ${{ steps.download.outcome }}
114
+ REVIEW_RESULT: ${{ needs.review.result }}
115
+ REVIEW_VERDICT: ${{ needs.review.outputs.verdict }}
116
+ with:
117
+ # Kept in sync by hand with cli/examples/pr-test-review.yml's
118
+ # "Find-and-update the review comment" step: same comment-building
119
+ # logic, duplicated because that file is a standalone copy-paste
120
+ # template for other repos and cannot depend on this one.
121
+ script: |
122
+ const fs = require("fs");
123
+ const marker = "<!-- tea-test-review -->";
124
+ const artifactsUrl = `https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}#artifacts`;
125
+ // GitHub caps a comment body at 65536 chars; this leaves headroom
126
+ // for the digest and wrapper around the inlined report.
127
+ const MAX_INLINE_REPORT_CHARS = 40000;
128
+
129
+ let body = null;
130
+ const artifactAvailable = process.env.DOWNLOAD_OUTCOME === "success" && fs.existsSync("test-review.json");
131
+ if (artifactAvailable) {
132
+ try {
133
+ const verdict = JSON.parse(fs.readFileSync("test-review.json", "utf8"));
134
+ if (verdict.skipped) {
135
+ body = [
136
+ marker,
137
+ "## TEA Test Review: skipped",
138
+ "",
139
+ `${verdict.reason ?? "No changed test files in this PR"}.`,
140
+ "",
141
+ `[Review job artifacts](${artifactsUrl})`,
142
+ ].join("\n");
143
+ } else {
144
+ const counts = verdict.violations ?? {};
145
+ const violations = `${counts.critical ?? 0} Critical / ${counts.high ?? 0} High / ${counts.medium ?? 0} Medium / ${counts.low ?? 0} Low`;
146
+ const weaknesses = (verdict.keyWeaknesses ?? []).slice(0, 3);
147
+
148
+ const lines = [
149
+ marker,
150
+ `## TEA Test Review: ${verdict.recommendation}`,
151
+ "",
152
+ `- **Quality score**: ${verdict.qualityScore ?? "n/a"}/100`,
153
+ `- **Recommendation**: ${verdict.recommendation}`,
154
+ `- **Violations**: ${violations}`,
155
+ `- **Reviewed files**: ${(verdict.reviewedFiles ?? []).length}`,
156
+ ];
157
+ if (weaknesses.length > 0) {
158
+ lines.push("", "**Key weaknesses**:", ...weaknesses.map((w) => `- ${w}`));
159
+ }
160
+ lines.push("");
161
+
162
+ // Inline the full report (not just the digest above) so a
163
+ // reviewer can paste it straight into an AI coding agent to
164
+ // apply the fixes, no artifact download required. Falls back
165
+ // to the artifact link alone if the report is too large or
166
+ // missing from the download.
167
+ const reportText = fs.existsSync("test-review.md") ? fs.readFileSync("test-review.md", "utf8") : null;
168
+ if (reportText && reportText.length <= MAX_INLINE_REPORT_CHARS) {
169
+ // A reviewed file's own content can contain a literal
170
+ // </details>; inserting a zero-width space breaks that as
171
+ // an HTML closing tag while leaving the visible text
172
+ // effectively unchanged, so it cannot end this block early
173
+ // and spill raw markdown into the rest of the comment.
174
+ const safeReportText = reportText.replace(/<\/details>/gi, "<\u200B/details>");
175
+ lines.push(
176
+ "<details>",
177
+ "<summary>Full report (paste into your AI coding agent to apply the fixes)</summary>",
178
+ "",
179
+ safeReportText,
180
+ "",
181
+ "</details>",
182
+ "",
183
+ `[Verdict JSON](${artifactsUrl})`,
184
+ );
185
+ } else {
186
+ lines.push(`[Full report and verdict JSON](${artifactsUrl})`);
187
+ }
188
+ body = lines.join("\n");
189
+ }
190
+ } catch {
191
+ body = null;
192
+ }
193
+ }
194
+ if (body === null) {
195
+ body = [
196
+ marker,
197
+ "## TEA Test Review: infrastructure failure",
198
+ "",
199
+ "The review job did not produce a readable verdict artifact.",
200
+ "This is **not** a review verdict; treat the gate as broken, not as approved tests.",
201
+ "",
202
+ `Review job result: ${process.env.REVIEW_RESULT} (verdict output: ${process.env.REVIEW_VERDICT || "n/a"}).`,
203
+ `[Review job artifacts](${artifactsUrl})`,
204
+ ].join("\n");
205
+ }
206
+
207
+ const { data: comments } = await github.rest.issues.listComments({
208
+ owner: context.repo.owner,
209
+ repo: context.repo.repo,
210
+ issue_number: context.issue.number,
211
+ per_page: 100,
212
+ });
213
+ const existing = comments.find((comment) => comment.body && comment.body.includes(marker));
214
+ if (existing) {
215
+ await github.rest.issues.updateComment({
216
+ owner: context.repo.owner,
217
+ repo: context.repo.repo,
218
+ comment_id: existing.id,
219
+ body,
220
+ });
221
+ } else {
222
+ await github.rest.issues.createComment({
223
+ owner: context.repo.owner,
224
+ repo: context.repo.repo,
225
+ issue_number: context.issue.number,
226
+ body,
227
+ });
228
+ }
package/CHANGELOG.md CHANGED
@@ -7,8 +7,24 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ### Added
11
+
12
+ - New `tea-test-review` CLI (`bin` entry) — headless runner for the `bmad-testarch-test-review` skill: changed-test scoping from the PR diff (`--base`, or an explicit `--files` list), prompt-only mode (`--agent none`), JSON verdict (`--json`), and CI exit codes (`--fail-on request-changes|block`). Hardened for required-gate use: stdin-delivered prompt, filesystem isolation (`--isolate`, on by default in CI, `--no-isolate` to opt out), extra test-file matchers (`--test-glob`), a quality-score floor (`--min-score`), non-pass on skips (`--fail-on-skip`) and deletions-only diffs, and strict report validation (dual-section Recommendation, bounded score, violations line, frontmatter, and a `## Reviewed Files` manifest). Ships with `cli/examples/pr-test-review.yml`, a two-job (review + comment) starting template for a required test-review gate, plus `cli/examples/README.md` covering two real adaptations: a central reusable-workflows repo, and a repo already running a third-party review bot.
13
+ - `tea-test-review` is no longer Claude-only: `--agent` resolves against a per-vendor adapter table (`cli/lib/agent-adapters.js`) instead of hardcoding `claude -p`'s argv at every call site `--agent-cmd` could override. `--agent codex` spawns `codex exec --sandbox workspace-write`, live-verified with a real review against a real Playwright spec (`codex-cli` 0.146.0) whose report `parseReport()` accepted; that run wrote Key Strengths/Weaknesses as plain bullets instead of the `✅`/`❌`-prefixed form claude reliably produces, which the existing best-effort extraction already tolerates by design (`plain-bullets-key-strengths.md` fixture). A drafted `--agent gemini` adapter was not shipped: this account's `gemini` CLI OAuth login is on a deprecated Code Assist tier with no fallback API key configured, so it was never verified end-to-end.
14
+ - Deterministic TEA config in headless runs: `tea-test-review` now resolves `tea_use_playwright_utils`, `tea_use_pactjs_utils`, and `tea_pact_mcp` through an explicit precedence chain (new `--use-playwright-utils` / `--no-use-playwright-utils`, `--use-pactjs-utils` / `--no-use-pactjs-utils`, and `--pact-mcp <mcp|none>` flags, then the project's `_bmad/tea/config.yaml`, then the `src/module.yaml` default) and states all of them in the prompt. `steps-c/step-01-load-context.md` branches on these keys to pick its knowledge fragments, and CI installs the skill from a tarball without running the installer, so `config.yaml` is typically absent: previously the three keys were unstated and the agent settled them per run, meaning two runs over identical files could review against different knowledge, and a contract-testing repository could load `contract-testing.md` instead of the six `pactjs-utils-*` and `pact-*` fragments. Unusable config content is an environment error (exit 2); a missing file is not. The CLI's copy of the module defaults is asserted equal to `src/module.yaml` in the test suite so the two cannot drift.
15
+ - Gate semantics for the CLI: `--waive <reason>` with mandatory `--waive-until <YYYY-MM-DD>` expiry (verdict-fails waivable, environment/agent/parse failures never), `--min-files <n>` minimum-evidence floor, `--max-critical <n>` violation cap, inconsistent-verdict rejection (Critical violations with an Approve recommendation fail parsing), and `--skill-root <path>` for an explicit trusted skill source outside the PR checkout.
16
+ - First-class headless contract in the `test-review` workflow: new `headless`, `review_files`, `output_file_override`, and `generate_inline_comments` inputs (`workflow.yaml`, `customize.toml`), a Headless mode section in `SKILL.md`, and `review_files` as an authoritative file-set source in the discovery step, so headless runs no longer depend on prompt prose overriding the interactive flow.
17
+ - Docs: new `tea-test-review` CLI reference page (`docs/reference/tea-test-review-cli.md`) covering flags, exit codes, the JSON verdict schema, the skill prerequisite, and the security model. `test/README.md` now covers every suite and the `fixtures/test-review-cli/` layout, and drops a stale reference to a `test-cli-integration.sh` that no longer exists.
18
+ - Docs: new explanation page `docs/explanation/test-review-cli-architecture.md` on how an interactive skill is wrapped into a headless CI gate — the five modules and the pipeline order, how a workflow is made headless without discarding its customization chain, why the prompt contract and the report parser must be edited together (a strict check absent from the prompt is a false failure, not a gate), why exit 1 is separated from exits 2 and 3, why the CLI must version with the skill, and what the fixture suite can and cannot prove.
19
+ - Docs: the CLI reference now states that the reviewed repository never has to commit BMAD files, add a dependency, or install the TEA module. The skill only has to be present in the workspace when the CLI runs, which CI does as a build step from a pinned tarball. The previous "Installed in the consuming project" wording read as a repository prerequisite.
20
+
10
21
  ### Changed
11
22
 
23
+ - Removed `test:cli` from the default `npm test` script (and Husky pre-commit hook) to keep local git hooks fast, running `test:cli` as part of CI validation in `quality.yaml` and `publish.yaml`.
24
+ - `test-review` now has a single scoring model. The deduction ledger printed in `test-review-template.md` (Critical -10, High -5, Medium -2, Low -1, plus six bonus categories worth 0 or 5 each) is authoritative, and `steps-c/step-03f-aggregate-scores.md` no longer computes a competing weighted average of the four quality dimensions. Grades are limited to A/B/C/D/F. Two live runs over an identical file set had returned 83 and 92 under the old ambiguity, one of them printing a breakdown that did not sum to its own total.
25
+ - `tea-test-review` recomputes the ledger from the report's own violation counts and rejects a report whose published score contradicts its breakdown, whose bonus total is not a multiple of 5 within 0-30, or that omits the `## Quality Score Breakdown` section. The prompt states the same arithmetic, so the strict check never demands a shape the reviewer was not told to produce.
26
+ - `tea-test-review` no longer forbids the scratch files the skill itself requires: the prompt permits the `/tmp/tea-test-review-*.json` outputs that `steps-c/step-03*` write and that `step-03` aborts without, while still forbidding every other write, including the test files under review.
27
+ - `output_file_override` is now honored where reports are actually written. Each step that resolves `{outputFile}` states that a non-empty override replaces the frontmatter default, so the input works for native skill runs and not only through the CLI prompt.
12
28
  - NFR workflow boundary clarified: `test-design` now owns NFR planning (thresholds, planned evidence, NFR-derived risks) and `nfr-assess` is reframed as NFR Evidence Audit — evaluating implementation evidence against planned thresholds after code exists.
13
29
  - `nfr-assess` step-02 now checks for an existing `test-design` NFR plan first and uses it as the primary threshold source, falling back to raw documents only for missing or UNKNOWN thresholds.
14
30
  - TEA agent menu gains a `GATE` routing intent that guides users through the release gate sequence (optional test-review → optional nfr-assess → trace Phase 2 gate) without merging those workflows.
@@ -23,6 +39,16 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
23
39
  - Publish releases now use `[Unreleased]` changelog notes before falling back to generated GitHub release notes when an exact version section is missing.
24
40
  - Documented workflow-local knowledge resources as intentional self-contained skill packaging and added validation for workflow-local knowledge indexes.
25
41
 
42
+ ### Fixed
43
+
44
+ - Normalized the `test-review` checklist's Recommendation vocabulary to the canonical four-value enum (`checklist.md`).
45
+ - `test-review` score aggregation now emits a CRITICAL severity tier (`step-03f`), mapped to the report's `Critical Issues (Must Fix)` section, matching the template's four-tier violations line; `generate_inline_comments` is now a defined workflow input (default `false`) instead of an unresolved reference in the checklist.
46
+ - `test-review` reports now carry a machine-readable `## Reviewed Files` manifest section in `test-review-template.md`, and every step's first-save frontmatter snippet declares `workflowType: 'testarch-test-review'`. A report produced from the template alone now satisfies the headless verdict schema, so a clean review can no longer be reported as a parse failure when the agent follows the template rather than prompt prose.
47
+ - `tea-test-review` isolation and agent environment corrections: artifacts written directly to the project root are copied back under the chmod isolation fallback (previously `EACCES`, surfacing a clean review as exit 3), the macOS sandbox profile permits the `/tmp` subagent output files the workflow's own step contract requires, and the minimal agent environment keeps `USER`, `LOGNAME`, and `CLAUDE_CODE_OAUTH_TOKEN` so a subscription or token login stays authenticated.
48
+ - `tea-test-review` chmod isolation now restores the project tree's exact permission bits from a snapshot taken before the lock. The previous `chmod -R u+w` restore is not an inverse of `chmod -R a-w`: it stripped group and other write bits and left deliberately read-only files writable.
49
+ - `tea-test-review` reviewed-files manifest ignores prose lines and strips inline markup, so a sentence inside the report's `## Reviewed Files` section can no longer inflate the `--min-files` evidence floor; a section with no file paths is a parse failure rather than a pass.
50
+ - `tea-test-review` no longer false-fails a valid report whose `stepsCompleted` frontmatter is a YAML flow sequence wrapped across several lines, which is the shape a formatter produces once the list outgrows one line. A live run produced an otherwise complete 742-line report and the CLI rejected it with exit 3.
51
+
26
52
  ---
27
53
 
28
54
  ## [1.16.0] - 2026-05-08
@@ -0,0 +1,53 @@
1
+ # CLI examples
2
+
3
+ `pr-test-review.yml` is the full annotated template, start there. These two cover real adaptations.
4
+
5
+ ## A central reusable-workflows repo
6
+
7
+ Some orgs centralize CI logic: one repo owns `workflow_call` workflows, consuming repos call them. If that repo already has a comment-triggered `@claude` reviewer (opt-in, advisory, fired from `issue_comment`), don't graft TEA onto it. TEA needs to run on every PR automatically and gate merges, a different job than an on-demand advisory review. Add it as its own pair, same shape as whatever pattern you already use:
8
+
9
+ - **Reusable workflow** (central repo, e.g. `rwf-tea-test-review.yml`): copy the `review` and `comment` jobs from `pr-test-review.yml` into a `workflow_call` workflow. Promote `--min-score`, `--max-critical`, `--min-files`, and the pinned `TEA_VERSION` to `inputs:`, and the Anthropic key to a required secret.
10
+ - **Caller** (each consuming repo, or the central repo itself for dogfooding): a thin `pull_request`-triggered workflow that does `uses: <org>/<central-repo>/.github/workflows/rwf-tea-test-review.yml@<ref>`.
11
+
12
+ Keep the trigger on `pull_request`. That's what makes it a required check: it runs automatically, no one has to remember to summon it.
13
+
14
+ ```yaml
15
+ # rwf-tea-test-review.yml (central repo): only what differs from pr-test-review.yml
16
+ on:
17
+ workflow_call:
18
+ inputs:
19
+ min_score: { type: number, required: false, default: 80 }
20
+ secrets:
21
+ anthropic_api_key:
22
+ required: true
23
+ jobs:
24
+ review:
25
+ # ...same steps as pr-test-review.yml's `review` job...
26
+ run: tea-test-review --base "$BASE_REF" --min-score ${{ inputs.min_score }} --agent claude --skill-root "$GITHUB_WORKSPACE/_bmad/tea/workflows/testarch/bmad-testarch-test-review" --output test-review.md --json test-review.json
27
+ ```
28
+
29
+ ```yaml
30
+ # .github/workflows/tea-test-review.yml (caller, per consuming repo)
31
+ on:
32
+ pull_request:
33
+ types: [opened, synchronize, reopened]
34
+ jobs:
35
+ tea-test-review:
36
+ uses: <org>/<central-repo>/.github/workflows/rwf-tea-test-review.yml@v1
37
+ with:
38
+ min_score: 80
39
+ secrets:
40
+ anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }}
41
+ ```
42
+
43
+ ## A repo already using a third-party review bot (CodeRabbit, etc.)
44
+
45
+ Third-party review bots are configured entirely through their own SaaS-side file, there's no hook in there for invoking an external CLI. Leave that file alone. Add a new, independent `pull_request`-triggered workflow (same shape as `pr-test-review.yml`) next to it.
46
+
47
+ The two don't compete. Check whether the bot's config sets a required commit status or a request-changes gate. If it doesn't (most default/free configs are advisory-only, commenting on the diff without blocking merges), TEA test-review can be the actual required check that config deliberately leaves open, scoped specifically to test quality rather than the whole diff.
48
+
49
+ Adjust flags to your layout, for example a monorepo with tests outside the default directory:
50
+
51
+ ```yaml
52
+ run: tea-test-review --base "$BASE_REF" --test-dir playwright --min-score 80 --agent claude --skill-root "$GITHUB_WORKSPACE/_bmad/tea/workflows/testarch/bmad-testarch-test-review" --output test-review.md --json test-review.json
53
+ ```
@@ -0,0 +1,279 @@
1
+ # tea-test-review — a starting template for a required test-review gate.
2
+ #
3
+ # Two jobs: `review` runs the headless TEA test review against the PR's
4
+ # changed test files and uploads the report artifacts; `comment` publishes the
5
+ # outcome as a single upserted PR comment. Make the `review` job a required
6
+ # status check to gate merges on the review verdict.
7
+ #
8
+ # Prerequisites:
9
+ # - An ANTHROPIC_API_KEY repository secret (this template's review step uses
10
+ # --agent claude, executed by the claude CLI installed below). --agent
11
+ # codex is also supported (OPENAI_API_KEY instead) — see the --agent flag
12
+ # in docs/reference/tea-test-review-cli.md and swap the installed package
13
+ # and secret accordingly.
14
+ # - A vetted version of the bmad-method-test-architecture-enterprise npm
15
+ # package, pinned exactly (see TEA_VERSION below).
16
+ #
17
+ # Honest limitations — read before requiring this check:
18
+ # - This workflow cannot be the sole required check for repositories that
19
+ # accept fork pull requests: forks receive no secrets by design, so the
20
+ # fork guard below skips the review for them and the gate never runs.
21
+ # Fork coverage needs a separate privileged design (for example a
22
+ # pull_request_target workflow with strict controls), which is out of
23
+ # scope for this template.
24
+ # - The review skill is installed from the pinned npm package below, not
25
+ # from the PR checkout, and the run step pins it explicitly with
26
+ # --skill-root — the CLI never probes the PR checkout for the reviewer,
27
+ # closing the vendored-checkout trust gap for this shipped path. The
28
+ # residual trust assumption: you are trusting the pinned version as
29
+ # published on the npm registry — vet that version once, pin it exactly,
30
+ # and bump the pin deliberately.
31
+ #
32
+ # Gate-policy flags (all optional; tune to your gate policy):
33
+ # --min-score <n> fail below a quality-score floor (0-100); tune to your gate policy
34
+ # --max-critical <n> fail above a Critical-violation cap (default: no cap); tune to your gate policy
35
+ # --min-files <n> fail when fewer than n files were reviewed (default 1); tune to your gate policy
36
+ # --waive/--waive-until record a time-boxed waiver: any verdict failure exits 0 with a WAIVED
37
+ # banner and waived fields in the JSON (exit 2/3 are never waivable);
38
+ # tune to your gate policy
39
+ name: TEA Test Review
40
+
41
+ on:
42
+ pull_request:
43
+ types: [opened, synchronize, reopened]
44
+
45
+ # Default deny; each job grants exactly the permissions it needs.
46
+ permissions: {}
47
+
48
+ jobs:
49
+ review:
50
+ name: Headless test review
51
+ runs-on: ubuntu-latest
52
+ # Forks receive no secrets, so the review cannot run for them; skip
53
+ # honestly instead of failing (see the header comment).
54
+ if: github.event.pull_request.head.repo.full_name == github.repository
55
+ permissions:
56
+ contents: read
57
+ outputs:
58
+ verdict: ${{ steps.verdict.outputs.verdict }}
59
+ env:
60
+ # Single source of truth for both install steps below, so bumping the
61
+ # pin can't update one and silently leave the other on the old version.
62
+ TEA_VERSION: 1.19.1
63
+ steps:
64
+ - name: Checkout (full history for the PR diff)
65
+ uses: actions/checkout@v5
66
+ with:
67
+ fetch-depth: 0
68
+ # The review only reads the tree; never hand the PR checkout a
69
+ # credentials-bearing git config.
70
+ persist-credentials: false
71
+
72
+ - name: Setup Node
73
+ uses: actions/setup-node@v6
74
+ with:
75
+ node-version: 22
76
+
77
+ - name: Install the review skill from the pinned package
78
+ # Deterministic, non-interactive install: the skill content the CLI
79
+ # discovers under _bmad/ comes from the vetted package tarball, never
80
+ # from the PR checkout.
81
+ run: |
82
+ npm pack "bmad-method-test-architecture-enterprise@${TEA_VERSION}"
83
+ tar -xzf "bmad-method-test-architecture-enterprise-${TEA_VERSION}.tgz"
84
+ mkdir -p _bmad/tea/workflows/testarch
85
+ cp -R package/src/workflows/testarch/bmad-testarch-test-review _bmad/tea/workflows/testarch/bmad-testarch-test-review
86
+
87
+ - name: Install the pinned CLI and agent
88
+ env:
89
+ # Both pinned: the reviewer's control plane must be the exact code
90
+ # you vetted, not whatever `latest` resolves to when a PR lands.
91
+ CLAUDE_CODE_VERSION: 2.1.220
92
+ run: npm install --global "bmad-method-test-architecture-enterprise@${TEA_VERSION}" "@anthropic-ai/claude-code@${CLAUDE_CODE_VERSION}"
93
+
94
+ - name: Run headless test review
95
+ id: run-review
96
+ env:
97
+ ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
98
+ BASE_REF: origin/${{ github.base_ref }}
99
+ # No ${{ }} expansion inside run text: the base ref arrives via the
100
+ # environment so a crafted branch name cannot inject shell.
101
+ # --skill-root pins the reviewer to the pinned npm copy installed
102
+ # above: the CLI skips probing the PR checkout for the skill, so a PR
103
+ # that edits its own vendored _bmad/ copy cannot rewrite the reviewer.
104
+ #
105
+ # TEA config: this job installs the skill from a tarball rather than
106
+ # running the interactive installer, so _bmad/tea/config.yaml does not
107
+ # exist and the module defaults apply (Playwright Utils on, Pact off,
108
+ # no Pact MCP). Those defaults are stated in the prompt, so the run is
109
+ # deterministic either way. If this repository uses contract testing,
110
+ # say so or the review loads the generic contract-testing fragment
111
+ # instead of the pactjs-utils set:
112
+ # --use-pactjs-utils [--pact-mcp mcp]
113
+ # Committing _bmad/tea/config.yaml works too; the flags win over it.
114
+ run: tea-test-review --base "$BASE_REF" --agent claude --skill-root "$GITHUB_WORKSPACE/_bmad/tea/workflows/testarch/bmad-testarch-test-review" --output test-review.md --json test-review.json
115
+
116
+ - name: Upload review artifacts
117
+ if: always()
118
+ uses: actions/upload-artifact@v4
119
+ with:
120
+ name: tea-test-review
121
+ path: |
122
+ test-review.md
123
+ test-review.json
124
+
125
+ - name: Export the verdict as a job output
126
+ id: verdict
127
+ if: always()
128
+ env:
129
+ # The CLI exit code is the verdict (it already folds in --fail-on,
130
+ # --min-score and deletions-only handling), so the step outcome is
131
+ # the faithful pass/fail signal; the JSON carries the detail.
132
+ REVIEW_OUTCOME: ${{ steps.run-review.outcome }}
133
+ run: |
134
+ verdict="failed"
135
+ if [ "$REVIEW_OUTCOME" = "success" ]; then
136
+ verdict="passed"
137
+ fi
138
+ echo "verdict=$verdict" >> "$GITHUB_OUTPUT"
139
+
140
+ comment:
141
+ name: Comment the outcome
142
+ needs: review
143
+ # Always run after the review (success or failure) so a failed review is
144
+ # still reported; never run for forks (nothing was reviewed there).
145
+ if: always() && github.event.pull_request.head.repo.full_name == github.repository
146
+ runs-on: ubuntu-latest
147
+ permissions:
148
+ pull-requests: write
149
+ steps:
150
+ # No PR checkout and no ANTHROPIC_API_KEY in this job: it only reads
151
+ # the uploaded artifacts and writes a PR comment.
152
+ - name: Download review artifacts
153
+ id: download
154
+ uses: actions/download-artifact@v4
155
+ # A missing artifact is reported below as an infrastructure failure,
156
+ # not raised as a step failure.
157
+ continue-on-error: true
158
+ with:
159
+ name: tea-test-review
160
+
161
+ - name: Find-and-update the review comment
162
+ uses: actions/github-script@v7
163
+ env:
164
+ DOWNLOAD_OUTCOME: ${{ steps.download.outcome }}
165
+ REVIEW_RESULT: ${{ needs.review.result }}
166
+ REVIEW_VERDICT: ${{ needs.review.outputs.verdict }}
167
+ with:
168
+ # Kept in sync by hand with .github/workflows/tea-test-review.yaml's
169
+ # "Find-and-update the review comment" step: same comment-building
170
+ # logic, duplicated because this file is a standalone copy-paste
171
+ # template for other repos and cannot depend on that one.
172
+ script: |
173
+ const fs = require("fs");
174
+ const marker = "<!-- tea-test-review -->";
175
+ const artifactsUrl = `https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}#artifacts`;
176
+ // GitHub caps a comment body at 65536 chars; this leaves headroom
177
+ // for the digest and wrapper around the inlined report.
178
+ const MAX_INLINE_REPORT_CHARS = 40000;
179
+
180
+ let body = null;
181
+ const artifactAvailable = process.env.DOWNLOAD_OUTCOME === "success" && fs.existsSync("test-review.json");
182
+ if (artifactAvailable) {
183
+ try {
184
+ const verdict = JSON.parse(fs.readFileSync("test-review.json", "utf8"));
185
+ if (verdict.skipped) {
186
+ body = [
187
+ marker,
188
+ "## TEA Test Review: skipped",
189
+ "",
190
+ `${verdict.reason ?? "No changed test files in this PR"}.`,
191
+ "",
192
+ `[Review job artifacts](${artifactsUrl})`,
193
+ ].join("\n");
194
+ } else {
195
+ const counts = verdict.violations ?? {};
196
+ const violations = `${counts.critical ?? 0} Critical / ${counts.high ?? 0} High / ${counts.medium ?? 0} Medium / ${counts.low ?? 0} Low`;
197
+ const weaknesses = (verdict.keyWeaknesses ?? []).slice(0, 3);
198
+
199
+ const lines = [
200
+ marker,
201
+ `## TEA Test Review: ${verdict.recommendation}`,
202
+ "",
203
+ `- **Quality score**: ${verdict.qualityScore ?? "n/a"}/100`,
204
+ `- **Recommendation**: ${verdict.recommendation}`,
205
+ `- **Violations**: ${violations}`,
206
+ `- **Reviewed files**: ${(verdict.reviewedFiles ?? []).length}`,
207
+ ];
208
+ if (weaknesses.length > 0) {
209
+ lines.push("", "**Key weaknesses**:", ...weaknesses.map((w) => `- ${w}`));
210
+ }
211
+ lines.push("");
212
+
213
+ // Inline the full report (not just the digest above) so a
214
+ // reviewer can paste it straight into an AI coding agent to
215
+ // apply the fixes, no artifact download required. Falls back
216
+ // to the artifact link alone if the report is too large or
217
+ // missing from the download.
218
+ const reportText = fs.existsSync("test-review.md") ? fs.readFileSync("test-review.md", "utf8") : null;
219
+ if (reportText && reportText.length <= MAX_INLINE_REPORT_CHARS) {
220
+ // A reviewed file's own content can contain a literal
221
+ // </details>; inserting a zero-width space breaks that as
222
+ // an HTML closing tag while leaving the visible text
223
+ // effectively unchanged, so it cannot end this block early
224
+ // and spill raw markdown into the rest of the comment.
225
+ const safeReportText = reportText.replace(/<\/details>/gi, "<\u200B/details>");
226
+ lines.push(
227
+ "<details>",
228
+ "<summary>Full report (paste into your AI coding agent to apply the fixes)</summary>",
229
+ "",
230
+ safeReportText,
231
+ "",
232
+ "</details>",
233
+ "",
234
+ `[Verdict JSON](${artifactsUrl})`,
235
+ );
236
+ } else {
237
+ lines.push(`[Full report and verdict JSON](${artifactsUrl})`);
238
+ }
239
+ body = lines.join("\n");
240
+ }
241
+ } catch {
242
+ body = null;
243
+ }
244
+ }
245
+ if (body === null) {
246
+ body = [
247
+ marker,
248
+ "## TEA Test Review: infrastructure failure",
249
+ "",
250
+ "The review job did not produce a readable verdict artifact.",
251
+ "This is **not** a review verdict — treat the gate as broken, not as approved tests.",
252
+ "",
253
+ `Review job result: ${process.env.REVIEW_RESULT} (verdict output: ${process.env.REVIEW_VERDICT || "n/a"}).`,
254
+ `[Review job artifacts](${artifactsUrl})`,
255
+ ].join("\n");
256
+ }
257
+
258
+ const { data: comments } = await github.rest.issues.listComments({
259
+ owner: context.repo.owner,
260
+ repo: context.repo.repo,
261
+ issue_number: context.issue.number,
262
+ per_page: 100,
263
+ });
264
+ const existing = comments.find((comment) => comment.body && comment.body.includes(marker));
265
+ if (existing) {
266
+ await github.rest.issues.updateComment({
267
+ owner: context.repo.owner,
268
+ repo: context.repo.repo,
269
+ comment_id: existing.id,
270
+ body,
271
+ });
272
+ } else {
273
+ await github.rest.issues.createComment({
274
+ owner: context.repo.owner,
275
+ repo: context.repo.repo,
276
+ issue_number: context.issue.number,
277
+ body,
278
+ });
279
+ }