bmad-method-test-architecture-enterprise 1.26.0 → 1.26.1-next.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/.claude-plugin/marketplace.json +1 -1
  2. package/.github/workflows/quality.yaml +6 -0
  3. package/CHANGELOG.md +62 -0
  4. package/docs/explanation/eval-quality-command-adapter.md +19 -15
  5. package/docs/explanation/eval-quality-roadmap.md +2 -2
  6. package/package.json +5 -3
  7. package/test/contracts/fragment-selection/bmad-testarch-atdd.contract.json +1 -1
  8. package/test/contracts/fragment-selection/bmad-testarch-automate.contract.json +1 -1
  9. package/test/contracts/fragment-selection/bmad-testarch-ci.contract.json +1 -1
  10. package/test/contracts/fragment-selection/bmad-testarch-framework.contract.json +1 -1
  11. package/test/contracts/fragment-selection/bmad-testarch-nfr.contract.json +1 -1
  12. package/test/contracts/fragment-selection/bmad-testarch-test-design.contract.json +1 -1
  13. package/test/contracts/fragment-selection/bmad-testarch-test-review.contract.json +1 -1
  14. package/test/contracts/fragment-selection/bmad-testarch-trace.contract.json +1 -1
  15. package/test/contracts/test-review.contract.json +1 -1
  16. package/test/contracts/trace.contract.json +1 -1
  17. package/test/eval-contract-strength.js +25 -6
  18. package/test/eval-fragment-selection.js +8 -2
  19. package/test/eval-test-review.js +5 -2
  20. package/test/eval-trace.js +4 -1
  21. package/test/lib/eval-quality-inputs.js +19 -10
  22. package/test/lib/probe-targets.js +154 -29
  23. package/test/probes/fragment-selection/bmad-testarch-atdd.probes.json +4 -3
  24. package/test/probes/fragment-selection/bmad-testarch-automate.probes.json +4 -3
  25. package/test/probes/fragment-selection/bmad-testarch-ci.probes.json +4 -3
  26. package/test/probes/fragment-selection/bmad-testarch-framework.probes.json +4 -3
  27. package/test/probes/fragment-selection/bmad-testarch-nfr.probes.json +4 -3
  28. package/test/probes/fragment-selection/bmad-testarch-test-design.probes.json +4 -3
  29. package/test/probes/fragment-selection/bmad-testarch-test-review.probes.json +4 -3
  30. package/test/probes/fragment-selection/bmad-testarch-trace.probes.json +4 -3
  31. package/test/probes/test-review.probes.json +31 -21
  32. package/test/probes/trace.probes.json +10 -7
  33. package/test/test-eval-quality-corpus.js +301 -0
  34. package/test/test-port-totality.js +297 -0
  35. package/test/test-probe-conformance.js +21 -4
  36. package/test/test-probe-targets.js +120 -2
  37. package/tools/generate-contracts.js +10 -3
  38. package/tools/generate-probes.js +12 -2
@@ -31,7 +31,7 @@
31
31
  "name": "bmad-method-test-architecture-enterprise",
32
32
  "source": "./",
33
33
  "description": "Master Test Architect module for quality strategy, test automation, CI/CD quality gates, and structured testing education. Part of the BMad Method ecosystem.",
34
- "version": "1.26.0",
34
+ "version": "1.26.1-next.0",
35
35
  "author": {
36
36
  "name": "Murat K Ozcan (TEA Creator) & Brian (BMad) Madison"
37
37
  },
@@ -154,6 +154,9 @@ jobs:
154
154
  - name: Replay stored eval outputs against the scorers
155
155
  run: npm run test:eval-replay
156
156
 
157
+ - name: Compile eval-quality's own published corpus against the pinned release
158
+ run: npm run test:eval-quality-corpus
159
+
157
160
  - name: Check the eval contracts against the sources they are generated from
158
161
  run: npm run test:contract-sources
159
162
 
@@ -175,6 +178,9 @@ jobs:
175
178
  - name: Probe every TEA command target through the eval-quality adapter
176
179
  run: npm run test:probe-targets
177
180
 
181
+ - name: Check TEA's branches are total over the published port vocabularies
182
+ run: npm run test:port-totality
183
+
178
184
  - name: Test changelog stamping
179
185
  run: npm run test:changelog
180
186
 
package/CHANGELOG.md CHANGED
@@ -7,6 +7,68 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ### Added
11
+
12
+ - `npm run test:eval-quality-corpus` compiles `eval-quality`'s own published corpus against the pinned release, before any TEA artifact is asked to run on it.
13
+ TEA's every other check feeds the package TEA's own bytes, so a package regression and a bad migration read identically, and this upgrade moved 31 probes, 10 contracts and both command policies at once.
14
+ This check feeds the package nothing of TEA's. The `./corpus/*` subpath ships `corpus/dev/index.json`, which names 27 entries with a `sha256:` digest each: 24 contracts under `contracts/`, the compile-and-seal example's contract and its sealed brief, and the corpus README. The index also records the failure code each of the three deliberately-refused contracts raises, and it carries no digest of itself, which is the one corpus file it cannot list.
15
+ So the corpus states the expected result and the package computes the observed one. Measured against 3.0.0: 27 entry digests match the shipped bytes, 22 contracts compile, 3 are refused with the code the index records, and the sealed brief the example produces is byte for byte the shipped `brief.json`.
16
+ It also holds the resolved version against the exact pin in `package.json`, so a tree that is not the tree the repository declares fails here rather than somewhere downstream.
17
+ Exit 1 is a moved status, a moved digest or a moved brief; exit 2 is a package, subpath or index that could not be read, an index naming no contract, or a tree resolving a version other than the pin. So an unresolvable install, an empty corpus and a tree nobody installed are each reported as nothing measured rather than as a package regression.
18
+ - `npm run test:port-totality` holds TEA's branches against the port vocabularies the installed package declares.
19
+ `ProbeRequest` and `ProbeObservation` are three-member unions tagged by `kind` and the conformance surface publishes six arms, and TEA is CommonJS with no typechecker, so no compiler and no reviewer will notice the day a fourth member arrives: the failure mode is code that keeps working on the members it knows.
20
+ Every member is now handled or declined with the reason recorded, every arm is run or recorded as not yet run, and a member or arm in neither fails the check by name.
21
+ `probeCommand` narrows every observation to the `cli` member and raises a named error otherwise, which the check proves by calling the narrowing with the two members TEA declines.
22
+ Three of the four harnesses reach the port through `probeCommand`; `test/eval-contract-strength.js` caches every observation itself, so it narrows at its own boundary, on the live answer and on the cached one.
23
+ The narrowing throws outside the fault handler on purpose: a member TEA cannot read is a defect in TEA, and classifying it as an environment failure would file that defect as a lost run.
24
+
25
+ ### Changed
26
+
27
+ - `eval-quality` moves from 1.4.0 to 3.0.0.
28
+ The pin alone left four checks red, which is why every migration below lands with it.
29
+ - Every `CommandTargetPolicy` TEA builds declares `permittedEnvironmentKeys`, required on 3.0.0 with no default.
30
+ The environment channel was the one command channel with no operator bound: the contract author declared it and everything declared reached the process.
31
+ The list is stated per target rather than once for every command TEA ships, because a key a command never reads has no business reaching it. All three permit exactly what their own contracts declare, the vendor names plus `HOME` and `USER`, read from `vendorEnvironmentNames()` so the contract and the policy cannot drift.
32
+ `PATH` is on no list, refused where the widening happens as well as by the schema and by the adapter, because `target` may name a bare command and a declared `PATH` would then choose which binary runs.
33
+ A per-interface override naming an interface the policy does not authorize is refused too: it widens nothing, and unnoticed it leaves every request of that harness carrying names the authorization never permitted.
34
+ A caller widens an authorization one call at a time through `environmentKeys`, which is how an operator's `--env-pass` names and this repository's two stub variables reach a process.
35
+ - `CI` reaches no measured run, and its absence is what makes a review reproducible.
36
+ `cli/test-review.js` reads it to decide filesystem isolation when `--isolate` is not stated, and no contract declares it, so passing it through let the host decide how the measured run executed: isolation on in GitHub Actions, off on a laptop, with the two sealed records indistinguishable afterwards.
37
+ TEA's harness had been sending it since before the environment channel had an operator bound, and `test/eval-contract-strength.js` had been filtering it back out against the contract, so the two halves of the same suite were already measuring different commands.
38
+ A command run through the port now sees no `CI` at all and isolates the same way on every host; `test/eval-test-review.js` states `--isolate` explicitly and never depended on the variable.
39
+ `test/test-probe-targets.js` holds each contract's declared keys equal to its authorization's permitted keys, in both directions, so neither side can drift from the other again.
40
+ - All 31 probes carry `schemaVersion: 5` and all 10 contracts carry `5`, written by their generators rather than by hand.
41
+ On 3.0.0 a stale stamp is a `schema-version-mismatch` runtime fault at exit 5 in both `preflight` and `score`, so every probe corpus and every contract compile was failing on the version alone.
42
+ The load-bearing edit is the ninth input channel: each of the 21 `inputBinding` selectors gains `"arguments": null`, which version 5 added so a signature against a tool call can filter on what the call supplied.
43
+ `tools/generate-contracts.js` now names its stamp `EVAL_CONTRACT_SCHEMA_VERSION` instead of repeating the literal in three places.
44
+ - TEA's sealed run records carry `schemaVersion: 6` and no `invalidReason`, and their observations carry the ninth `arguments` call-input channel.
45
+ Version 6 drops `invalidReason` because nothing in the package ever read it, so a caller attesting that a run was invalid was ignored by every stage while the field looked like a supported channel.
46
+ `SCHEMA_VERSIONS` in `test/lib/eval-quality-inputs.js` read 3 for both the probe and the record; the record entry is the one with a reader, and it was emitting version-3 records against a version-6 parser.
47
+ - The command-probe conformance arm reports 16 outcomes where it reported 15. The count is the package's and moved with the version; what TEA supplies is `unauthorizedEnvironmentKeyRequest`, which is what makes the sixteenth assertion pass rather than fail.
48
+ That request is authorized in every other respect and declares exactly one environment key its authorization omits, and the suite checks that the refusal costs zero calls into the underlying mechanism, so no process is spawned.
49
+ TEA reads the expected count from `CONFORMANCE_OUTCOME_COUNTS` rather than stating it, so the next assertion the package adds fails this check until TEA answers it.
50
+ The conformance policy permits no environment key at all, which is both the honest allowlist for a fixture command that reads none and AD-35's default-deny base case.
51
+
52
+ - Scores computed before and after this release are comparable on every input, with one exception that nothing recorded depends on.
53
+ Every replay-scored result is comparable across this change, and none of them moved.
54
+ `npm run test:probe-corpus` recomputes each probe's pre-flight, verdict, exit code, basis, qualification and strength vector from the stored evidence under 3.0.0, against `test/probes/expected-strength.json` unchanged by this release.
55
+ The probe and contract bytes moved, so `corpusDigest` and `contractDigest` move with them at scoring time, and nothing under `test/replay/` pins either or pins a scoring version, which is why no stored score moved.
56
+ No digest stored inside a probe or a contract moved either: those are taken over the harness implementation and its artifacts rather than over the files themselves.
57
+ The live `eval:test-review` numbers in `docs/explanation/eval-quality-roadmap.md` are comparable.
58
+ That harness states `--isolate` explicitly on every run, so its filesystem isolation never depended on `CI`, before or after.
59
+ A live `eval:preflight` or `eval:contract-strength` measurement of `tea-test-review` taken on a host where `CI` was set is not comparable with one taken after this release, because such a run isolated where this one does not.
60
+ No such measurement is recorded here: no workflow invokes any `eval:` script, the pre-flight cache under `test/eval-artifacts/` is not committed, and every contract-strength result this repository records is a finding rather than a figure.
61
+ So nothing recorded is invalidated; what changes is that the next contract-strength run and the next eval run measure the same command on every host.
62
+ `tea-fragment-selection-runner` and `tea-trace-runner` stop carrying `CI` as well, and nothing observable moves.
63
+ Only `cli/test-review.js` reads it, and `buildMinimalEnv` in `cli/lib/run-agent.js` never passes it to a vendor, so no measured run of those two could have read it.
64
+
65
+ ### Fixed
66
+
67
+ - Two statements in `docs/explanation/eval-quality-roadmap.md` and `docs/explanation/eval-quality-command-adapter.md` that this upgrade falsified or that were already false.
68
+ The roadmap counted the published conformance suite at fifteen assertions, which is sixteen now.
69
+ The adapter page said `tea-test-review` permits neither `HOME` nor `USER`, so its live pre-flight must use an API key. That stopped being true one release earlier when its request shape moved to `vendorEnvironmentNames()`, and the same page recorded the correction two hundred lines further down while the first statement stayed, so the page contradicted itself.
70
+ The page's adoption inventory also listed `serializeArtifact` and `digestBytes` as unused, which the new corpus check falsifies, and its "done, and covered by npm test" section named none of the three checks this change adds.
71
+
10
72
  ## [1.26.0] - 2026-09-09
11
73
 
12
74
  ### Changed
@@ -7,7 +7,7 @@ description: 'Why TEA probes every measured command through eval-quality, what t
7
7
 
8
8
  TEA measures a skill by running a command and reading what it wrote. Every harness owned that mechanism itself: its own `spawnSync`, argv, timeout, and `existsSync` plus `JSON.parse` over the file the run produced. Three copies, disagreeing about what a command may do, none capping output.
9
9
 
10
- `eval-quality` 1.0.0 ships that mechanism as `createCommandLineAdapter`, a real `EnvironmentProbePort` over a child process with 15 conformance outcomes behind it. TEA uses it and deletes what it invented. See the [roadmap](./eval-quality-roadmap.md) for the surrounding plan and `test/contracts/README.md` for the contracts.
10
+ `eval-quality` 1.0.0 ships that mechanism as `createCommandLineAdapter`, a real `EnvironmentProbePort` over a child process with a conformance arm behind it, 16 outcomes on the 3.0.0 TEA now runs. TEA uses it and deletes what it invented. See the [roadmap](./eval-quality-roadmap.md) for the surrounding plan and `test/contracts/README.md` for the contracts.
11
11
 
12
12
  "Runner" has meant three processes here, which is most of why the boundary was unclear: the vendor agent doing the skill's work, the TEA command under evaluation, and `eval-all.js` orchestrating one child per suite. `eval-quality` has an opinion about the middle one only. It probes a declared interface and returns an observation, never launching an agent and holding no view on vendor routing or credential shape.
13
13
 
@@ -36,7 +36,7 @@ A contract names a logical executable and `ProbeRequest` enforces it: `executabl
36
36
  Two couplings the seam does not remove, both found by running it:
37
37
 
38
38
  - **`--json` and `--output` resolve against `--project-root`; the artifact map resolves against the policy `cwd`.** `test-review.contract.json`'s witness legs pass a bare `verdict.json` and name no project root, so its live pre-flight writes into whichever directory also has to satisfy its repository-relative `--files`. Those legs need an explicit project root and run-scoped artifact paths.
39
- - **The child environment is closed** to `PATH` plus what the request declares. Both vendors resolve a stored login through `HOME`, so the selection runner permits `HOME` and `USER`. `tea-test-review` does not, so its live pre-flight must use an API key.
39
+ - **The child environment is closed** to `PATH` plus what the request declares, and from `eval-quality` 3.0.0 the authorization declares which of those keys a request may carry at all. All three commands permit the vendor names plus `HOME` and `USER`, because both vendors resolve a stored login through `HOME`; `tea-test-review` permits `CI` on top of them, which it reads to decide filesystem isolation. A key outside a command's list is refused before a process spawns, and `PATH` is on no list: `target` may name a bare command, so a declared `PATH` would choose which binary runs. The adapter supplies its own.
40
40
 
41
41
  ## Reaching more than one skill
42
42
 
@@ -64,18 +64,19 @@ The package has three stages: compile a contract, probe an environment, then sco
64
64
  TEA used the first two and none of the third. It uses all three now, and this section is the
65
65
  inventory, kept honest by being a list of what is still unused rather than a list of what is.
66
66
 
67
- | Published surface | TEA's use |
68
- | ---------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
69
- | `compile`, through the `eval-quality` binary | `npm run test:contracts` compiles all ten contracts against `test/contracts/expected-status.json` |
70
- | `createCommandLineAdapter`, `nodeCommandMechanism`, `CommandTargetPolicy` | `test/lib/probe-targets.js` maps three logical executables to three real commands |
71
- | `runPreflight` | `npm run eval:preflight` drives every contract's witness legs through the adapter for real |
72
- | `runScore` | `npm run test:probe-corpus` scores 31 probes across ten corpora; `npm run eval:contract-strength` scores them under a live pre-flight verdict |
73
- | `seal` | one sealed evaluator brief per contract, written by the same script |
74
- | `digestArtifact` | every artifact digest the run record and the isolation manifest declare |
75
- | the published JSON Schemas | `test/lib/eval-quality-inputs.js` validates every artifact TEA builds or receives against `eval-quality/schemas/*` |
76
- | `preflightFromObservations` | unused, and it cannot be used: a caller has to key its observations by leg identifier, and the leg identifiers are minted by the plan `runPreflight` builds. TEA holds a port for both halves, so the port entry point answers the same question with no ordering problem |
77
- | `validateLineageChain`, `INTERCHANGE_ARTIFACT_KEYS`, `serializeArtifact`, `digestBytes`, `digestComposite` | unused. Every TEA artifact is `revisionCount: 0` with a null parent, so there is no chain to validate, and the digest helpers TEA needs are the artifact one and its own file digest |
78
- | `eval-quality/conformance` | `npm run test:probe-conformance` runs the published command-line arm against a fixture command |
67
+ | Published surface | TEA's use |
68
+ | --------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
69
+ | `compile`, through the `eval-quality` binary | `npm run test:contracts` compiles all ten contracts against `test/contracts/expected-status.json` |
70
+ | `createCommandLineAdapter`, `nodeCommandMechanism`, `CommandTargetPolicy` | `test/lib/probe-targets.js` maps three logical executables to three real commands |
71
+ | `runPreflight` | `npm run eval:preflight` drives every contract's witness legs through the adapter for real |
72
+ | `runScore` | `npm run test:probe-corpus` scores 31 probes across ten corpora; `npm run eval:contract-strength` scores them under a live pre-flight verdict |
73
+ | `seal` | one sealed evaluator brief per contract, written by the same script |
74
+ | `digestArtifact` | every artifact digest the run record and the isolation manifest declare |
75
+ | the published JSON Schemas | `test/lib/eval-quality-inputs.js` validates every artifact TEA builds or receives against `eval-quality/schemas/*` |
76
+ | `preflightFromObservations` | unused, and it cannot be used: a caller has to key its observations by leg identifier, and the leg identifiers are minted by the plan `runPreflight` builds. TEA holds a port for both halves, so the port entry point answers the same question with no ordering problem |
77
+ | `validateLineageChain`, `INTERCHANGE_ARTIFACT_KEYS`, `digestComposite` | unused. Every TEA artifact is `revisionCount: 0` with a null parent, so there is no chain to validate, and the digest helper TEA needs is the artifact one |
78
+ | `serializeArtifact`, `digestBytes`, `StructuralFailure`, the `./corpus/*` subpath | `npm run test:eval-quality-corpus` seals the package's own compile-and-seal example and compares the serialized bytes with the shipped brief, digests every corpus file against the digest `corpus/dev/index.json` records, and reads a refused compile by its failure class rather than by duck-typing |
79
+ | `eval-quality/conformance` | `npm run test:probe-conformance` runs the published command-line arm against a fixture command |
79
80
 
80
81
  ### What the corpus is
81
82
 
@@ -379,12 +380,15 @@ for the same correctness.
379
380
  The adapter closes the child environment to `PATH` plus what the request declares, and
380
381
  `cli/test-review.js` resolves a stored login through `HOME`, so a machine with a keychain login could
381
382
  not run that contract's own pre-flight. The shape is read from `vendorEnvironmentNames()` now, the
382
- same source the other two commands use.
383
+ same source the other two commands use, and on 3.0.0 the authorization permits exactly that list,
384
+ asserted equal in both directions by `npm run test:probe-targets`.
383
385
 
384
386
  ## Done, and owed
385
387
 
386
388
  Done, and covered by `npm test`: the registry, policy, port, and fault-to-failure-class mapping in `test/lib/probe-targets.js`; the runner, whose request shape and default agent `tools/generate-contracts.js` reads rather than transcribes; and `npm run test:probe-targets`, which drives all three real commands through the real adapter against checked-in fixtures with a stub vendor. It asserts default-deny, the observation shape, artifact read-back, an absent artifact, a real budget kill classified as a timeout, and contract-to-registry agreement both ways, with no model call and no credential.
387
389
 
390
+ Three more checks joined that list with the move to 3.0.0. `npm run test:probe-targets` now also holds each contract's declared environment keys equal to its authorization's permitted keys in both directions, and asserts that an unpermitted key is denied before a process spawns and that a malformed key fails at the port parse. `npm run test:eval-quality-corpus` compiles the package's own published corpus, which is the one check here that feeds the package nothing of TEA's. `npm run test:port-totality` holds TEA's branches total over both probe unions and all six published conformance arms.
391
+
388
392
  Also done: `eval-test-review.js`'s `runReview` and `promptDigestFromCli` drive `tea-test-review` through the port, and `eval-fragment-selection.js` runs `cli/fragment-selection-runner.js` rather than calling `runAgent` in process, so its live path and its contract describe one thing. Both were verified end to end against their stub agents, with no model call: the review harness reads a real verdict artifact back as JSON and scores it, and the selection harness scores a real reply off stdout.
389
393
 
390
394
  Two path couplings closed with them. `runReview` states `--project-root` because the process no longer runs in the repository, and it passes absolute artifact paths so the CLI's `--project-root` resolution and the policy's `cwd` resolution cannot disagree. Each fragment-selection run's scratch directory is the authorization's `cwd`, so the `read-only` declaration is enforced by the policy rather than by the caller remembering to pass one.
@@ -166,9 +166,9 @@ Items 1 through 8 are done. Item 9 is what remains:
166
166
  - **trace**: green on both sets, and the corpus had to be corrected to get there. The seeded set scores 10 of 10 criteria with the expected FAIL gate; the clean set scores 5 of 5 with the expected PASS gate; every one of the sixteen thresholds is met, with zero clean false positives, zero invented criteria, zero duplicate sections, and zero fixture mutations.
167
167
  The clean set failed on the first three measured runs, and the reason is the record worth keeping. Its discriminating criterion AC-4 came back PARTIAL, then INTEGRATION-ONLY, then PARTIAL again, against a declared truth of FULL, and the mismatch false-positived the gate to FAIL. The first fix inlined the coverage-classification rule into step-03, on the theory that a run citing `checklist.md` by name had to guess the semantics from the enum labels. Re-measuring at `--runs 3` did not move it. That second measurement is what showed the runs were right: AC-4 claimed a console list shows an expired token while its only evidence was two API tests asserting what `GET /tenants/{tenant}/tokens` returns, and every other criterion in that set naming a rendered surface carries component or e2e evidence. The criterion was rewritten to say what the tests establish, and the clean set passed.
168
168
  Measured against `codex` rather than `claude`/`sonnet`, because that account was rate limited when the confirming run was due. A second vendor agreeing is worth more here than a matched one would have been.
169
- 7. Adopt `eval-quality`'s command-line adapter and retire the process-probing machinery this repository invented. `eval-quality` 1.0.0 ships `createCommandLineAdapter`, `nodeCommandMechanism`, and a deny-by-default `CommandTargetPolicy`, which is a real `EnvironmentProbePort` over a child process, plus a `cli` arm in its own conformance suite. `test/lib/probe-targets.js` and `npm run test:probe-targets` drive all three of TEA's commands through it: `tea-fragment-selection-runner` exists now, so the eight fragment-selection contracts name a command TEA ships, and `tea-trace-runner` exists, so the trace suite has a command and a contract of its own. All three harnesses probe through the port as of `eval-quality` 1.2.0, which added the repeatable-option spelling the harnesses needed to forward `--env-pass` and `--agent-arg` at all; each was verified end to end against its stub agent with no model call, the trace harness by spawning it whole and reading its result record back. Reaching every skill therefore means giving the remaining skills real CLI entry points, which is already the direction the TEA CLI rollout records. This is owed work with a design document of its own.
169
+ 7. Adopt `eval-quality`'s command-line adapter and retire the process-probing machinery this repository invented. `eval-quality` 1.0.0 ships `createCommandLineAdapter`, `nodeCommandMechanism`, and a deny-by-default `CommandTargetPolicy`, which is a real `EnvironmentProbePort` over a child process, plus a `cli` arm in its own conformance suite. `test/lib/probe-targets.js` and `npm run test:probe-targets` drive all three of TEA's commands through it: `tea-fragment-selection-runner` exists now, so the eight fragment-selection contracts name a command TEA ships, and `tea-trace-runner` exists, so the trace suite has a command and a contract of its own. All three harnesses probe through the port as of `eval-quality` 1.2.0, which added the repeatable-option spelling the harnesses needed to forward `--env-pass` and `--agent-arg` at all; the pin is 3.0.0 now, and on that release every authorization also declares which environment keys its requests may carry, so the channel the contract author declares is bounded by the operator's mapping rather than passed through whole; each was verified end to end against its stub agent with no model call, the trace harness by spawning it whole and reading its result record back. Reaching every skill therefore means giving the remaining skills real CLI entry points, which is already the direction the TEA CLI rollout records. This is owed work with a design document of its own.
170
170
 
171
- 8. ~~Adopt the scoring half.~~ `runPreflight`, `runScore` and `seal` all run. `tools/generate-probes.js` writes 31 probes across ten corpora from the ground truth this repository already keeps, `npm run test:probe-corpus` scores every one of them against the outputs under `test/replay/` with no model call and validates every artifact against the schemas `eval-quality` publishes, and `npm run eval:preflight` drives the witness legs through the real command-line adapter, which is the first pre-flight this repository has run. `npm run test:probe-conformance` runs the package's own published port conformance suite against the adapter, fifteen assertions. What the scoring half found is recorded in `docs/explanation/eval-quality-command-adapter.md` under "How much of eval-quality TEA actually uses": three contract-strength findings, two limits in the probe vocabulary, and one contract-authoring defect that had made the whole measurement meaningless.
171
+ 8. ~~Adopt the scoring half.~~ `runPreflight`, `runScore` and `seal` all run. `tools/generate-probes.js` writes 31 probes across ten corpora from the ground truth this repository already keeps, `npm run test:probe-corpus` scores every one of them against the outputs under `test/replay/` with no model call and validates every artifact against the schemas `eval-quality` publishes, and `npm run eval:preflight` drives the witness legs through the real command-line adapter, which is the first pre-flight this repository has run. `npm run test:probe-conformance` runs the package's own published port conformance suite against the adapter, sixteen assertions. `npm run test:eval-quality-corpus` compiles the package's own published corpus against the pinned release, which is the one check here that feeds the package nothing of TEA's, and `npm run test:port-totality` holds TEA's branches total over both probe unions and all six published conformance arms. What the scoring half found is recorded in `docs/explanation/eval-quality-command-adapter.md` under "How much of eval-quality TEA actually uses": three contract-strength findings, two limits in the probe vocabulary, and one contract-authoring defect that had made the whole measurement meaningless.
172
172
 
173
173
  9. Give the remaining eight skills a behavioral suite, in the evidence order section 3 sets out. Everything above is measurement machinery, and it measures two skills.
174
174
 
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "$schema": "https://json.schemastore.org/package.json",
3
3
  "name": "bmad-method-test-architecture-enterprise",
4
- "version": "1.26.0",
4
+ "version": "1.26.1-next.0",
5
5
  "description": "Master Test Architect for quality strategy, test automation, and release gates",
6
6
  "keywords": [
7
7
  "bmad",
@@ -58,7 +58,7 @@
58
58
  "release:minor": "gh workflow run publish.yaml -f channel=latest -f bump=minor",
59
59
  "release:next": "gh workflow run publish.yaml -f channel=next",
60
60
  "release:patch": "gh workflow run publish.yaml -f channel=latest -f bump=patch",
61
- "test": "npm run test:schemas && npm run test:install && npm run test:knowledge && npm run test:criteria-fragments && npm run test:enforce-hook && npm run test:eval-data && npm run test:eval-trace-data && npm run test:eval-schemas && npm run test:eval-replay && npm run test:contract-sources && npm run test:contracts && npm run test:contract-oracles && npm run test:probe-sources && npm run test:probe-corpus && npm run test:probe-conformance && npm run test:probe-targets && npm run test:ci-coverage && npm run test:release-metadata && npm run test:changelog && npm run test:tea-workflow-descriptions && npm run validate:schemas && npm run lint && npm run lint:md && npm run format:check",
61
+ "test": "npm run test:schemas && npm run test:install && npm run test:knowledge && npm run test:criteria-fragments && npm run test:enforce-hook && npm run test:eval-data && npm run test:eval-trace-data && npm run test:eval-schemas && npm run test:eval-replay && npm run test:eval-quality-corpus && npm run test:contract-sources && npm run test:contracts && npm run test:contract-oracles && npm run test:probe-sources && npm run test:probe-corpus && npm run test:probe-conformance && npm run test:probe-targets && npm run test:port-totality && npm run test:ci-coverage && npm run test:release-metadata && npm run test:changelog && npm run test:tea-workflow-descriptions && npm run validate:schemas && npm run lint && npm run lint:md && npm run format:check",
62
62
  "test:changelog": "node test/test-stamp-changelog.js",
63
63
  "test:ci-coverage": "node tools/validate-ci-coverage.js",
64
64
  "test:cli": "node test/test-test-review-cli.js",
@@ -69,11 +69,13 @@
69
69
  "test:criteria-fragments": "node tools/validate-criteria-fragments.js",
70
70
  "test:enforce-hook": "node test/test-enforce-hook.js",
71
71
  "test:eval-data": "node test/eval-fragment-selection.js --validate-only",
72
+ "test:eval-quality-corpus": "node test/test-eval-quality-corpus.js",
72
73
  "test:eval-replay": "node test/test-eval-replay.js",
73
74
  "test:eval-schemas": "node tools/validate-eval-schemas.js",
74
75
  "test:eval-trace-data": "node test/eval-trace.js --validate-only",
75
76
  "test:install": "node test/test-installation-components.js",
76
77
  "test:knowledge": "node test/test-knowledge-base.js",
78
+ "test:port-totality": "node test/test-port-totality.js",
77
79
  "test:probe-conformance": "node test/test-probe-conformance.js",
78
80
  "test:probe-corpus": "node test/test-probe-corpus.js",
79
81
  "test:probe-sources": "node tools/generate-probes.js --check",
@@ -137,7 +139,7 @@
137
139
  "eslint-plugin-n": "^17.21.3",
138
140
  "eslint-plugin-unicorn": "^60.0.0",
139
141
  "eslint-plugin-yml": "^1.18.0",
140
- "eval-quality": "1.4.0",
142
+ "eval-quality": "3.0.0",
141
143
  "husky": "^9.1.7",
142
144
  "jest": "^30.0.4",
143
145
  "lint-staged": "^16.1.1",
@@ -1,5 +1,5 @@
1
1
  {
2
- "schemaVersion": 4,
2
+ "schemaVersion": 5,
3
3
  "parentDigest": null,
4
4
  "revisionCount": 0,
5
5
  "contractId": "tea-fragment-selection-atdd",
@@ -1,5 +1,5 @@
1
1
  {
2
- "schemaVersion": 4,
2
+ "schemaVersion": 5,
3
3
  "parentDigest": null,
4
4
  "revisionCount": 0,
5
5
  "contractId": "tea-fragment-selection-automate",
@@ -1,5 +1,5 @@
1
1
  {
2
- "schemaVersion": 4,
2
+ "schemaVersion": 5,
3
3
  "parentDigest": null,
4
4
  "revisionCount": 0,
5
5
  "contractId": "tea-fragment-selection-ci",
@@ -1,5 +1,5 @@
1
1
  {
2
- "schemaVersion": 4,
2
+ "schemaVersion": 5,
3
3
  "parentDigest": null,
4
4
  "revisionCount": 0,
5
5
  "contractId": "tea-fragment-selection-framework",
@@ -1,5 +1,5 @@
1
1
  {
2
- "schemaVersion": 4,
2
+ "schemaVersion": 5,
3
3
  "parentDigest": null,
4
4
  "revisionCount": 0,
5
5
  "contractId": "tea-fragment-selection-nfr",
@@ -1,5 +1,5 @@
1
1
  {
2
- "schemaVersion": 4,
2
+ "schemaVersion": 5,
3
3
  "parentDigest": null,
4
4
  "revisionCount": 0,
5
5
  "contractId": "tea-fragment-selection-test-design",
@@ -1,5 +1,5 @@
1
1
  {
2
- "schemaVersion": 4,
2
+ "schemaVersion": 5,
3
3
  "parentDigest": null,
4
4
  "revisionCount": 0,
5
5
  "contractId": "tea-fragment-selection-test-review",
@@ -1,5 +1,5 @@
1
1
  {
2
- "schemaVersion": 4,
2
+ "schemaVersion": 5,
3
3
  "parentDigest": null,
4
4
  "revisionCount": 0,
5
5
  "contractId": "tea-fragment-selection-trace",
@@ -1,5 +1,5 @@
1
1
  {
2
- "schemaVersion": 4,
2
+ "schemaVersion": 5,
3
3
  "parentDigest": null,
4
4
  "revisionCount": 0,
5
5
  "contractId": "tea-test-review-behavioral",
@@ -1,5 +1,5 @@
1
1
  {
2
- "schemaVersion": 4,
2
+ "schemaVersion": 5,
3
3
  "parentDigest": null,
4
4
  "revisionCount": 0,
5
5
  "contractId": "tea-trace-behavioral",
@@ -62,7 +62,7 @@ const path = require('node:path');
62
62
 
63
63
  const { digest } = require('./lib/eval-record');
64
64
  const { validateArtifact } = require('./lib/eval-quality-inputs');
65
- const { createProbePort, hostEnvironment } = require('./lib/probe-targets');
65
+ const { cliObservation, createProbePort, readEnvironment } = require('./lib/probe-targets');
66
66
  const { runSuite, sealContract, suites } = require('./lib/probe-scoring');
67
67
  const { stageWorkspace, traceArtifactPaths } = require('./eval-trace');
68
68
 
@@ -204,11 +204,18 @@ function stagedWorkspaceFor(suiteId, request) {
204
204
  return { root: dir, cwd: dir, artifacts: {} };
205
205
  }
206
206
 
207
- /** The environment names this operation declares it accepts, intersected with what this machine has. */
207
+ /**
208
+ * The environment names this operation declares it accepts, with the values this
209
+ * machine has for them.
210
+ *
211
+ * Read straight off the contract's own declaration. The authorization permits
212
+ * the same names, because both derive from the command's one allowlist in
213
+ * `cli/lib/runner-exit-codes.js`, so a leg built here carries no key the adapter
214
+ * refuses.
215
+ */
208
216
  function permittedEnvironment(contract, operationId) {
209
217
  const operation = contract.permittedInterfaces.flatMap((iface) => iface.operations).find((entry) => entry.operationId === operationId);
210
- const permitted = new Set(operation?.requestShape?.environment?.permittedKeys ?? []);
211
- return Object.fromEntries(Object.entries(hostEnvironment()).filter(([name]) => permitted.has(name)));
218
+ return readEnvironment(operation?.requestShape?.environment?.permittedKeys ?? []);
212
219
  }
213
220
 
214
221
  /** The cache key for one probe request: everything about it except which leg asked. */
@@ -246,7 +253,16 @@ function cachingPort({ makePort, contract, cacheDir, agent, force, counters, log
246
253
  for (const counter of counters) counter.hits += 1;
247
254
  log(` ${colors.dim}cached${colors.reset} leg ${request.probeId} (${key})`);
248
255
  const cached = JSON.parse(fs.readFileSync(file, 'utf8'));
249
- return { ...cached.observation, probeId: request.probeId, interfaceId: request.interfaceId, operationId: request.operationId };
256
+ // Narrowed like a live one. A cache written by an older build, or by a
257
+ // port that answered in another member, is a shape this harness cannot
258
+ // read, and reading it as a cli observation would score `undefined` as
259
+ // an exit code.
260
+ return cliObservation({
261
+ ...cached.observation,
262
+ probeId: request.probeId,
263
+ interfaceId: request.interfaceId,
264
+ operationId: request.operationId,
265
+ });
250
266
  }
251
267
  const augmented = {
252
268
  ...request,
@@ -261,7 +277,10 @@ function cachingPort({ makePort, contract, cacheDir, agent, force, counters, log
261
277
  const startedAt = Date.now();
262
278
  let observation;
263
279
  try {
264
- observation = await realPort.probe(augmented, signal);
280
+ // This harness is the fourth port caller and the one that does not go
281
+ // through `probeCommand`, because it caches every observation itself.
282
+ // It narrows at the same boundary for the same reason.
283
+ observation = cliObservation(await realPort.probe(augmented, signal));
265
284
  } finally {
266
285
  fs.rmSync(workspace.root, { recursive: true, force: true });
267
286
  }
@@ -589,7 +589,7 @@ async function main() {
589
589
  // authorization's maxElapsedMs is a minute longer and SIGKILLs, so the inner
590
590
  // bound is the one that fires and the classification survives.
591
591
  runnerOption['timeout-ms'] = String(RUN_TIMEOUT_MS);
592
- const runnerEnvironment = hostEnvironment(options.envPass);
592
+ const runnerEnvironment = hostEnvironment('tea-fragment-selection-runner', options.envPass);
593
593
 
594
594
  const runners = [];
595
595
 
@@ -627,7 +627,13 @@ async function main() {
627
627
  let written = [];
628
628
  let treeChanges = [];
629
629
  try {
630
- const { port } = await createProbePort({ cwd: scratch, interfaceIds: ['tea-fragment-selection-runner'] });
630
+ const { port } = await createProbePort({
631
+ cwd: scratch,
632
+ interfaceIds: ['tea-fragment-selection-runner'],
633
+ // The operator's own pass-through names, so the authorization
634
+ // permits exactly what the request above declares.
635
+ environmentKeys: { 'tea-fragment-selection-runner': options.envPass },
636
+ });
631
637
  result = await probeCommand(
632
638
  port,
633
639
  probeRequest({
@@ -425,6 +425,9 @@ async function runReview(agent, runIndex, runner = {}) {
425
425
  cwd: runDir,
426
426
  interfaceIds: ['tea-test-review'],
427
427
  artifacts: { 'tea-test-review': { verdict: jsonPath, report: reportPath } },
428
+ // The operator's own pass-through names, so the authorization permits
429
+ // exactly what the request below declares.
430
+ environmentKeys: { 'tea-test-review': runner.envPass ?? [] },
428
431
  });
429
432
 
430
433
  // No --fail-on override: the enum is request-changes|block, so there is no "never".
@@ -458,7 +461,7 @@ async function runReview(agent, runIndex, runner = {}) {
458
461
  interfaceId: 'tea-test-review',
459
462
  operationId: 'review-test-files',
460
463
  option,
461
- environment: hostEnvironment(runner.envPass ?? []),
464
+ environment: hostEnvironment('tea-test-review', runner.envPass ?? []),
462
465
  }),
463
466
  new AbortController().signal,
464
467
  );
@@ -690,7 +693,7 @@ async function promptDigestFromCli() {
690
693
  'project-root': PROJECT_ROOT,
691
694
  output: path.join(probeDir, 'test-review.md'),
692
695
  },
693
- environment: hostEnvironment(),
696
+ environment: hostEnvironment('tea-test-review'),
694
697
  }),
695
698
  new AbortController().signal,
696
699
  );
@@ -1915,6 +1915,9 @@ async function runCase(set, options, agent, runIndex, tolerance, pctTolerance) {
1915
1915
  cwd: workspace.dir,
1916
1916
  interfaceIds: [TRACE_INTERFACE],
1917
1917
  artifacts: { [TRACE_INTERFACE]: traceArtifactPaths(set) },
1918
+ // The operator's own pass-through names, so the authorization permits
1919
+ // exactly what the request below declares.
1920
+ environmentKeys: { [TRACE_INTERFACE]: options.envPass },
1918
1921
  });
1919
1922
  const result = await probeCommand(
1920
1923
  port,
@@ -1923,7 +1926,7 @@ async function runCase(set, options, agent, runIndex, tolerance, pctTolerance) {
1923
1926
  interfaceId: TRACE_INTERFACE,
1924
1927
  operationId: TRACE_OPERATION,
1925
1928
  option: { ...runnerOptions(options), agent },
1926
- environment: hostEnvironment(options.envPass),
1929
+ environment: hostEnvironment(TRACE_INTERFACE, options.envPass),
1927
1930
  stdin: { kind: 'text', value: buildPrompt(set) },
1928
1931
  }),
1929
1932
  new AbortController().signal,
@@ -37,18 +37,23 @@ const SCHEMA_ROOT = path.join(PROJECT_ROOT, 'node_modules', 'eval-quality', 'sch
37
37
  const POLICY_PATH = path.join(PROJECT_ROOT, 'test', 'probes', 'scoring-policy.json');
38
38
 
39
39
  /**
40
- * The `schemaVersion` each artifact carries.
40
+ * The `schemaVersion` each artifact this file builds carries.
41
41
  *
42
42
  * Stated rather than derived, because the published JSON Schemas declare
43
43
  * `schemaVersion` as a plain integer and name no accepted value; the package's
44
44
  * own readers throw `schema-version-mismatch` on a number they do not read. A
45
45
  * bump therefore arrives as a loud fault on the first run after an upgrade, which
46
46
  * is where a table like this is supposed to fail.
47
+ *
48
+ * Three entries, one per artifact built below. It carried two more, `probe` and
49
+ * `scoringPolicy`, which nothing read: the probe stamp that lands in bytes is
50
+ * `tools/generate-probes.js`'s, and TEA's scoring policy is a file on disk. A
51
+ * version stated in a second place is a version that drifts, and this table
52
+ * drifted exactly that way on the upgrade to 3.0.0, where its record entry read
53
+ * 3 against a parser that reads 6.
47
54
  */
48
55
  const SCHEMA_VERSIONS = {
49
- probe: 3,
50
- sealedRunRecord: 3,
51
- scoringPolicy: 2,
56
+ sealedRunRecord: 6,
52
57
  isolationManifest: 1,
53
58
  evaluatorConfiguration: 1,
54
59
  };
@@ -207,10 +212,15 @@ function isolationManifest({
207
212
  /**
208
213
  * One observation inside a sealed run record.
209
214
  *
210
- * The ten channels are total in the schema, so a caller naming only what it saw
211
- * would fail to parse. `stdout`, `stderr` and each artifact are tagged bodies,
212
- * the same three tags `createCommandLineAdapter` returns, so an observation built
213
- * here and one read off the port have one shape.
215
+ * The eleven channels are total in the schema, so a caller naming only what it
216
+ * saw would fail to parse. `arguments` is the ninth call-input channel, added by
217
+ * the record's version 5 bump so what a tool call supplied has somewhere to
218
+ * live. TEA observes spawned commands and never a tool call, so it is stated as
219
+ * null the way the four HTTP channels are.
220
+ *
221
+ * `stdout`, `stderr` and each artifact are tagged bodies, the same three tags
222
+ * `createCommandLineAdapter` returns, so an observation built here and one read
223
+ * off the port have one shape.
214
224
  *
215
225
  * `provenance` defaults to `evaluator-chosen` because that is what every run this
216
226
  * repository measures is: the harness chose the invocation. `matchProbeWitness`
@@ -244,6 +254,7 @@ function recordObservation({
244
254
  option: callInputs.option ?? null,
245
255
  environment: callInputs.environment ?? null,
246
256
  stdin: callInputs.stdin ?? null,
257
+ arguments: callInputs.arguments ?? null,
247
258
  },
248
259
  responseBody: null,
249
260
  responseHeaders: null,
@@ -294,7 +305,6 @@ function sealedRunRecord({
294
305
  resourceUse,
295
306
  truncationBound = null,
296
307
  reportedIncomplete = false,
297
- invalidReason = null,
298
308
  }) {
299
309
  return {
300
310
  schemaVersion: SCHEMA_VERSIONS.sealedRunRecord,
@@ -319,7 +329,6 @@ function sealedRunRecord({
319
329
  isolationManifestArtifact,
320
330
  resourceUse,
321
331
  evidenceDisclosure: { truncationBound, reportedIncomplete },
322
- invalidReason,
323
332
  };
324
333
  }
325
334