bmad-method-test-architecture-enterprise 1.26.0 → 1.26.1-next.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/.github/workflows/quality.yaml +6 -0
- package/CHANGELOG.md +62 -0
- package/docs/explanation/eval-quality-command-adapter.md +19 -15
- package/docs/explanation/eval-quality-roadmap.md +2 -2
- package/package.json +5 -3
- package/test/contracts/fragment-selection/bmad-testarch-atdd.contract.json +1 -1
- package/test/contracts/fragment-selection/bmad-testarch-automate.contract.json +1 -1
- package/test/contracts/fragment-selection/bmad-testarch-ci.contract.json +1 -1
- package/test/contracts/fragment-selection/bmad-testarch-framework.contract.json +1 -1
- package/test/contracts/fragment-selection/bmad-testarch-nfr.contract.json +1 -1
- package/test/contracts/fragment-selection/bmad-testarch-test-design.contract.json +1 -1
- package/test/contracts/fragment-selection/bmad-testarch-test-review.contract.json +1 -1
- package/test/contracts/fragment-selection/bmad-testarch-trace.contract.json +1 -1
- package/test/contracts/test-review.contract.json +1 -1
- package/test/contracts/trace.contract.json +1 -1
- package/test/eval-contract-strength.js +25 -6
- package/test/eval-fragment-selection.js +8 -2
- package/test/eval-test-review.js +5 -2
- package/test/eval-trace.js +4 -1
- package/test/lib/eval-quality-inputs.js +19 -10
- package/test/lib/probe-targets.js +154 -29
- package/test/probes/fragment-selection/bmad-testarch-atdd.probes.json +4 -3
- package/test/probes/fragment-selection/bmad-testarch-automate.probes.json +4 -3
- package/test/probes/fragment-selection/bmad-testarch-ci.probes.json +4 -3
- package/test/probes/fragment-selection/bmad-testarch-framework.probes.json +4 -3
- package/test/probes/fragment-selection/bmad-testarch-nfr.probes.json +4 -3
- package/test/probes/fragment-selection/bmad-testarch-test-design.probes.json +4 -3
- package/test/probes/fragment-selection/bmad-testarch-test-review.probes.json +4 -3
- package/test/probes/fragment-selection/bmad-testarch-trace.probes.json +4 -3
- package/test/probes/test-review.probes.json +31 -21
- package/test/probes/trace.probes.json +10 -7
- package/test/test-eval-quality-corpus.js +301 -0
- package/test/test-port-totality.js +297 -0
- package/test/test-probe-conformance.js +21 -4
- package/test/test-probe-targets.js +120 -2
- package/tools/generate-contracts.js +10 -3
- package/tools/generate-probes.js +12 -2
|
@@ -31,7 +31,7 @@
|
|
|
31
31
|
"name": "bmad-method-test-architecture-enterprise",
|
|
32
32
|
"source": "./",
|
|
33
33
|
"description": "Master Test Architect module for quality strategy, test automation, CI/CD quality gates, and structured testing education. Part of the BMad Method ecosystem.",
|
|
34
|
-
"version": "1.26.0",
|
|
34
|
+
"version": "1.26.1-next.0",
|
|
35
35
|
"author": {
|
|
36
36
|
"name": "Murat K Ozcan (TEA Creator) & Brian (BMad) Madison"
|
|
37
37
|
},
|
|
@@ -154,6 +154,9 @@ jobs:
|
|
|
154
154
|
- name: Replay stored eval outputs against the scorers
|
|
155
155
|
run: npm run test:eval-replay
|
|
156
156
|
|
|
157
|
+
- name: Compile eval-quality's own published corpus against the pinned release
|
|
158
|
+
run: npm run test:eval-quality-corpus
|
|
159
|
+
|
|
157
160
|
- name: Check the eval contracts against the sources they are generated from
|
|
158
161
|
run: npm run test:contract-sources
|
|
159
162
|
|
|
@@ -175,6 +178,9 @@ jobs:
|
|
|
175
178
|
- name: Probe every TEA command target through the eval-quality adapter
|
|
176
179
|
run: npm run test:probe-targets
|
|
177
180
|
|
|
181
|
+
- name: Check TEA's branches are total over the published port vocabularies
|
|
182
|
+
run: npm run test:port-totality
|
|
183
|
+
|
|
178
184
|
- name: Test changelog stamping
|
|
179
185
|
run: npm run test:changelog
|
|
180
186
|
|
package/CHANGELOG.md
CHANGED
|
@@ -7,6 +7,68 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
### Added
|
|
11
|
+
|
|
12
|
+
- `npm run test:eval-quality-corpus` compiles `eval-quality`'s own published corpus against the pinned release, before any TEA artifact is asked to run on it.
|
|
13
|
+
TEA's every other check feeds the package TEA's own bytes, so a package regression and a bad migration read identically, and this upgrade moved 31 probes, 10 contracts and both command policies at once.
|
|
14
|
+
This check feeds the package nothing of TEA's. The `./corpus/*` subpath ships `corpus/dev/index.json`, which names 27 entries with a `sha256:` digest each: 24 contracts under `contracts/`, the compile-and-seal example's contract and its sealed brief, and the corpus README. The index also records the failure code each of the three deliberately-refused contracts raises, and it carries no digest of itself, which is the one corpus file it cannot list.
|
|
15
|
+
So the corpus states the expected result and the package computes the observed one. Measured against 3.0.0: 27 entry digests match the shipped bytes, 22 contracts compile, 3 are refused with the code the index records, and the sealed brief the example produces is byte for byte the shipped `brief.json`.
|
|
16
|
+
It also holds the resolved version against the exact pin in `package.json`, so a tree that is not the tree the repository declares fails here rather than somewhere downstream.
|
|
17
|
+
Exit 1 is a moved status, a moved digest or a moved brief; exit 2 is a package, subpath or index that could not be read, an index naming no contract, or a tree resolving a version other than the pin. So an unresolvable install, an empty corpus and a tree nobody installed are each reported as nothing measured rather than as a package regression.
|
|
18
|
+
- `npm run test:port-totality` holds TEA's branches against the port vocabularies the installed package declares.
|
|
19
|
+
`ProbeRequest` and `ProbeObservation` are three-member unions tagged by `kind` and the conformance surface publishes six arms, and TEA is CommonJS with no typechecker, so no compiler and no reviewer will notice the day a fourth member arrives: the failure mode is code that keeps working on the members it knows.
|
|
20
|
+
Every member is now handled or declined with the reason recorded, every arm is run or recorded as not yet run, and a member or arm in neither fails the check by name.
|
|
21
|
+
`probeCommand` narrows every observation to the `cli` member and raises a named error otherwise, which the check proves by calling the narrowing with the two members TEA declines.
|
|
22
|
+
Three of the four harnesses reach the port through `probeCommand`; `test/eval-contract-strength.js` caches every observation itself, so it narrows at its own boundary, on the live answer and on the cached one.
|
|
23
|
+
The narrowing throws outside the fault handler on purpose: a member TEA cannot read is a defect in TEA, and classifying it as an environment failure would file that defect as a lost run.
|
|
24
|
+
|
|
25
|
+
### Changed
|
|
26
|
+
|
|
27
|
+
- `eval-quality` moves from 1.4.0 to 3.0.0.
|
|
28
|
+
The pin alone left four checks red, which is why every migration below lands with it.
|
|
29
|
+
- Every `CommandTargetPolicy` TEA builds declares `permittedEnvironmentKeys`, required on 3.0.0 with no default.
|
|
30
|
+
The environment channel was the one command channel with no operator bound: the contract author declared it and everything declared reached the process.
|
|
31
|
+
The list is stated per target rather than once for every command TEA ships, because a key a command never reads has no business reaching it. All three permit exactly what their own contracts declare, the vendor names plus `HOME` and `USER`, read from `vendorEnvironmentNames()` so the contract and the policy cannot drift.
|
|
32
|
+
`PATH` is on no list, refused where the widening happens as well as by the schema and by the adapter, because `target` may name a bare command and a declared `PATH` would then choose which binary runs.
|
|
33
|
+
A per-interface override naming an interface the policy does not authorize is refused too: it widens nothing, and unnoticed it leaves every request of that harness carrying names the authorization never permitted.
|
|
34
|
+
A caller widens an authorization one call at a time through `environmentKeys`, which is how an operator's `--env-pass` names and this repository's two stub variables reach a process.
|
|
35
|
+
- `CI` reaches no measured run, and its absence is what makes a review reproducible.
|
|
36
|
+
`cli/test-review.js` reads it to decide filesystem isolation when `--isolate` is not stated, and no contract declares it, so passing it through let the host decide how the measured run executed: isolation on in GitHub Actions, off on a laptop, with the two sealed records indistinguishable afterwards.
|
|
37
|
+
TEA's harness had been sending it since before the environment channel had an operator bound, and `test/eval-contract-strength.js` had been filtering it back out against the contract, so the two halves of the same suite were already measuring different commands.
|
|
38
|
+
A command run through the port now sees no `CI` at all and isolates the same way on every host; `test/eval-test-review.js` states `--isolate` explicitly and never depended on the variable.
|
|
39
|
+
`test/test-probe-targets.js` holds each contract's declared keys equal to its authorization's permitted keys, in both directions, so neither side can drift from the other again.
|
|
40
|
+
- All 31 probes carry `schemaVersion: 5` and all 10 contracts carry `5`, written by their generators rather than by hand.
|
|
41
|
+
On 3.0.0 a stale stamp is a `schema-version-mismatch` runtime fault at exit 5 in both `preflight` and `score`, so every probe corpus and every contract compile was failing on the version alone.
|
|
42
|
+
The load-bearing edit is the ninth input channel: each of the 21 `inputBinding` selectors gains `"arguments": null`, which version 5 added so a signature against a tool call can filter on what the call supplied.
|
|
43
|
+
`tools/generate-contracts.js` now names its stamp `EVAL_CONTRACT_SCHEMA_VERSION` instead of repeating the literal in three places.
|
|
44
|
+
- TEA's sealed run records carry `schemaVersion: 6` and no `invalidReason`, and their observations carry the ninth `arguments` call-input channel.
|
|
45
|
+
Version 6 drops `invalidReason` because nothing in the package ever read it, so a caller attesting that a run was invalid was ignored by every stage while the field looked like a supported channel.
|
|
46
|
+
`SCHEMA_VERSIONS` in `test/lib/eval-quality-inputs.js` read 3 for both the probe and the record; the record entry is the one with a reader, and it was emitting version-3 records against a version-6 parser.
|
|
47
|
+
- The command-probe conformance arm reports 16 outcomes where it reported 15. The count is the package's and moved with the version; what TEA supplies is `unauthorizedEnvironmentKeyRequest`, which is what makes the sixteenth assertion pass rather than fail.
|
|
48
|
+
That request is authorized in every other respect and declares exactly one environment key its authorization omits, and the suite checks that the refusal costs zero calls into the underlying mechanism, so no process is spawned.
|
|
49
|
+
TEA reads the expected count from `CONFORMANCE_OUTCOME_COUNTS` rather than stating it, so the next assertion the package adds fails this check until TEA answers it.
|
|
50
|
+
The conformance policy permits no environment key at all, which is both the honest allowlist for a fixture command that reads none and AD-35's default-deny base case.
|
|
51
|
+
|
|
52
|
+
- Scores computed before and after this release are comparable on every input, with one exception that nothing recorded depends on.
|
|
53
|
+
Every replay-scored result is comparable across this change, and none of them moved.
|
|
54
|
+
`npm run test:probe-corpus` recomputes each probe's pre-flight, verdict, exit code, basis, qualification and strength vector from the stored evidence under 3.0.0, against `test/probes/expected-strength.json` unchanged by this release.
|
|
55
|
+
The probe and contract bytes moved, so `corpusDigest` and `contractDigest` move with them at scoring time, and nothing under `test/replay/` pins either or pins a scoring version, which is why no stored score moved.
|
|
56
|
+
No digest stored inside a probe or a contract moved either: those are taken over the harness implementation and its artifacts rather than over the files themselves.
|
|
57
|
+
The live `eval:test-review` numbers in `docs/explanation/eval-quality-roadmap.md` are comparable.
|
|
58
|
+
That harness states `--isolate` explicitly on every run, so its filesystem isolation never depended on `CI`, before or after.
|
|
59
|
+
A live `eval:preflight` or `eval:contract-strength` measurement of `tea-test-review` taken on a host where `CI` was set is not comparable with one taken after this release, because such a run isolated where this one does not.
|
|
60
|
+
No such measurement is recorded here: no workflow invokes any `eval:` script, the pre-flight cache under `test/eval-artifacts/` is not committed, and every contract-strength result this repository records is a finding rather than a figure.
|
|
61
|
+
So nothing recorded is invalidated; what changes is that the next contract-strength run and the next eval run measure the same command on every host.
|
|
62
|
+
`tea-fragment-selection-runner` and `tea-trace-runner` stop carrying `CI` as well, and nothing observable moves.
|
|
63
|
+
Only `cli/test-review.js` reads it, and `buildMinimalEnv` in `cli/lib/run-agent.js` never passes it to a vendor, so no measured run of those two could have read it.
|
|
64
|
+
|
|
65
|
+
### Fixed
|
|
66
|
+
|
|
67
|
+
- Two statements in `docs/explanation/eval-quality-roadmap.md` and `docs/explanation/eval-quality-command-adapter.md` that this upgrade falsified or that were already false.
|
|
68
|
+
The roadmap counted the published conformance suite at fifteen assertions, which is sixteen now.
|
|
69
|
+
The adapter page said `tea-test-review` permits neither `HOME` nor `USER`, so its live pre-flight must use an API key. That stopped being true one release earlier when its request shape moved to `vendorEnvironmentNames()`, and the same page recorded the correction two hundred lines further down while the first statement stayed, so the page contradicted itself.
|
|
70
|
+
The page's adoption inventory also listed `serializeArtifact` and `digestBytes` as unused, which the new corpus check falsifies, and its "done, and covered by npm test" section named none of the three checks this change adds.
|
|
71
|
+
|
|
10
72
|
## [1.26.0] - 2026-09-09
|
|
11
73
|
|
|
12
74
|
### Changed
|
|
@@ -7,7 +7,7 @@ description: 'Why TEA probes every measured command through eval-quality, what t
|
|
|
7
7
|
|
|
8
8
|
TEA measures a skill by running a command and reading what it wrote. Every harness owned that mechanism itself: its own `spawnSync`, argv, timeout, and `existsSync` plus `JSON.parse` over the file the run produced. Three copies, disagreeing about what a command may do, none capping output.
|
|
9
9
|
|
|
10
|
-
`eval-quality` 1.0.0 ships that mechanism as `createCommandLineAdapter`, a real `EnvironmentProbePort` over a child process with
|
|
10
|
+
`eval-quality` 1.0.0 ships that mechanism as `createCommandLineAdapter`, a real `EnvironmentProbePort` over a child process with a conformance arm behind it, 16 outcomes on the 3.0.0 TEA now runs. TEA uses it and deletes what it invented. See the [roadmap](./eval-quality-roadmap.md) for the surrounding plan and `test/contracts/README.md` for the contracts.
|
|
11
11
|
|
|
12
12
|
"Runner" has meant three processes here, which is most of why the boundary was unclear: the vendor agent doing the skill's work, the TEA command under evaluation, and `eval-all.js` orchestrating one child per suite. `eval-quality` has an opinion about the middle one only. It probes a declared interface and returns an observation, never launching an agent and holding no view on vendor routing or credential shape.
|
|
13
13
|
|
|
@@ -36,7 +36,7 @@ A contract names a logical executable and `ProbeRequest` enforces it: `executabl
|
|
|
36
36
|
Two couplings the seam does not remove, both found by running it:
|
|
37
37
|
|
|
38
38
|
- **`--json` and `--output` resolve against `--project-root`; the artifact map resolves against the policy `cwd`.** `test-review.contract.json`'s witness legs pass a bare `verdict.json` and name no project root, so its live pre-flight writes into whichever directory also has to satisfy its repository-relative `--files`. Those legs need an explicit project root and run-scoped artifact paths.
|
|
39
|
-
- **The child environment is closed** to `PATH` plus what the request declares.
|
|
39
|
+
- **The child environment is closed** to `PATH` plus what the request declares, and from `eval-quality` 3.0.0 the authorization declares which of those keys a request may carry at all. All three commands permit the vendor names plus `HOME` and `USER`, because both vendors resolve a stored login through `HOME`; `tea-test-review` permits `CI` on top of them, which it reads to decide filesystem isolation. A key outside a command's list is refused before a process spawns, and `PATH` is on no list: `target` may name a bare command, so a declared `PATH` would choose which binary runs. The adapter supplies its own.
|
|
40
40
|
|
|
41
41
|
## Reaching more than one skill
|
|
42
42
|
|
|
@@ -64,18 +64,19 @@ The package has three stages: compile a contract, probe an environment, then sco
|
|
|
64
64
|
TEA used the first two and none of the third. It uses all three now, and this section is the
|
|
65
65
|
inventory, kept honest by being a list of what is still unused rather than a list of what is.
|
|
66
66
|
|
|
67
|
-
| Published surface
|
|
68
|
-
|
|
|
69
|
-
| `compile`, through the `eval-quality` binary
|
|
70
|
-
| `createCommandLineAdapter`, `nodeCommandMechanism`, `CommandTargetPolicy`
|
|
71
|
-
| `runPreflight`
|
|
72
|
-
| `runScore`
|
|
73
|
-
| `seal`
|
|
74
|
-
| `digestArtifact`
|
|
75
|
-
| the published JSON Schemas
|
|
76
|
-
| `preflightFromObservations`
|
|
77
|
-
| `validateLineageChain`, `INTERCHANGE_ARTIFACT_KEYS`, `
|
|
78
|
-
| `
|
|
67
|
+
| Published surface | TEA's use |
|
|
68
|
+
| --------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
69
|
+
| `compile`, through the `eval-quality` binary | `npm run test:contracts` compiles all ten contracts against `test/contracts/expected-status.json` |
|
|
70
|
+
| `createCommandLineAdapter`, `nodeCommandMechanism`, `CommandTargetPolicy` | `test/lib/probe-targets.js` maps three logical executables to three real commands |
|
|
71
|
+
| `runPreflight` | `npm run eval:preflight` drives every contract's witness legs through the adapter for real |
|
|
72
|
+
| `runScore` | `npm run test:probe-corpus` scores 31 probes across ten corpora; `npm run eval:contract-strength` scores them under a live pre-flight verdict |
|
|
73
|
+
| `seal` | one sealed evaluator brief per contract, written by the same script |
|
|
74
|
+
| `digestArtifact` | every artifact digest the run record and the isolation manifest declare |
|
|
75
|
+
| the published JSON Schemas | `test/lib/eval-quality-inputs.js` validates every artifact TEA builds or receives against `eval-quality/schemas/*` |
|
|
76
|
+
| `preflightFromObservations` | unused, and it cannot be used: a caller has to key its observations by leg identifier, and the leg identifiers are minted by the plan `runPreflight` builds. TEA holds a port for both halves, so the port entry point answers the same question with no ordering problem |
|
|
77
|
+
| `validateLineageChain`, `INTERCHANGE_ARTIFACT_KEYS`, `digestComposite` | unused. Every TEA artifact is `revisionCount: 0` with a null parent, so there is no chain to validate, and the digest helper TEA needs is the artifact one |
|
|
78
|
+
| `serializeArtifact`, `digestBytes`, `StructuralFailure`, the `./corpus/*` subpath | `npm run test:eval-quality-corpus` seals the package's own compile-and-seal example and compares the serialized bytes with the shipped brief, digests every corpus file against the digest `corpus/dev/index.json` records, and reads a refused compile by its failure class rather than by duck-typing |
|
|
79
|
+
| `eval-quality/conformance` | `npm run test:probe-conformance` runs the published command-line arm against a fixture command |
|
|
79
80
|
|
|
80
81
|
### What the corpus is
|
|
81
82
|
|
|
@@ -379,12 +380,15 @@ for the same correctness.
|
|
|
379
380
|
The adapter closes the child environment to `PATH` plus what the request declares, and
|
|
380
381
|
`cli/test-review.js` resolves a stored login through `HOME`, so a machine with a keychain login could
|
|
381
382
|
not run that contract's own pre-flight. The shape is read from `vendorEnvironmentNames()` now, the
|
|
382
|
-
same source the other two commands use.
|
|
383
|
+
same source the other two commands use, and on 3.0.0 the authorization permits exactly that list,
|
|
384
|
+
asserted equal in both directions by `npm run test:probe-targets`.
|
|
383
385
|
|
|
384
386
|
## Done, and owed
|
|
385
387
|
|
|
386
388
|
Done, and covered by `npm test`: the registry, policy, port, and fault-to-failure-class mapping in `test/lib/probe-targets.js`; the runner, whose request shape and default agent `tools/generate-contracts.js` reads rather than transcribes; and `npm run test:probe-targets`, which drives all three real commands through the real adapter against checked-in fixtures with a stub vendor. It asserts default-deny, the observation shape, artifact read-back, an absent artifact, a real budget kill classified as a timeout, and contract-to-registry agreement both ways, with no model call and no credential.
|
|
387
389
|
|
|
390
|
+
Three more checks joined that list with the move to 3.0.0. `npm run test:probe-targets` now also holds each contract's declared environment keys equal to its authorization's permitted keys in both directions, and asserts that an unpermitted key is denied before a process spawns and that a malformed key fails at the port parse. `npm run test:eval-quality-corpus` compiles the package's own published corpus, which is the one check here that feeds the package nothing of TEA's. `npm run test:port-totality` holds TEA's branches total over both probe unions and all six published conformance arms.
|
|
391
|
+
|
|
388
392
|
Also done: `eval-test-review.js`'s `runReview` and `promptDigestFromCli` drive `tea-test-review` through the port, and `eval-fragment-selection.js` runs `cli/fragment-selection-runner.js` rather than calling `runAgent` in process, so its live path and its contract describe one thing. Both were verified end to end against their stub agents, with no model call: the review harness reads a real verdict artifact back as JSON and scores it, and the selection harness scores a real reply off stdout.
|
|
389
393
|
|
|
390
394
|
Two path couplings closed with them. `runReview` states `--project-root` because the process no longer runs in the repository, and it passes absolute artifact paths so the CLI's `--project-root` resolution and the policy's `cwd` resolution cannot disagree. Each fragment-selection run's scratch directory is the authorization's `cwd`, so the `read-only` declaration is enforced by the policy rather than by the caller remembering to pass one.
|
|
@@ -166,9 +166,9 @@ Items 1 through 8 are done. Item 9 is what remains:
|
|
|
166
166
|
- **trace**: green on both sets, and the corpus had to be corrected to get there. The seeded set scores 10 of 10 criteria with the expected FAIL gate; the clean set scores 5 of 5 with the expected PASS gate; every one of the sixteen thresholds is met, with zero clean false positives, zero invented criteria, zero duplicate sections, and zero fixture mutations.
|
|
167
167
|
The clean set failed on the first three measured runs, and the reason is the record worth keeping. Its discriminating criterion AC-4 came back PARTIAL, then INTEGRATION-ONLY, then PARTIAL again, against a declared truth of FULL, and the mismatch false-positived the gate to FAIL. The first fix inlined the coverage-classification rule into step-03, on the theory that a run citing `checklist.md` by name had to guess the semantics from the enum labels. Re-measuring at `--runs 3` did not move it. That second measurement is what showed the runs were right: AC-4 claimed a console list shows an expired token while its only evidence was two API tests asserting what `GET /tenants/{tenant}/tokens` returns, and every other criterion in that set naming a rendered surface carries component or e2e evidence. The criterion was rewritten to say what the tests establish, and the clean set passed.
|
|
168
168
|
Measured against `codex` rather than `claude`/`sonnet`, because that account was rate limited when the confirming run was due. A second vendor agreeing is worth more here than a matched one would have been.
|
|
169
|
-
7. Adopt `eval-quality`'s command-line adapter and retire the process-probing machinery this repository invented. `eval-quality` 1.0.0 ships `createCommandLineAdapter`, `nodeCommandMechanism`, and a deny-by-default `CommandTargetPolicy`, which is a real `EnvironmentProbePort` over a child process, plus a `cli` arm in its own conformance suite. `test/lib/probe-targets.js` and `npm run test:probe-targets` drive all three of TEA's commands through it: `tea-fragment-selection-runner` exists now, so the eight fragment-selection contracts name a command TEA ships, and `tea-trace-runner` exists, so the trace suite has a command and a contract of its own. All three harnesses probe through the port as of `eval-quality` 1.2.0, which added the repeatable-option spelling the harnesses needed to forward `--env-pass` and `--agent-arg` at all; each was verified end to end against its stub agent with no model call, the trace harness by spawning it whole and reading its result record back. Reaching every skill therefore means giving the remaining skills real CLI entry points, which is already the direction the TEA CLI rollout records. This is owed work with a design document of its own.
|
|
169
|
+
7. Adopt `eval-quality`'s command-line adapter and retire the process-probing machinery this repository invented. `eval-quality` 1.0.0 ships `createCommandLineAdapter`, `nodeCommandMechanism`, and a deny-by-default `CommandTargetPolicy`, which is a real `EnvironmentProbePort` over a child process, plus a `cli` arm in its own conformance suite. `test/lib/probe-targets.js` and `npm run test:probe-targets` drive all three of TEA's commands through it: `tea-fragment-selection-runner` exists now, so the eight fragment-selection contracts name a command TEA ships, and `tea-trace-runner` exists, so the trace suite has a command and a contract of its own. All three harnesses probe through the port as of `eval-quality` 1.2.0, which added the repeatable-option spelling the harnesses needed to forward `--env-pass` and `--agent-arg` at all; the pin is 3.0.0 now, and on that release every authorization also declares which environment keys its requests may carry, so the channel the contract author declares is bounded by the operator's mapping rather than passed through whole; each was verified end to end against its stub agent with no model call, the trace harness by spawning it whole and reading its result record back. Reaching every skill therefore means giving the remaining skills real CLI entry points, which is already the direction the TEA CLI rollout records. This is owed work with a design document of its own.
|
|
170
170
|
|
|
171
|
-
8. ~~Adopt the scoring half.~~ `runPreflight`, `runScore` and `seal` all run. `tools/generate-probes.js` writes 31 probes across ten corpora from the ground truth this repository already keeps, `npm run test:probe-corpus` scores every one of them against the outputs under `test/replay/` with no model call and validates every artifact against the schemas `eval-quality` publishes, and `npm run eval:preflight` drives the witness legs through the real command-line adapter, which is the first pre-flight this repository has run. `npm run test:probe-conformance` runs the package's own published port conformance suite against the adapter,
|
|
171
|
+
8. ~~Adopt the scoring half.~~ `runPreflight`, `runScore` and `seal` all run. `tools/generate-probes.js` writes 31 probes across ten corpora from the ground truth this repository already keeps, `npm run test:probe-corpus` scores every one of them against the outputs under `test/replay/` with no model call and validates every artifact against the schemas `eval-quality` publishes, and `npm run eval:preflight` drives the witness legs through the real command-line adapter, which is the first pre-flight this repository has run. `npm run test:probe-conformance` runs the package's own published port conformance suite against the adapter, sixteen assertions. `npm run test:eval-quality-corpus` compiles the package's own published corpus against the pinned release, which is the one check here that feeds the package nothing of TEA's, and `npm run test:port-totality` holds TEA's branches total over both probe unions and all six published conformance arms. What the scoring half found is recorded in `docs/explanation/eval-quality-command-adapter.md` under "How much of eval-quality TEA actually uses": three contract-strength findings, two limits in the probe vocabulary, and one contract-authoring defect that had made the whole measurement meaningless.
|
|
172
172
|
|
|
173
173
|
9. Give the remaining eight skills a behavioral suite, in the evidence order section 3 sets out. Everything above is measurement machinery, and it measures two skills.
|
|
174
174
|
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"$schema": "https://json.schemastore.org/package.json",
|
|
3
3
|
"name": "bmad-method-test-architecture-enterprise",
|
|
4
|
-
"version": "1.26.0",
|
|
4
|
+
"version": "1.26.1-next.0",
|
|
5
5
|
"description": "Master Test Architect for quality strategy, test automation, and release gates",
|
|
6
6
|
"keywords": [
|
|
7
7
|
"bmad",
|
|
@@ -58,7 +58,7 @@
|
|
|
58
58
|
"release:minor": "gh workflow run publish.yaml -f channel=latest -f bump=minor",
|
|
59
59
|
"release:next": "gh workflow run publish.yaml -f channel=next",
|
|
60
60
|
"release:patch": "gh workflow run publish.yaml -f channel=latest -f bump=patch",
|
|
61
|
-
"test": "npm run test:schemas && npm run test:install && npm run test:knowledge && npm run test:criteria-fragments && npm run test:enforce-hook && npm run test:eval-data && npm run test:eval-trace-data && npm run test:eval-schemas && npm run test:eval-replay && npm run test:contract-sources && npm run test:contracts && npm run test:contract-oracles && npm run test:probe-sources && npm run test:probe-corpus && npm run test:probe-conformance && npm run test:probe-targets && npm run test:ci-coverage && npm run test:release-metadata && npm run test:changelog && npm run test:tea-workflow-descriptions && npm run validate:schemas && npm run lint && npm run lint:md && npm run format:check",
|
|
61
|
+
"test": "npm run test:schemas && npm run test:install && npm run test:knowledge && npm run test:criteria-fragments && npm run test:enforce-hook && npm run test:eval-data && npm run test:eval-trace-data && npm run test:eval-schemas && npm run test:eval-replay && npm run test:eval-quality-corpus && npm run test:contract-sources && npm run test:contracts && npm run test:contract-oracles && npm run test:probe-sources && npm run test:probe-corpus && npm run test:probe-conformance && npm run test:probe-targets && npm run test:port-totality && npm run test:ci-coverage && npm run test:release-metadata && npm run test:changelog && npm run test:tea-workflow-descriptions && npm run validate:schemas && npm run lint && npm run lint:md && npm run format:check",
|
|
62
62
|
"test:changelog": "node test/test-stamp-changelog.js",
|
|
63
63
|
"test:ci-coverage": "node tools/validate-ci-coverage.js",
|
|
64
64
|
"test:cli": "node test/test-test-review-cli.js",
|
|
@@ -69,11 +69,13 @@
|
|
|
69
69
|
"test:criteria-fragments": "node tools/validate-criteria-fragments.js",
|
|
70
70
|
"test:enforce-hook": "node test/test-enforce-hook.js",
|
|
71
71
|
"test:eval-data": "node test/eval-fragment-selection.js --validate-only",
|
|
72
|
+
"test:eval-quality-corpus": "node test/test-eval-quality-corpus.js",
|
|
72
73
|
"test:eval-replay": "node test/test-eval-replay.js",
|
|
73
74
|
"test:eval-schemas": "node tools/validate-eval-schemas.js",
|
|
74
75
|
"test:eval-trace-data": "node test/eval-trace.js --validate-only",
|
|
75
76
|
"test:install": "node test/test-installation-components.js",
|
|
76
77
|
"test:knowledge": "node test/test-knowledge-base.js",
|
|
78
|
+
"test:port-totality": "node test/test-port-totality.js",
|
|
77
79
|
"test:probe-conformance": "node test/test-probe-conformance.js",
|
|
78
80
|
"test:probe-corpus": "node test/test-probe-corpus.js",
|
|
79
81
|
"test:probe-sources": "node tools/generate-probes.js --check",
|
|
@@ -137,7 +139,7 @@
|
|
|
137
139
|
"eslint-plugin-n": "^17.21.3",
|
|
138
140
|
"eslint-plugin-unicorn": "^60.0.0",
|
|
139
141
|
"eslint-plugin-yml": "^1.18.0",
|
|
140
|
-
"eval-quality": "
|
|
142
|
+
"eval-quality": "3.0.0",
|
|
141
143
|
"husky": "^9.1.7",
|
|
142
144
|
"jest": "^30.0.4",
|
|
143
145
|
"lint-staged": "^16.1.1",
|
|
@@ -62,7 +62,7 @@ const path = require('node:path');
|
|
|
62
62
|
|
|
63
63
|
const { digest } = require('./lib/eval-record');
|
|
64
64
|
const { validateArtifact } = require('./lib/eval-quality-inputs');
|
|
65
|
-
const { createProbePort,
|
|
65
|
+
const { cliObservation, createProbePort, readEnvironment } = require('./lib/probe-targets');
|
|
66
66
|
const { runSuite, sealContract, suites } = require('./lib/probe-scoring');
|
|
67
67
|
const { stageWorkspace, traceArtifactPaths } = require('./eval-trace');
|
|
68
68
|
|
|
@@ -204,11 +204,18 @@ function stagedWorkspaceFor(suiteId, request) {
|
|
|
204
204
|
return { root: dir, cwd: dir, artifacts: {} };
|
|
205
205
|
}
|
|
206
206
|
|
|
207
|
-
/**
|
|
207
|
+
/**
|
|
208
|
+
* The environment names this operation declares it accepts, with the values this
|
|
209
|
+
* machine has for them.
|
|
210
|
+
*
|
|
211
|
+
* Read straight off the contract's own declaration. The authorization permits
|
|
212
|
+
* the same names, because both derive from the command's one allowlist in
|
|
213
|
+
* `cli/lib/runner-exit-codes.js`, so a leg built here carries no key the adapter
|
|
214
|
+
* refuses.
|
|
215
|
+
*/
|
|
208
216
|
function permittedEnvironment(contract, operationId) {
|
|
209
217
|
const operation = contract.permittedInterfaces.flatMap((iface) => iface.operations).find((entry) => entry.operationId === operationId);
|
|
210
|
-
|
|
211
|
-
return Object.fromEntries(Object.entries(hostEnvironment()).filter(([name]) => permitted.has(name)));
|
|
218
|
+
return readEnvironment(operation?.requestShape?.environment?.permittedKeys ?? []);
|
|
212
219
|
}
|
|
213
220
|
|
|
214
221
|
/** The cache key for one probe request: everything about it except which leg asked. */
|
|
@@ -246,7 +253,16 @@ function cachingPort({ makePort, contract, cacheDir, agent, force, counters, log
|
|
|
246
253
|
for (const counter of counters) counter.hits += 1;
|
|
247
254
|
log(` ${colors.dim}cached${colors.reset} leg ${request.probeId} (${key})`);
|
|
248
255
|
const cached = JSON.parse(fs.readFileSync(file, 'utf8'));
|
|
249
|
-
|
|
256
|
+
// Narrowed like a live one. A cache written by an older build, or by a
|
|
257
|
+
// port that answered in another member, is a shape this harness cannot
|
|
258
|
+
// read, and reading it as a cli observation would score `undefined` as
|
|
259
|
+
// an exit code.
|
|
260
|
+
return cliObservation({
|
|
261
|
+
...cached.observation,
|
|
262
|
+
probeId: request.probeId,
|
|
263
|
+
interfaceId: request.interfaceId,
|
|
264
|
+
operationId: request.operationId,
|
|
265
|
+
});
|
|
250
266
|
}
|
|
251
267
|
const augmented = {
|
|
252
268
|
...request,
|
|
@@ -261,7 +277,10 @@ function cachingPort({ makePort, contract, cacheDir, agent, force, counters, log
|
|
|
261
277
|
const startedAt = Date.now();
|
|
262
278
|
let observation;
|
|
263
279
|
try {
|
|
264
|
-
|
|
280
|
+
// This harness is the fourth port caller and the one that does not go
|
|
281
|
+
// through `probeCommand`, because it caches every observation itself.
|
|
282
|
+
// It narrows at the same boundary for the same reason.
|
|
283
|
+
observation = cliObservation(await realPort.probe(augmented, signal));
|
|
265
284
|
} finally {
|
|
266
285
|
fs.rmSync(workspace.root, { recursive: true, force: true });
|
|
267
286
|
}
|
|
@@ -589,7 +589,7 @@ async function main() {
|
|
|
589
589
|
// authorization's maxElapsedMs is a minute longer and SIGKILLs, so the inner
|
|
590
590
|
// bound is the one that fires and the classification survives.
|
|
591
591
|
runnerOption['timeout-ms'] = String(RUN_TIMEOUT_MS);
|
|
592
|
-
const runnerEnvironment = hostEnvironment(options.envPass);
|
|
592
|
+
const runnerEnvironment = hostEnvironment('tea-fragment-selection-runner', options.envPass);
|
|
593
593
|
|
|
594
594
|
const runners = [];
|
|
595
595
|
|
|
@@ -627,7 +627,13 @@ async function main() {
|
|
|
627
627
|
let written = [];
|
|
628
628
|
let treeChanges = [];
|
|
629
629
|
try {
|
|
630
|
-
const { port } = await createProbePort({
|
|
630
|
+
const { port } = await createProbePort({
|
|
631
|
+
cwd: scratch,
|
|
632
|
+
interfaceIds: ['tea-fragment-selection-runner'],
|
|
633
|
+
// The operator's own pass-through names, so the authorization
|
|
634
|
+
// permits exactly what the request above declares.
|
|
635
|
+
environmentKeys: { 'tea-fragment-selection-runner': options.envPass },
|
|
636
|
+
});
|
|
631
637
|
result = await probeCommand(
|
|
632
638
|
port,
|
|
633
639
|
probeRequest({
|
package/test/eval-test-review.js
CHANGED
|
@@ -425,6 +425,9 @@ async function runReview(agent, runIndex, runner = {}) {
|
|
|
425
425
|
cwd: runDir,
|
|
426
426
|
interfaceIds: ['tea-test-review'],
|
|
427
427
|
artifacts: { 'tea-test-review': { verdict: jsonPath, report: reportPath } },
|
|
428
|
+
// The operator's own pass-through names, so the authorization permits
|
|
429
|
+
// exactly what the request below declares.
|
|
430
|
+
environmentKeys: { 'tea-test-review': runner.envPass ?? [] },
|
|
428
431
|
});
|
|
429
432
|
|
|
430
433
|
// No --fail-on override: the enum is request-changes|block, so there is no "never".
|
|
@@ -458,7 +461,7 @@ async function runReview(agent, runIndex, runner = {}) {
|
|
|
458
461
|
interfaceId: 'tea-test-review',
|
|
459
462
|
operationId: 'review-test-files',
|
|
460
463
|
option,
|
|
461
|
-
environment: hostEnvironment(runner.envPass ?? []),
|
|
464
|
+
environment: hostEnvironment('tea-test-review', runner.envPass ?? []),
|
|
462
465
|
}),
|
|
463
466
|
new AbortController().signal,
|
|
464
467
|
);
|
|
@@ -690,7 +693,7 @@ async function promptDigestFromCli() {
|
|
|
690
693
|
'project-root': PROJECT_ROOT,
|
|
691
694
|
output: path.join(probeDir, 'test-review.md'),
|
|
692
695
|
},
|
|
693
|
-
environment: hostEnvironment(),
|
|
696
|
+
environment: hostEnvironment('tea-test-review'),
|
|
694
697
|
}),
|
|
695
698
|
new AbortController().signal,
|
|
696
699
|
);
|
package/test/eval-trace.js
CHANGED
|
@@ -1915,6 +1915,9 @@ async function runCase(set, options, agent, runIndex, tolerance, pctTolerance) {
|
|
|
1915
1915
|
cwd: workspace.dir,
|
|
1916
1916
|
interfaceIds: [TRACE_INTERFACE],
|
|
1917
1917
|
artifacts: { [TRACE_INTERFACE]: traceArtifactPaths(set) },
|
|
1918
|
+
// The operator's own pass-through names, so the authorization permits
|
|
1919
|
+
// exactly what the request below declares.
|
|
1920
|
+
environmentKeys: { [TRACE_INTERFACE]: options.envPass },
|
|
1918
1921
|
});
|
|
1919
1922
|
const result = await probeCommand(
|
|
1920
1923
|
port,
|
|
@@ -1923,7 +1926,7 @@ async function runCase(set, options, agent, runIndex, tolerance, pctTolerance) {
|
|
|
1923
1926
|
interfaceId: TRACE_INTERFACE,
|
|
1924
1927
|
operationId: TRACE_OPERATION,
|
|
1925
1928
|
option: { ...runnerOptions(options), agent },
|
|
1926
|
-
environment: hostEnvironment(options.envPass),
|
|
1929
|
+
environment: hostEnvironment(TRACE_INTERFACE, options.envPass),
|
|
1927
1930
|
stdin: { kind: 'text', value: buildPrompt(set) },
|
|
1928
1931
|
}),
|
|
1929
1932
|
new AbortController().signal,
|
|
@@ -37,18 +37,23 @@ const SCHEMA_ROOT = path.join(PROJECT_ROOT, 'node_modules', 'eval-quality', 'sch
|
|
|
37
37
|
const POLICY_PATH = path.join(PROJECT_ROOT, 'test', 'probes', 'scoring-policy.json');
|
|
38
38
|
|
|
39
39
|
/**
|
|
40
|
-
* The `schemaVersion` each artifact carries.
|
|
40
|
+
* The `schemaVersion` each artifact this file builds carries.
|
|
41
41
|
*
|
|
42
42
|
* Stated rather than derived, because the published JSON Schemas declare
|
|
43
43
|
* `schemaVersion` as a plain integer and name no accepted value; the package's
|
|
44
44
|
* own readers throw `schema-version-mismatch` on a number they do not read. A
|
|
45
45
|
* bump therefore arrives as a loud fault on the first run after an upgrade, which
|
|
46
46
|
* is where a table like this is supposed to fail.
|
|
47
|
+
*
|
|
48
|
+
* Three entries, one per artifact built below. It carried two more, `probe` and
|
|
49
|
+
* `scoringPolicy`, which nothing read: the probe stamp that lands in bytes is
|
|
50
|
+
* `tools/generate-probes.js`'s, and TEA's scoring policy is a file on disk. A
|
|
51
|
+
* version stated in a second place is a version that drifts, and this table
|
|
52
|
+
* drifted exactly that way on the upgrade to 3.0.0, where its record entry read
|
|
53
|
+
* 3 against a parser that reads 6.
|
|
47
54
|
*/
|
|
48
55
|
const SCHEMA_VERSIONS = {
|
|
49
|
-
|
|
50
|
-
sealedRunRecord: 3,
|
|
51
|
-
scoringPolicy: 2,
|
|
56
|
+
sealedRunRecord: 6,
|
|
52
57
|
isolationManifest: 1,
|
|
53
58
|
evaluatorConfiguration: 1,
|
|
54
59
|
};
|
|
@@ -207,10 +212,15 @@ function isolationManifest({
|
|
|
207
212
|
/**
|
|
208
213
|
* One observation inside a sealed run record.
|
|
209
214
|
*
|
|
210
|
-
* The
|
|
211
|
-
* would fail to parse. `
|
|
212
|
-
* the
|
|
213
|
-
*
|
|
215
|
+
* The eleven channels are total in the schema, so a caller naming only what it
|
|
216
|
+
* saw would fail to parse. `arguments` is the ninth call-input channel, added by
|
|
217
|
+
* the record's version 5 bump so what a tool call supplied has somewhere to
|
|
218
|
+
* live. TEA observes spawned commands and never a tool call, so it is stated as
|
|
219
|
+
* null the way the four HTTP channels are.
|
|
220
|
+
*
|
|
221
|
+
* `stdout`, `stderr` and each artifact are tagged bodies, the same three tags
|
|
222
|
+
* `createCommandLineAdapter` returns, so an observation built here and one read
|
|
223
|
+
* off the port have one shape.
|
|
214
224
|
*
|
|
215
225
|
* `provenance` defaults to `evaluator-chosen` because that is what every run this
|
|
216
226
|
* repository measures is: the harness chose the invocation. `matchProbeWitness`
|
|
@@ -244,6 +254,7 @@ function recordObservation({
|
|
|
244
254
|
option: callInputs.option ?? null,
|
|
245
255
|
environment: callInputs.environment ?? null,
|
|
246
256
|
stdin: callInputs.stdin ?? null,
|
|
257
|
+
arguments: callInputs.arguments ?? null,
|
|
247
258
|
},
|
|
248
259
|
responseBody: null,
|
|
249
260
|
responseHeaders: null,
|
|
@@ -294,7 +305,6 @@ function sealedRunRecord({
|
|
|
294
305
|
resourceUse,
|
|
295
306
|
truncationBound = null,
|
|
296
307
|
reportedIncomplete = false,
|
|
297
|
-
invalidReason = null,
|
|
298
308
|
}) {
|
|
299
309
|
return {
|
|
300
310
|
schemaVersion: SCHEMA_VERSIONS.sealedRunRecord,
|
|
@@ -319,7 +329,6 @@ function sealedRunRecord({
|
|
|
319
329
|
isolationManifestArtifact,
|
|
320
330
|
resourceUse,
|
|
321
331
|
evidenceDisclosure: { truncationBound, reportedIncomplete },
|
|
322
|
-
invalidReason,
|
|
323
332
|
};
|
|
324
333
|
}
|
|
325
334
|
|