bmad-method-test-architecture-enterprise 1.25.0 → 1.25.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/CHANGELOG.md +14 -0
- package/docs/explanation/eval-quality-command-adapter.md +147 -55
- package/package.json +2 -2
- package/test/contracts/trace.contract.json +5 -5
- package/test/eval-contract-strength.js +25 -22
- package/test/eval-trace.js +102 -29
- package/test/fixtures/trace-eval/ground-truth.json +2 -0
- package/test/fixtures/trace-runner/stub-agent.js +12 -1
- package/test/lib/probe-scoring.js +18 -4
- package/test/probes/README.md +45 -6
- package/test/probes/expected-strength.json +124 -33
- package/test/probes/trace.probes.json +29 -29
- package/test/replay/trace/clean-ac4-name-matched/test-artifacts/e2e-trace-summary.json +2 -2
- package/test/replay/trace/clean-correct-run/test-artifacts/e2e-trace-summary.json +2 -2
- package/test/replay/trace/clean-empty-waivers-block/test-artifacts/e2e-trace-summary.json +2 -2
- package/test/replay/trace/clean-matrix-without-sections/test-artifacts/e2e-trace-summary.json +2 -2
- package/test/replay/trace/seeded-ac2-title-matched/test-artifacts/e2e-trace-summary.json +4 -4
- package/test/replay/trace/seeded-arithmetic-off/test-artifacts/e2e-trace-summary.json +5 -5
- package/test/replay/trace/seeded-correct-run/test-artifacts/e2e-trace-summary.json +5 -5
- package/test/replay/trace/seeded-invented-and-duplicate-sections/test-artifacts/e2e-trace-summary.json +5 -5
- package/test/replay/trace/seeded-live-unverifiable/test-artifacts/e2e-trace-summary.json +5 -5
- package/test/replay/trace/seeded-matrix-lines-rejected/test-artifacts/e2e-trace-summary.json +5 -5
- package/test/replay/trace/seeded-missing-oracle-source/test-artifacts/e2e-trace-summary.json +4 -4
- package/test/replay/trace/seeded-rejected-evidence-omitted/test-artifacts/e2e-trace-summary.json +4 -4
- package/test/replay/trace/seeded-summary-schema-0-2/test-artifacts/e2e-trace-summary.json +4 -4
- package/test/replay/trace/seeded-waiver-misjudged/test-artifacts/e2e-trace-summary.json +5 -5
- package/test/test-probe-corpus.js +24 -7
- package/tools/generate-contracts.js +43 -34
- package/tools/generate-probes.js +10 -7
|
@@ -31,7 +31,7 @@
|
|
|
31
31
|
"name": "bmad-method-test-architecture-enterprise",
|
|
32
32
|
"source": "./",
|
|
33
33
|
"description": "Master Test Architect module for quality strategy, test automation, CI/CD quality gates, and structured testing education. Part of the BMad Method ecosystem.",
|
|
34
|
-
"version": "1.25.
|
|
34
|
+
"version": "1.25.1",
|
|
35
35
|
"author": {
|
|
36
36
|
"name": "Murat K Ozcan (TEA Creator) & Brian (BMad) Madison"
|
|
37
37
|
},
|
package/CHANGELOG.md
CHANGED
|
@@ -7,6 +7,20 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [1.25.1] - 2026-09-09
|
|
11
|
+
|
|
12
|
+
### Changed
|
|
13
|
+
|
|
14
|
+
- `eval-quality` moves from 1.3.0 to 1.4.0, which closes three findings this repository raised against it, and the probe corpora record what changed. Measured on the stored replay across the two versions: `test-review`'s five plants that had failed pre-flight on `seeded-faults-scoped` with a null verdict and exit 3 now pass pre-flight and score, taking the suite's defect class from four exercised and four caught to nine and nine at a rate of 1; `trace`'s O-023 and O-024 move from `abstained` to `passed-clean-control`, because a bare `count-tolerance` with `expected: 0` now counts a collection observed to be present and empty; and `trace`'s three defect probes report `condition-artifact-channel-contract-local`, the reason AD-9's gate refused them, where before a rejected probe surfaced only as `infrastructure-error` and exit 3. `expected-strength.json` records a `qualification` field per probe, so a rejection that changes its reason shows up as a diff.
|
|
15
|
+
- Each trace fixture set is staged under its own project root, declared as `projectRoot` in `test/fixtures/trace-eval/ground-truth.json` and named for the epic the set traces: `tenant-data-export` and `api-token-lifecycle`. Both sets used to stage at one `project/` root and `buildPrompt` read nothing off the set it was given, so the two sets sent byte-identical prompts and the staged workspace, which is the run's real input, was nameable from no request. The whole prompt is written against the set's root now, so a leg can ask for the set it means: `stagedWorkspaceFor` stages the set the prompt names, the stored-evidence port answers with that set's stored run, and the stub agent resolves `{project-root}` off the prompt the way a real agent does. `validateCorpus` fails a corpus whose two sets share a root or whose root names the set's role, because a run that read `clean` in the directory it works in would have been handed the answer.
|
|
16
|
+
|
|
17
|
+
### Fixed
|
|
18
|
+
|
|
19
|
+
- Four passages describing limitations that no longer exist. `test/probes/README.md` and `docs/explanation/eval-quality-command-adapter.md` each said a rejected probe carries no reason across the boundary; the adapter document also said "this collection is empty" has no spelling, and argued that no leg TEA could add repairs `seeded-faults-scoped` firing on a leg carrying the fault leg's own request. All four are rewritten against what 1.4.0 does, with the measured before and after. One finding stays open and belongs to TEA rather than to `eval-quality`: `trace`'s three defect probes address a file the command wrote, which the qualification gate refuses by design. The other is closed below.
|
|
20
|
+
- `trace`'s three defect probes failed pre-flight on `seeded-faults-scoped`, reported as `D-001: the manifestation witness fires on clean leg "witness-gate-withheld"`. The witness was right and the leg was not clean. AD-10 reads every other leg of an operation as a clean leg, `trace-fixture-set` declares `stateChangeMarker: true` so no control leg is planned for it, and the only other legs it had were its two sensitivity witness legs, both staging the seeded set. The manifestation witness asserts a coverage number the seeded workspace produces and `allow_gate` does not move, so it fired there truthfully. Both witness legs stage the clean set now, where P0 coverage is 100 and the relation asserting 50 resolves false, so the check is satisfied on evidence rather than on an authored relation. The `allow_gate` differential is unchanged and holds on either set. Measured on the stored replay, the three probes move from `preflight: failed: seeded-faults-scoped` to `preflight: passed` and lose the `pre-flight verdict did not pass` line from their basis; they are still refused by the qualification gate, so their verdict, exit code and strength are unchanged. All thirty-one probes in the corpus pre-flight now.
|
|
21
|
+
- Every signature channel `tea-trace-runner` could carry, measured through `runScore` rather than asserted: an `artifact` pointer is refused as `condition-artifact-channel-contract-local`, a `stdout` pointer as `condition-pointer-unwritable`, and an `exit-code` condition is admitted and discriminates nothing, since `cli/lib/runner-exit-codes.js` gives 0 to every run whose agent completed. The three probes keep the signature that states the truth about the plant, and `test/probes/README.md` and `docs/explanation/eval-quality-command-adapter.md` name the reason code rather than describing the refusal in prose.
|
|
22
|
+
- Stale numbers and claims throughout `docs/explanation/eval-quality-command-adapter.md`, checked against `test/probes/expected-strength.json` and against the contracts rather than reread. Its corpus table was pinned at `eval-quality` 1.3.0 and said `test-review` caught four of four with five plants firing on a witness leg; the baseline says nine of nine. It said eight probes cannot be pre-flighted; none is left. It said seven of `trace`'s twenty-six oracles abstain on the clean control; five do, measured on `P-004`, and all five are `for-any` quantifiers over the seeded export named in the artifact's own `verdictBasis`. It said no request shape names a trace fixture set, which the project root now does. The dated live pre-flight table is kept as the measurement it was, with what has moved under it since recorded beside it, including that `trace` now spawns three legs where it spawned two because its manifestation witnesses no longer share a request with a witness leg.
|
|
23
|
+
|
|
10
24
|
## [1.25.0] - 2026-09-09
|
|
11
25
|
|
|
12
26
|
### Added
|
|
@@ -27,7 +27,7 @@ That last one had no TEA equivalent at all. A harness could spawn anything.
|
|
|
27
27
|
|
|
28
28
|
**`eval-all.js`'s child spawn stays.** It uses `stdio: 'inherit'` so a forty-minute matrix prints as it goes. A probe captures and returns at the end, which is wrong for an operator watching one.
|
|
29
29
|
|
|
30
|
-
**The `--version`, `git`, and keychain probes stay.** They interrogate the environment rather than probe a system under test, so the adapter does not cover them. None passed a timeout, so any one could hang CI rather than fail it; all
|
|
30
|
+
**The `--version`, `git`, and keychain probes stay.** They interrogate the environment rather than probe a system under test, so the adapter does not cover them. None passed a timeout, so any one could hang CI rather than fail it; all eight now go through `test/lib/bounded-probe.js`, which is a ten-second deadline, a SIGKILL, and a reason the caller can act on. Eight is four `--version` probes, three `git` reads, and the keychain lookup.
|
|
31
31
|
|
|
32
32
|
## The policy is the seam
|
|
33
33
|
|
|
@@ -90,13 +90,19 @@ Every probe names the oracle that catches it, and that is enforced rather than i
|
|
|
90
90
|
generator reads a probe's `behaviorId` out of the contract and refuses one whose behavior discharges
|
|
91
91
|
more than a single oracle.
|
|
92
92
|
|
|
93
|
-
What the corpus scores,
|
|
93
|
+
What the corpus scores, read off `test/probes/expected-strength.json` as it stands:
|
|
94
94
|
|
|
95
95
|
| Contract | defect | gameability | Not scored, and why |
|
|
96
96
|
| -------------------------------------- | ------------- | --------------------- | ------------------------------------------------------------------------------------------------------- |
|
|
97
|
-
| `test-review` |
|
|
97
|
+
| `test-review` | 9 of 9 caught | refused | the gameability signature reads a written file, so AD-9's gate refuses it |
|
|
98
98
|
| the eight fragment-selection contracts | none authored | 1 of 1 caught on each | fragment selection seeds no defect; it is a routing measurement with a required set and a forbidden set |
|
|
99
|
-
| `trace` | refused | none authored | all three
|
|
99
|
+
| `trace` | refused | none authored | all three signatures read a written file, so AD-9's gate refuses them |
|
|
100
|
+
|
|
101
|
+
The two numbers that moved and what moved them: `test-review`'s defect class went from four exercised
|
|
102
|
+
and four caught to nine and nine when `eval-quality` 1.4.0 dropped a clean leg that had issued the
|
|
103
|
+
fault leg's own request, and `trace`'s three plants stopped failing pre-flight when its witness legs
|
|
104
|
+
moved off the seeded set. Both are `seeded-faults-scoped`, and both are below. Every probe in the
|
|
105
|
+
corpus now pre-flights, so what is left unscored is the qualification gate alone.
|
|
100
106
|
|
|
101
107
|
A clean control never enters the vector, which is AD-7's rule rather than a gap: what it establishes
|
|
102
108
|
is that the contract does not fire where there is nothing to find.
|
|
@@ -104,9 +110,10 @@ is that the contract does not fire where there is nothing to find.
|
|
|
104
110
|
`npm run eval:preflight` and `npm run eval:contract-strength` read their exit code against
|
|
105
111
|
`test/probes/expected-strength.json` rather than against pass and fail directly: 0 when every probe
|
|
106
112
|
reached the outcome the corpus records, 1 when a verdict moved, 2 when a pre-flight outcome moved.
|
|
107
|
-
Eight probes
|
|
108
|
-
|
|
109
|
-
|
|
113
|
+
Eight probes could not be pre-flighted when that rule was written, and both scripts would have been
|
|
114
|
+
red on every run, which is how a script stops being read before the day it means something. All
|
|
115
|
+
thirty-one pre-flight now, so the baseline is what says a green run is green rather than what excuses
|
|
116
|
+
a red one, and the rule is what catches the first probe to stop.
|
|
110
117
|
|
|
111
118
|
### The behavior grouping was a defect, and it is fixed
|
|
112
119
|
|
|
@@ -132,7 +139,9 @@ probe there would have been voted by `O-001` whatever criterion it withheld.
|
|
|
132
139
|
### What the probe vocabulary cannot say about a command
|
|
133
140
|
|
|
134
141
|
Four limits, all measured against the installed package rather than inferred, and all recorded in
|
|
135
|
-
`test/probes/expected-strength.json` so the day one closes is visible.
|
|
142
|
+
`test/probes/expected-strength.json` so the day one closes is visible. Three have closed since they
|
|
143
|
+
were written: two upstream in `eval-quality` 1.4.0 and one here, in the trace contract. Each is kept
|
|
144
|
+
with what closed it, because a limit that vanishes silently teaches nobody why it was there.
|
|
136
145
|
|
|
137
146
|
**A defect signature cannot address a file a command wrote.** `qualifyProbe` refuses an `artifact`
|
|
138
147
|
pointer outright as `condition-artifact-channel-contract-local`: an artifact identifier is minted per
|
|
@@ -141,7 +150,9 @@ contract, so a signature carrying one resolves only against the contract it was
|
|
|
141
150
|
output as its descriptor channel. Measured across TEA's three commands: a structured stdout signature
|
|
142
151
|
against `tea-fragment-selection-runner` qualifies, the same shape against `tea-test-review` does not,
|
|
143
152
|
an artifact signature against either is refused, and an `exit-code` signature qualifies against all
|
|
144
|
-
three.
|
|
153
|
+
three. Measured again on `tea-trace-runner` through `runScore`, one channel at a time over the same
|
|
154
|
+
stored evidence: `artifact` is refused as `condition-artifact-channel-contract-local`, `stdout` as
|
|
155
|
+
`condition-pointer-unwritable`, and `exit-code` is admitted.
|
|
145
156
|
|
|
146
157
|
What is left is the exit code, and whether that is honest depends on the command. For
|
|
147
158
|
`tea-test-review` it discriminates: a review that finds a gating defect exits 1 and one that finds
|
|
@@ -153,43 +164,105 @@ since every completed trace run exits 0 whatever it wrote, so its three probes k
|
|
|
153
164
|
that states the truth about the plant and are refused rather than given one that would qualify and
|
|
154
165
|
mean nothing.
|
|
155
166
|
|
|
156
|
-
**`seeded-faults-scoped` compares a run against itself.** AD-10 asks whether a seeded fault fires
|
|
157
|
-
it should not, and `planPreflight` answers it against every leg already registered for the
|
|
158
|
-
which for a TEA contract is its sensitivity witness legs. `test-review`'s differential
|
|
159
|
-
at a seeded fixture and `trace`'s
|
|
160
|
-
|
|
161
|
-
trace plants: the live pre-flight
|
|
162
|
-
"witness-gate-evaluated"`. The four review plants in the
|
|
163
|
-
and
|
|
167
|
+
**`seeded-faults-scoped` compares a run against itself.** AD-10 asks whether a seeded fault fires
|
|
168
|
+
anywhere it should not, and `planPreflight` answers it against every leg already registered for the
|
|
169
|
+
operation, which for a TEA contract is its sensitivity witness legs. `test-review`'s differential
|
|
170
|
+
drove one leg at a seeded fixture and `trace`'s drove both at the seeded set, so at `eval-quality`
|
|
171
|
+
1.3.0 a plant in a file a witness leg read fired on a leg the plan called clean. Five of the nine
|
|
172
|
+
review plants landed there, and all three trace plants: the live pre-flight reported `D-001: the
|
|
173
|
+
manifestation witness fires on clean leg "witness-gate-evaluated"`. The four review plants in the
|
|
174
|
+
file no witness leg reads pre-flighted cleanly and scored, and the defect class caught all four.
|
|
164
175
|
|
|
165
|
-
No leg TEA
|
|
176
|
+
No leg TEA could add repaired it, and the reason was sharper than "a plant sits in a file a witness
|
|
166
177
|
reads". For those five plants the fault leg's request is byte for byte the sensitivity leg's request:
|
|
167
178
|
the same executable, the same `--files`, the same `--json`, the same `--agent`. The observation cache
|
|
168
|
-
collapses them into one spawn for exactly that reason. So the check
|
|
179
|
+
collapses them into one spawn for exactly that reason. So the check resolved the manifestation
|
|
169
180
|
relation against a run identical to the fault run, and no relation true on the one can be false on
|
|
170
|
-
the other. Adding a leg that reads an unplanted fixture
|
|
171
|
-
leg that fires rather than on the absence of a leg that does not, and adding legs
|
|
172
|
-
to fire. Making the seeded witness leg read an unplanted fixture
|
|
181
|
+
the other. Adding a leg that reads an unplanted fixture changed nothing, because the check fails on a
|
|
182
|
+
leg that fires rather than on the absence of a leg that does not, and adding legs could only add ways
|
|
183
|
+
to fire. Making the seeded witness leg read an unplanted fixture was not available either: the
|
|
173
184
|
differential it asserts is that two file lists produce different severity counts, and two unplanted
|
|
174
185
|
lists produce the same counts, so the witness would fail instead. The one remaining shape, a witness
|
|
175
186
|
whose relation compares the `reviewedFiles` the verdict echoes back, is the "the evidence contains
|
|
176
187
|
the string I sent" condition the probe-side qualification gate exists to reject, and authoring it
|
|
177
188
|
contract-side to get a green pre-flight would be gaming the witness.
|
|
178
189
|
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
the
|
|
192
|
-
|
|
190
|
+
TEA raised it upstream, and 1.4.0 fixes it in the reducer. A clean leg is dropped when it issued the
|
|
191
|
+
fault leg's request and received the fault leg's answer, compared over the request with the
|
|
192
|
+
correlation identifier neutralised and over the projected evidence with the observation identifier
|
|
193
|
+
neutralised. Both halves are required: dropping on the answer alone would discard AD-10's own worked
|
|
194
|
+
example of two distinct nonexistent identifiers both returning 404, which are the legs the check
|
|
195
|
+
exists to read. A check left with no clean leg to examine now fails and names why, where it reported
|
|
196
|
+
satisfied before.
|
|
197
|
+
|
|
198
|
+
Measured on the stored replay across the two versions, `test-review`'s five plants move from
|
|
199
|
+
`preflight: failed: seeded-faults-scoped` with a null verdict and exit 3 to `preflight: passed`,
|
|
200
|
+
`CONCERNS`, exit 0, each carrying its own strength vector. The suite's defect class goes from four
|
|
201
|
+
exercised and four caught to nine and nine, still at a rate of 1. `expected-strength.json` records
|
|
202
|
+
the move.
|
|
203
|
+
|
|
204
|
+
`trace`'s three plants failed on that same run, and that was the fix working: their witness fired on
|
|
205
|
+
`witness-gate-withheld`, a leg that issues a different request and receives a different answer, so
|
|
206
|
+
the reducer kept it in the examined set and the check reported a real scoping problem in the trace
|
|
207
|
+
contract. That problem is now closed, and the shape of it is worth keeping.
|
|
208
|
+
|
|
209
|
+
The witness was right and the leg was not clean. AD-10 reads every other leg of an operation as a
|
|
210
|
+
clean leg. `trace-fixture-set` carries `stateChangeMarker: true`, so `selectControl` plans no control
|
|
211
|
+
leg for it, and the only other legs it had were its two sensitivity witness legs, both staged against
|
|
212
|
+
the seeded set. `D-001` asserts that P0 coverage is 50, which is what the seeded workspace produces
|
|
213
|
+
and what `allow_gate` does not move, so the relation was true on a seeded run with the gate withheld.
|
|
214
|
+
It fired there because the plant was there.
|
|
215
|
+
|
|
216
|
+
Nothing in the request could say otherwise. The run's real input is the staged workspace, both sets
|
|
217
|
+
were staged at one `project/` root, and `buildPrompt` took a fixture set and read nothing off it, so
|
|
218
|
+
the two sets sent byte-identical prompts and no leg could ask for the set without the plant. Each set
|
|
219
|
+
now declares a `projectRoot` in `test/fixtures/trace-eval/ground-truth.json`,
|
|
220
|
+
`tenant-data-export` and `api-token-lifecycle`, and the whole prompt is written against it:
|
|
221
|
+
`{project-root}`, `{config_source}`, `{test_artifacts}`, `{test_dir}`, `{source_dir}`, the epic
|
|
222
|
+
directory, and both deliverable paths. `stagedWorkspaceFor` stages the set the leg's prompt names,
|
|
223
|
+
the stored-evidence port answers with that set's stored run, and the stub agent resolves
|
|
224
|
+
`{project-root}` off the prompt the way a real agent does.
|
|
225
|
+
|
|
226
|
+
The contract's two witness legs now trace the clean set, which is the set that establishes what
|
|
227
|
+
"clean leg" means for this operation. P0 coverage there is 100, so `D-001`'s relation resolves false
|
|
228
|
+
on both legs and the check is satisfied on evidence. The `allow_gate` differential is untouched and
|
|
229
|
+
holds on either set: the clean set writes `gate_basis: "priority_thresholds"` when the gate is
|
|
230
|
+
allowed and `"none"` when it is withheld, the same pair the seeded set writes. Measured on the stored
|
|
231
|
+
replay, the three probes move from `preflight: failed: seeded-faults-scoped` to `preflight: passed`,
|
|
232
|
+
and their basis loses the `pre-flight verdict did not pass` line. They are still refused by the
|
|
233
|
+
qualification gate, so their verdict, exit code and strength are unchanged.
|
|
234
|
+
|
|
235
|
+
The project roots name the epic each set traces and not the set's role. A run that read `clean` in
|
|
236
|
+
the directory it works in would have been handed the answer, which is the rule the corpus already
|
|
237
|
+
follows when it keeps every inline label out of the fixture files, so `validateCorpus` fails a
|
|
238
|
+
project root carrying `seeded`, `clean`, `control`, `planted` or `gap`.
|
|
239
|
+
|
|
240
|
+
**"This collection is empty" has a spelling, as of 1.4.0.** Through 1.3.0 the evaluator intercepted
|
|
241
|
+
an empty array on every quantifier and every single-operand leaf and returned `insufficient-evidence`
|
|
242
|
+
with an `empty-collection` condition before the operator ran, so `count-tolerance` with `expected: 0`
|
|
243
|
+
never counted. `trace`'s O-023 and O-024 make exactly that claim about its clean set, and both
|
|
244
|
+
abstained on the run they were written to confirm. TEA raised it upstream and 1.4.0 exempts the three
|
|
245
|
+
operators that read a property of the collection itself: `count-tolerance` reads its cardinality,
|
|
246
|
+
`existence` and `absence` read its presence. Measured on the stored replay across the two versions,
|
|
247
|
+
O-023 and O-024 move from `abstained` to `passed-clean-control`, so `count-tolerance` was the right
|
|
248
|
+
spelling to have committed to.
|
|
249
|
+
|
|
250
|
+
One limit is worth knowing before writing a new oracle. Every quantifier still abstains over an empty
|
|
251
|
+
collection, which is AD-4's whole purpose and is why `trace`'s five `for-any` oracles over the seeded
|
|
252
|
+
export still abstain on a clean run: they ask whether some element exists, and an empty collection is
|
|
253
|
+
an honest "nothing was checked". `deep-equality` against a literal `[]` also still abstains, so the
|
|
254
|
+
two spellings of "this collection is empty" disagree. eval-quality records that disagreement in AD-4
|
|
255
|
+
rather than hiding it. The bare `count-tolerance` assertion is the one to write.
|
|
256
|
+
|
|
257
|
+
**A rejected probe names its reason, as of 1.4.0.** The qualification gate computes a closed list of
|
|
258
|
+
twenty reason codes. Through 1.3.0 none of them reached the evidence artifact or any published
|
|
259
|
+
export, so a probe the gate rejected surfaced as `infrastructure-error` and exit 3 and a corpus
|
|
260
|
+
author reading that had nothing to act on. Reading the reasons meant calling `qualifyProbe`, which
|
|
261
|
+
was off the exports map, and taking it would have been the third reach into `dist/` this document
|
|
262
|
+
already records two of. TEA raised that upstream instead. `runScore` now returns `qualification`
|
|
263
|
+
beside the artifact and the ladder, `QUALIFICATION_FAILURES` publishes the closed set, and
|
|
264
|
+
`test-probe-corpus` records the codes per probe in `expected-strength.json`. No reach into `dist/`
|
|
265
|
+
was needed and the count stays at two.
|
|
193
266
|
|
|
194
267
|
### What the scoring half says about TEA's contracts
|
|
195
268
|
|
|
@@ -200,16 +273,25 @@ Three findings, all measured, none of them tuned away.
|
|
|
200
273
|
before. `test-review` leaves `whole-body`, `malformed-input` and `state-change-read-back`
|
|
201
274
|
unsatisfied; every fragment-selection contract leaves `malformed-input` unsatisfied. Each scores the
|
|
202
275
|
run down to CONCERNS without blocking it, which is exactly the weight AD-20 gives a coverage gap.
|
|
203
|
-
- **`trace`'s clean control scores FAIL, and it is
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
276
|
+
- **`trace`'s clean control scores FAIL, and what remains of it is one thing.** Measured on the
|
|
277
|
+
stored replay, five of its twenty-six oracles abstain on `P-004`: `O-009`, `O-010`, `O-011`,
|
|
278
|
+
`O-013` and `O-014`, every one a `for-any` quantifier over the seeded export, and every one named
|
|
279
|
+
in the artifact's `verdictBasis`. It was seven. `O-023` and `O-024`, the two asserting that a
|
|
280
|
+
collection is empty, moved to `passed-clean-control` when `eval-quality` 1.4.0 gave that claim a
|
|
281
|
+
spelling.
|
|
282
|
+
|
|
283
|
+
The five abstain because the record carries only the clean set's observation and both plan steps
|
|
284
|
+
select it. The reason recorded here used to be that no request shape names a fixture set; that is
|
|
285
|
+
no longer true, since each set is staged under its own project root and the prompt is written
|
|
286
|
+
against it. What is left is the binding: `stdin.prompt` is bound `{matcher: 'any'}`, so one
|
|
287
|
+
observation is selected by both steps, which is the same limit as the bullet below. The trace
|
|
288
|
+
prompt is 1.8 kilobytes rather than fragment selection's 28, so the size argument against a literal
|
|
289
|
+
does not apply here either. What a literal binding needs beside it is a record carrying both sets'
|
|
290
|
+
observations, and whether one probe's record may carry the run of a set it did not seed is a
|
|
291
|
+
question about what a clean control means rather than a mechanical change, so it is not made here.
|
|
292
|
+
Neither half is a case for giving the clean set a member so an oracle has something to read: that
|
|
293
|
+
would mutate a control to make the instrument work.
|
|
294
|
+
|
|
213
295
|
- **A plan cannot tell two steps apart when both bind their inputs by matcher.** Each
|
|
214
296
|
fragment-selection contract declares one plan step per case, distinguished only by the prompt, and
|
|
215
297
|
the prompt is bound `{matcher: 'any'}` because the alternative is a 28-kilobyte literal per step. A
|
|
@@ -221,11 +303,12 @@ Three findings, all measured, none of them tuned away.
|
|
|
221
303
|
|
|
222
304
|
### What the live pre-flight measured
|
|
223
305
|
|
|
224
|
-
The first pre-flight this repository has ever run, on 2026-09-09 against `claude
|
|
225
|
-
spawned, 55 minutes of model time, and every leg cached under a digest of its
|
|
226
|
-
invocation pays for nothing it has already answered. The three manifestation
|
|
227
|
-
|
|
228
|
-
already
|
|
306
|
+
The first pre-flight this repository has ever run, on 2026-09-09 against `claude`, at `eval-quality`
|
|
307
|
+
1.3.0. Twenty-one legs spawned, 55 minutes of model time, and every leg cached under a digest of its
|
|
308
|
+
request so a second invocation pays for nothing it has already answered. The three manifestation
|
|
309
|
+
witnesses `trace` gained at 1.3.0 cost nothing on that run: their request was the one its
|
|
310
|
+
`allow_gate: true` witness leg already sent, and the cache is keyed on the request. The table is that
|
|
311
|
+
run and has not been re-measured; what has changed under it since is below.
|
|
229
312
|
|
|
230
313
|
| Suite | Legs spawned | Model time | Pre-flight |
|
|
231
314
|
| -------------------------------------- | -----------: | ---------: | ------------------------------------------------------------------------------------ |
|
|
@@ -241,6 +324,15 @@ the `false` leg wrote `gate_basis: "none"` and no gate at all, which is what ste
|
|
|
241
324
|
`test-review`'s differential is the file list: the seeded fixture drew five findings and exit 1, the
|
|
242
325
|
clean control drew none and exit 0.
|
|
243
326
|
|
|
327
|
+
Two things about that table have moved since, both on the stored replay rather than on a repeat of
|
|
328
|
+
the live run. `test-review`'s pre-flight outcome is now nine of nine defect probes rather than four,
|
|
329
|
+
because `eval-quality` 1.4.0 drops a clean leg that issued the fault leg's own request. `trace`'s is
|
|
330
|
+
now all four probes passing, because its witness legs moved to the clean set. That move also costs a
|
|
331
|
+
leg: the three manifestation witnesses trace the seeded set while the two witness legs trace the
|
|
332
|
+
clean set, so their request is no longer one the cache already holds and `trace` spawns three legs
|
|
333
|
+
where it spawned two. The three still share one request between them, so it is one extra spawn and
|
|
334
|
+
not three.
|
|
335
|
+
|
|
244
336
|
Two things the live run found that no deterministic check could.
|
|
245
337
|
|
|
246
338
|
**The two witness legs shared a staged workspace, and the second read the first's file.** The first
|
|
@@ -265,8 +357,8 @@ prompt the harness assembles for case X", which parses, compiles, and is schedul
|
|
|
265
357
|
then measures nothing when a real agent is finally handed it. `tools/generate-contracts.js` reads
|
|
266
358
|
`buildPrompt` out of `test/eval-fragment-selection.js` now, the same way the trace witness already
|
|
267
359
|
read its two prompts from its own harness, so a leg sends the prompt the suite sends. The contracts
|
|
268
|
-
grew from around 20 kilobytes to
|
|
269
|
-
same correctness.
|
|
360
|
+
grew from around 20 kilobytes to between 59 and 109, which is what the trace contract already paid
|
|
361
|
+
for the same correctness.
|
|
270
362
|
|
|
271
363
|
`test-review`'s request shape declared an environment permitting three API keys and forbidding `HOME`.
|
|
272
364
|
The adapter closes the child environment to `PATH` plus what the request declares, and
|
|
@@ -284,9 +376,9 @@ Two path couplings closed with them. `runReview` states `--project-root` because
|
|
|
284
376
|
|
|
285
377
|
Also done: `eval-trace.js`'s `runCase` probes `tea-trace-runner` (`cli/trace-runner.js`), the command the trace suite had no equivalent of. The runner's whole surface is a prompt on standard input and an agent run in the working directory. It builds no prompt and knows no vendor, and it declares `scoped-artifact-writes`, the one capability a trace run needs and a selection does not. The two workflow artifacts are the authorization's artifact map, supplied per run because each case stages its own workspace, and they come back tagged, so a file the run never wrote is `absent` and a missing artifact rather than an `existsSync` race. Both runner commands share one exit-code table in `cli/lib/runner-exit-codes.js`, because a caller holding an observation cannot tell which command produced it. `trace.contract.json` is the tenth contract, generated from `test/fixtures/trace-eval/ground-truth.json` with 26 oracles over the summary artifact, and `npm run test:contract-oracles` evaluates every one of them over the fourteen stored trace runs, under both fixture sets' oracles, and compares each with the `scoreRun` check it restates. The whole chain was verified with a stub vendor: `npm run test:probe-targets` spawns the harness itself, reads its result record back, and sees every threshold met on a correct run, a quality failure on a run that wrote a test, and a missing-artifact failure on a run that wrote nothing.
|
|
286
378
|
|
|
287
|
-
The trace witness is a differential over standard input on one prompt value, `allow_gate`. The
|
|
379
|
+
The trace witness is a differential over standard input on one prompt value, `allow_gate`. The differential is on that value rather than between the two fixture sets, because the seeded and clean summaries differ from the staged files rather than from the prompt, so a differential between the sets would attribute to the prompt a difference the prompt did not cause, and an invariance claim would be false because the two summaries differ. `allow_gate` is the one prompt value the corpus establishes an effect for: step-05 evaluates a gate only when it is true and writes `gate_basis` as `none` otherwise, so two prompts differing in that value over one staged set produce two `gate_basis` values. Both witness legs stage the clean set, which is what makes them clean legs for `seeded-faults-scoped`.
|
|
288
380
|
|
|
289
|
-
Two more couplings, both found by running it. A relative `--agent-cmd` passed the harness pre-flight, which probes it from the harness's own directory, and then failed every run, because the runner executes in the staged workspace and a relative path resolves there; `parseArgs` resolves a path against the operator's directory now. And the trace witness legs are runnable only against a staged workspace
|
|
381
|
+
Two more couplings, both found by running it. A relative `--agent-cmd` passed the harness pre-flight, which probes it from the harness's own directory, and then failed every run, because the runner executes in the staged workspace and a relative path resolves there; `parseArgs` resolves a path against the operator's directory now. And the trace witness legs are runnable only against a staged workspace, because the run's real input is the working directory; that is the same coupling `test-review`'s legs have with `--project-root`. Which set that workspace holds is in the request now: each fixture set declares its own `projectRoot`, the prompt is written against it, and `stagedWorkspaceFor` stages the set the leg names. Staging one set for every leg is what made the two witness legs seeded runs, which is the scoping failure `seeded-faults-scoped` reported against all three defect probes.
|
|
290
382
|
|
|
291
383
|
One coupling worth naming, and it did not change with the scoring half: two of the entry points TEA
|
|
292
384
|
reaches are addressed by file path into `dist/`, because neither the compiler CLI nor the evaluator is
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"$schema": "https://json.schemastore.org/package.json",
|
|
3
3
|
"name": "bmad-method-test-architecture-enterprise",
|
|
4
|
-
"version": "1.25.
|
|
4
|
+
"version": "1.25.1",
|
|
5
5
|
"description": "Master Test Architect for quality strategy, test automation, and release gates",
|
|
6
6
|
"keywords": [
|
|
7
7
|
"bmad",
|
|
@@ -137,7 +137,7 @@
|
|
|
137
137
|
"eslint-plugin-n": "^17.21.3",
|
|
138
138
|
"eslint-plugin-unicorn": "^60.0.0",
|
|
139
139
|
"eslint-plugin-yml": "^1.18.0",
|
|
140
|
-
"eval-quality": "1.
|
|
140
|
+
"eval-quality": "1.4.0",
|
|
141
141
|
"husky": "^9.1.7",
|
|
142
142
|
"jest": "^30.0.4",
|
|
143
143
|
"lint-staged": "^16.1.1",
|
|
@@ -2640,7 +2640,7 @@
|
|
|
2640
2640
|
"environment": {},
|
|
2641
2641
|
"stdin": {
|
|
2642
2642
|
"kind": "text",
|
|
2643
|
-
"value": "You are running the TEA workflow `bmad-testarch-trace` against the project in `
|
|
2643
|
+
"value": "You are running the TEA workflow `bmad-testarch-trace` against the project in `api-token-lifecycle/`.\n\nThe workflow is in `skill/`. Read `skill/instructions.md` first, then execute every step file it\nnames in order, in full, without skipping or reordering. The step files are under `skill/steps-c/`.\n\n----- run configuration -----\nResolve the workflow placeholders to these values:\n\n- `{project-root}`: `api-token-lifecycle`\n- `{config_source}`: `api-token-lifecycle/_bmad/tea/config.yaml`\n- `{test_artifacts}`: `api-token-lifecycle/test-artifacts`\n- `{test_dir}`: `api-token-lifecycle/tests`\n- `{source_dir}`: `api-token-lifecycle/src`\n- `{skill-root}`: `skill`\n- `gate_type`: `epic`\n- `decision_mode`: `deterministic`\n- `collection_mode`: `contract_static`\n- `allow_gate`: `true`\n- `coverage_basis`: `auto`\n- `summary_confidence`: `auto`\n- `coverage_levels`: `e2e,api,component,unit,live`\n\nThe trace target is the epic under `api-token-lifecycle/docs/epics/`. Resolve the coverage oracle from it the\nway step-01 says to.\n\n----- what to produce -----\nWrite both deliverables the workflow declares:\n\n- `api-token-lifecycle/test-artifacts/traceability-matrix.md`, from `skill/trace-template.md`, carrying the\n detailed mapping with one section per criterion, each stating its coverage status and the tests\n that establish it as `file:line`.\n- `api-token-lifecycle/test-artifacts/e2e-trace-summary.json` at schema_version 0.3.0, exactly as\n `skill/steps-c/step-05-gate-decision.md` section 3b defines it.\n\nDo not add, edit, or delete any file under `api-token-lifecycle/docs/`, `api-token-lifecycle/src/`, or `api-token-lifecycle/tests/`.\nThis workflow does not generate tests.\n\nWhen you are done, print one line naming the two files you wrote. Nothing else you print is read."
|
|
2644
2644
|
}
|
|
2645
2645
|
}
|
|
2646
2646
|
},
|
|
@@ -2654,7 +2654,7 @@
|
|
|
2654
2654
|
"environment": {},
|
|
2655
2655
|
"stdin": {
|
|
2656
2656
|
"kind": "text",
|
|
2657
|
-
"value": "You are running the TEA workflow `bmad-testarch-trace` against the project in `
|
|
2657
|
+
"value": "You are running the TEA workflow `bmad-testarch-trace` against the project in `api-token-lifecycle/`.\n\nThe workflow is in `skill/`. Read `skill/instructions.md` first, then execute every step file it\nnames in order, in full, without skipping or reordering. The step files are under `skill/steps-c/`.\n\n----- run configuration -----\nResolve the workflow placeholders to these values:\n\n- `{project-root}`: `api-token-lifecycle`\n- `{config_source}`: `api-token-lifecycle/_bmad/tea/config.yaml`\n- `{test_artifacts}`: `api-token-lifecycle/test-artifacts`\n- `{test_dir}`: `api-token-lifecycle/tests`\n- `{source_dir}`: `api-token-lifecycle/src`\n- `{skill-root}`: `skill`\n- `gate_type`: `epic`\n- `decision_mode`: `deterministic`\n- `collection_mode`: `contract_static`\n- `allow_gate`: `false`\n- `coverage_basis`: `auto`\n- `summary_confidence`: `auto`\n- `coverage_levels`: `e2e,api,component,unit,live`\n\nThe trace target is the epic under `api-token-lifecycle/docs/epics/`. Resolve the coverage oracle from it the\nway step-01 says to.\n\n----- what to produce -----\nWrite both deliverables the workflow declares:\n\n- `api-token-lifecycle/test-artifacts/traceability-matrix.md`, from `skill/trace-template.md`, carrying the\n detailed mapping with one section per criterion, each stating its coverage status and the tests\n that establish it as `file:line`.\n- `api-token-lifecycle/test-artifacts/e2e-trace-summary.json` at schema_version 0.3.0, exactly as\n `skill/steps-c/step-05-gate-decision.md` section 3b defines it.\n\nDo not add, edit, or delete any file under `api-token-lifecycle/docs/`, `api-token-lifecycle/src/`, or `api-token-lifecycle/tests/`.\nThis workflow does not generate tests.\n\nWhen you are done, print one line naming the two files you wrote. Nothing else you print is read."
|
|
2658
2658
|
}
|
|
2659
2659
|
}
|
|
2660
2660
|
}
|
|
@@ -2738,7 +2738,7 @@
|
|
|
2738
2738
|
"human-labels"
|
|
2739
2739
|
],
|
|
2740
2740
|
"testData": {
|
|
2741
|
-
"setup": "Each plan step stages one fixture set from test/fixtures/trace-eval/ into a disposable workspace: the set's files under project
|
|
2741
|
+
"setup": "Each plan step stages one fixture set from test/fixtures/trace-eval/ into a disposable workspace: the set's files under its own project root (tenant-data-export/ for seeded-tenant-data-export, api-token-lifecycle/ for clean-api-token-lifecycle), a resolved _bmad/tea/config.yaml whose test_artifacts points inside that workspace, and the bmad-testarch-trace workflow under skill/. ground-truth.json is never staged, and the harness asserts that no staged file carries its bytes or its keys before the run. The workspace is the authorization's working directory, and the prompt on standard input names the project root and skill/ and resolves every placeholder against them. The project root is the one fact about the set the prompt carries, and it carries the epic the set traces rather than the set's role, so it says which set a leg is asking for and suggests no coverage status. The sensitivity witness differs its two legs on allow_gate rather than between the two sets: the seeded and clean summaries differ because of the staged workspace, so a differential between the sets would attribute to the prompt a difference the staged files produced, and an invariance claim would be false. allow_gate is the one prompt value the ground truth establishes an effect for, through skillRuleCitations.gateEligibility: step-05 evaluates a gate only when it is true and writes gate_basis as none otherwise, so two prompts differing in that value, over one staged fixture set, produce two gate_basis values. That is the same reasoning that gives the fragment-selection contract for this workflow an invariance witness: the claim is moved onto an input the run demonstrably reads, and stated as what it is. Both witness legs stage the clean set, because they are the only other legs this operation has and AD-10 reads them as its clean legs when it asks whether a seeded fault is scoped to its own leg.",
|
|
2742
2742
|
"cleanup": "Delete the workspace. The corpus under test/fixtures/trace-eval/ is read-only and the harness digests it before and after every run.",
|
|
2743
2743
|
"principals": null,
|
|
2744
2744
|
"resources": null
|
|
@@ -2749,8 +2749,8 @@
|
|
|
2749
2749
|
"maxCostUsd": "8.00"
|
|
2750
2750
|
},
|
|
2751
2751
|
"safetyLimits": [
|
|
2752
|
-
"The runner writes only inside the staged workspace, and only its two deliverables under
|
|
2753
|
-
"The run adds, edits, and deletes nothing under
|
|
2752
|
+
"The runner writes only inside the staged workspace, and only its two deliverables under the fixture set's own test-artifacts/; the harness fails a run that changed the repository or the staged corpus.",
|
|
2753
|
+
"The run adds, edits, and deletes nothing under the fixture set's docs/, src/, or tests/. The workflow does not generate tests, and a run that did has moved the benchmark.",
|
|
2754
2754
|
"No credential value appears in a prompt, an artifact, a log, or a result file."
|
|
2755
2755
|
],
|
|
2756
2756
|
"requiredEvidence": [
|
|
@@ -36,9 +36,9 @@
|
|
|
36
36
|
* test-review leg names its fixtures by repository-relative path and writes
|
|
37
37
|
* `verdict.json` beside them, and the policy's `cwd` is what both resolve
|
|
38
38
|
* against, so the fixtures are copied into the run directory and the artifact
|
|
39
|
-
* lands there. A trace leg needs
|
|
40
|
-
* test/eval-trace.js already does for its own runs
|
|
41
|
-
* disk at all.
|
|
39
|
+
* lands there. A trace leg needs its own fixture set staged, which is what
|
|
40
|
+
* test/eval-trace.js already does for its own runs, and the leg's prompt says which
|
|
41
|
+
* set that is. A selection needs nothing on disk at all.
|
|
42
42
|
*
|
|
43
43
|
* Usage:
|
|
44
44
|
* npm run eval:preflight # the live pre-flight, cached, nothing scored
|
|
@@ -49,9 +49,9 @@
|
|
|
49
49
|
* Exit codes are read against `test/probes/expected-strength.json`, the same
|
|
50
50
|
* baseline the deterministic gate compares to: 0 when every probe reached the
|
|
51
51
|
* outcome the corpus records, 1 when a verdict moved, 2 when a pre-flight outcome
|
|
52
|
-
* moved. Eight of the thirty-one probes
|
|
53
|
-
*
|
|
54
|
-
*
|
|
52
|
+
* moved. Eight of the thirty-one probes could not be pre-flighted when that rule
|
|
53
|
+
* was written, and the baseline said so; all thirty-one pre-flight now, so a
|
|
54
|
+
* failure here is news and the baseline is what says so.
|
|
55
55
|
*/
|
|
56
56
|
|
|
57
57
|
'use strict';
|
|
@@ -64,7 +64,7 @@ const { digest } = require('./lib/eval-record');
|
|
|
64
64
|
const { validateArtifact } = require('./lib/eval-quality-inputs');
|
|
65
65
|
const { createProbePort, hostEnvironment } = require('./lib/probe-targets');
|
|
66
66
|
const { runSuite, sealContract, suites } = require('./lib/probe-scoring');
|
|
67
|
-
const { stageWorkspace } = require('./eval-trace');
|
|
67
|
+
const { stageWorkspace, traceArtifactPaths } = require('./eval-trace');
|
|
68
68
|
|
|
69
69
|
const PROJECT_ROOT = path.join(__dirname, '..');
|
|
70
70
|
const REVIEW_FIXTURE_DIR = path.join('test', 'fixtures', 'test-review-eval');
|
|
@@ -160,9 +160,13 @@ const slug = (suiteId) => suiteId.replaceAll(/[^a-z\d]+/gi, '-');
|
|
|
160
160
|
* The run directory one suite's legs execute in, and the artifact paths that
|
|
161
161
|
* directory makes true.
|
|
162
162
|
*
|
|
163
|
-
* A trace leg is staged by the trace harness itself, so the
|
|
164
|
-
*
|
|
165
|
-
*
|
|
163
|
+
* A trace leg is staged by the trace harness itself, so the set arrives exactly as
|
|
164
|
+
* `npm run eval:trace` stages it, under the same project root its artifact override
|
|
165
|
+
* names. Which set that is comes out of the leg's own prompt: each fixture set has
|
|
166
|
+
* its own project root and the prompt is written against it, so a leg that traces
|
|
167
|
+
* the clean set asks for the clean set. Staging one set for every leg is what made
|
|
168
|
+
* the contract's two witness legs seeded runs, which is the scoping failure
|
|
169
|
+
* `seeded-faults-scoped` reported against all three defect probes.
|
|
166
170
|
*
|
|
167
171
|
* A test-review leg is the coupling `docs/explanation/eval-quality-command-adapter.md`
|
|
168
172
|
* records: its `--files` are repository-relative, its `--json` is a bare
|
|
@@ -177,20 +181,19 @@ const slug = (suiteId) => suiteId.replaceAll(/[^a-z\d]+/gi, '-');
|
|
|
177
181
|
* A selection leg gets an empty directory, which is what its `read-only`
|
|
178
182
|
* declaration is for.
|
|
179
183
|
*/
|
|
180
|
-
function stagedWorkspaceFor(suiteId) {
|
|
184
|
+
function stagedWorkspaceFor(suiteId, request) {
|
|
181
185
|
if (suiteId === 'trace') {
|
|
182
186
|
const groundTruth = JSON.parse(fs.readFileSync(TRACE_GROUND_TRUTH, 'utf8'));
|
|
183
|
-
const
|
|
184
|
-
const
|
|
187
|
+
const prompt = String(request?.channels?.stdin?.value ?? '');
|
|
188
|
+
const set = groundTruth.fixtureSets.find((entry) => prompt.includes(`\`{project-root}\`: \`${entry.projectRoot}\``));
|
|
189
|
+
if (set === undefined) {
|
|
190
|
+
throw new Error('a trace leg sent a prompt naming no fixture set project root, so there is no set to stage for it');
|
|
191
|
+
}
|
|
192
|
+
const staged = stageWorkspace(set);
|
|
185
193
|
return {
|
|
186
194
|
root: staged.dir,
|
|
187
195
|
cwd: staged.dir,
|
|
188
|
-
artifacts: {
|
|
189
|
-
'tea-trace-runner': {
|
|
190
|
-
summary: path.join('project', 'test-artifacts', 'e2e-trace-summary.json'),
|
|
191
|
-
matrix: path.join('project', 'test-artifacts', 'traceability-matrix.md'),
|
|
192
|
-
},
|
|
193
|
-
},
|
|
196
|
+
artifacts: { 'tea-trace-runner': traceArtifactPaths(set) },
|
|
194
197
|
};
|
|
195
198
|
}
|
|
196
199
|
const dir = fs.mkdtempSync(path.join(os.tmpdir(), `tea-${slug(suiteId)}-preflight-`));
|
|
@@ -254,7 +257,7 @@ function cachingPort({ makePort, contract, cacheDir, agent, force, counters, log
|
|
|
254
257
|
},
|
|
255
258
|
};
|
|
256
259
|
log(` ${colors.yellow}running${colors.reset} leg ${request.probeId} (${key})`);
|
|
257
|
-
const { port: realPort, workspace } = await makePort();
|
|
260
|
+
const { port: realPort, workspace } = await makePort(augmented);
|
|
258
261
|
const startedAt = Date.now();
|
|
259
262
|
let observation;
|
|
260
263
|
try {
|
|
@@ -334,8 +337,8 @@ async function runOneSuite(suite, options, stats) {
|
|
|
334
337
|
port = cacheOnlyPort(cacheDir, [stats, suiteStats]);
|
|
335
338
|
} else {
|
|
336
339
|
port = cachingPort({
|
|
337
|
-
makePort: async () => {
|
|
338
|
-
const staged = stagedWorkspaceFor(suite.id);
|
|
340
|
+
makePort: async (request) => {
|
|
341
|
+
const staged = stagedWorkspaceFor(suite.id, request);
|
|
339
342
|
const { port: realPort } = await createProbePort({ cwd: staged.cwd, interfaceIds, artifacts: staged.artifacts });
|
|
340
343
|
return { port: realPort, workspace: staged };
|
|
341
344
|
},
|