bmad-method-test-architecture-enterprise 1.25.0 → 1.25.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. package/.claude-plugin/marketplace.json +1 -1
  2. package/CHANGELOG.md +14 -0
  3. package/docs/explanation/eval-quality-command-adapter.md +147 -55
  4. package/package.json +2 -2
  5. package/test/contracts/trace.contract.json +5 -5
  6. package/test/eval-contract-strength.js +25 -22
  7. package/test/eval-trace.js +102 -29
  8. package/test/fixtures/trace-eval/ground-truth.json +2 -0
  9. package/test/fixtures/trace-runner/stub-agent.js +12 -1
  10. package/test/lib/probe-scoring.js +18 -4
  11. package/test/probes/README.md +45 -6
  12. package/test/probes/expected-strength.json +124 -33
  13. package/test/probes/trace.probes.json +29 -29
  14. package/test/replay/trace/clean-ac4-name-matched/test-artifacts/e2e-trace-summary.json +2 -2
  15. package/test/replay/trace/clean-correct-run/test-artifacts/e2e-trace-summary.json +2 -2
  16. package/test/replay/trace/clean-empty-waivers-block/test-artifacts/e2e-trace-summary.json +2 -2
  17. package/test/replay/trace/clean-matrix-without-sections/test-artifacts/e2e-trace-summary.json +2 -2
  18. package/test/replay/trace/seeded-ac2-title-matched/test-artifacts/e2e-trace-summary.json +4 -4
  19. package/test/replay/trace/seeded-arithmetic-off/test-artifacts/e2e-trace-summary.json +5 -5
  20. package/test/replay/trace/seeded-correct-run/test-artifacts/e2e-trace-summary.json +5 -5
  21. package/test/replay/trace/seeded-invented-and-duplicate-sections/test-artifacts/e2e-trace-summary.json +5 -5
  22. package/test/replay/trace/seeded-live-unverifiable/test-artifacts/e2e-trace-summary.json +5 -5
  23. package/test/replay/trace/seeded-matrix-lines-rejected/test-artifacts/e2e-trace-summary.json +5 -5
  24. package/test/replay/trace/seeded-missing-oracle-source/test-artifacts/e2e-trace-summary.json +4 -4
  25. package/test/replay/trace/seeded-rejected-evidence-omitted/test-artifacts/e2e-trace-summary.json +4 -4
  26. package/test/replay/trace/seeded-summary-schema-0-2/test-artifacts/e2e-trace-summary.json +4 -4
  27. package/test/replay/trace/seeded-waiver-misjudged/test-artifacts/e2e-trace-summary.json +5 -5
  28. package/test/test-probe-corpus.js +24 -7
  29. package/tools/generate-contracts.js +43 -34
  30. package/tools/generate-probes.js +10 -7
@@ -31,7 +31,7 @@
31
31
  "name": "bmad-method-test-architecture-enterprise",
32
32
  "source": "./",
33
33
  "description": "Master Test Architect module for quality strategy, test automation, CI/CD quality gates, and structured testing education. Part of the BMad Method ecosystem.",
34
- "version": "1.25.0",
34
+ "version": "1.25.1",
35
35
  "author": {
36
36
  "name": "Murat K Ozcan (TEA Creator) & Brian (BMad) Madison"
37
37
  },
package/CHANGELOG.md CHANGED
@@ -7,6 +7,20 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [1.25.1] - 2026-09-09
11
+
12
+ ### Changed
13
+
14
+ - `eval-quality` moves from 1.3.0 to 1.4.0, which closes three findings this repository raised against it, and the probe corpora record what changed. Measured on the stored replay across the two versions: `test-review`'s five plants that had failed pre-flight on `seeded-faults-scoped` with a null verdict and exit 3 now pass pre-flight and score, taking the suite's defect class from four exercised and four caught to nine and nine at a rate of 1; `trace`'s O-023 and O-024 move from `abstained` to `passed-clean-control`, because a bare `count-tolerance` with `expected: 0` now counts a collection observed to be present and empty; and `trace`'s three defect probes report `condition-artifact-channel-contract-local`, the reason AD-9's gate refused them, where before a rejected probe surfaced only as `infrastructure-error` and exit 3. `expected-strength.json` records a `qualification` field per probe, so a rejection that changes its reason shows up as a diff.
15
+ - Each trace fixture set is staged under its own project root, declared as `projectRoot` in `test/fixtures/trace-eval/ground-truth.json` and named for the epic the set traces: `tenant-data-export` and `api-token-lifecycle`. Both sets used to stage at one `project/` root and `buildPrompt` read nothing off the set it was given, so the two sets sent byte-identical prompts and the staged workspace, which is the run's real input, was nameable from no request. The whole prompt is written against the set's root now, so a leg can ask for the set it means: `stagedWorkspaceFor` stages the set the prompt names, the stored-evidence port answers with that set's stored run, and the stub agent resolves `{project-root}` off the prompt the way a real agent does. `validateCorpus` fails a corpus whose two sets share a root or whose root names the set's role, because a run that read `clean` in the directory it works in would have been handed the answer.
16
+
17
+ ### Fixed
18
+
19
+ - Four passages describing limitations that no longer exist. `test/probes/README.md` and `docs/explanation/eval-quality-command-adapter.md` each said a rejected probe carries no reason across the boundary; the adapter document also said "this collection is empty" has no spelling, and argued that no leg TEA could add repairs `seeded-faults-scoped` firing on a leg carrying the fault leg's own request. All four are rewritten against what 1.4.0 does, with the measured before and after. One finding stays open and belongs to TEA rather than to `eval-quality`: `trace`'s three defect probes address a file the command wrote, which the qualification gate refuses by design. The other is closed below.
20
+ - `trace`'s three defect probes failed pre-flight on `seeded-faults-scoped`, reported as `D-001: the manifestation witness fires on clean leg "witness-gate-withheld"`. The witness was right and the leg was not clean. AD-10 reads every other leg of an operation as a clean leg, `trace-fixture-set` declares `stateChangeMarker: true` so no control leg is planned for it, and the only other legs it had were its two sensitivity witness legs, both staging the seeded set. The manifestation witness asserts a coverage number the seeded workspace produces and `allow_gate` does not move, so it fired there truthfully. Both witness legs stage the clean set now, where P0 coverage is 100 and the relation asserting 50 resolves false, so the check is satisfied on evidence rather than on an authored relation. The `allow_gate` differential is unchanged and holds on either set. Measured on the stored replay, the three probes move from `preflight: failed: seeded-faults-scoped` to `preflight: passed` and lose the `pre-flight verdict did not pass` line from their basis; they are still refused by the qualification gate, so their verdict, exit code and strength are unchanged. All thirty-one probes in the corpus pre-flight now.
21
+ - Every signature channel `tea-trace-runner` could carry, measured through `runScore` rather than asserted: an `artifact` pointer is refused as `condition-artifact-channel-contract-local`, a `stdout` pointer as `condition-pointer-unwritable`, and an `exit-code` condition is admitted and discriminates nothing, since `cli/lib/runner-exit-codes.js` gives 0 to every run whose agent completed. The three probes keep the signature that states the truth about the plant, and `test/probes/README.md` and `docs/explanation/eval-quality-command-adapter.md` name the reason code rather than describing the refusal in prose.
22
+ - Stale numbers and claims throughout `docs/explanation/eval-quality-command-adapter.md`, checked against `test/probes/expected-strength.json` and against the contracts rather than reread. Its corpus table was pinned at `eval-quality` 1.3.0 and said `test-review` caught four of four with five plants firing on a witness leg; the baseline says nine of nine. It said eight probes cannot be pre-flighted; none is left. It said seven of `trace`'s twenty-six oracles abstain on the clean control; five do, measured on `P-004`, and all five are `for-any` quantifiers over the seeded export named in the artifact's own `verdictBasis`. It said no request shape names a trace fixture set, which the project root now does. The dated live pre-flight table is kept as the measurement it was, with what has moved under it since recorded beside it, including that `trace` now spawns three legs where it spawned two because its manifestation witnesses no longer share a request with a witness leg.
23
+
10
24
  ## [1.25.0] - 2026-09-09
11
25
 
12
26
  ### Added
@@ -27,7 +27,7 @@ That last one had no TEA equivalent at all. A harness could spawn anything.
27
27
 
28
28
  **`eval-all.js`'s child spawn stays.** It uses `stdio: 'inherit'` so a forty-minute matrix prints as it goes. A probe captures and returns at the end, which is wrong for an operator watching one.
29
29
 
30
- **The `--version`, `git`, and keychain probes stay.** They interrogate the environment rather than probe a system under test, so the adapter does not cover them. None passed a timeout, so any one could hang CI rather than fail it; all seven now go through `test/lib/bounded-probe.js`, which is a ten-second deadline, a SIGKILL, and a reason the caller can act on.
30
+ **The `--version`, `git`, and keychain probes stay.** They interrogate the environment rather than probe a system under test, so the adapter does not cover them. None passed a timeout, so any one could hang CI rather than fail it; all eight now go through `test/lib/bounded-probe.js`, which is a ten-second deadline, a SIGKILL, and a reason the caller can act on. Eight is four `--version` probes, three `git` reads, and the keychain lookup.
31
31
 
32
32
  ## The policy is the seam
33
33
 
@@ -90,13 +90,19 @@ Every probe names the oracle that catches it, and that is enforced rather than i
90
90
  generator reads a probe's `behaviorId` out of the contract and refuses one whose behavior discharges
91
91
  more than a single oracle.
92
92
 
93
- What the corpus scores, at `eval-quality` 1.3.0:
93
+ What the corpus scores, read off `test/probes/expected-strength.json` as it stands:
94
94
 
95
95
  | Contract | defect | gameability | Not scored, and why |
96
96
  | -------------------------------------- | ------------- | --------------------- | ------------------------------------------------------------------------------------------------------- |
97
- | `test-review` | 4 of 4 caught | refused | five plants fire on a witness leg the plan calls clean; the gameability signature reads a written file |
97
+ | `test-review` | 9 of 9 caught | refused | the gameability signature reads a written file, so AD-9's gate refuses it |
98
98
  | the eight fragment-selection contracts | none authored | 1 of 1 caught on each | fragment selection seeds no defect; it is a routing measurement with a required set and a forbidden set |
99
- | `trace` | refused | none authored | all three plants fire on a witness leg the plan calls clean, and the signature reads a written file |
99
+ | `trace` | refused | none authored | all three signatures read a written file, so AD-9's gate refuses them |
100
+
101
+ The two numbers that moved and what moved them: `test-review`'s defect class went from four exercised
102
+ and four caught to nine and nine when `eval-quality` 1.4.0 dropped a clean leg that had issued the
103
+ fault leg's own request, and `trace`'s three plants stopped failing pre-flight when its witness legs
104
+ moved off the seeded set. Both are `seeded-faults-scoped`, and both are below. Every probe in the
105
+ corpus now pre-flights, so what is left unscored is the qualification gate alone.
100
106
 
101
107
  A clean control never enters the vector, which is AD-7's rule rather than a gap: what it establishes
102
108
  is that the contract does not fire where there is nothing to find.
@@ -104,9 +110,10 @@ is that the contract does not fire where there is nothing to find.
104
110
  `npm run eval:preflight` and `npm run eval:contract-strength` read their exit code against
105
111
  `test/probes/expected-strength.json` rather than against pass and fail directly: 0 when every probe
106
112
  reached the outcome the corpus records, 1 when a verdict moved, 2 when a pre-flight outcome moved.
107
- Eight probes that cannot be pre-flighted would otherwise make both scripts red on every run, and a
108
- script that is always red stops being read before the day it means something, which is a declaration
109
- nothing enforces arriving from the other direction.
113
+ Eight probes could not be pre-flighted when that rule was written, and both scripts would have been
114
+ red on every run, which is how a script stops being read before the day it means something. All
115
+ thirty-one pre-flight now, so the baseline is what says a green run is green rather than what excuses
116
+ a red one, and the rule is what catches the first probe to stop.
110
117
 
111
118
  ### The behavior grouping was a defect, and it is fixed
112
119
 
@@ -132,7 +139,9 @@ probe there would have been voted by `O-001` whatever criterion it withheld.
132
139
  ### What the probe vocabulary cannot say about a command
133
140
 
134
141
  Four limits, all measured against the installed package rather than inferred, and all recorded in
135
- `test/probes/expected-strength.json` so the day one closes is visible.
142
+ `test/probes/expected-strength.json` so the day one closes is visible. Three have closed since they
143
+ were written: two upstream in `eval-quality` 1.4.0 and one here, in the trace contract. Each is kept
144
+ with what closed it, because a limit that vanishes silently teaches nobody why it was there.
136
145
 
137
146
  **A defect signature cannot address a file a command wrote.** `qualifyProbe` refuses an `artifact`
138
147
  pointer outright as `condition-artifact-channel-contract-local`: an artifact identifier is minted per
@@ -141,7 +150,9 @@ contract, so a signature carrying one resolves only against the contract it was
141
150
  output as its descriptor channel. Measured across TEA's three commands: a structured stdout signature
142
151
  against `tea-fragment-selection-runner` qualifies, the same shape against `tea-test-review` does not,
143
152
  an artifact signature against either is refused, and an `exit-code` signature qualifies against all
144
- three.
153
+ three. Measured again on `tea-trace-runner` through `runScore`, one channel at a time over the same
154
+ stored evidence: `artifact` is refused as `condition-artifact-channel-contract-local`, `stdout` as
155
+ `condition-pointer-unwritable`, and `exit-code` is admitted.
145
156
 
146
157
  What is left is the exit code, and whether that is honest depends on the command. For
147
158
  `tea-test-review` it discriminates: a review that finds a gating defect exits 1 and one that finds
@@ -153,43 +164,105 @@ since every completed trace run exits 0 whatever it wrote, so its three probes k
153
164
  that states the truth about the plant and are refused rather than given one that would qualify and
154
165
  mean nothing.
155
166
 
156
- **`seeded-faults-scoped` compares a run against itself.** AD-10 asks whether a seeded fault fires anywhere
157
- it should not, and `planPreflight` answers it against every leg already registered for the operation,
158
- which for a TEA contract is its sensitivity witness legs. `test-review`'s differential drives one leg
159
- at a seeded fixture and `trace`'s drives both at the seeded set, so a plant in a file a witness leg
160
- reads fires on a leg the plan calls clean. Five of the nine review plants land there, and all three
161
- trace plants: the live pre-flight reports `D-001: the manifestation witness fires on clean leg
162
- "witness-gate-evaluated"`. The four review plants in the file no witness leg reads pre-flight cleanly
163
- and score, and the defect class catches all four.
167
+ **`seeded-faults-scoped` compares a run against itself.** AD-10 asks whether a seeded fault fires
168
+ anywhere it should not, and `planPreflight` answers it against every leg already registered for the
169
+ operation, which for a TEA contract is its sensitivity witness legs. `test-review`'s differential
170
+ drove one leg at a seeded fixture and `trace`'s drove both at the seeded set, so at `eval-quality`
171
+ 1.3.0 a plant in a file a witness leg read fired on a leg the plan called clean. Five of the nine
172
+ review plants landed there, and all three trace plants: the live pre-flight reported `D-001: the
173
+ manifestation witness fires on clean leg "witness-gate-evaluated"`. The four review plants in the
174
+ file no witness leg reads pre-flighted cleanly and scored, and the defect class caught all four.
164
175
 
165
- No leg TEA can add repairs it, and the reason is sharper than "a plant sits in a file a witness
176
+ No leg TEA could add repaired it, and the reason was sharper than "a plant sits in a file a witness
166
177
  reads". For those five plants the fault leg's request is byte for byte the sensitivity leg's request:
167
178
  the same executable, the same `--files`, the same `--json`, the same `--agent`. The observation cache
168
- collapses them into one spawn for exactly that reason. So the check resolves the manifestation
179
+ collapses them into one spawn for exactly that reason. So the check resolved the manifestation
169
180
  relation against a run identical to the fault run, and no relation true on the one can be false on
170
- the other. Adding a leg that reads an unplanted fixture changes nothing, because the check fails on a
171
- leg that fires rather than on the absence of a leg that does not, and adding legs can only add ways
172
- to fire. Making the seeded witness leg read an unplanted fixture is not available either: the
181
+ the other. Adding a leg that reads an unplanted fixture changed nothing, because the check fails on a
182
+ leg that fires rather than on the absence of a leg that does not, and adding legs could only add ways
183
+ to fire. Making the seeded witness leg read an unplanted fixture was not available either: the
173
184
  differential it asserts is that two file lists produce different severity counts, and two unplanted
174
185
  lists produce the same counts, so the witness would fail instead. The one remaining shape, a witness
175
186
  whose relation compares the `reviewedFiles` the verdict echoes back, is the "the evidence contains
176
187
  the string I sent" condition the probe-side qualification gate exists to reject, and authoring it
177
188
  contract-side to get a green pre-flight would be gaming the witness.
178
189
 
179
- **"This collection is empty" has no spelling.** The evaluator intercepts an empty array on every
180
- quantifier and on every single-operand leaf and returns `insufficient-evidence` with an
181
- `empty-collection` condition before the operator runs, so `count-tolerance` with `expected: 0` never
182
- counts and `deep-equality` against a literal `[]` never compares. Two of `trace`'s oracles make
183
- exactly that claim about its clean set, and both abstain on the run they were written to confirm.
184
- `count-tolerance` is the spelling this repository now uses, because it is the claim the oracle is
185
- making and it starts working the day an empty collection counts as evidence for a cardinality check.
186
-
187
- **A rejected probe carries no reason across the boundary.** The qualification gate computes a closed
188
- list of twenty reason codes and none of them reaches the evidence artifact or any published export.
189
- A probe the gate rejects surfaces as `infrastructure-error` on every oracle and exit 3, and a corpus
190
- author reading that has nothing to act on. Reading the reasons needs `qualifyProbe`, which is not on
191
- the exports map, and that would be the third reach into `dist/` this document already records two of.
192
- TEA does not take it.
190
+ TEA raised it upstream, and 1.4.0 fixes it in the reducer. A clean leg is dropped when it issued the
191
+ fault leg's request and received the fault leg's answer, compared over the request with the
192
+ correlation identifier neutralised and over the projected evidence with the observation identifier
193
+ neutralised. Both halves are required: dropping on the answer alone would discard AD-10's own worked
194
+ example of two distinct nonexistent identifiers both returning 404, which are the legs the check
195
+ exists to read. A check left with no clean leg to examine now fails and names why, where it reported
196
+ satisfied before.
197
+
198
+ Measured on the stored replay across the two versions, `test-review`'s five plants move from
199
+ `preflight: failed: seeded-faults-scoped` with a null verdict and exit 3 to `preflight: passed`,
200
+ `CONCERNS`, exit 0, each carrying its own strength vector. The suite's defect class goes from four
201
+ exercised and four caught to nine and nine, still at a rate of 1. `expected-strength.json` records
202
+ the move.
203
+
204
+ `trace`'s three plants failed on that same run, and that was the fix working: their witness fired on
205
+ `witness-gate-withheld`, a leg that issues a different request and receives a different answer, so
206
+ the reducer kept it in the examined set and the check reported a real scoping problem in the trace
207
+ contract. That problem is now closed, and the shape of it is worth keeping.
208
+
209
+ The witness was right and the leg was not clean. AD-10 reads every other leg of an operation as a
210
+ clean leg. `trace-fixture-set` carries `stateChangeMarker: true`, so `selectControl` plans no control
211
+ leg for it, and the only other legs it had were its two sensitivity witness legs, both staged against
212
+ the seeded set. `D-001` asserts that P0 coverage is 50, which is what the seeded workspace produces
213
+ and what `allow_gate` does not move, so the relation was true on a seeded run with the gate withheld.
214
+ It fired there because the plant was there.
215
+
216
+ Nothing in the request could say otherwise. The run's real input is the staged workspace, both sets
217
+ were staged at one `project/` root, and `buildPrompt` took a fixture set and read nothing off it, so
218
+ the two sets sent byte-identical prompts and no leg could ask for the set without the plant. Each set
219
+ now declares a `projectRoot` in `test/fixtures/trace-eval/ground-truth.json`,
220
+ `tenant-data-export` and `api-token-lifecycle`, and the whole prompt is written against it:
221
+ `{project-root}`, `{config_source}`, `{test_artifacts}`, `{test_dir}`, `{source_dir}`, the epic
222
+ directory, and both deliverable paths. `stagedWorkspaceFor` stages the set the leg's prompt names,
223
+ the stored-evidence port answers with that set's stored run, and the stub agent resolves
224
+ `{project-root}` off the prompt the way a real agent does.
225
+
226
+ The contract's two witness legs now trace the clean set, which is the set that establishes what
227
+ "clean leg" means for this operation. P0 coverage there is 100, so `D-001`'s relation resolves false
228
+ on both legs and the check is satisfied on evidence. The `allow_gate` differential is untouched and
229
+ holds on either set: the clean set writes `gate_basis: "priority_thresholds"` when the gate is
230
+ allowed and `"none"` when it is withheld, the same pair the seeded set writes. Measured on the stored
231
+ replay, the three probes move from `preflight: failed: seeded-faults-scoped` to `preflight: passed`,
232
+ and their basis loses the `pre-flight verdict did not pass` line. They are still refused by the
233
+ qualification gate, so their verdict, exit code and strength are unchanged.
234
+
235
+ The project roots name the epic each set traces and not the set's role. A run that read `clean` in
236
+ the directory it works in would have been handed the answer, which is the rule the corpus already
237
+ follows when it keeps every inline label out of the fixture files, so `validateCorpus` fails a
238
+ project root carrying `seeded`, `clean`, `control`, `planted` or `gap`.
239
+
240
+ **"This collection is empty" has a spelling, as of 1.4.0.** Through 1.3.0 the evaluator intercepted
241
+ an empty array on every quantifier and every single-operand leaf and returned `insufficient-evidence`
242
+ with an `empty-collection` condition before the operator ran, so `count-tolerance` with `expected: 0`
243
+ never counted. `trace`'s O-023 and O-024 make exactly that claim about its clean set, and both
244
+ abstained on the run they were written to confirm. TEA raised it upstream and 1.4.0 exempts the three
245
+ operators that read a property of the collection itself: `count-tolerance` reads its cardinality,
246
+ `existence` and `absence` read its presence. Measured on the stored replay across the two versions,
247
+ O-023 and O-024 move from `abstained` to `passed-clean-control`, so `count-tolerance` was the right
248
+ spelling to have committed to.
249
+
250
+ One limit is worth knowing before writing a new oracle. Every quantifier still abstains over an empty
251
+ collection, which is AD-4's whole purpose and is why `trace`'s five `for-any` oracles over the seeded
252
+ export still abstain on a clean run: they ask whether some element exists, and an empty collection is
253
+ an honest "nothing was checked". `deep-equality` against a literal `[]` also still abstains, so the
254
+ two spellings of "this collection is empty" disagree. eval-quality records that disagreement in AD-4
255
+ rather than hiding it. The bare `count-tolerance` assertion is the one to write.
256
+
257
+ **A rejected probe names its reason, as of 1.4.0.** The qualification gate computes a closed list of
258
+ twenty reason codes. Through 1.3.0 none of them reached the evidence artifact or any published
259
+ export, so a probe the gate rejected surfaced as `infrastructure-error` and exit 3 and a corpus
260
+ author reading that had nothing to act on. Reading the reasons meant calling `qualifyProbe`, which
261
+ was off the exports map, and taking it would have been the third reach into `dist/` this document
262
+ already records two of. TEA raised that upstream instead. `runScore` now returns `qualification`
263
+ beside the artifact and the ladder, `QUALIFICATION_FAILURES` publishes the closed set, and
264
+ `test-probe-corpus` records the codes per probe in `expected-strength.json`. No reach into `dist/`
265
+ was needed and the count stays at two.
193
266
 
194
267
  ### What the scoring half says about TEA's contracts
195
268
 
@@ -200,16 +273,25 @@ Three findings, all measured, none of them tuned away.
200
273
  before. `test-review` leaves `whole-body`, `malformed-input` and `state-change-read-back`
201
274
  unsatisfied; every fragment-selection contract leaves `malformed-input` unsatisfied. Each scores the
202
275
  run down to CONCERNS without blocking it, which is exactly the weight AD-20 gives a coverage gap.
203
- - **`trace`'s clean control scores FAIL, and it is two different things.** Seven of its twenty-six
204
- oracles abstain. Five belong to the seeded set and resolve against a record carrying only the clean
205
- set's observation, and TEA cannot repair that: both plan steps declare one operation and send the
206
- same prompt, because what makes a trace run the seeded set or the clean set is the staged workspace
207
- and no request shape names one, so a record carrying both runs is ambiguous under `exactly-one` and
208
- a record carrying one leaves the other set's oracles reading nothing. Binding the prompt literally
209
- was tried and the generator's comment records why it does not work. The other two assert that a
210
- collection is empty, which the vocabulary cannot say at all. Neither half is a case for giving the
211
- clean set a member so an oracle has something to read: that would mutate a control to make the
212
- instrument work.
276
+ - **`trace`'s clean control scores FAIL, and what remains of it is one thing.** Measured on the
277
+ stored replay, five of its twenty-six oracles abstain on `P-004`: `O-009`, `O-010`, `O-011`,
278
+ `O-013` and `O-014`, every one a `for-any` quantifier over the seeded export, and every one named
279
+ in the artifact's `verdictBasis`. It was seven. `O-023` and `O-024`, the two asserting that a
280
+ collection is empty, moved to `passed-clean-control` when `eval-quality` 1.4.0 gave that claim a
281
+ spelling.
282
+
283
+ The five abstain because the record carries only the clean set's observation and both plan steps
284
+ select it. The reason recorded here used to be that no request shape names a fixture set; that is
285
+ no longer true, since each set is staged under its own project root and the prompt is written
286
+ against it. What is left is the binding: `stdin.prompt` is bound `{matcher: 'any'}`, so one
287
+ observation is selected by both steps, which is the same limit as the bullet below. The trace
288
+ prompt is 1.8 kilobytes rather than fragment selection's 28, so the size argument against a literal
289
+ does not apply here either. What a literal binding needs beside it is a record carrying both sets'
290
+ observations, and whether one probe's record may carry the run of a set it did not seed is a
291
+ question about what a clean control means rather than a mechanical change, so it is not made here.
292
+ Neither half is a case for giving the clean set a member so an oracle has something to read: that
293
+ would mutate a control to make the instrument work.
294
+
213
295
  - **A plan cannot tell two steps apart when both bind their inputs by matcher.** Each
214
296
  fragment-selection contract declares one plan step per case, distinguished only by the prompt, and
215
297
  the prompt is bound `{matcher: 'any'}` because the alternative is a 28-kilobyte literal per step. A
@@ -221,11 +303,12 @@ Three findings, all measured, none of them tuned away.
221
303
 
222
304
  ### What the live pre-flight measured
223
305
 
224
- The first pre-flight this repository has ever run, on 2026-09-09 against `claude`. Twenty-one legs
225
- spawned, 55 minutes of model time, and every leg cached under a digest of its request so a second
226
- invocation pays for nothing it has already answered. The three manifestation witnesses `trace` gained
227
- at `eval-quality` 1.3.0 cost nothing: their request is the one its `allow_gate: true` witness leg
228
- already sends, and the cache is keyed on the request.
306
+ The first pre-flight this repository has ever run, on 2026-09-09 against `claude`, at `eval-quality`
307
+ 1.3.0. Twenty-one legs spawned, 55 minutes of model time, and every leg cached under a digest of its
308
+ request so a second invocation pays for nothing it has already answered. The three manifestation
309
+ witnesses `trace` gained at 1.3.0 cost nothing on that run: their request was the one its
310
+ `allow_gate: true` witness leg already sent, and the cache is keyed on the request. The table is that
311
+ run and has not been re-measured; what has changed under it since is below.
229
312
 
230
313
  | Suite | Legs spawned | Model time | Pre-flight |
231
314
  | -------------------------------------- | -----------: | ---------: | ------------------------------------------------------------------------------------ |
@@ -241,6 +324,15 @@ the `false` leg wrote `gate_basis: "none"` and no gate at all, which is what ste
241
324
  `test-review`'s differential is the file list: the seeded fixture drew five findings and exit 1, the
242
325
  clean control drew none and exit 0.
243
326
 
327
+ Two things about that table have moved since, both on the stored replay rather than on a repeat of
328
+ the live run. `test-review`'s pre-flight outcome is now nine of nine defect probes rather than four,
329
+ because `eval-quality` 1.4.0 drops a clean leg that issued the fault leg's own request. `trace`'s is
330
+ now all four probes passing, because its witness legs moved to the clean set. That move also costs a
331
+ leg: the three manifestation witnesses trace the seeded set while the two witness legs trace the
332
+ clean set, so their request is no longer one the cache already holds and `trace` spawns three legs
333
+ where it spawned two. The three still share one request between them, so it is one extra spawn and
334
+ not three.
335
+
244
336
  Two things the live run found that no deterministic check could.
245
337
 
246
338
  **The two witness legs shared a staged workspace, and the second read the first's file.** The first
@@ -265,8 +357,8 @@ prompt the harness assembles for case X", which parses, compiles, and is schedul
265
357
  then measures nothing when a real agent is finally handed it. `tools/generate-contracts.js` reads
266
358
  `buildPrompt` out of `test/eval-fragment-selection.js` now, the same way the trace witness already
267
359
  read its two prompts from its own harness, so a leg sends the prompt the suite sends. The contracts
268
- grew from around 20 kilobytes to around 90, which is what the trace contract already paid for the
269
- same correctness.
360
+ grew from around 20 kilobytes to between 59 and 109, which is what the trace contract already paid
361
+ for the same correctness.
270
362
 
271
363
  `test-review`'s request shape declared an environment permitting three API keys and forbidding `HOME`.
272
364
  The adapter closes the child environment to `PATH` plus what the request declares, and
@@ -284,9 +376,9 @@ Two path couplings closed with them. `runReview` states `--project-root` because
284
376
 
285
377
  Also done: `eval-trace.js`'s `runCase` probes `tea-trace-runner` (`cli/trace-runner.js`), the command the trace suite had no equivalent of. The runner's whole surface is a prompt on standard input and an agent run in the working directory. It builds no prompt and knows no vendor, and it declares `scoped-artifact-writes`, the one capability a trace run needs and a selection does not. The two workflow artifacts are the authorization's artifact map, supplied per run because each case stages its own workspace, and they come back tagged, so a file the run never wrote is `absent` and a missing artifact rather than an `existsSync` race. Both runner commands share one exit-code table in `cli/lib/runner-exit-codes.js`, because a caller holding an observation cannot tell which command produced it. `trace.contract.json` is the tenth contract, generated from `test/fixtures/trace-eval/ground-truth.json` with 26 oracles over the summary artifact, and `npm run test:contract-oracles` evaluates every one of them over the fourteen stored trace runs, under both fixture sets' oracles, and compares each with the `scoreRun` check it restates. The whole chain was verified with a stub vendor: `npm run test:probe-targets` spawns the harness itself, reads its result record back, and sees every threshold met on a correct run, a quality failure on a run that wrote a test, and a missing-artifact failure on a run that wrote nothing.
286
378
 
287
- The trace witness is a differential over standard input on one prompt value, `allow_gate`. The two plan steps share one prompt on purpose, since the harness names no fact about either set in it, so a differential between the two steps would attribute to the prompt a difference the staged workspace produced, and an invariance claim would be false because the two summaries differ. `allow_gate` is the one prompt value the corpus establishes an effect for: step-05 evaluates a gate only when it is true and writes `gate_basis` as `none` otherwise, so two prompts differing in that value in one workspace produce two `gate_basis` values.
379
+ The trace witness is a differential over standard input on one prompt value, `allow_gate`. The differential is on that value rather than between the two fixture sets, because the seeded and clean summaries differ from the staged files rather than from the prompt, so a differential between the sets would attribute to the prompt a difference the prompt did not cause, and an invariance claim would be false because the two summaries differ. `allow_gate` is the one prompt value the corpus establishes an effect for: step-05 evaluates a gate only when it is true and writes `gate_basis` as `none` otherwise, so two prompts differing in that value over one staged set produce two `gate_basis` values. Both witness legs stage the clean set, which is what makes them clean legs for `seeded-faults-scoped`.
288
380
 
289
- Two more couplings, both found by running it. A relative `--agent-cmd` passed the harness pre-flight, which probes it from the harness's own directory, and then failed every run, because the runner executes in the staged workspace and a relative path resolves there; `parseArgs` resolves a path against the operator's directory now. And the trace witness legs are runnable only against a staged workspace of the seeded set, because the run's real input is the working directory, which the request shape cannot name; that is the same coupling `test-review`'s legs have with `--project-root`.
381
+ Two more couplings, both found by running it. A relative `--agent-cmd` passed the harness pre-flight, which probes it from the harness's own directory, and then failed every run, because the runner executes in the staged workspace and a relative path resolves there; `parseArgs` resolves a path against the operator's directory now. And the trace witness legs are runnable only against a staged workspace, because the run's real input is the working directory; that is the same coupling `test-review`'s legs have with `--project-root`. Which set that workspace holds is in the request now: each fixture set declares its own `projectRoot`, the prompt is written against it, and `stagedWorkspaceFor` stages the set the leg names. Staging one set for every leg is what made the two witness legs seeded runs, which is the scoping failure `seeded-faults-scoped` reported against all three defect probes.
290
382
 
291
383
  One coupling worth naming, and it did not change with the scoring half: two of the entry points TEA
292
384
  reaches are addressed by file path into `dist/`, because neither the compiler CLI nor the evaluator is
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "$schema": "https://json.schemastore.org/package.json",
3
3
  "name": "bmad-method-test-architecture-enterprise",
4
- "version": "1.25.0",
4
+ "version": "1.25.1",
5
5
  "description": "Master Test Architect for quality strategy, test automation, and release gates",
6
6
  "keywords": [
7
7
  "bmad",
@@ -137,7 +137,7 @@
137
137
  "eslint-plugin-n": "^17.21.3",
138
138
  "eslint-plugin-unicorn": "^60.0.0",
139
139
  "eslint-plugin-yml": "^1.18.0",
140
- "eval-quality": "1.3.0",
140
+ "eval-quality": "1.4.0",
141
141
  "husky": "^9.1.7",
142
142
  "jest": "^30.0.4",
143
143
  "lint-staged": "^16.1.1",
@@ -2640,7 +2640,7 @@
2640
2640
  "environment": {},
2641
2641
  "stdin": {
2642
2642
  "kind": "text",
2643
- "value": "You are running the TEA workflow `bmad-testarch-trace` against the project in `project/`.\n\nThe workflow is in `skill/`. Read `skill/instructions.md` first, then execute every step file it\nnames in order, in full, without skipping or reordering. The step files are under `skill/steps-c/`.\n\n----- run configuration -----\nResolve the workflow placeholders to these values:\n\n- `{project-root}`: `project`\n- `{config_source}`: `project/_bmad/tea/config.yaml`\n- `{test_artifacts}`: `project/test-artifacts`\n- `{test_dir}`: `project/tests`\n- `{source_dir}`: `project/src`\n- `{skill-root}`: `skill`\n- `gate_type`: `epic`\n- `decision_mode`: `deterministic`\n- `collection_mode`: `contract_static`\n- `allow_gate`: `true`\n- `coverage_basis`: `auto`\n- `summary_confidence`: `auto`\n- `coverage_levels`: `e2e,api,component,unit,live`\n\nThe trace target is the epic under `project/docs/epics/`. Resolve the coverage oracle from it the\nway step-01 says to.\n\n----- what to produce -----\nWrite both deliverables the workflow declares:\n\n- `project/test-artifacts/traceability-matrix.md`, from `skill/trace-template.md`, carrying the\n detailed mapping with one section per criterion, each stating its coverage status and the tests\n that establish it as `file:line`.\n- `project/test-artifacts/e2e-trace-summary.json` at schema_version 0.3.0, exactly as\n `skill/steps-c/step-05-gate-decision.md` section 3b defines it.\n\nDo not add, edit, or delete any file under `project/docs/`, `project/src/`, or `project/tests/`.\nThis workflow does not generate tests.\n\nWhen you are done, print one line naming the two files you wrote. Nothing else you print is read."
2643
+ "value": "You are running the TEA workflow `bmad-testarch-trace` against the project in `api-token-lifecycle/`.\n\nThe workflow is in `skill/`. Read `skill/instructions.md` first, then execute every step file it\nnames in order, in full, without skipping or reordering. The step files are under `skill/steps-c/`.\n\n----- run configuration -----\nResolve the workflow placeholders to these values:\n\n- `{project-root}`: `api-token-lifecycle`\n- `{config_source}`: `api-token-lifecycle/_bmad/tea/config.yaml`\n- `{test_artifacts}`: `api-token-lifecycle/test-artifacts`\n- `{test_dir}`: `api-token-lifecycle/tests`\n- `{source_dir}`: `api-token-lifecycle/src`\n- `{skill-root}`: `skill`\n- `gate_type`: `epic`\n- `decision_mode`: `deterministic`\n- `collection_mode`: `contract_static`\n- `allow_gate`: `true`\n- `coverage_basis`: `auto`\n- `summary_confidence`: `auto`\n- `coverage_levels`: `e2e,api,component,unit,live`\n\nThe trace target is the epic under `api-token-lifecycle/docs/epics/`. Resolve the coverage oracle from it the\nway step-01 says to.\n\n----- what to produce -----\nWrite both deliverables the workflow declares:\n\n- `api-token-lifecycle/test-artifacts/traceability-matrix.md`, from `skill/trace-template.md`, carrying the\n detailed mapping with one section per criterion, each stating its coverage status and the tests\n that establish it as `file:line`.\n- `api-token-lifecycle/test-artifacts/e2e-trace-summary.json` at schema_version 0.3.0, exactly as\n `skill/steps-c/step-05-gate-decision.md` section 3b defines it.\n\nDo not add, edit, or delete any file under `api-token-lifecycle/docs/`, `api-token-lifecycle/src/`, or `api-token-lifecycle/tests/`.\nThis workflow does not generate tests.\n\nWhen you are done, print one line naming the two files you wrote. Nothing else you print is read."
2644
2644
  }
2645
2645
  }
2646
2646
  },
@@ -2654,7 +2654,7 @@
2654
2654
  "environment": {},
2655
2655
  "stdin": {
2656
2656
  "kind": "text",
2657
- "value": "You are running the TEA workflow `bmad-testarch-trace` against the project in `project/`.\n\nThe workflow is in `skill/`. Read `skill/instructions.md` first, then execute every step file it\nnames in order, in full, without skipping or reordering. The step files are under `skill/steps-c/`.\n\n----- run configuration -----\nResolve the workflow placeholders to these values:\n\n- `{project-root}`: `project`\n- `{config_source}`: `project/_bmad/tea/config.yaml`\n- `{test_artifacts}`: `project/test-artifacts`\n- `{test_dir}`: `project/tests`\n- `{source_dir}`: `project/src`\n- `{skill-root}`: `skill`\n- `gate_type`: `epic`\n- `decision_mode`: `deterministic`\n- `collection_mode`: `contract_static`\n- `allow_gate`: `false`\n- `coverage_basis`: `auto`\n- `summary_confidence`: `auto`\n- `coverage_levels`: `e2e,api,component,unit,live`\n\nThe trace target is the epic under `project/docs/epics/`. Resolve the coverage oracle from it the\nway step-01 says to.\n\n----- what to produce -----\nWrite both deliverables the workflow declares:\n\n- `project/test-artifacts/traceability-matrix.md`, from `skill/trace-template.md`, carrying the\n detailed mapping with one section per criterion, each stating its coverage status and the tests\n that establish it as `file:line`.\n- `project/test-artifacts/e2e-trace-summary.json` at schema_version 0.3.0, exactly as\n `skill/steps-c/step-05-gate-decision.md` section 3b defines it.\n\nDo not add, edit, or delete any file under `project/docs/`, `project/src/`, or `project/tests/`.\nThis workflow does not generate tests.\n\nWhen you are done, print one line naming the two files you wrote. Nothing else you print is read."
2657
+ "value": "You are running the TEA workflow `bmad-testarch-trace` against the project in `api-token-lifecycle/`.\n\nThe workflow is in `skill/`. Read `skill/instructions.md` first, then execute every step file it\nnames in order, in full, without skipping or reordering. The step files are under `skill/steps-c/`.\n\n----- run configuration -----\nResolve the workflow placeholders to these values:\n\n- `{project-root}`: `api-token-lifecycle`\n- `{config_source}`: `api-token-lifecycle/_bmad/tea/config.yaml`\n- `{test_artifacts}`: `api-token-lifecycle/test-artifacts`\n- `{test_dir}`: `api-token-lifecycle/tests`\n- `{source_dir}`: `api-token-lifecycle/src`\n- `{skill-root}`: `skill`\n- `gate_type`: `epic`\n- `decision_mode`: `deterministic`\n- `collection_mode`: `contract_static`\n- `allow_gate`: `false`\n- `coverage_basis`: `auto`\n- `summary_confidence`: `auto`\n- `coverage_levels`: `e2e,api,component,unit,live`\n\nThe trace target is the epic under `api-token-lifecycle/docs/epics/`. Resolve the coverage oracle from it the\nway step-01 says to.\n\n----- what to produce -----\nWrite both deliverables the workflow declares:\n\n- `api-token-lifecycle/test-artifacts/traceability-matrix.md`, from `skill/trace-template.md`, carrying the\n detailed mapping with one section per criterion, each stating its coverage status and the tests\n that establish it as `file:line`.\n- `api-token-lifecycle/test-artifacts/e2e-trace-summary.json` at schema_version 0.3.0, exactly as\n `skill/steps-c/step-05-gate-decision.md` section 3b defines it.\n\nDo not add, edit, or delete any file under `api-token-lifecycle/docs/`, `api-token-lifecycle/src/`, or `api-token-lifecycle/tests/`.\nThis workflow does not generate tests.\n\nWhen you are done, print one line naming the two files you wrote. Nothing else you print is read."
2658
2658
  }
2659
2659
  }
2660
2660
  }
@@ -2738,7 +2738,7 @@
2738
2738
  "human-labels"
2739
2739
  ],
2740
2740
  "testData": {
2741
- "setup": "Each plan step stages one fixture set from test/fixtures/trace-eval/ into a disposable workspace: the set's files under project/, a resolved _bmad/tea/config.yaml whose test_artifacts points inside that workspace, and the bmad-testarch-trace workflow under skill/. ground-truth.json is never staged, and the harness asserts that no staged file carries its bytes or its keys before the run. The workspace is the authorization's working directory, and the prompt on standard input names project/ and skill/ and resolves every placeholder; it is the same text for both steps, because it names no fact about either set. That shared prompt is why the sensitivity witness differs its two legs on allow_gate rather than between the two steps: the seeded and clean summaries differ because of the staged workspace, so a differential between the steps would attribute to the prompt a difference the prompt did not cause, and an invariance claim would be false. allow_gate is the one prompt value the ground truth establishes an effect for, through skillRuleCitations.gateEligibility: step-05 evaluates a gate only when it is true and writes gate_basis as none otherwise, so two prompts differing in that value, in one staged workspace, produce two gate_basis values. That is the same reasoning that gives the fragment-selection contract for this workflow an invariance witness: the claim is moved onto an input the run demonstrably reads, and stated as what it is.",
2741
+ "setup": "Each plan step stages one fixture set from test/fixtures/trace-eval/ into a disposable workspace: the set's files under its own project root (tenant-data-export/ for seeded-tenant-data-export, api-token-lifecycle/ for clean-api-token-lifecycle), a resolved _bmad/tea/config.yaml whose test_artifacts points inside that workspace, and the bmad-testarch-trace workflow under skill/. ground-truth.json is never staged, and the harness asserts that no staged file carries its bytes or its keys before the run. The workspace is the authorization's working directory, and the prompt on standard input names the project root and skill/ and resolves every placeholder against them. The project root is the one fact about the set the prompt carries, and it carries the epic the set traces rather than the set's role, so it says which set a leg is asking for and suggests no coverage status. The sensitivity witness differs its two legs on allow_gate rather than between the two sets: the seeded and clean summaries differ because of the staged workspace, so a differential between the sets would attribute to the prompt a difference the staged files produced, and an invariance claim would be false. allow_gate is the one prompt value the ground truth establishes an effect for, through skillRuleCitations.gateEligibility: step-05 evaluates a gate only when it is true and writes gate_basis as none otherwise, so two prompts differing in that value, over one staged fixture set, produce two gate_basis values. That is the same reasoning that gives the fragment-selection contract for this workflow an invariance witness: the claim is moved onto an input the run demonstrably reads, and stated as what it is. Both witness legs stage the clean set, because they are the only other legs this operation has and AD-10 reads them as its clean legs when it asks whether a seeded fault is scoped to its own leg.",
2742
2742
  "cleanup": "Delete the workspace. The corpus under test/fixtures/trace-eval/ is read-only and the harness digests it before and after every run.",
2743
2743
  "principals": null,
2744
2744
  "resources": null
@@ -2749,8 +2749,8 @@
2749
2749
  "maxCostUsd": "8.00"
2750
2750
  },
2751
2751
  "safetyLimits": [
2752
- "The runner writes only inside the staged workspace, and only its two deliverables under project/test-artifacts/; the harness fails a run that changed the repository or the staged corpus.",
2753
- "The run adds, edits, and deletes nothing under project/docs/, project/src/, or project/tests/. The workflow does not generate tests, and a run that did has moved the benchmark.",
2752
+ "The runner writes only inside the staged workspace, and only its two deliverables under the fixture set's own test-artifacts/; the harness fails a run that changed the repository or the staged corpus.",
2753
+ "The run adds, edits, and deletes nothing under the fixture set's docs/, src/, or tests/. The workflow does not generate tests, and a run that did has moved the benchmark.",
2754
2754
  "No credential value appears in a prompt, an artifact, a log, or a result file."
2755
2755
  ],
2756
2756
  "requiredEvidence": [
@@ -36,9 +36,9 @@
36
36
  * test-review leg names its fixtures by repository-relative path and writes
37
37
  * `verdict.json` beside them, and the policy's `cwd` is what both resolve
38
38
  * against, so the fixtures are copied into the run directory and the artifact
39
- * lands there. A trace leg needs the seeded set staged, which is what
40
- * test/eval-trace.js already does for its own runs. A selection needs nothing on
41
- * disk at all.
39
+ * lands there. A trace leg needs its own fixture set staged, which is what
40
+ * test/eval-trace.js already does for its own runs, and the leg's prompt says which
41
+ * set that is. A selection needs nothing on disk at all.
42
42
  *
43
43
  * Usage:
44
44
  * npm run eval:preflight # the live pre-flight, cached, nothing scored
@@ -49,9 +49,9 @@
49
49
  * Exit codes are read against `test/probes/expected-strength.json`, the same
50
50
  * baseline the deterministic gate compares to: 0 when every probe reached the
51
51
  * outcome the corpus records, 1 when a verdict moved, 2 when a pre-flight outcome
52
- * moved. Eight of the thirty-one probes cannot be pre-flighted today and the
53
- * baseline says so, so their failure is not news and does not colour the run;
54
- * one of them starting to pass is news, and so is one that stops.
52
+ * moved. Eight of the thirty-one probes could not be pre-flighted when that rule
53
+ * was written, and the baseline said so; all thirty-one pre-flight now, so a
54
+ * failure here is news and the baseline is what says so.
55
55
  */
56
56
 
57
57
  'use strict';
@@ -64,7 +64,7 @@ const { digest } = require('./lib/eval-record');
64
64
  const { validateArtifact } = require('./lib/eval-quality-inputs');
65
65
  const { createProbePort, hostEnvironment } = require('./lib/probe-targets');
66
66
  const { runSuite, sealContract, suites } = require('./lib/probe-scoring');
67
- const { stageWorkspace } = require('./eval-trace');
67
+ const { stageWorkspace, traceArtifactPaths } = require('./eval-trace');
68
68
 
69
69
  const PROJECT_ROOT = path.join(__dirname, '..');
70
70
  const REVIEW_FIXTURE_DIR = path.join('test', 'fixtures', 'test-review-eval');
@@ -160,9 +160,13 @@ const slug = (suiteId) => suiteId.replaceAll(/[^a-z\d]+/gi, '-');
160
160
  * The run directory one suite's legs execute in, and the artifact paths that
161
161
  * directory makes true.
162
162
  *
163
- * A trace leg is staged by the trace harness itself, so the seeded set arrives
164
- * exactly as `npm run eval:trace` stages it, under the same `project/` prefix its
165
- * artifact override names.
163
+ * A trace leg is staged by the trace harness itself, so the set arrives exactly as
164
+ * `npm run eval:trace` stages it, under the same project root its artifact override
165
+ * names. Which set that is comes out of the leg's own prompt: each fixture set has
166
+ * its own project root and the prompt is written against it, so a leg that traces
167
+ * the clean set asks for the clean set. Staging one set for every leg is what made
168
+ * the contract's two witness legs seeded runs, which is the scoping failure
169
+ * `seeded-faults-scoped` reported against all three defect probes.
166
170
  *
167
171
  * A test-review leg is the coupling `docs/explanation/eval-quality-command-adapter.md`
168
172
  * records: its `--files` are repository-relative, its `--json` is a bare
@@ -177,20 +181,19 @@ const slug = (suiteId) => suiteId.replaceAll(/[^a-z\d]+/gi, '-');
177
181
  * A selection leg gets an empty directory, which is what its `read-only`
178
182
  * declaration is for.
179
183
  */
180
- function stagedWorkspaceFor(suiteId) {
184
+ function stagedWorkspaceFor(suiteId, request) {
181
185
  if (suiteId === 'trace') {
182
186
  const groundTruth = JSON.parse(fs.readFileSync(TRACE_GROUND_TRUTH, 'utf8'));
183
- const seeded = groundTruth.fixtureSets.find((set) => set.id.startsWith('seeded'));
184
- const staged = stageWorkspace(seeded);
187
+ const prompt = String(request?.channels?.stdin?.value ?? '');
188
+ const set = groundTruth.fixtureSets.find((entry) => prompt.includes(`\`{project-root}\`: \`${entry.projectRoot}\``));
189
+ if (set === undefined) {
190
+ throw new Error('a trace leg sent a prompt naming no fixture set project root, so there is no set to stage for it');
191
+ }
192
+ const staged = stageWorkspace(set);
185
193
  return {
186
194
  root: staged.dir,
187
195
  cwd: staged.dir,
188
- artifacts: {
189
- 'tea-trace-runner': {
190
- summary: path.join('project', 'test-artifacts', 'e2e-trace-summary.json'),
191
- matrix: path.join('project', 'test-artifacts', 'traceability-matrix.md'),
192
- },
193
- },
196
+ artifacts: { 'tea-trace-runner': traceArtifactPaths(set) },
194
197
  };
195
198
  }
196
199
  const dir = fs.mkdtempSync(path.join(os.tmpdir(), `tea-${slug(suiteId)}-preflight-`));
@@ -254,7 +257,7 @@ function cachingPort({ makePort, contract, cacheDir, agent, force, counters, log
254
257
  },
255
258
  };
256
259
  log(` ${colors.yellow}running${colors.reset} leg ${request.probeId} (${key})`);
257
- const { port: realPort, workspace } = await makePort();
260
+ const { port: realPort, workspace } = await makePort(augmented);
258
261
  const startedAt = Date.now();
259
262
  let observation;
260
263
  try {
@@ -334,8 +337,8 @@ async function runOneSuite(suite, options, stats) {
334
337
  port = cacheOnlyPort(cacheDir, [stats, suiteStats]);
335
338
  } else {
336
339
  port = cachingPort({
337
- makePort: async () => {
338
- const staged = stagedWorkspaceFor(suite.id);
340
+ makePort: async (request) => {
341
+ const staged = stagedWorkspaceFor(suite.id, request);
339
342
  const { port: realPort } = await createProbePort({ cwd: staged.cwd, interfaceIds, artifacts: staged.artifacts });
340
343
  return { port: realPort, workspace: staged };
341
344
  },