bmad-method-test-architecture-enterprise 1.27.3-next.21 → 1.27.3-next.22

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/.claude-plugin/marketplace.json +1 -1
  2. package/README.md +5 -5
  3. package/cli/lib/evaluate/arm.js +5 -0
  4. package/cli/lib/evaluate/check.js +82 -0
  5. package/cli/lib/evaluate/ci.js +1 -1
  6. package/cli/lib/evaluate/command-evaluator.js +88 -16
  7. package/cli/lib/evaluate/confinement.js +237 -14
  8. package/cli/lib/evaluate/evaluators.js +52 -4
  9. package/cli/lib/evaluate/frameworks.js +324 -0
  10. package/cli/lib/evaluate/preflight.js +2 -0
  11. package/cli/lib/evaluate/registry.js +35 -4
  12. package/cli/lib/evaluate/run.js +87 -11
  13. package/cli/lib/evaluate/schemas/evaluation-ci-plan.schema.json +1 -1
  14. package/cli/lib/evaluate/schemas/evaluator-frameworks.schema.json +37 -0
  15. package/cli/lib/evaluate/schemas/framework-versions.schema.json +39 -0
  16. package/cli/lib/evaluate/workspace.js +1 -23
  17. package/package.json +3 -2
  18. package/src/workflows/testarch/bmad-testarch-ci/SKILL.md +2 -0
  19. package/src/workflows/testarch/bmad-testarch-ci/checklist.md +9 -0
  20. package/src/workflows/testarch/bmad-testarch-ci/github-actions-template.yaml +73 -0
  21. package/src/workflows/testarch/bmad-testarch-ci/instructions.md +5 -1
  22. package/src/workflows/testarch/bmad-testarch-ci/resources/ci-pipeline-progress.example.md +12 -1
  23. package/src/workflows/testarch/bmad-testarch-ci/steps-c/step-01b-resume.md +5 -3
  24. package/src/workflows/testarch/bmad-testarch-ci/steps-c/step-03-configure-quality-gates.md +1 -1
  25. package/src/workflows/testarch/bmad-testarch-ci/steps-c/step-03b-render-evaluation-plans.md +142 -0
  26. package/src/workflows/testarch/bmad-testarch-ci/steps-c/step-04-validate-and-summary.md +2 -0
  27. package/src/workflows/testarch/bmad-testarch-ci/steps-e/step-01-assess.md +7 -2
  28. package/src/workflows/testarch/bmad-testarch-ci/steps-e/step-02-apply-edit.md +4 -1
  29. package/src/workflows/testarch/bmad-testarch-ci/steps-v/step-01-validate.md +4 -0
  30. package/src/workflows/testarch/bmad-testarch-evaluate/assets/README.md +2 -2
  31. package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/LEARNED.md +2 -1
  32. package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/agentevals-frameworks.json +10 -0
  33. package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/agentevals-trajectory.mjs +1 -0
  34. package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/command-evaluator.mjs +1 -0
  35. package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/frameworks.json +4 -0
  36. package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/installed-version.mjs +42 -0
  37. package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/promptfoo-assertions.mjs +1 -0
  38. package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/promptfoo-frameworks.json +10 -0
  39. package/src/workflows/testarch/bmad-testarch-evaluate/references/evaluator.md +51 -7
  40. package/src/workflows/testarch/bmad-testarch-evaluate/references/gaps.md +15 -15
  41. package/src/workflows/testarch/bmad-testarch-evaluate/references/harness.md +23 -0
@@ -31,7 +31,7 @@
31
31
  "name": "bmad-method-test-architecture-enterprise",
32
32
  "source": "./",
33
33
  "description": "Master Test Architect module for quality strategy, test automation, CI/CD quality gates, and structured testing education. Part of the BMad Method ecosystem.",
34
- "version": "1.27.3-next.21",
34
+ "version": "1.27.3-next.22",
35
35
  "author": {
36
36
  "name": "Murat K Ozcan (TEA Creator) & Brian (BMad) Madison"
37
37
  },
package/README.md CHANGED
@@ -383,7 +383,7 @@ The `atdd`, `automate`, `ci`, `framework`, `nfr`, `teach-me-testing`, `test-desi
383
383
  | `bmad-teach-me-testing` | N/A | Yes; multi-turn session with seeded wrong quiz answer, review loop, and progress |
384
384
  | `bmad-testarch-atdd` | 3 | Yes; one unimplemented fixture and five acceptance criteria |
385
385
  | `bmad-testarch-automate` | 5 | Yes; four hand-authored spec sets against a fixed service and its seeded regression |
386
- | `bmad-testarch-ci` | 2 | Yes; one full request and one minimal request |
386
+ | `bmad-testarch-ci` | 2 | Yes; one full request, one minimal request and one request over an evaluation plan |
387
387
  | `bmad-testarch-evaluate` | N/A | Yes; Evaluate-authored, a gap-guide class swap seeded and a second held out |
388
388
  | `bmad-testarch-framework` | 3 | Yes; install and smoke-test generated scaffold in network-isolated sandbox |
389
389
  | `bmad-testarch-nfr` | 2 | Yes; one evidence bundle with known gaps and one clean bundle |
@@ -395,7 +395,7 @@ A passing fragment-selection eval means the workflow loaded the right knowledge.
395
395
 
396
396
  ### Deterministic Checks
397
397
 
398
- `npm test` chains 106 checks. That count covers the whole chain: every entry in it is deterministic and credential-free, so the chain and its credential-free subset are the same list. `npm run test:ci-coverage` derives the count from `package.json` and prints it. Twelve of the 106 keep the rules, guidance, hook, eval data, eval contracts, diagnostics, and documentation aligned:
398
+ `npm test` chains 107 checks. That count covers the whole chain: every entry in it is deterministic and credential-free, so the chain and its credential-free subset are the same list. `npm run test:ci-coverage` derives the count from `package.json` and prints it. Twelve of the 107 keep the rules, guidance, hook, eval data, eval contracts, diagnostics, and documentation aligned:
399
399
 
400
400
  - `test:criteria-fragments` fails when a registry row is neither mapped to a knowledge fragment nor declared a known gap. A rule the reviewer scores but no fragment teaches is a rule TEA punishes without ever having explained it. All 36 rows are currently mapped across 50 anchors. Because the declared-gap list is empty, the validator feeds itself a synthetic unmapped row on every run to prove that path still works.
401
401
  - `test:doc-counts` runs `eval-quality-gates doc-counts`, which holds a hand-written count on a published page against the source that computes it: the roadmap's per-suite `eval:all` call counts, the knowledge-fragment tier breakdown, this section's own npm-test-chain length, and the fragment-selection case count. A pattern matching no sentence, or more than one, fails the same way a wrong number does, so the entry cannot go stale by drifting out from under its own pattern either.
@@ -432,7 +432,7 @@ npm run eval:all -- --agent agy
432
432
  npm run eval:all -- --agent agy --agent claude --agent codex
433
433
  ```
434
434
 
435
- `eval:all` uses two repetitions per fragment-selection case and per routing intent, three repetitions for `test-review`, and two repetitions per `nfr` evidence bundle, per `ci` project, per `test-design` epic, per `trace` fixture set, and per `atdd` story. One runner makes 107 agent calls: 48 fragment selections, 38 routing intents, 3 reviews, 4 audits, 4 pipelines, 4 test designs, 4 traces, and 2 ATDD generations. All three built-in runners make 321 calls.
435
+ `eval:all` uses two repetitions per fragment-selection case and per routing intent, three repetitions for `test-review`, and two repetitions per `nfr` evidence bundle, per `ci` project, per `test-design` epic, per `trace` fixture set, and per `atdd` story. One runner makes 109 agent calls: 48 fragment selections, 38 routing intents, 3 reviews, 4 audits, 6 pipelines, 4 test designs, 4 traces, and 2 ATDD generations. All three built-in runners make 327 calls.
436
436
 
437
437
  Check the data, executable, login, fixtures, and expected results without making a model call:
438
438
 
@@ -558,7 +558,7 @@ Every row below is a declared gate. Ten of TEA's eleven skills have behavioral s
558
558
 
559
559
  | Eval | Declared threshold | Declared volume |
560
560
  | --------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------- |
561
- | `npm run eval:all -- --agent ...` | All eight live suites below meet their thresholds for the selected runner | 48 selections, 38 routing calls, 3 reviews, 4 audits, 4 pipelines, 4 test designs, 4 traces, 2 generations |
561
+ | `npm run eval:all -- --agent ...` | All eight live suites below meet their thresholds for the selected runner | 48 selections, 38 routing calls, 3 reviews, 4 audits, 6 pipelines, 4 test designs, 4 traces, 2 generations |
562
562
  | Fragment selection | At least 90% required-fragment recall, at most 10% forbidden-fragment selection, and stable choices across repeated cases | 24 cases twice: 48 calls |
563
563
  | Test review | At least 70% overall recall, 100% CRITICAL recall, an 80% non-false-positive rate, score standard deviation no higher than 3, and a stable verdict | Three complete reviews |
564
564
  | Trace | At least 90% criterion-status accuracy and 90% evidence-citation precision; 100% on both discriminating criteria, the gate decision, the gate criteria, the coverage arithmetic, the oracle resolution, the rejected evidence, the waiver validity, and the live evidence; no clean-set false positive, no unstable case, and no fixture mutation | Two fixture sets twice: 4 calls |
@@ -596,7 +596,7 @@ npm ci
596
596
  npm run test:eval-data # fragment-selection corpus, static
597
597
  npm run test:eval-trace-data # trace corpus, static
598
598
  npm run test:eval-schemas # manifest against harness constants, and the preflight argv
599
- npm run test:eval-replay # 123 stored outputs against the scorers
599
+ npm run test:eval-replay # stored outputs against the scorers
600
600
  ```
601
601
 
602
602
  Run live evals in a scheduled or manually triggered CI job after installing and authenticating the selected agent CLI:
@@ -379,6 +379,8 @@ function bodyValue(body) {
379
379
  */
380
380
  function hostEnvironmentPort({ port, registry }) {
381
381
  return {
382
+ /** Empties the private home the port's sandbox keeps, where it has one (`registry.js` `createProbePort`). */
383
+ resetHome: () => port.resetHome?.(),
382
384
  async probe(request, signal) {
383
385
  const registered = request?.kind === 'cli' && registry.targetFor(request.interfaceId, request.executable) !== undefined;
384
386
  const injected = registered ? registry.hostEnvironment(request.interfaceId, [], request.executable) : {};
@@ -720,6 +722,9 @@ async function runArm({
720
722
  signal,
721
723
  seed = 'tea-evaluate-default-seed',
722
724
  }) {
725
+ // An arm is independent of the arms before it on a shared port, so its target starts with an empty private home; the
726
+ // steps of this arm share it.
727
+ port.resetHome?.();
723
728
  const operations = operationsById(contract);
724
729
  const plan = contract.interactionPlan ?? [];
725
730
  const declared = new Set(plan.map((step) => step.stepId));
@@ -70,6 +70,10 @@
70
70
  * - `evaluator` (Story 1.34): a `sealed-brief-agent` evaluator whose `evaluation.json` has no
71
71
  * `evaluatorQualification` (`attempts`, `minimumAgreement`), which `run` needs to qualify the agent
72
72
  * before its verdicts count; and an `evaluatorQualification` beside any other kind, where nothing would use it.
73
+ * - `evaluator` (Story 1.44, AD-21): a `command` evaluator has no tracked `evaluator/frameworks.json` (its declared
74
+ * installed frameworks, an empty list for none), one whose version probe is not a tracked regular executable under
75
+ * `evaluator/`, or an `evaluator/LEARNED.md` whose `package@version` records disagree with a declaration; a declaration
76
+ * that fails its shape is reported under `schema` (or `json`). `check` runs no probe (`run` does, `frameworks.js`).
73
77
  * - `evaluator` (Story 1.17, AD-21): a `command` or `sealed-brief-agent` evaluator has no
74
78
  * `evaluator/mapping.json`, or one binding an oracle, behavior or rubric criterion the contract does not
75
79
  * declare, an oracle to a behavior that does not declare it, levels other than the criterion's anchored
@@ -140,6 +144,7 @@ const {
140
144
  } = require('./registry');
141
145
  const { AGENT_ADAPTERS, bridgedArgsRefused, resolveModel } = require('../agent-adapters');
142
146
  const { EVALUATOR_DIRECTORY, EvaluatorLayerError, evaluatorFiles, evaluatorOf, isKnownEvaluator } = require('./evaluators');
147
+ const { FRAMEWORKS_PATH, LEARNED_PATH, declarationProblems, declaredFrameworks, learnedProblems } = require('./frameworks');
143
148
  const { answeredKind, degenerateResponsePath } = require('./gameability');
144
149
 
145
150
  /** How a finding names a call of each interface kind. */
@@ -1239,6 +1244,82 @@ function checkRecordsCalibration(report, folder, evaluator, evaluation, contract
1239
1244
  /** The mode bits that let anyone execute a file. */
1240
1245
  const EXECUTE_BITS = 0o111;
1241
1246
 
1247
+ /**
1248
+ * A `command` evaluator's declared framework dependencies (Story 1.44):
1249
+ * `evaluator/frameworks.json` must be there, tracked, and meet its shape (an
1250
+ * empty list for an evaluator with no installed framework), every version
1251
+ * probe must be a tracked regular executable of the layer, and
1252
+ * `evaluator/LEARNED.md` must record the version each nonempty declaration
1253
+ * names. `run` observes what is installed; `check` runs no probe.
1254
+ */
1255
+ function checkFrameworks(report, folder, layer, untracked) {
1256
+ if (untracked(FRAMEWORKS_PATH)) {
1257
+ report.add(
1258
+ FRAMEWORKS_PATH,
1259
+ 'evaluator',
1260
+ `${FRAMEWORKS_PATH} is not tracked by git, and a run reads only the files git tracks under evaluator/; git add it`,
1261
+ );
1262
+ return;
1263
+ }
1264
+ if (!fs.existsSync(path.join(folder, ...FRAMEWORKS_PATH.split('/')))) {
1265
+ report.add(
1266
+ FRAMEWORKS_PATH,
1267
+ 'evaluator',
1268
+ `evaluation.json's evaluator is command, which declares the installed frameworks it depends on in ${FRAMEWORKS_PATH}, and the folder has none; declare each dependency, or an empty list for an evaluator with none`,
1269
+ );
1270
+ return;
1271
+ }
1272
+ const declaration = parseInto(report, folder, FRAMEWORKS_PATH);
1273
+ if (declaration === undefined) return;
1274
+ const shape = declarationProblems(declaration);
1275
+ for (const problem of shape) report.add(FRAMEWORKS_PATH, 'schema', problem);
1276
+ if (shape.length > 0) return;
1277
+ const frameworks = declaredFrameworks(declaration);
1278
+ for (const { package: name, probe } of frameworks) {
1279
+ if (untracked(probe.command)) {
1280
+ report.add(
1281
+ FRAMEWORKS_PATH,
1282
+ 'evaluator',
1283
+ `the version probe of ${name} names ${probe.command}, which git does not track, and a run reads only the files git tracks under evaluator/; git add it`,
1284
+ );
1285
+ continue;
1286
+ }
1287
+ let stats;
1288
+ try {
1289
+ stats = fs.lstatSync(path.join(folder, ...probe.command.split('/')));
1290
+ } catch {
1291
+ stats = null;
1292
+ }
1293
+ if (stats === null || !stats.isFile()) {
1294
+ report.add(
1295
+ FRAMEWORKS_PATH,
1296
+ 'evaluator',
1297
+ `the version probe of ${name} names ${probe.command}, which is not a regular file the evaluation folder holds`,
1298
+ );
1299
+ } else if (process.platform !== 'win32' && (stats.mode & EXECUTE_BITS) === 0) {
1300
+ report.add(
1301
+ FRAMEWORKS_PATH,
1302
+ 'evaluator',
1303
+ `the version probe of ${name} names ${probe.command}, which is not executable; set its execute bit`,
1304
+ );
1305
+ }
1306
+ }
1307
+ // Where the layer could not be read, the layer's own finding already says so.
1308
+ if (layer === null) return;
1309
+ if (untracked(LEARNED_PATH)) {
1310
+ report.add(
1311
+ LEARNED_PATH,
1312
+ 'evaluator',
1313
+ `${LEARNED_PATH} is not tracked by git, and a run reads only the files git tracks under evaluator/; git add it`,
1314
+ );
1315
+ return;
1316
+ }
1317
+ const learned = layer.files.find((file) => file.path === LEARNED_PATH);
1318
+ for (const problem of learnedProblems(frameworks, learned === undefined ? null : learned.bytes.toString('utf8'))) {
1319
+ report.add(LEARNED_PATH, 'evaluator', problem);
1320
+ }
1321
+ }
1322
+
1242
1323
  /**
1243
1324
  * `evaluation.json`'s `evaluator` (AD-21) held to the folder and the
1244
1325
  * contract: a `command` or `sealed-brief-agent` evaluator needs
@@ -1339,6 +1420,7 @@ function checkEvaluator(report, folder, evaluation, contract, conditions, engine
1339
1420
  `evaluation.json's evaluator is ${kind}, whose judgment rows convert through ${MAPPING_PATH}, and the folder has none`,
1340
1421
  );
1341
1422
  }
1423
+ if (kind === 'command') checkFrameworks(report, folder, layer, untracked);
1342
1424
  if (kind === 'command') {
1343
1425
  if (typeof evaluator.command !== 'string') return;
1344
1426
  if (untracked(evaluator.command)) {
@@ -5,7 +5,7 @@
5
5
  * The plan (`ci-plan.js`) is the only definition of tier membership. The runtime reads it, validates its placement
6
6
  * rules (exit 10 on a finding, 64 when the plan is absent) and runs exactly the checks whose `placement.tier` is the
7
7
  * tier asked for, in plan order. Every check runs, whether or not an earlier one failed, so the evidence bundle is
8
- * complete. An `evaluate` check is run by its id; its `command` is the `tea-evaluate` argv a pipeline step renders. A
8
+ * complete. An `evaluate` check is run by its id; its `command` records the `tea-evaluate` argv a reader can run by hand, and a pipeline runs `tea-evaluate ci --tier <tier>` once per tier. A
9
9
  * `gate` check is an `eval-quality-gates` command the adopter adopted, run as a child process with no shell and the
10
10
  * plan's argv, in the evaluation folder, as the leader of a process group of its own: it ends at the plan check's
11
11
  * `timeoutMs` or past 64 MiB of output (exit 12, what it printed kept), and a signal to `ci` reaches the group first.
@@ -27,6 +27,11 @@
27
27
  * (`confinement.js` `layerPrefix`), so neither it nor any process it starts
28
28
  * can write under `evaluator/` (Story 1.31).
29
29
  *
30
+ * The frameworks it depends on are declared in `evaluator/frameworks.json` and read through the same launch path
31
+ * (`observeFrameworks`, Story 1.44): each declared version probe starts as the evaluator does, with the same
32
+ * environment, an empty private working directory of its own and the same confinement, and prints the installed
33
+ * package identity and version (`frameworks.js`).
34
+ *
30
35
  * Its stdout is read as UTF-8 (a byte sequence that is not UTF-8 reads as
31
36
  * U+FFFD), and what it printed is kept as the bytes it wrote. One that cannot
32
37
  * start, is still running at its wall clock, exits other than 0, or prints
@@ -39,9 +44,81 @@
39
44
  const path = require('node:path');
40
45
 
41
46
  const { buildMinimalEnv, runSupervised } = require('../run-agent');
47
+ const { readProbeAnswer, stderrNote } = require('./frameworks');
42
48
  const { EvaluatorError, readAnswer } = require('./judgment-rows');
43
49
  const { releaseScratchDirectory, makeScratchDirectory } = require('./workspace');
44
50
 
51
+ /**
52
+ * Starts one executable under the evaluation folder's `evaluator/` exactly as
53
+ * the evaluator itself starts: through `spawnPrefix` when the run confines it,
54
+ * under the supervisor, with an empty private working directory in `scratch`
55
+ * removed afterwards, the base environment plus `evaluator.environmentKeys`,
56
+ * and `evaluator.timeoutMs` as its wall clock. The evaluator's launch and a
57
+ * framework's version probe (`observeFrameworks`) both go through it, so a
58
+ * probe sees what the evaluator sees.
59
+ *
60
+ * @returns {Promise<object>} `runSupervised`'s report: the outcome and the streams
61
+ */
62
+ async function launchExecutable({ folder, evaluator, command, args, input, scratch, env, spawnPrefix }) {
63
+ const executable = path.join(folder, ...command.split('/'));
64
+ const cwd = makeScratchDirectory(scratch, 'tea-evaluate-command-');
65
+ try {
66
+ return await runSupervised({
67
+ command: spawnPrefix.length === 0 ? executable : spawnPrefix[0],
68
+ args: [...spawnPrefix.slice(1), ...(spawnPrefix.length === 0 ? [] : [executable]), ...args],
69
+ input,
70
+ cwd,
71
+ env: buildMinimalEnv(evaluator.environmentKeys ?? [], env),
72
+ timeout: evaluator.timeoutMs,
73
+ });
74
+ } finally {
75
+ releaseScratchDirectory(scratch, cwd);
76
+ }
77
+ }
78
+
79
+ /**
80
+ * Reads the installed version of each declared framework (Story 1.44): runs
81
+ * the dependency's version probe through the evaluator's own launch path
82
+ * (`launchExecutable`) and reads the `{ package, version }` it prints
83
+ * (`frameworks.js`). A probe that cannot start, outlives the timeout, exits
84
+ * other than 0 or prints another shape yields no observation for that
85
+ * dependency and a `fault` naming why, with what it printed. Nothing is
86
+ * thrown for a dependency that is missing: the caller compares the entries
87
+ * with the declaration.
88
+ *
89
+ * @param {object} options
90
+ * @param {string} options.folder the evaluation folder
91
+ * @param {object} options.evaluator `evaluation.json`'s `command` evaluator
92
+ * @param {Array<{ package: string, probe: { command: string, args: string[] } }>} options.frameworks
93
+ * @param {string[]} options.scratch
94
+ * @param {NodeJS.ProcessEnv} [options.env]
95
+ * @param {string[]} [options.spawnPrefix]
96
+ * @returns {Promise<Array<{ package: string, observed: { package: string, version: string }|null, fault: string|null, stdout: string, stderr: string }>>}
97
+ * one entry per declared framework, in the order given
98
+ */
99
+ async function observeFrameworks({ folder, evaluator, frameworks, scratch, env = process.env, spawnPrefix = [] }) {
100
+ const entries = [];
101
+ for (const framework of frameworks) {
102
+ const { command, args } = framework.probe;
103
+ const label = `the version probe of ${framework.package} (${command})`;
104
+ const ended = await launchExecutable({ folder, evaluator, command, args, input: '', scratch, env, spawnPrefix });
105
+ const { outcome, stdout, stderr } = ended;
106
+ const entry = { package: framework.package, observed: null, fault: null, stdout, stderr };
107
+ if (outcome.spawnError) entry.fault = `${label} could not start: ${outcome.spawnError.message}`;
108
+ else if (outcome.timedOut) entry.fault = `${label} was still running at the evaluator's ${evaluator.timeoutMs}ms timeout`;
109
+ else if (outcome.failure) entry.fault = `${label} did not finish: ${outcome.failure}`;
110
+ else if (outcome.status === 0) {
111
+ const answer = readProbeAnswer(framework.package, stdout);
112
+ if ('fault' in answer) entry.fault = `${label}: ${answer.fault}`;
113
+ else entry.observed = answer;
114
+ } else {
115
+ entry.fault = `${label} ${outcome.signal ? `was killed by signal ${outcome.signal}` : `exited ${outcome.status}`}${stderrNote(stderr)}`;
116
+ }
117
+ entries.push(entry);
118
+ }
119
+ return entries;
120
+ }
121
+
45
122
  /**
46
123
  * Runs the evaluator over one trial.
47
124
  *
@@ -70,21 +147,16 @@ async function runCommandEvaluator({
70
147
  env = process.env,
71
148
  spawnPrefix = [],
72
149
  }) {
73
- const executable = path.join(folder, ...evaluator.command.split('/'));
74
- const cwd = makeScratchDirectory(scratch, 'tea-evaluate-command-');
75
- let ended;
76
- try {
77
- ended = await runSupervised({
78
- command: spawnPrefix.length === 0 ? executable : spawnPrefix[0],
79
- args: [...spawnPrefix.slice(1), ...(spawnPrefix.length === 0 ? [] : [executable]), ...(evaluator.args ?? [])],
80
- input: `${JSON.stringify({ sealedBrief, observations })}\n`,
81
- cwd,
82
- env: buildMinimalEnv(evaluator.environmentKeys ?? [], env),
83
- timeout: evaluator.timeoutMs,
84
- });
85
- } finally {
86
- releaseScratchDirectory(scratch, cwd);
87
- }
150
+ const ended = await launchExecutable({
151
+ folder,
152
+ evaluator,
153
+ command: evaluator.command,
154
+ args: evaluator.args ?? [],
155
+ input: `${JSON.stringify({ sealedBrief, observations })}\n`,
156
+ scratch,
157
+ env,
158
+ spawnPrefix,
159
+ });
88
160
  const { outcome, stdout, stderr, stdoutBytes, stderrBytes } = ended;
89
161
  const streams = { stdout, stderr, stdoutBytes, stderrBytes };
90
162
  const fail = (message) => Object.assign(new EvaluatorError(message, streams), { outcome });
@@ -108,4 +180,4 @@ async function runCommandEvaluator({
108
180
  }
109
181
  }
110
182
 
111
- module.exports = { runCommandEvaluator };
183
+ module.exports = { observeFrameworks, runCommandEvaluator };