bmad-method-test-architecture-enterprise 1.27.3-next.21 → 1.27.3-next.23

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/.claude-plugin/marketplace.json +1 -1
  2. package/README.md +5 -5
  3. package/cli/evaluate.js +30 -3
  4. package/cli/lib/evaluate/arm.js +5 -0
  5. package/cli/lib/evaluate/calibration.js +7 -1
  6. package/cli/lib/evaluate/check.js +86 -12
  7. package/cli/lib/evaluate/ci.js +1 -1
  8. package/cli/lib/evaluate/command-evaluator.js +88 -16
  9. package/cli/lib/evaluate/confinement.js +237 -14
  10. package/cli/lib/evaluate/evaluators.js +52 -4
  11. package/cli/lib/evaluate/frameworks.js +324 -0
  12. package/cli/lib/evaluate/preflight.js +2 -0
  13. package/cli/lib/evaluate/records-calibration.js +174 -20
  14. package/cli/lib/evaluate/records-evaluator.js +30 -7
  15. package/cli/lib/evaluate/registry.js +35 -4
  16. package/cli/lib/evaluate/run.js +89 -12
  17. package/cli/lib/evaluate/schemas/evaluation-ci-plan.schema.json +1 -1
  18. package/cli/lib/evaluate/schemas/evaluator-frameworks.schema.json +37 -0
  19. package/cli/lib/evaluate/schemas/framework-versions.schema.json +39 -0
  20. package/cli/lib/evaluate/workspace.js +1 -23
  21. package/package.json +3 -2
  22. package/src/workflows/testarch/bmad-testarch-ci/SKILL.md +2 -0
  23. package/src/workflows/testarch/bmad-testarch-ci/checklist.md +9 -0
  24. package/src/workflows/testarch/bmad-testarch-ci/github-actions-template.yaml +73 -0
  25. package/src/workflows/testarch/bmad-testarch-ci/instructions.md +5 -1
  26. package/src/workflows/testarch/bmad-testarch-ci/resources/ci-pipeline-progress.example.md +12 -1
  27. package/src/workflows/testarch/bmad-testarch-ci/steps-c/step-01b-resume.md +5 -3
  28. package/src/workflows/testarch/bmad-testarch-ci/steps-c/step-03-configure-quality-gates.md +1 -1
  29. package/src/workflows/testarch/bmad-testarch-ci/steps-c/step-03b-render-evaluation-plans.md +142 -0
  30. package/src/workflows/testarch/bmad-testarch-ci/steps-c/step-04-validate-and-summary.md +2 -0
  31. package/src/workflows/testarch/bmad-testarch-ci/steps-e/step-01-assess.md +7 -2
  32. package/src/workflows/testarch/bmad-testarch-ci/steps-e/step-02-apply-edit.md +4 -1
  33. package/src/workflows/testarch/bmad-testarch-ci/steps-v/step-01-validate.md +4 -0
  34. package/src/workflows/testarch/bmad-testarch-evaluate/assets/README.md +2 -2
  35. package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/LEARNED.md +2 -1
  36. package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/agentevals-frameworks.json +10 -0
  37. package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/agentevals-trajectory.mjs +1 -0
  38. package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/command-evaluator.mjs +1 -0
  39. package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/frameworks.json +4 -0
  40. package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/installed-version.mjs +42 -0
  41. package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/promptfoo-assertions.mjs +1 -0
  42. package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/promptfoo-frameworks.json +10 -0
  43. package/src/workflows/testarch/bmad-testarch-evaluate/references/evaluator.md +57 -9
  44. package/src/workflows/testarch/bmad-testarch-evaluate/references/gaps.md +15 -15
  45. package/src/workflows/testarch/bmad-testarch-evaluate/references/harness.md +23 -0
@@ -31,7 +31,7 @@
31
31
  "name": "bmad-method-test-architecture-enterprise",
32
32
  "source": "./",
33
33
  "description": "Master Test Architect module for quality strategy, test automation, CI/CD quality gates, and structured testing education. Part of the BMad Method ecosystem.",
34
- "version": "1.27.3-next.21",
34
+ "version": "1.27.3-next.23",
35
35
  "author": {
36
36
  "name": "Murat K Ozcan (TEA Creator) & Brian (BMad) Madison"
37
37
  },
package/README.md CHANGED
@@ -383,7 +383,7 @@ The `atdd`, `automate`, `ci`, `framework`, `nfr`, `teach-me-testing`, `test-desi
383
383
  | `bmad-teach-me-testing` | N/A | Yes; multi-turn session with seeded wrong quiz answer, review loop, and progress |
384
384
  | `bmad-testarch-atdd` | 3 | Yes; one unimplemented fixture and five acceptance criteria |
385
385
  | `bmad-testarch-automate` | 5 | Yes; four hand-authored spec sets against a fixed service and its seeded regression |
386
- | `bmad-testarch-ci` | 2 | Yes; one full request and one minimal request |
386
+ | `bmad-testarch-ci` | 2 | Yes; one full request, one minimal request and one request over an evaluation plan |
387
387
  | `bmad-testarch-evaluate` | N/A | Yes; Evaluate-authored, a gap-guide class swap seeded and a second held out |
388
388
  | `bmad-testarch-framework` | 3 | Yes; install and smoke-test generated scaffold in network-isolated sandbox |
389
389
  | `bmad-testarch-nfr` | 2 | Yes; one evidence bundle with known gaps and one clean bundle |
@@ -395,7 +395,7 @@ A passing fragment-selection eval means the workflow loaded the right knowledge.
395
395
 
396
396
  ### Deterministic Checks
397
397
 
398
- `npm test` chains 106 checks. That count covers the whole chain: every entry in it is deterministic and credential-free, so the chain and its credential-free subset are the same list. `npm run test:ci-coverage` derives the count from `package.json` and prints it. Twelve of the 106 keep the rules, guidance, hook, eval data, eval contracts, diagnostics, and documentation aligned:
398
+ `npm test` chains 107 checks. That count covers the whole chain: every entry in it is deterministic and credential-free, so the chain and its credential-free subset are the same list. `npm run test:ci-coverage` derives the count from `package.json` and prints it. Twelve of the 107 keep the rules, guidance, hook, eval data, eval contracts, diagnostics, and documentation aligned:
399
399
 
400
400
  - `test:criteria-fragments` fails when a registry row is neither mapped to a knowledge fragment nor declared a known gap. A rule the reviewer scores but no fragment teaches is a rule TEA punishes without ever having explained it. All 36 rows are currently mapped across 50 anchors. Because the declared-gap list is empty, the validator feeds itself a synthetic unmapped row on every run to prove that path still works.
401
401
  - `test:doc-counts` runs `eval-quality-gates doc-counts`, which holds a hand-written count on a published page against the source that computes it: the roadmap's per-suite `eval:all` call counts, the knowledge-fragment tier breakdown, this section's own npm-test-chain length, and the fragment-selection case count. A pattern matching no sentence, or more than one, fails the same way a wrong number does, so the entry cannot go stale by drifting out from under its own pattern either.
@@ -432,7 +432,7 @@ npm run eval:all -- --agent agy
432
432
  npm run eval:all -- --agent agy --agent claude --agent codex
433
433
  ```
434
434
 
435
- `eval:all` uses two repetitions per fragment-selection case and per routing intent, three repetitions for `test-review`, and two repetitions per `nfr` evidence bundle, per `ci` project, per `test-design` epic, per `trace` fixture set, and per `atdd` story. One runner makes 107 agent calls: 48 fragment selections, 38 routing intents, 3 reviews, 4 audits, 4 pipelines, 4 test designs, 4 traces, and 2 ATDD generations. All three built-in runners make 321 calls.
435
+ `eval:all` uses two repetitions per fragment-selection case and per routing intent, three repetitions for `test-review`, and two repetitions per `nfr` evidence bundle, per `ci` project, per `test-design` epic, per `trace` fixture set, and per `atdd` story. One runner makes 109 agent calls: 48 fragment selections, 38 routing intents, 3 reviews, 4 audits, 6 pipelines, 4 test designs, 4 traces, and 2 ATDD generations. All three built-in runners make 327 calls.
436
436
 
437
437
  Check the data, executable, login, fixtures, and expected results without making a model call:
438
438
 
@@ -558,7 +558,7 @@ Every row below is a declared gate. Ten of TEA's eleven skills have behavioral s
558
558
 
559
559
  | Eval | Declared threshold | Declared volume |
560
560
  | --------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------- |
561
- | `npm run eval:all -- --agent ...` | All eight live suites below meet their thresholds for the selected runner | 48 selections, 38 routing calls, 3 reviews, 4 audits, 4 pipelines, 4 test designs, 4 traces, 2 generations |
561
+ | `npm run eval:all -- --agent ...` | All eight live suites below meet their thresholds for the selected runner | 48 selections, 38 routing calls, 3 reviews, 4 audits, 6 pipelines, 4 test designs, 4 traces, 2 generations |
562
562
  | Fragment selection | At least 90% required-fragment recall, at most 10% forbidden-fragment selection, and stable choices across repeated cases | 24 cases twice: 48 calls |
563
563
  | Test review | At least 70% overall recall, 100% CRITICAL recall, an 80% non-false-positive rate, score standard deviation no higher than 3, and a stable verdict | Three complete reviews |
564
564
  | Trace | At least 90% criterion-status accuracy and 90% evidence-citation precision; 100% on both discriminating criteria, the gate decision, the gate criteria, the coverage arithmetic, the oracle resolution, the rejected evidence, the waiver validity, and the live evidence; no clean-set false positive, no unstable case, and no fixture mutation | Two fixture sets twice: 4 calls |
@@ -596,7 +596,7 @@ npm ci
596
596
  npm run test:eval-data # fragment-selection corpus, static
597
597
  npm run test:eval-trace-data # trace corpus, static
598
598
  npm run test:eval-schemas # manifest against harness constants, and the preflight argv
599
- npm run test:eval-replay # 123 stored outputs against the scorers
599
+ npm run test:eval-replay # stored outputs against the scorers
600
600
  ```
601
601
 
602
602
  Run live evals in a scheduled or manually triggered CI job after installing and authenticating the selected agent CLI:
package/cli/evaluate.js CHANGED
@@ -4,7 +4,12 @@
4
4
  *
5
5
  * Subcommands in this release:
6
6
  * tea-evaluate check --evaluation <path> validate the folder; exit 10 on any authoring defect
7
- * tea-evaluate digest --evaluation <path> write corpus-index.json and print corpusDigest
7
+ * tea-evaluate digest --evaluation <path> [--calibration-inputs]
8
+ * write corpus-index.json and print corpusDigest; with
9
+ * --calibration-inputs write nothing and print, as JSON, the values a
10
+ * records harness copies into calibration-judgments.json: the labelled
11
+ * file's digest, the scorer configuration digest and each labelled
12
+ * item's label-free scorerInput
8
13
  * tea-evaluate preflight --evaluation <path> [--from-working-tree]
9
14
  * qualify the seeded probes in a disposable workspace, drive the
10
15
  * preflight legs and take the verdict from eval-quality
@@ -38,7 +43,9 @@
38
43
  * 2-5 also ci: the exit of each stage it runs (compile, seal, the replay's preflight and score, a live check's
39
44
  * preflight, run and score), passed through the same way; 2 is also a probe class below its strength floor
40
45
  * on the release tier
41
- * 10 authoring defect: every finding is printed, one per line (digest: an indexed entry it cannot digest;
46
+ * 10 authoring defect: every finding is printed, one per line (digest: an indexed entry it cannot digest, and with
47
+ * --calibration-inputs an evaluator that is not records, a contract with no rubric, an unusable labelled file
48
+ * or a harness configuration that is absent, a link, not JSON or not an EvaluatorConfiguration;
42
49
  * preflight: a leg the registry does not authorize, or a mutation whose find text does not occur
43
50
  * exactly once)
44
51
  * 10 also compare: a run whose scores or members cannot be read as files the run wrote, a baseline holding a link
@@ -103,6 +110,7 @@ const { runScoreCommand } = require('./lib/evaluate/score');
103
110
  const { runCiCommand } = require('./lib/evaluate/ci');
104
111
  const { TIERS } = require('./lib/evaluate/ci-plan');
105
112
  const { escapeUnprintable, findingLine } = require('./lib/evaluate/finding-lines');
113
+ const { calibrationInputsOf } = require('./lib/evaluate/records-calibration');
106
114
 
107
115
  const EXIT_CODES = {
108
116
  ok: 0,
@@ -135,8 +143,21 @@ async function runCheck(options) {
135
143
  return EXIT_CODES.authoring;
136
144
  }
137
145
 
146
+ /** Prints the values a records harness copies into its judgments file; writes nothing under the folder. */
147
+ async function runCalibrationInputs(folder) {
148
+ const { problems, inputs } = await calibrationInputsOf(folder);
149
+ if (inputs === null) {
150
+ for (const problem of problems) process.stdout.write(findingLine(problem.file, 'judge-calibration', problem.message));
151
+ process.stderr.write(`${NAME} digest --calibration-inputs: ${problems.length} authoring defect(s) in ${folder}\n`);
152
+ return EXIT_CODES.authoring;
153
+ }
154
+ process.stdout.write(`${JSON.stringify(inputs, null, 2)}\n`);
155
+ return EXIT_CODES.ok;
156
+ }
157
+
138
158
  async function runDigest(options) {
139
159
  const folder = folderFrom(options);
160
+ if (options.calibrationInputs === true) return runCalibrationInputs(folder);
140
161
  const { index, corpusDigest, indexPath } = await writeCorpusIndex(folder);
141
162
  process.stderr.write(`${NAME} digest: wrote ${indexPath} (${index.length} file(s))\n`);
142
163
  process.stdout.write(`${corpusDigest}\n`);
@@ -220,8 +241,14 @@ function buildProgram(run) {
220
241
  .action((options) => run(runCheck, options));
221
242
  program
222
243
  .command('digest')
223
- .description('Write corpus-index.json over corpus/, probes/ and mutations/, and print corpusDigest.')
244
+ .description(
245
+ 'Write corpus-index.json over corpus/, probes/ and mutations/, and print corpusDigest; with --calibration-inputs, print the values a records harness copies instead.',
246
+ )
224
247
  .option('--evaluation <path>', 'the evaluation folder, or its evaluation.json')
248
+ .option(
249
+ '--calibration-inputs',
250
+ "print, and write nothing: the labelled file's digest, the scorer configuration digest and each item's scorerInput",
251
+ )
225
252
  .action((options) => run(runDigest, options));
226
253
  program
227
254
  .command('preflight')
@@ -379,6 +379,8 @@ function bodyValue(body) {
379
379
  */
380
380
  function hostEnvironmentPort({ port, registry }) {
381
381
  return {
382
+ /** Empties the private home the port's sandbox keeps, where it has one (`registry.js` `createProbePort`). */
383
+ resetHome: () => port.resetHome?.(),
382
384
  async probe(request, signal) {
383
385
  const registered = request?.kind === 'cli' && registry.targetFor(request.interfaceId, request.executable) !== undefined;
384
386
  const injected = registered ? registry.hostEnvironment(request.interfaceId, [], request.executable) : {};
@@ -720,6 +722,9 @@ async function runArm({
720
722
  signal,
721
723
  seed = 'tea-evaluate-default-seed',
722
724
  }) {
725
+ // An arm is independent of the arms before it on a shared port, so its target starts with an empty private home; the
726
+ // steps of this arm share it.
727
+ port.resetHome?.();
723
728
  const operations = operationsById(contract);
724
729
  const plan = contract.interactionPlan ?? [];
725
730
  const declared = new Set(plan.map((step) => step.stepId));
@@ -203,6 +203,11 @@ function readCalibration(folder) {
203
203
  }
204
204
  }
205
205
 
206
+ /** The digest of the labelled file's bytes: what a configuration binds as `tea.judgeCalibrationDigest` and a run records. */
207
+ function labelledDigest(labelled, engine) {
208
+ return engine.digestBytes(labelled.bytes);
209
+ }
210
+
206
211
  /**
207
212
  * The criteria of a calibration report whose agreement is below its `minimumAgreement`, each as a sentence; empty when
208
213
  * every criterion meets it. `run` stops with exit 11 on any, and `tea-evaluate ci` reads the same report with this.
@@ -258,7 +263,7 @@ async function runCalibration({ calibration, evaluation, contract, engine, write
258
263
  exitCode: 11,
259
264
  message: `judge calibration agreement fell below ${report.minimumAgreement}; see judge-calibration.json`,
260
265
  });
261
- return { digest: engine.digestBytes(calibration.bytes), report };
266
+ return { digest: labelledDigest(calibration, engine), report };
262
267
  }
263
268
 
264
269
  module.exports = {
@@ -267,6 +272,7 @@ module.exports = {
267
272
  calibrationOperationId,
268
273
  calibrationProblems,
269
274
  calibrationShortfalls,
275
+ labelledDigest,
270
276
  readCalibration,
271
277
  runCalibration,
272
278
  };
@@ -70,6 +70,10 @@
70
70
  * - `evaluator` (Story 1.34): a `sealed-brief-agent` evaluator whose `evaluation.json` has no
71
71
  * `evaluatorQualification` (`attempts`, `minimumAgreement`), which `run` needs to qualify the agent
72
72
  * before its verdicts count; and an `evaluatorQualification` beside any other kind, where nothing would use it.
73
+ * - `evaluator` (Story 1.44, AD-21): a `command` evaluator has no tracked `evaluator/frameworks.json` (its declared
74
+ * installed frameworks, an empty list for none), one whose version probe is not a tracked regular executable under
75
+ * `evaluator/`, or an `evaluator/LEARNED.md` whose `package@version` records disagree with a declaration; a declaration
76
+ * that fails its shape is reported under `schema` (or `json`). `check` runs no probe (`run` does, `frameworks.js`).
73
77
  * - `evaluator` (Story 1.17, AD-21): a `command` or `sealed-brief-agent` evaluator has no
74
78
  * `evaluator/mapping.json`, or one binding an oracle, behavior or rubric criterion the contract does not
75
79
  * declare, an oracle to a behavior that does not declare it, levels other than the criterion's anchored
@@ -123,7 +127,7 @@ const AjvModule = require('ajv/dist/2020');
123
127
 
124
128
  const { engineSchemaPath, loadEngine, schemaVersionProblems } = require('./engine');
125
129
  const { CALIBRATION_PATH, calibrationProblems, readCalibration } = require('./calibration');
126
- const { verifyRecordsCalibration } = require('./records-calibration');
130
+ const { readConfiguration, recordsDirectory, verifyRecordsCalibration } = require('./records-calibration');
127
131
  const { readPlan } = require('./ci-plan');
128
132
  const { MANIFEST_NAME } = require('./folder');
129
133
  const { addFormats } = require('./formats');
@@ -140,6 +144,7 @@ const {
140
144
  } = require('./registry');
141
145
  const { AGENT_ADAPTERS, bridgedArgsRefused, resolveModel } = require('../agent-adapters');
142
146
  const { EVALUATOR_DIRECTORY, EvaluatorLayerError, evaluatorFiles, evaluatorOf, isKnownEvaluator } = require('./evaluators');
147
+ const { FRAMEWORKS_PATH, LEARNED_PATH, declarationProblems, declaredFrameworks, learnedProblems } = require('./frameworks');
143
148
  const { answeredKind, degenerateResponsePath } = require('./gameability');
144
149
 
145
150
  /** How a finding names a call of each interface kind. */
@@ -1213,9 +1218,7 @@ function checkRecordsCalibration(report, folder, evaluator, evaluation, contract
1213
1218
  if (calibrationProblems(evaluation, contract, labelled?.value, engine).length > 0) return;
1214
1219
  let configuration;
1215
1220
  try {
1216
- const stats = fs.lstatSync(path.join(root, 'evaluator-configuration.json'));
1217
- if (!stats.isFile()) throw new Error('it is not a regular file');
1218
- configuration = JSON.parse(fs.readFileSync(path.join(root, 'evaluator-configuration.json'), 'utf8'));
1221
+ configuration = readConfiguration(root);
1219
1222
  } catch (error) {
1220
1223
  report.add(
1221
1224
  configurationFile,
@@ -1239,6 +1242,82 @@ function checkRecordsCalibration(report, folder, evaluator, evaluation, contract
1239
1242
  /** The mode bits that let anyone execute a file. */
1240
1243
  const EXECUTE_BITS = 0o111;
1241
1244
 
1245
+ /**
1246
+ * A `command` evaluator's declared framework dependencies (Story 1.44):
1247
+ * `evaluator/frameworks.json` must be there, tracked, and meet its shape (an
1248
+ * empty list for an evaluator with no installed framework), every version
1249
+ * probe must be a tracked regular executable of the layer, and
1250
+ * `evaluator/LEARNED.md` must record the version each nonempty declaration
1251
+ * names. `run` observes what is installed; `check` runs no probe.
1252
+ */
1253
+ function checkFrameworks(report, folder, layer, untracked) {
1254
+ if (untracked(FRAMEWORKS_PATH)) {
1255
+ report.add(
1256
+ FRAMEWORKS_PATH,
1257
+ 'evaluator',
1258
+ `${FRAMEWORKS_PATH} is not tracked by git, and a run reads only the files git tracks under evaluator/; git add it`,
1259
+ );
1260
+ return;
1261
+ }
1262
+ if (!fs.existsSync(path.join(folder, ...FRAMEWORKS_PATH.split('/')))) {
1263
+ report.add(
1264
+ FRAMEWORKS_PATH,
1265
+ 'evaluator',
1266
+ `evaluation.json's evaluator is command, which declares the installed frameworks it depends on in ${FRAMEWORKS_PATH}, and the folder has none; declare each dependency, or an empty list for an evaluator with none`,
1267
+ );
1268
+ return;
1269
+ }
1270
+ const declaration = parseInto(report, folder, FRAMEWORKS_PATH);
1271
+ if (declaration === undefined) return;
1272
+ const shape = declarationProblems(declaration);
1273
+ for (const problem of shape) report.add(FRAMEWORKS_PATH, 'schema', problem);
1274
+ if (shape.length > 0) return;
1275
+ const frameworks = declaredFrameworks(declaration);
1276
+ for (const { package: name, probe } of frameworks) {
1277
+ if (untracked(probe.command)) {
1278
+ report.add(
1279
+ FRAMEWORKS_PATH,
1280
+ 'evaluator',
1281
+ `the version probe of ${name} names ${probe.command}, which git does not track, and a run reads only the files git tracks under evaluator/; git add it`,
1282
+ );
1283
+ continue;
1284
+ }
1285
+ let stats;
1286
+ try {
1287
+ stats = fs.lstatSync(path.join(folder, ...probe.command.split('/')));
1288
+ } catch {
1289
+ stats = null;
1290
+ }
1291
+ if (stats === null || !stats.isFile()) {
1292
+ report.add(
1293
+ FRAMEWORKS_PATH,
1294
+ 'evaluator',
1295
+ `the version probe of ${name} names ${probe.command}, which is not a regular file the evaluation folder holds`,
1296
+ );
1297
+ } else if (process.platform !== 'win32' && (stats.mode & EXECUTE_BITS) === 0) {
1298
+ report.add(
1299
+ FRAMEWORKS_PATH,
1300
+ 'evaluator',
1301
+ `the version probe of ${name} names ${probe.command}, which is not executable; set its execute bit`,
1302
+ );
1303
+ }
1304
+ }
1305
+ // Where the layer could not be read, the layer's own finding already says so.
1306
+ if (layer === null) return;
1307
+ if (untracked(LEARNED_PATH)) {
1308
+ report.add(
1309
+ LEARNED_PATH,
1310
+ 'evaluator',
1311
+ `${LEARNED_PATH} is not tracked by git, and a run reads only the files git tracks under evaluator/; git add it`,
1312
+ );
1313
+ return;
1314
+ }
1315
+ const learned = layer.files.find((file) => file.path === LEARNED_PATH);
1316
+ for (const problem of learnedProblems(frameworks, learned === undefined ? null : learned.bytes.toString('utf8'))) {
1317
+ report.add(LEARNED_PATH, 'evaluator', problem);
1318
+ }
1319
+ }
1320
+
1242
1321
  /**
1243
1322
  * `evaluation.json`'s `evaluator` (AD-21) held to the folder and the
1244
1323
  * contract: a `command` or `sealed-brief-agent` evaluator needs
@@ -1287,14 +1366,8 @@ function checkEvaluator(report, folder, evaluation, contract, conditions, engine
1287
1366
  if (kind === 'records') {
1288
1367
  if (typeof evaluator.records !== 'string') return;
1289
1368
  // Only a directory inside the folder, reached through no link, is the folder's own.
1290
- const spelled = path.join(fs.realpathSync(folder), ...evaluator.records.split('/'));
1291
- let real;
1292
- try {
1293
- real = fs.realpathSync(path.join(folder, ...evaluator.records.split('/')));
1294
- } catch {
1295
- real = null;
1296
- }
1297
- if (real !== spelled || !fs.statSync(real).isDirectory()) {
1369
+ const real = recordsDirectory(folder, evaluator.records);
1370
+ if (real === null) {
1298
1371
  report.add(
1299
1372
  MANIFEST_NAME,
1300
1373
  'evaluator',
@@ -1339,6 +1412,7 @@ function checkEvaluator(report, folder, evaluation, contract, conditions, engine
1339
1412
  `evaluation.json's evaluator is ${kind}, whose judgment rows convert through ${MAPPING_PATH}, and the folder has none`,
1340
1413
  );
1341
1414
  }
1415
+ if (kind === 'command') checkFrameworks(report, folder, layer, untracked);
1342
1416
  if (kind === 'command') {
1343
1417
  if (typeof evaluator.command !== 'string') return;
1344
1418
  if (untracked(evaluator.command)) {
@@ -5,7 +5,7 @@
5
5
  * The plan (`ci-plan.js`) is the only definition of tier membership. The runtime reads it, validates its placement
6
6
  * rules (exit 10 on a finding, 64 when the plan is absent) and runs exactly the checks whose `placement.tier` is the
7
7
  * tier asked for, in plan order. Every check runs, whether or not an earlier one failed, so the evidence bundle is
8
- * complete. An `evaluate` check is run by its id; its `command` is the `tea-evaluate` argv a pipeline step renders. A
8
+ * complete. An `evaluate` check is run by its id; its `command` records the `tea-evaluate` argv a reader can run by hand, and a pipeline runs `tea-evaluate ci --tier <tier>` once per tier. A
9
9
  * `gate` check is an `eval-quality-gates` command the adopter adopted, run as a child process with no shell and the
10
10
  * plan's argv, in the evaluation folder, as the leader of a process group of its own: it ends at the plan check's
11
11
  * `timeoutMs` or past 64 MiB of output (exit 12, what it printed kept), and a signal to `ci` reaches the group first.
@@ -27,6 +27,11 @@
27
27
  * (`confinement.js` `layerPrefix`), so neither it nor any process it starts
28
28
  * can write under `evaluator/` (Story 1.31).
29
29
  *
30
+ * The frameworks it depends on are declared in `evaluator/frameworks.json` and read through the same launch path
31
+ * (`observeFrameworks`, Story 1.44): each declared version probe starts as the evaluator does, with the same
32
+ * environment, an empty private working directory of its own and the same confinement, and prints the installed
33
+ * package identity and version (`frameworks.js`).
34
+ *
30
35
  * Its stdout is read as UTF-8 (a byte sequence that is not UTF-8 reads as
31
36
  * U+FFFD), and what it printed is kept as the bytes it wrote. One that cannot
32
37
  * start, is still running at its wall clock, exits other than 0, or prints
@@ -39,9 +44,81 @@
39
44
  const path = require('node:path');
40
45
 
41
46
  const { buildMinimalEnv, runSupervised } = require('../run-agent');
47
+ const { readProbeAnswer, stderrNote } = require('./frameworks');
42
48
  const { EvaluatorError, readAnswer } = require('./judgment-rows');
43
49
  const { releaseScratchDirectory, makeScratchDirectory } = require('./workspace');
44
50
 
51
+ /**
52
+ * Starts one executable under the evaluation folder's `evaluator/` exactly as
53
+ * the evaluator itself starts: through `spawnPrefix` when the run confines it,
54
+ * under the supervisor, with an empty private working directory in `scratch`
55
+ * removed afterwards, the base environment plus `evaluator.environmentKeys`,
56
+ * and `evaluator.timeoutMs` as its wall clock. The evaluator's launch and a
57
+ * framework's version probe (`observeFrameworks`) both go through it, so a
58
+ * probe sees what the evaluator sees.
59
+ *
60
+ * @returns {Promise<object>} `runSupervised`'s report: the outcome and the streams
61
+ */
62
+ async function launchExecutable({ folder, evaluator, command, args, input, scratch, env, spawnPrefix }) {
63
+ const executable = path.join(folder, ...command.split('/'));
64
+ const cwd = makeScratchDirectory(scratch, 'tea-evaluate-command-');
65
+ try {
66
+ return await runSupervised({
67
+ command: spawnPrefix.length === 0 ? executable : spawnPrefix[0],
68
+ args: [...spawnPrefix.slice(1), ...(spawnPrefix.length === 0 ? [] : [executable]), ...args],
69
+ input,
70
+ cwd,
71
+ env: buildMinimalEnv(evaluator.environmentKeys ?? [], env),
72
+ timeout: evaluator.timeoutMs,
73
+ });
74
+ } finally {
75
+ releaseScratchDirectory(scratch, cwd);
76
+ }
77
+ }
78
+
79
+ /**
80
+ * Reads the installed version of each declared framework (Story 1.44): runs
81
+ * the dependency's version probe through the evaluator's own launch path
82
+ * (`launchExecutable`) and reads the `{ package, version }` it prints
83
+ * (`frameworks.js`). A probe that cannot start, outlives the timeout, exits
84
+ * other than 0 or prints another shape yields no observation for that
85
+ * dependency and a `fault` naming why, with what it printed. Nothing is
86
+ * thrown for a dependency that is missing: the caller compares the entries
87
+ * with the declaration.
88
+ *
89
+ * @param {object} options
90
+ * @param {string} options.folder the evaluation folder
91
+ * @param {object} options.evaluator `evaluation.json`'s `command` evaluator
92
+ * @param {Array<{ package: string, probe: { command: string, args: string[] } }>} options.frameworks
93
+ * @param {string[]} options.scratch
94
+ * @param {NodeJS.ProcessEnv} [options.env]
95
+ * @param {string[]} [options.spawnPrefix]
96
+ * @returns {Promise<Array<{ package: string, observed: { package: string, version: string }|null, fault: string|null, stdout: string, stderr: string }>>}
97
+ * one entry per declared framework, in the order given
98
+ */
99
+ async function observeFrameworks({ folder, evaluator, frameworks, scratch, env = process.env, spawnPrefix = [] }) {
100
+ const entries = [];
101
+ for (const framework of frameworks) {
102
+ const { command, args } = framework.probe;
103
+ const label = `the version probe of ${framework.package} (${command})`;
104
+ const ended = await launchExecutable({ folder, evaluator, command, args, input: '', scratch, env, spawnPrefix });
105
+ const { outcome, stdout, stderr } = ended;
106
+ const entry = { package: framework.package, observed: null, fault: null, stdout, stderr };
107
+ if (outcome.spawnError) entry.fault = `${label} could not start: ${outcome.spawnError.message}`;
108
+ else if (outcome.timedOut) entry.fault = `${label} was still running at the evaluator's ${evaluator.timeoutMs}ms timeout`;
109
+ else if (outcome.failure) entry.fault = `${label} did not finish: ${outcome.failure}`;
110
+ else if (outcome.status === 0) {
111
+ const answer = readProbeAnswer(framework.package, stdout);
112
+ if ('fault' in answer) entry.fault = `${label}: ${answer.fault}`;
113
+ else entry.observed = answer;
114
+ } else {
115
+ entry.fault = `${label} ${outcome.signal ? `was killed by signal ${outcome.signal}` : `exited ${outcome.status}`}${stderrNote(stderr)}`;
116
+ }
117
+ entries.push(entry);
118
+ }
119
+ return entries;
120
+ }
121
+
45
122
  /**
46
123
  * Runs the evaluator over one trial.
47
124
  *
@@ -70,21 +147,16 @@ async function runCommandEvaluator({
70
147
  env = process.env,
71
148
  spawnPrefix = [],
72
149
  }) {
73
- const executable = path.join(folder, ...evaluator.command.split('/'));
74
- const cwd = makeScratchDirectory(scratch, 'tea-evaluate-command-');
75
- let ended;
76
- try {
77
- ended = await runSupervised({
78
- command: spawnPrefix.length === 0 ? executable : spawnPrefix[0],
79
- args: [...spawnPrefix.slice(1), ...(spawnPrefix.length === 0 ? [] : [executable]), ...(evaluator.args ?? [])],
80
- input: `${JSON.stringify({ sealedBrief, observations })}\n`,
81
- cwd,
82
- env: buildMinimalEnv(evaluator.environmentKeys ?? [], env),
83
- timeout: evaluator.timeoutMs,
84
- });
85
- } finally {
86
- releaseScratchDirectory(scratch, cwd);
87
- }
150
+ const ended = await launchExecutable({
151
+ folder,
152
+ evaluator,
153
+ command: evaluator.command,
154
+ args: evaluator.args ?? [],
155
+ input: `${JSON.stringify({ sealedBrief, observations })}\n`,
156
+ scratch,
157
+ env,
158
+ spawnPrefix,
159
+ });
88
160
  const { outcome, stdout, stderr, stdoutBytes, stderrBytes } = ended;
89
161
  const streams = { stdout, stderr, stdoutBytes, stderrBytes };
90
162
  const fail = (message) => Object.assign(new EvaluatorError(message, streams), { outcome });
@@ -108,4 +180,4 @@ async function runCommandEvaluator({
108
180
  }
109
181
  }
110
182
 
111
- module.exports = { runCommandEvaluator };
183
+ module.exports = { observeFrameworks, runCommandEvaluator };