bmad-method-test-architecture-enterprise 1.27.3-next.21 → 1.27.3-next.22
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/README.md +5 -5
- package/cli/lib/evaluate/arm.js +5 -0
- package/cli/lib/evaluate/check.js +82 -0
- package/cli/lib/evaluate/ci.js +1 -1
- package/cli/lib/evaluate/command-evaluator.js +88 -16
- package/cli/lib/evaluate/confinement.js +237 -14
- package/cli/lib/evaluate/evaluators.js +52 -4
- package/cli/lib/evaluate/frameworks.js +324 -0
- package/cli/lib/evaluate/preflight.js +2 -0
- package/cli/lib/evaluate/registry.js +35 -4
- package/cli/lib/evaluate/run.js +87 -11
- package/cli/lib/evaluate/schemas/evaluation-ci-plan.schema.json +1 -1
- package/cli/lib/evaluate/schemas/evaluator-frameworks.schema.json +37 -0
- package/cli/lib/evaluate/schemas/framework-versions.schema.json +39 -0
- package/cli/lib/evaluate/workspace.js +1 -23
- package/package.json +3 -2
- package/src/workflows/testarch/bmad-testarch-ci/SKILL.md +2 -0
- package/src/workflows/testarch/bmad-testarch-ci/checklist.md +9 -0
- package/src/workflows/testarch/bmad-testarch-ci/github-actions-template.yaml +73 -0
- package/src/workflows/testarch/bmad-testarch-ci/instructions.md +5 -1
- package/src/workflows/testarch/bmad-testarch-ci/resources/ci-pipeline-progress.example.md +12 -1
- package/src/workflows/testarch/bmad-testarch-ci/steps-c/step-01b-resume.md +5 -3
- package/src/workflows/testarch/bmad-testarch-ci/steps-c/step-03-configure-quality-gates.md +1 -1
- package/src/workflows/testarch/bmad-testarch-ci/steps-c/step-03b-render-evaluation-plans.md +142 -0
- package/src/workflows/testarch/bmad-testarch-ci/steps-c/step-04-validate-and-summary.md +2 -0
- package/src/workflows/testarch/bmad-testarch-ci/steps-e/step-01-assess.md +7 -2
- package/src/workflows/testarch/bmad-testarch-ci/steps-e/step-02-apply-edit.md +4 -1
- package/src/workflows/testarch/bmad-testarch-ci/steps-v/step-01-validate.md +4 -0
- package/src/workflows/testarch/bmad-testarch-evaluate/assets/README.md +2 -2
- package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/LEARNED.md +2 -1
- package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/agentevals-frameworks.json +10 -0
- package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/agentevals-trajectory.mjs +1 -0
- package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/command-evaluator.mjs +1 -0
- package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/frameworks.json +4 -0
- package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/installed-version.mjs +42 -0
- package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/promptfoo-assertions.mjs +1 -0
- package/src/workflows/testarch/bmad-testarch-evaluate/assets/evaluators/promptfoo-frameworks.json +10 -0
- package/src/workflows/testarch/bmad-testarch-evaluate/references/evaluator.md +51 -7
- package/src/workflows/testarch/bmad-testarch-evaluate/references/gaps.md +15 -15
- package/src/workflows/testarch/bmad-testarch-evaluate/references/harness.md +23 -0
|
@@ -31,7 +31,7 @@
|
|
|
31
31
|
"name": "bmad-method-test-architecture-enterprise",
|
|
32
32
|
"source": "./",
|
|
33
33
|
"description": "Master Test Architect module for quality strategy, test automation, CI/CD quality gates, and structured testing education. Part of the BMad Method ecosystem.",
|
|
34
|
-
"version": "1.27.3-next.
|
|
34
|
+
"version": "1.27.3-next.22",
|
|
35
35
|
"author": {
|
|
36
36
|
"name": "Murat K Ozcan (TEA Creator) & Brian (BMad) Madison"
|
|
37
37
|
},
|
package/README.md
CHANGED
|
@@ -383,7 +383,7 @@ The `atdd`, `automate`, `ci`, `framework`, `nfr`, `teach-me-testing`, `test-desi
|
|
|
383
383
|
| `bmad-teach-me-testing` | N/A | Yes; multi-turn session with seeded wrong quiz answer, review loop, and progress |
|
|
384
384
|
| `bmad-testarch-atdd` | 3 | Yes; one unimplemented fixture and five acceptance criteria |
|
|
385
385
|
| `bmad-testarch-automate` | 5 | Yes; four hand-authored spec sets against a fixed service and its seeded regression |
|
|
386
|
-
| `bmad-testarch-ci` | 2 | Yes; one full request and one
|
|
386
|
+
| `bmad-testarch-ci` | 2 | Yes; one full request, one minimal request and one request over an evaluation plan |
|
|
387
387
|
| `bmad-testarch-evaluate` | N/A | Yes; Evaluate-authored, a gap-guide class swap seeded and a second held out |
|
|
388
388
|
| `bmad-testarch-framework` | 3 | Yes; install and smoke-test generated scaffold in network-isolated sandbox |
|
|
389
389
|
| `bmad-testarch-nfr` | 2 | Yes; one evidence bundle with known gaps and one clean bundle |
|
|
@@ -395,7 +395,7 @@ A passing fragment-selection eval means the workflow loaded the right knowledge.
|
|
|
395
395
|
|
|
396
396
|
### Deterministic Checks
|
|
397
397
|
|
|
398
|
-
`npm test` chains
|
|
398
|
+
`npm test` chains 107 checks. That count covers the whole chain: every entry in it is deterministic and credential-free, so the chain and its credential-free subset are the same list. `npm run test:ci-coverage` derives the count from `package.json` and prints it. Twelve of the 107 keep the rules, guidance, hook, eval data, eval contracts, diagnostics, and documentation aligned:
|
|
399
399
|
|
|
400
400
|
- `test:criteria-fragments` fails when a registry row is neither mapped to a knowledge fragment nor declared a known gap. A rule the reviewer scores but no fragment teaches is a rule TEA punishes without ever having explained it. All 36 rows are currently mapped across 50 anchors. Because the declared-gap list is empty, the validator feeds itself a synthetic unmapped row on every run to prove that path still works.
|
|
401
401
|
- `test:doc-counts` runs `eval-quality-gates doc-counts`, which holds a hand-written count on a published page against the source that computes it: the roadmap's per-suite `eval:all` call counts, the knowledge-fragment tier breakdown, this section's own npm-test-chain length, and the fragment-selection case count. A pattern matching no sentence, or more than one, fails the same way a wrong number does, so the entry cannot go stale by drifting out from under its own pattern either.
|
|
@@ -432,7 +432,7 @@ npm run eval:all -- --agent agy
|
|
|
432
432
|
npm run eval:all -- --agent agy --agent claude --agent codex
|
|
433
433
|
```
|
|
434
434
|
|
|
435
|
-
`eval:all` uses two repetitions per fragment-selection case and per routing intent, three repetitions for `test-review`, and two repetitions per `nfr` evidence bundle, per `ci` project, per `test-design` epic, per `trace` fixture set, and per `atdd` story. One runner makes
|
|
435
|
+
`eval:all` uses two repetitions per fragment-selection case and per routing intent, three repetitions for `test-review`, and two repetitions per `nfr` evidence bundle, per `ci` project, per `test-design` epic, per `trace` fixture set, and per `atdd` story. One runner makes 109 agent calls: 48 fragment selections, 38 routing intents, 3 reviews, 4 audits, 6 pipelines, 4 test designs, 4 traces, and 2 ATDD generations. All three built-in runners make 327 calls.
|
|
436
436
|
|
|
437
437
|
Check the data, executable, login, fixtures, and expected results without making a model call:
|
|
438
438
|
|
|
@@ -558,7 +558,7 @@ Every row below is a declared gate. Ten of TEA's eleven skills have behavioral s
|
|
|
558
558
|
|
|
559
559
|
| Eval | Declared threshold | Declared volume |
|
|
560
560
|
| --------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------- |
|
|
561
|
-
| `npm run eval:all -- --agent ...` | All eight live suites below meet their thresholds for the selected runner | 48 selections, 38 routing calls, 3 reviews, 4 audits,
|
|
561
|
+
| `npm run eval:all -- --agent ...` | All eight live suites below meet their thresholds for the selected runner | 48 selections, 38 routing calls, 3 reviews, 4 audits, 6 pipelines, 4 test designs, 4 traces, 2 generations |
|
|
562
562
|
| Fragment selection | At least 90% required-fragment recall, at most 10% forbidden-fragment selection, and stable choices across repeated cases | 24 cases twice: 48 calls |
|
|
563
563
|
| Test review | At least 70% overall recall, 100% CRITICAL recall, an 80% non-false-positive rate, score standard deviation no higher than 3, and a stable verdict | Three complete reviews |
|
|
564
564
|
| Trace | At least 90% criterion-status accuracy and 90% evidence-citation precision; 100% on both discriminating criteria, the gate decision, the gate criteria, the coverage arithmetic, the oracle resolution, the rejected evidence, the waiver validity, and the live evidence; no clean-set false positive, no unstable case, and no fixture mutation | Two fixture sets twice: 4 calls |
|
|
@@ -596,7 +596,7 @@ npm ci
|
|
|
596
596
|
npm run test:eval-data # fragment-selection corpus, static
|
|
597
597
|
npm run test:eval-trace-data # trace corpus, static
|
|
598
598
|
npm run test:eval-schemas # manifest against harness constants, and the preflight argv
|
|
599
|
-
npm run test:eval-replay #
|
|
599
|
+
npm run test:eval-replay # stored outputs against the scorers
|
|
600
600
|
```
|
|
601
601
|
|
|
602
602
|
Run live evals in a scheduled or manually triggered CI job after installing and authenticating the selected agent CLI:
|
package/cli/lib/evaluate/arm.js
CHANGED
|
@@ -379,6 +379,8 @@ function bodyValue(body) {
|
|
|
379
379
|
*/
|
|
380
380
|
function hostEnvironmentPort({ port, registry }) {
|
|
381
381
|
return {
|
|
382
|
+
/** Empties the private home the port's sandbox keeps, where it has one (`registry.js` `createProbePort`). */
|
|
383
|
+
resetHome: () => port.resetHome?.(),
|
|
382
384
|
async probe(request, signal) {
|
|
383
385
|
const registered = request?.kind === 'cli' && registry.targetFor(request.interfaceId, request.executable) !== undefined;
|
|
384
386
|
const injected = registered ? registry.hostEnvironment(request.interfaceId, [], request.executable) : {};
|
|
@@ -720,6 +722,9 @@ async function runArm({
|
|
|
720
722
|
signal,
|
|
721
723
|
seed = 'tea-evaluate-default-seed',
|
|
722
724
|
}) {
|
|
725
|
+
// An arm is independent of the arms before it on a shared port, so its target starts with an empty private home; the
|
|
726
|
+
// steps of this arm share it.
|
|
727
|
+
port.resetHome?.();
|
|
723
728
|
const operations = operationsById(contract);
|
|
724
729
|
const plan = contract.interactionPlan ?? [];
|
|
725
730
|
const declared = new Set(plan.map((step) => step.stepId));
|
|
@@ -70,6 +70,10 @@
|
|
|
70
70
|
* - `evaluator` (Story 1.34): a `sealed-brief-agent` evaluator whose `evaluation.json` has no
|
|
71
71
|
* `evaluatorQualification` (`attempts`, `minimumAgreement`), which `run` needs to qualify the agent
|
|
72
72
|
* before its verdicts count; and an `evaluatorQualification` beside any other kind, where nothing would use it.
|
|
73
|
+
* - `evaluator` (Story 1.44, AD-21): a `command` evaluator has no tracked `evaluator/frameworks.json` (its declared
|
|
74
|
+
* installed frameworks, an empty list for none), one whose version probe is not a tracked regular executable under
|
|
75
|
+
* `evaluator/`, or an `evaluator/LEARNED.md` whose `package@version` records disagree with a declaration; a declaration
|
|
76
|
+
* that fails its shape is reported under `schema` (or `json`). `check` runs no probe (`run` does, `frameworks.js`).
|
|
73
77
|
* - `evaluator` (Story 1.17, AD-21): a `command` or `sealed-brief-agent` evaluator has no
|
|
74
78
|
* `evaluator/mapping.json`, or one binding an oracle, behavior or rubric criterion the contract does not
|
|
75
79
|
* declare, an oracle to a behavior that does not declare it, levels other than the criterion's anchored
|
|
@@ -140,6 +144,7 @@ const {
|
|
|
140
144
|
} = require('./registry');
|
|
141
145
|
const { AGENT_ADAPTERS, bridgedArgsRefused, resolveModel } = require('../agent-adapters');
|
|
142
146
|
const { EVALUATOR_DIRECTORY, EvaluatorLayerError, evaluatorFiles, evaluatorOf, isKnownEvaluator } = require('./evaluators');
|
|
147
|
+
const { FRAMEWORKS_PATH, LEARNED_PATH, declarationProblems, declaredFrameworks, learnedProblems } = require('./frameworks');
|
|
143
148
|
const { answeredKind, degenerateResponsePath } = require('./gameability');
|
|
144
149
|
|
|
145
150
|
/** How a finding names a call of each interface kind. */
|
|
@@ -1239,6 +1244,82 @@ function checkRecordsCalibration(report, folder, evaluator, evaluation, contract
|
|
|
1239
1244
|
/** The mode bits that let anyone execute a file. */
|
|
1240
1245
|
const EXECUTE_BITS = 0o111;
|
|
1241
1246
|
|
|
1247
|
+
/**
|
|
1248
|
+
* A `command` evaluator's declared framework dependencies (Story 1.44):
|
|
1249
|
+
* `evaluator/frameworks.json` must be there, tracked, and meet its shape (an
|
|
1250
|
+
* empty list for an evaluator with no installed framework), every version
|
|
1251
|
+
* probe must be a tracked regular executable of the layer, and
|
|
1252
|
+
* `evaluator/LEARNED.md` must record the version each nonempty declaration
|
|
1253
|
+
* names. `run` observes what is installed; `check` runs no probe.
|
|
1254
|
+
*/
|
|
1255
|
+
function checkFrameworks(report, folder, layer, untracked) {
|
|
1256
|
+
if (untracked(FRAMEWORKS_PATH)) {
|
|
1257
|
+
report.add(
|
|
1258
|
+
FRAMEWORKS_PATH,
|
|
1259
|
+
'evaluator',
|
|
1260
|
+
`${FRAMEWORKS_PATH} is not tracked by git, and a run reads only the files git tracks under evaluator/; git add it`,
|
|
1261
|
+
);
|
|
1262
|
+
return;
|
|
1263
|
+
}
|
|
1264
|
+
if (!fs.existsSync(path.join(folder, ...FRAMEWORKS_PATH.split('/')))) {
|
|
1265
|
+
report.add(
|
|
1266
|
+
FRAMEWORKS_PATH,
|
|
1267
|
+
'evaluator',
|
|
1268
|
+
`evaluation.json's evaluator is command, which declares the installed frameworks it depends on in ${FRAMEWORKS_PATH}, and the folder has none; declare each dependency, or an empty list for an evaluator with none`,
|
|
1269
|
+
);
|
|
1270
|
+
return;
|
|
1271
|
+
}
|
|
1272
|
+
const declaration = parseInto(report, folder, FRAMEWORKS_PATH);
|
|
1273
|
+
if (declaration === undefined) return;
|
|
1274
|
+
const shape = declarationProblems(declaration);
|
|
1275
|
+
for (const problem of shape) report.add(FRAMEWORKS_PATH, 'schema', problem);
|
|
1276
|
+
if (shape.length > 0) return;
|
|
1277
|
+
const frameworks = declaredFrameworks(declaration);
|
|
1278
|
+
for (const { package: name, probe } of frameworks) {
|
|
1279
|
+
if (untracked(probe.command)) {
|
|
1280
|
+
report.add(
|
|
1281
|
+
FRAMEWORKS_PATH,
|
|
1282
|
+
'evaluator',
|
|
1283
|
+
`the version probe of ${name} names ${probe.command}, which git does not track, and a run reads only the files git tracks under evaluator/; git add it`,
|
|
1284
|
+
);
|
|
1285
|
+
continue;
|
|
1286
|
+
}
|
|
1287
|
+
let stats;
|
|
1288
|
+
try {
|
|
1289
|
+
stats = fs.lstatSync(path.join(folder, ...probe.command.split('/')));
|
|
1290
|
+
} catch {
|
|
1291
|
+
stats = null;
|
|
1292
|
+
}
|
|
1293
|
+
if (stats === null || !stats.isFile()) {
|
|
1294
|
+
report.add(
|
|
1295
|
+
FRAMEWORKS_PATH,
|
|
1296
|
+
'evaluator',
|
|
1297
|
+
`the version probe of ${name} names ${probe.command}, which is not a regular file the evaluation folder holds`,
|
|
1298
|
+
);
|
|
1299
|
+
} else if (process.platform !== 'win32' && (stats.mode & EXECUTE_BITS) === 0) {
|
|
1300
|
+
report.add(
|
|
1301
|
+
FRAMEWORKS_PATH,
|
|
1302
|
+
'evaluator',
|
|
1303
|
+
`the version probe of ${name} names ${probe.command}, which is not executable; set its execute bit`,
|
|
1304
|
+
);
|
|
1305
|
+
}
|
|
1306
|
+
}
|
|
1307
|
+
// Where the layer could not be read, the layer's own finding already says so.
|
|
1308
|
+
if (layer === null) return;
|
|
1309
|
+
if (untracked(LEARNED_PATH)) {
|
|
1310
|
+
report.add(
|
|
1311
|
+
LEARNED_PATH,
|
|
1312
|
+
'evaluator',
|
|
1313
|
+
`${LEARNED_PATH} is not tracked by git, and a run reads only the files git tracks under evaluator/; git add it`,
|
|
1314
|
+
);
|
|
1315
|
+
return;
|
|
1316
|
+
}
|
|
1317
|
+
const learned = layer.files.find((file) => file.path === LEARNED_PATH);
|
|
1318
|
+
for (const problem of learnedProblems(frameworks, learned === undefined ? null : learned.bytes.toString('utf8'))) {
|
|
1319
|
+
report.add(LEARNED_PATH, 'evaluator', problem);
|
|
1320
|
+
}
|
|
1321
|
+
}
|
|
1322
|
+
|
|
1242
1323
|
/**
|
|
1243
1324
|
* `evaluation.json`'s `evaluator` (AD-21) held to the folder and the
|
|
1244
1325
|
* contract: a `command` or `sealed-brief-agent` evaluator needs
|
|
@@ -1339,6 +1420,7 @@ function checkEvaluator(report, folder, evaluation, contract, conditions, engine
|
|
|
1339
1420
|
`evaluation.json's evaluator is ${kind}, whose judgment rows convert through ${MAPPING_PATH}, and the folder has none`,
|
|
1340
1421
|
);
|
|
1341
1422
|
}
|
|
1423
|
+
if (kind === 'command') checkFrameworks(report, folder, layer, untracked);
|
|
1342
1424
|
if (kind === 'command') {
|
|
1343
1425
|
if (typeof evaluator.command !== 'string') return;
|
|
1344
1426
|
if (untracked(evaluator.command)) {
|
package/cli/lib/evaluate/ci.js
CHANGED
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
* The plan (`ci-plan.js`) is the only definition of tier membership. The runtime reads it, validates its placement
|
|
6
6
|
* rules (exit 10 on a finding, 64 when the plan is absent) and runs exactly the checks whose `placement.tier` is the
|
|
7
7
|
* tier asked for, in plan order. Every check runs, whether or not an earlier one failed, so the evidence bundle is
|
|
8
|
-
* complete. An `evaluate` check is run by its id; its `command`
|
|
8
|
+
* complete. An `evaluate` check is run by its id; its `command` records the `tea-evaluate` argv a reader can run by hand, and a pipeline runs `tea-evaluate ci --tier <tier>` once per tier. A
|
|
9
9
|
* `gate` check is an `eval-quality-gates` command the adopter adopted, run as a child process with no shell and the
|
|
10
10
|
* plan's argv, in the evaluation folder, as the leader of a process group of its own: it ends at the plan check's
|
|
11
11
|
* `timeoutMs` or past 64 MiB of output (exit 12, what it printed kept), and a signal to `ci` reaches the group first.
|
|
@@ -27,6 +27,11 @@
|
|
|
27
27
|
* (`confinement.js` `layerPrefix`), so neither it nor any process it starts
|
|
28
28
|
* can write under `evaluator/` (Story 1.31).
|
|
29
29
|
*
|
|
30
|
+
* The frameworks it depends on are declared in `evaluator/frameworks.json` and read through the same launch path
|
|
31
|
+
* (`observeFrameworks`, Story 1.44): each declared version probe starts as the evaluator does, with the same
|
|
32
|
+
* environment, an empty private working directory of its own and the same confinement, and prints the installed
|
|
33
|
+
* package identity and version (`frameworks.js`).
|
|
34
|
+
*
|
|
30
35
|
* Its stdout is read as UTF-8 (a byte sequence that is not UTF-8 reads as
|
|
31
36
|
* U+FFFD), and what it printed is kept as the bytes it wrote. One that cannot
|
|
32
37
|
* start, is still running at its wall clock, exits other than 0, or prints
|
|
@@ -39,9 +44,81 @@
|
|
|
39
44
|
const path = require('node:path');
|
|
40
45
|
|
|
41
46
|
const { buildMinimalEnv, runSupervised } = require('../run-agent');
|
|
47
|
+
const { readProbeAnswer, stderrNote } = require('./frameworks');
|
|
42
48
|
const { EvaluatorError, readAnswer } = require('./judgment-rows');
|
|
43
49
|
const { releaseScratchDirectory, makeScratchDirectory } = require('./workspace');
|
|
44
50
|
|
|
51
|
+
/**
|
|
52
|
+
* Starts one executable under the evaluation folder's `evaluator/` exactly as
|
|
53
|
+
* the evaluator itself starts: through `spawnPrefix` when the run confines it,
|
|
54
|
+
* under the supervisor, with an empty private working directory in `scratch`
|
|
55
|
+
* removed afterwards, the base environment plus `evaluator.environmentKeys`,
|
|
56
|
+
* and `evaluator.timeoutMs` as its wall clock. The evaluator's launch and a
|
|
57
|
+
* framework's version probe (`observeFrameworks`) both go through it, so a
|
|
58
|
+
* probe sees what the evaluator sees.
|
|
59
|
+
*
|
|
60
|
+
* @returns {Promise<object>} `runSupervised`'s report: the outcome and the streams
|
|
61
|
+
*/
|
|
62
|
+
async function launchExecutable({ folder, evaluator, command, args, input, scratch, env, spawnPrefix }) {
|
|
63
|
+
const executable = path.join(folder, ...command.split('/'));
|
|
64
|
+
const cwd = makeScratchDirectory(scratch, 'tea-evaluate-command-');
|
|
65
|
+
try {
|
|
66
|
+
return await runSupervised({
|
|
67
|
+
command: spawnPrefix.length === 0 ? executable : spawnPrefix[0],
|
|
68
|
+
args: [...spawnPrefix.slice(1), ...(spawnPrefix.length === 0 ? [] : [executable]), ...args],
|
|
69
|
+
input,
|
|
70
|
+
cwd,
|
|
71
|
+
env: buildMinimalEnv(evaluator.environmentKeys ?? [], env),
|
|
72
|
+
timeout: evaluator.timeoutMs,
|
|
73
|
+
});
|
|
74
|
+
} finally {
|
|
75
|
+
releaseScratchDirectory(scratch, cwd);
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* Reads the installed version of each declared framework (Story 1.44): runs
|
|
81
|
+
* the dependency's version probe through the evaluator's own launch path
|
|
82
|
+
* (`launchExecutable`) and reads the `{ package, version }` it prints
|
|
83
|
+
* (`frameworks.js`). A probe that cannot start, outlives the timeout, exits
|
|
84
|
+
* other than 0 or prints another shape yields no observation for that
|
|
85
|
+
* dependency and a `fault` naming why, with what it printed. Nothing is
|
|
86
|
+
* thrown for a dependency that is missing: the caller compares the entries
|
|
87
|
+
* with the declaration.
|
|
88
|
+
*
|
|
89
|
+
* @param {object} options
|
|
90
|
+
* @param {string} options.folder the evaluation folder
|
|
91
|
+
* @param {object} options.evaluator `evaluation.json`'s `command` evaluator
|
|
92
|
+
* @param {Array<{ package: string, probe: { command: string, args: string[] } }>} options.frameworks
|
|
93
|
+
* @param {string[]} options.scratch
|
|
94
|
+
* @param {NodeJS.ProcessEnv} [options.env]
|
|
95
|
+
* @param {string[]} [options.spawnPrefix]
|
|
96
|
+
* @returns {Promise<Array<{ package: string, observed: { package: string, version: string }|null, fault: string|null, stdout: string, stderr: string }>>}
|
|
97
|
+
* one entry per declared framework, in the order given
|
|
98
|
+
*/
|
|
99
|
+
async function observeFrameworks({ folder, evaluator, frameworks, scratch, env = process.env, spawnPrefix = [] }) {
|
|
100
|
+
const entries = [];
|
|
101
|
+
for (const framework of frameworks) {
|
|
102
|
+
const { command, args } = framework.probe;
|
|
103
|
+
const label = `the version probe of ${framework.package} (${command})`;
|
|
104
|
+
const ended = await launchExecutable({ folder, evaluator, command, args, input: '', scratch, env, spawnPrefix });
|
|
105
|
+
const { outcome, stdout, stderr } = ended;
|
|
106
|
+
const entry = { package: framework.package, observed: null, fault: null, stdout, stderr };
|
|
107
|
+
if (outcome.spawnError) entry.fault = `${label} could not start: ${outcome.spawnError.message}`;
|
|
108
|
+
else if (outcome.timedOut) entry.fault = `${label} was still running at the evaluator's ${evaluator.timeoutMs}ms timeout`;
|
|
109
|
+
else if (outcome.failure) entry.fault = `${label} did not finish: ${outcome.failure}`;
|
|
110
|
+
else if (outcome.status === 0) {
|
|
111
|
+
const answer = readProbeAnswer(framework.package, stdout);
|
|
112
|
+
if ('fault' in answer) entry.fault = `${label}: ${answer.fault}`;
|
|
113
|
+
else entry.observed = answer;
|
|
114
|
+
} else {
|
|
115
|
+
entry.fault = `${label} ${outcome.signal ? `was killed by signal ${outcome.signal}` : `exited ${outcome.status}`}${stderrNote(stderr)}`;
|
|
116
|
+
}
|
|
117
|
+
entries.push(entry);
|
|
118
|
+
}
|
|
119
|
+
return entries;
|
|
120
|
+
}
|
|
121
|
+
|
|
45
122
|
/**
|
|
46
123
|
* Runs the evaluator over one trial.
|
|
47
124
|
*
|
|
@@ -70,21 +147,16 @@ async function runCommandEvaluator({
|
|
|
70
147
|
env = process.env,
|
|
71
148
|
spawnPrefix = [],
|
|
72
149
|
}) {
|
|
73
|
-
const
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
timeout: evaluator.timeoutMs,
|
|
84
|
-
});
|
|
85
|
-
} finally {
|
|
86
|
-
releaseScratchDirectory(scratch, cwd);
|
|
87
|
-
}
|
|
150
|
+
const ended = await launchExecutable({
|
|
151
|
+
folder,
|
|
152
|
+
evaluator,
|
|
153
|
+
command: evaluator.command,
|
|
154
|
+
args: evaluator.args ?? [],
|
|
155
|
+
input: `${JSON.stringify({ sealedBrief, observations })}\n`,
|
|
156
|
+
scratch,
|
|
157
|
+
env,
|
|
158
|
+
spawnPrefix,
|
|
159
|
+
});
|
|
88
160
|
const { outcome, stdout, stderr, stdoutBytes, stderrBytes } = ended;
|
|
89
161
|
const streams = { stdout, stderr, stdoutBytes, stderrBytes };
|
|
90
162
|
const fail = (message) => Object.assign(new EvaluatorError(message, streams), { outcome });
|
|
@@ -108,4 +180,4 @@ async function runCommandEvaluator({
|
|
|
108
180
|
}
|
|
109
181
|
}
|
|
110
182
|
|
|
111
|
-
module.exports = { runCommandEvaluator };
|
|
183
|
+
module.exports = { observeFrameworks, runCommandEvaluator };
|