@forwardimpact/libharness 1.3.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/fit-harness.js +18 -0
- package/bin/fit-trace.js +10 -0
- package/package.json +1 -1
- package/src/advisor.js +218 -0
- package/src/agent-runner.js +6 -0
- package/src/benchmark/grade.js +222 -0
- package/src/benchmark/hidden-tests.js +180 -0
- package/src/benchmark/invariants.js +9 -10
- package/src/benchmark/judge.js +9 -6
- package/src/benchmark/report.js +150 -56
- package/src/benchmark/result.js +43 -8
- package/src/benchmark/runner.js +99 -109
- package/src/benchmark/task-family.js +139 -26
- package/src/benchmark/trace-split.js +73 -0
- package/src/benchmark/workdir.js +6 -1
- package/src/commands/advisor-flags.js +28 -0
- package/src/commands/assert.js +72 -0
- package/src/commands/benchmark-definition.js +11 -6
- package/src/commands/benchmark-grade.js +82 -0
- package/src/commands/benchmark-report.js +12 -2
- package/src/commands/discuss.js +4 -0
- package/src/commands/facilitate.js +4 -0
- package/src/commands/run.js +162 -67
- package/src/commands/supervise.js +4 -0
- package/src/discuss-tools.js +2 -1
- package/src/discusser.js +65 -10
- package/src/facilitator.js +64 -9
- package/src/index.js +9 -0
- package/src/orchestration-toolkit.js +65 -7
- package/src/supervisor.js +72 -10
- package/src/transcript-recorder.js +94 -0
- package/src/commands/benchmark-invariants.js +0 -73
package/bin/fit-harness.js
CHANGED
|
@@ -27,6 +27,20 @@ const LEAD_OPTIONS = {
|
|
|
27
27
|
},
|
|
28
28
|
};
|
|
29
29
|
|
|
30
|
+
// Advisor consult flags, shared by all four session modes.
|
|
31
|
+
const ADVISOR_OPTIONS = {
|
|
32
|
+
"advisor-model": {
|
|
33
|
+
type: "string",
|
|
34
|
+
description:
|
|
35
|
+
"Claude model for advisor consults; omitting the flag disables the Advisor tool (default: off)",
|
|
36
|
+
},
|
|
37
|
+
"advisor-max-uses": {
|
|
38
|
+
type: "string",
|
|
39
|
+
description:
|
|
40
|
+
"Session-wide consult budget shared by all participants (default: 3; requires --advisor-model)",
|
|
41
|
+
},
|
|
42
|
+
};
|
|
43
|
+
|
|
30
44
|
// Shared task-input flags: --task-file (path), --task-text (inline), and
|
|
31
45
|
// --task-event (path to native GitHub event JSON composed into a task via
|
|
32
46
|
// libharness/src/events/github.js). Exactly one of the three is required.
|
|
@@ -94,6 +108,7 @@ const definition = {
|
|
|
94
108
|
description:
|
|
95
109
|
"Connect to the MCP service (e.g. --mcp-server=guide); adds mcp__<name>__* to allowed tools",
|
|
96
110
|
},
|
|
111
|
+
...ADVISOR_OPTIONS,
|
|
97
112
|
},
|
|
98
113
|
},
|
|
99
114
|
{
|
|
@@ -143,6 +158,7 @@ const definition = {
|
|
|
143
158
|
description:
|
|
144
159
|
"Connect to the MCP service (e.g. --mcp-server=guide); adds mcp__<name>__* to allowed tools",
|
|
145
160
|
},
|
|
161
|
+
...ADVISOR_OPTIONS,
|
|
146
162
|
},
|
|
147
163
|
},
|
|
148
164
|
{
|
|
@@ -185,6 +201,7 @@ const definition = {
|
|
|
185
201
|
description:
|
|
186
202
|
"Active work-item tracker (github|filesystem, default: github)",
|
|
187
203
|
},
|
|
204
|
+
...ADVISOR_OPTIONS,
|
|
188
205
|
},
|
|
189
206
|
},
|
|
190
207
|
{
|
|
@@ -231,6 +248,7 @@ const definition = {
|
|
|
231
248
|
description:
|
|
232
249
|
"Active work-item tracker (github|filesystem, default: github)",
|
|
233
250
|
},
|
|
251
|
+
...ADVISOR_OPTIONS,
|
|
234
252
|
},
|
|
235
253
|
},
|
|
236
254
|
{
|
package/bin/fit-trace.js
CHANGED
|
@@ -398,6 +398,16 @@ const definition = {
|
|
|
398
398
|
type: "string",
|
|
399
399
|
description: "Custom failure message",
|
|
400
400
|
},
|
|
401
|
+
gate: {
|
|
402
|
+
type: "boolean",
|
|
403
|
+
description:
|
|
404
|
+
"Mark the row a gate check (any failing gate fails the run)",
|
|
405
|
+
},
|
|
406
|
+
weight: {
|
|
407
|
+
type: "string",
|
|
408
|
+
description:
|
|
409
|
+
"Attach a numeric weight to the scored row; 0 marks the row diagnostic",
|
|
410
|
+
},
|
|
401
411
|
},
|
|
402
412
|
},
|
|
403
413
|
],
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@forwardimpact/libharness",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "2.0.0",
|
|
4
4
|
"description": "Autonomous agent team harness — coordinate a lead and participant agents in one async session, with eval, benchmark, and trace tooling to prove the changes worked.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"orchestration",
|
package/src/advisor.js
ADDED
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Advisor — the judge's mid-loop sibling: a solo, tool-restricted, one-shot
|
|
3
|
+
* `AgentRunner` session on a stronger model whose final text is the advice.
|
|
4
|
+
* Each consult forwards the caller's whole recorded context (system prompt,
|
|
5
|
+
* delivered prompts, transcript so far) plus a focused question; the advisor
|
|
6
|
+
* can inspect files read-only but holds no write, execute, subagent, or
|
|
7
|
+
* orchestration tools and never appears on the message bus.
|
|
8
|
+
*
|
|
9
|
+
* Consults are stateless — one fresh session per call, re-reading the
|
|
10
|
+
* caller's context as it stands — and fail-open: timeout, error, and abort
|
|
11
|
+
* all resolve to an in-band `{unavailable}` result so the caller's session
|
|
12
|
+
* never stalls or crashes on a consult.
|
|
13
|
+
*
|
|
14
|
+
* Follows OO+DI: factory function, tests inject a fake `query`.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import { Writable } from "node:stream";
|
|
18
|
+
|
|
19
|
+
import { createAgentRunner } from "./agent-runner.js";
|
|
20
|
+
import { composeSystemPrompt } from "./profile-prompt.js";
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* System-prompt trailer for the advisor session. Fixes the response
|
|
24
|
+
* contract (spec criterion "Advice is bounded"): assessment,
|
|
25
|
+
* recommendation, unsolicited findings, with a stated length ceiling.
|
|
26
|
+
*/
|
|
27
|
+
export const ADVISOR_SYSTEM_PROMPT =
|
|
28
|
+
"You are a consulted specialist, not a worker. " +
|
|
29
|
+
"Another agent paused its work to ask you one question; its full session context and the question are in the task. " +
|
|
30
|
+
"You may Read, Glob, and Grep the files the transcript names to ground your advice; never modify anything. " +
|
|
31
|
+
"Respond in one turn of prose — your final text is delivered to the caller verbatim. " +
|
|
32
|
+
"Structure the response as: assessment (what you see), recommendation (what to do and why), and unsolicited findings (anything important the caller did not ask about). " +
|
|
33
|
+
"Keep the whole response to at most three short paragraphs. " +
|
|
34
|
+
"Do not ask follow-up questions — the caller cannot reply.";
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Consult-guidance fragment for caller system prompts, present only when
|
|
38
|
+
* the session runs with an advisor model. Steers the caller's judgment; it
|
|
39
|
+
* mandates nothing.
|
|
40
|
+
* @param {number} maxUses - The session-wide consult budget.
|
|
41
|
+
* @returns {string}
|
|
42
|
+
*/
|
|
43
|
+
export function advisorGuidance(maxUses) {
|
|
44
|
+
return (
|
|
45
|
+
"An `Advisor` tool is available: one focused question per call, answered by a stronger model that sees your full session context. " +
|
|
46
|
+
"A consult pays off at hard decision points — architectural forks, unclear root causes, trade-offs you cannot rank — and early, before work builds on an unvalidated assumption. " +
|
|
47
|
+
"It does not pay off for routine reads, writes, or searches. " +
|
|
48
|
+
`The session-wide budget is ${maxUses} consult${maxUses === 1 ? "" : "s"}, shared across all participants. ` +
|
|
49
|
+
"Consulting is your judgment, never mandatory."
|
|
50
|
+
);
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* Create the session-wide consult budget, shared by every caller's tool
|
|
55
|
+
* handler. Enforced in code by the tool handler, not in the prompt.
|
|
56
|
+
* @param {number} maxUses
|
|
57
|
+
* @returns {{maxUses: number, used: number}}
|
|
58
|
+
*/
|
|
59
|
+
export function createAdvisorBudget(maxUses) {
|
|
60
|
+
return { maxUses, used: 0 };
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Fold the consult guidance into an existing run-specific amendment when
|
|
65
|
+
* the advisor is enabled (budget present); return the amendment unchanged
|
|
66
|
+
* otherwise, so advisor-off prompts stay byte-identical.
|
|
67
|
+
* @param {string|undefined} amend - The existing amendment, if any.
|
|
68
|
+
* @param {{maxUses: number}|null} budget
|
|
69
|
+
* @returns {string|undefined}
|
|
70
|
+
*/
|
|
71
|
+
export function withAdvisorGuidance(amend, budget) {
|
|
72
|
+
if (!budget) return amend;
|
|
73
|
+
return [amend, advisorGuidance(budget.maxUses)].filter(Boolean).join("\n\n");
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/** Consult timeout — generous for a read-a-few-files-and-answer session, and the universal guard in modes with no stop path. */
|
|
77
|
+
export const DEFAULT_CONSULT_TIMEOUT_MS = 300_000;
|
|
78
|
+
|
|
79
|
+
const ADVISOR_ALLOWED_TOOLS = ["Read", "Glob", "Grep"];
|
|
80
|
+
|
|
81
|
+
// Under the harness's always-on bypassPermissions, `allowedTools` alone is
|
|
82
|
+
// not structural — `disallowedTools` is what removes tools from the model's
|
|
83
|
+
// context, the same treatment the lead runners get. The advisor's contract
|
|
84
|
+
// is stricter than the leads' (read-only inspection, nothing else), so the
|
|
85
|
+
// list also removes the write-capable and non-inspection built-ins the
|
|
86
|
+
// lead convention leaves in.
|
|
87
|
+
const ADVISOR_DISALLOWED_TOOLS = [
|
|
88
|
+
"Bash",
|
|
89
|
+
"Write",
|
|
90
|
+
"Edit",
|
|
91
|
+
"NotebookEdit",
|
|
92
|
+
"Agent",
|
|
93
|
+
"Task",
|
|
94
|
+
"TaskOutput",
|
|
95
|
+
"TaskStop",
|
|
96
|
+
"TodoWrite",
|
|
97
|
+
"WebFetch",
|
|
98
|
+
"WebSearch",
|
|
99
|
+
];
|
|
100
|
+
|
|
101
|
+
const devNull = new Writable({
|
|
102
|
+
write(_chunk, _enc, cb) {
|
|
103
|
+
cb();
|
|
104
|
+
},
|
|
105
|
+
});
|
|
106
|
+
|
|
107
|
+
/**
|
|
108
|
+
* Create a per-caller advisor closed over that caller's transcript
|
|
109
|
+
* recorder.
|
|
110
|
+
*
|
|
111
|
+
* @param {object} deps
|
|
112
|
+
* @param {string} deps.model - Advisor model id.
|
|
113
|
+
* @param {string} deps.cwd - The caller's working directory, so read-only inspection sees the caller's files.
|
|
114
|
+
* @param {function} deps.query - SDK query function (injected for testing).
|
|
115
|
+
* @param {{render: () => string}} deps.recorder - The caller's transcript recorder.
|
|
116
|
+
* @param {import("./redaction.js").Redactor} deps.redactor
|
|
117
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} deps.runtime - Clock surface for timeout and duration.
|
|
118
|
+
* @param {function} deps.onLine - Re-emitter for the advisor session's NDJSON lines (tagged `source: "advisor"` by the caller).
|
|
119
|
+
* @param {number} [deps.maxTurns] - Default 5 — single-digit per the spec criterion.
|
|
120
|
+
* @param {number} [deps.timeoutMs] - Default `DEFAULT_CONSULT_TIMEOUT_MS`.
|
|
121
|
+
* @returns {{consult: (question: string) => Promise<{advice?: string, unavailable?: boolean, reason?: string, durationMs: number}>, abort: () => void}}
|
|
122
|
+
*/
|
|
123
|
+
export function createAdvisor({
|
|
124
|
+
model,
|
|
125
|
+
cwd,
|
|
126
|
+
query,
|
|
127
|
+
recorder,
|
|
128
|
+
redactor,
|
|
129
|
+
runtime,
|
|
130
|
+
onLine,
|
|
131
|
+
maxTurns,
|
|
132
|
+
timeoutMs,
|
|
133
|
+
}) {
|
|
134
|
+
if (!model) throw new Error("model is required");
|
|
135
|
+
if (!cwd) throw new Error("cwd is required");
|
|
136
|
+
if (!query) throw new Error("query is required");
|
|
137
|
+
if (!recorder) throw new Error("recorder is required");
|
|
138
|
+
if (!redactor) throw new Error("redactor is required");
|
|
139
|
+
if (!runtime) throw new Error("runtime is required");
|
|
140
|
+
if (!onLine) throw new Error("onLine is required");
|
|
141
|
+
const resolvedMaxTurns = maxTurns ?? 5;
|
|
142
|
+
const resolvedTimeoutMs = timeoutMs ?? DEFAULT_CONSULT_TIMEOUT_MS;
|
|
143
|
+
|
|
144
|
+
/** @type {import("./agent-runner.js").AgentRunner|null} */
|
|
145
|
+
let currentRunner = null;
|
|
146
|
+
|
|
147
|
+
return {
|
|
148
|
+
/**
|
|
149
|
+
* Run one fresh advisor session over the caller's context as it
|
|
150
|
+
* stands plus the question. Never rejects — every failure shape
|
|
151
|
+
* resolves to `{unavailable, reason}` (fail-open).
|
|
152
|
+
* @param {string} question
|
|
153
|
+
*/
|
|
154
|
+
async consult(question) {
|
|
155
|
+
const started = runtime.clock.now();
|
|
156
|
+
const runner = createAgentRunner({
|
|
157
|
+
cwd,
|
|
158
|
+
query,
|
|
159
|
+
output: devNull,
|
|
160
|
+
model,
|
|
161
|
+
maxTurns: resolvedMaxTurns,
|
|
162
|
+
allowedTools: ADVISOR_ALLOWED_TOOLS,
|
|
163
|
+
disallowedTools: ADVISOR_DISALLOWED_TOOLS,
|
|
164
|
+
onLine,
|
|
165
|
+
settingSources: ["project"],
|
|
166
|
+
systemPrompt: composeSystemPrompt({
|
|
167
|
+
role: "agent",
|
|
168
|
+
trailer: ADVISOR_SYSTEM_PROMPT,
|
|
169
|
+
runtime,
|
|
170
|
+
}),
|
|
171
|
+
redactor,
|
|
172
|
+
});
|
|
173
|
+
currentRunner = runner;
|
|
174
|
+
const task = `${recorder.render()}\n\n<consult_question>\n${question}\n</consult_question>`;
|
|
175
|
+
const timer = runtime.clock.setTimeout(
|
|
176
|
+
() => runner.currentAbortController?.abort(),
|
|
177
|
+
resolvedTimeoutMs,
|
|
178
|
+
);
|
|
179
|
+
try {
|
|
180
|
+
const result = await runner.run(task);
|
|
181
|
+
const durationMs = runtime.clock.now() - started;
|
|
182
|
+
if (result.aborted) {
|
|
183
|
+
return {
|
|
184
|
+
unavailable: true,
|
|
185
|
+
reason: "timed out or aborted",
|
|
186
|
+
durationMs,
|
|
187
|
+
};
|
|
188
|
+
}
|
|
189
|
+
if (!result.success) {
|
|
190
|
+
return {
|
|
191
|
+
unavailable: true,
|
|
192
|
+
reason: result.error?.message ?? "advisor session failed",
|
|
193
|
+
durationMs,
|
|
194
|
+
};
|
|
195
|
+
}
|
|
196
|
+
return { advice: result.text, durationMs };
|
|
197
|
+
} catch (err) {
|
|
198
|
+
return {
|
|
199
|
+
unavailable: true,
|
|
200
|
+
reason: err?.message ?? "advisor session failed",
|
|
201
|
+
durationMs: runtime.clock.now() - started,
|
|
202
|
+
};
|
|
203
|
+
} finally {
|
|
204
|
+
runtime.clock.clearTimeout(timer);
|
|
205
|
+
currentRunner = null;
|
|
206
|
+
}
|
|
207
|
+
},
|
|
208
|
+
|
|
209
|
+
/**
|
|
210
|
+
* Abort the in-flight consult, if any. A consult is a blocking tool
|
|
211
|
+
* call, so one caller cannot overlap its own consults; advisors are
|
|
212
|
+
* per-caller, so at most one runner is ever tracked.
|
|
213
|
+
*/
|
|
214
|
+
abort() {
|
|
215
|
+
currentRunner?.currentAbortController?.abort();
|
|
216
|
+
},
|
|
217
|
+
};
|
|
218
|
+
}
|
package/src/agent-runner.js
CHANGED
|
@@ -53,6 +53,7 @@ export class AgentRunner {
|
|
|
53
53
|
* @param {number} [deps.maxTurns] - Maximum agentic turns; 0 means unlimited
|
|
54
54
|
* @param {string[]} [deps.allowedTools] - Tools the agent may use
|
|
55
55
|
* @param {function} [deps.onLine] - Callback invoked with each NDJSON line as it's produced
|
|
56
|
+
* @param {function} [deps.onPrompt] - Callback invoked with the effective (amend-applied) prompt of each run/resume
|
|
56
57
|
* @param {string[]} [deps.settingSources] - SDK setting sources (e.g. ['project'] to load CLAUDE.md)
|
|
57
58
|
* @param {string|object} [deps.systemPrompt] - SDK system prompt (string replaces default; {type:'preset', preset:'claude_code', append} appends)
|
|
58
59
|
* @param {string[]} [deps.disallowedTools] - Tools to explicitly remove from the model's context
|
|
@@ -80,6 +81,9 @@ export class AgentRunner {
|
|
|
80
81
|
this.maxTurns = deps.maxTurns ?? 50;
|
|
81
82
|
this.allowedTools = deps.allowedTools ?? DEFAULT_ALLOWED_TOOLS;
|
|
82
83
|
this.onLine = deps.onLine ?? null;
|
|
84
|
+
// Optional; read only through a truthy guard in run()/resume(), so an
|
|
85
|
+
// absent value stays undefined rather than needing a `?? null` default.
|
|
86
|
+
this.onPrompt = deps.onPrompt;
|
|
83
87
|
this.settingSources = deps.settingSources ?? [];
|
|
84
88
|
this.systemPrompt = deps.systemPrompt ?? null;
|
|
85
89
|
this.disallowedTools = deps.disallowedTools ?? [];
|
|
@@ -106,6 +110,7 @@ export class AgentRunner {
|
|
|
106
110
|
? `${task}\n\n${this.taskAmend}`
|
|
107
111
|
: this.taskAmend
|
|
108
112
|
: task;
|
|
113
|
+
if (this.onPrompt) this.onPrompt(effectiveTask);
|
|
109
114
|
try {
|
|
110
115
|
const iterator = this.query({
|
|
111
116
|
prompt: effectiveTask,
|
|
@@ -125,6 +130,7 @@ export class AgentRunner {
|
|
|
125
130
|
async resume(prompt) {
|
|
126
131
|
const abortController = new AbortController();
|
|
127
132
|
this.currentAbortController = abortController;
|
|
133
|
+
if (this.onPrompt) this.onPrompt(prompt);
|
|
128
134
|
try {
|
|
129
135
|
const iterator = this.query({
|
|
130
136
|
prompt,
|
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Grading derivation — the sole home of the check-row arithmetic.
|
|
3
|
+
*
|
|
4
|
+
* Check rows are the single authoritative grading channel. Every row is a
|
|
5
|
+
* check by default; a row declares its role with its own fields, checked in
|
|
6
|
+
* order:
|
|
7
|
+
*
|
|
8
|
+
* 1. Gate — `gate` is exactly `true`, `pass` is boolean, and no
|
|
9
|
+
* `weight` key is present. Any failing gate → `gatesPass`
|
|
10
|
+
* false.
|
|
11
|
+
* 2. Diagnostic — no `gate` key and `weight` is exactly `0`. Free-form;
|
|
12
|
+
* never graded.
|
|
13
|
+
* 3. Scored — no `gate` key, boolean `pass`, `weight` absent (defaults
|
|
14
|
+
* to 1) or finite > 0.
|
|
15
|
+
* 4. Malformed — everything else: any `gate`+`weight` co-occurrence (a
|
|
16
|
+
* stray weight must never silently disarm a gate), a
|
|
17
|
+
* non-boolean `gate`, a missing or non-boolean `pass` on a
|
|
18
|
+
* graded row, an invalid `weight`, an fd-3 line that failed
|
|
19
|
+
* to parse, a non-object row. Counts as a **failing scored
|
|
20
|
+
* check** — dropping a defect could mint full marks;
|
|
21
|
+
* failing the whole run would zero completed work.
|
|
22
|
+
*
|
|
23
|
+
* The producers' `source` stamp is display metadata, never a grading input.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* @typedef {object} GradeResult
|
|
28
|
+
* @property {"pass" | "fail"} verdict - `healthy ∧ gatesPass ∧ fullMarks`.
|
|
29
|
+
* @property {boolean} gatesPass - Every gate row passes (vacuously true).
|
|
30
|
+
* @property {number | null} score - Weighted fraction of passing scored
|
|
31
|
+
* checks; `null` when the cell has zero scored checks (binary task).
|
|
32
|
+
* @property {boolean} fullMarks - Integer count predicate: no malformed rows
|
|
33
|
+
* and every scored check passes. Never a float comparison, so fractional
|
|
34
|
+
* weights carry no equality hazard. Vacuously true with zero scored checks.
|
|
35
|
+
* @property {number} malformed - Malformed row count.
|
|
36
|
+
*/
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Grade the merged check rows against grader health.
|
|
40
|
+
*
|
|
41
|
+
* `healthy` is the completion signal a crashed grader cannot fake: when it is
|
|
42
|
+
* false the verdict is `fail` whatever the rows say, so a hook that dies
|
|
43
|
+
* after emitting passing rows can never mint marks.
|
|
44
|
+
* @param {unknown[]} details - Merged check rows from both producers.
|
|
45
|
+
* @param {boolean} healthy - Invariants exited 0 AND the hidden-test engine
|
|
46
|
+
* did not throw.
|
|
47
|
+
* @returns {GradeResult}
|
|
48
|
+
*/
|
|
49
|
+
export function gradeChecks(details, healthy) {
|
|
50
|
+
const tally = {
|
|
51
|
+
gatesPass: true,
|
|
52
|
+
malformed: 0,
|
|
53
|
+
scored: 0,
|
|
54
|
+
passing: 0,
|
|
55
|
+
weightAll: 0,
|
|
56
|
+
weightPassing: 0,
|
|
57
|
+
};
|
|
58
|
+
for (const row of details) tallyRow(tally, row);
|
|
59
|
+
|
|
60
|
+
const score =
|
|
61
|
+
tally.scored + tally.malformed === 0
|
|
62
|
+
? null
|
|
63
|
+
: tally.weightPassing / tally.weightAll;
|
|
64
|
+
const fullMarks = tally.malformed === 0 && tally.passing === tally.scored;
|
|
65
|
+
const verdict = healthy && tally.gatesPass && fullMarks ? "pass" : "fail";
|
|
66
|
+
return {
|
|
67
|
+
verdict,
|
|
68
|
+
gatesPass: tally.gatesPass,
|
|
69
|
+
score,
|
|
70
|
+
fullMarks,
|
|
71
|
+
malformed: tally.malformed,
|
|
72
|
+
};
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* Fold one row into the running tally per its classified role.
|
|
77
|
+
* @param {{gatesPass: boolean, malformed: number, scored: number, passing: number, weightAll: number, weightPassing: number}} tally
|
|
78
|
+
* @param {unknown} row
|
|
79
|
+
*/
|
|
80
|
+
function tallyRow(tally, row) {
|
|
81
|
+
const role = classifyRow(row);
|
|
82
|
+
if (role === "gate") {
|
|
83
|
+
if (!row.pass) tally.gatesPass = false;
|
|
84
|
+
} else if (role === "scored") {
|
|
85
|
+
const weight = row.weight ?? 1;
|
|
86
|
+
tally.scored++;
|
|
87
|
+
tally.weightAll += weight;
|
|
88
|
+
if (row.pass) {
|
|
89
|
+
tally.passing++;
|
|
90
|
+
tally.weightPassing += weight;
|
|
91
|
+
}
|
|
92
|
+
} else if (role === "malformed") {
|
|
93
|
+
tally.malformed++;
|
|
94
|
+
tally.weightAll += malformedWeight(row);
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* Run both check-row producers and grade the merged rows — the one
|
|
100
|
+
* composition shared by the runner and the `grade` subcommand. An engine
|
|
101
|
+
* throw is grader fault: its message lands on the returned `engineError`
|
|
102
|
+
* and health fails, so a crashed grader can never mint marks from rows it
|
|
103
|
+
* happened to emit first.
|
|
104
|
+
* @param {import("./task-family.js").Task} task
|
|
105
|
+
* @param {{cwd: string, port: number, runDir: string, familyDir?: string|null}} ctx
|
|
106
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
107
|
+
* @param {{runInvariants: Function, runHiddenTests: Function}} producers -
|
|
108
|
+
* The two producer functions (real implementations or test seams).
|
|
109
|
+
* @returns {Promise<{invariants: object, hiddenRows: object[], engineError: Error|null, rows: unknown[], healthy: boolean, grade: object}>}
|
|
110
|
+
*/
|
|
111
|
+
export async function runProducersAndGrade(task, ctx, runtime, producers) {
|
|
112
|
+
const invariants = await producers.runInvariants(task, ctx, runtime);
|
|
113
|
+
let hiddenRows = [];
|
|
114
|
+
let engineError = null;
|
|
115
|
+
try {
|
|
116
|
+
const hidden = await producers.runHiddenTests(task, ctx, runtime);
|
|
117
|
+
hiddenRows = hidden.details;
|
|
118
|
+
} catch (e) {
|
|
119
|
+
engineError = e;
|
|
120
|
+
}
|
|
121
|
+
const rows = mergeRows(invariants.details, hiddenRows);
|
|
122
|
+
const healthy = invariants.exitCode === 0 && !engineError;
|
|
123
|
+
const grade = normalizeGrade(gradeChecks(rows, healthy));
|
|
124
|
+
return { invariants, hiddenRows, engineError, rows, healthy, grade };
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* Merge the two producers' rows (invariants first) and stamp each row's
|
|
129
|
+
* provenance. The stamp is display metadata, never a grading input, and
|
|
130
|
+
* non-object rows (malformed by contract) pass through verbatim.
|
|
131
|
+
* @param {unknown[]} invariantsDetails
|
|
132
|
+
* @param {unknown[]} hiddenDetails
|
|
133
|
+
* @returns {unknown[]}
|
|
134
|
+
*/
|
|
135
|
+
export function mergeRows(invariantsDetails, hiddenDetails) {
|
|
136
|
+
return [
|
|
137
|
+
...invariantsDetails.map((row) => stampSource(row, "invariants")),
|
|
138
|
+
...hiddenDetails.map((row) => stampSource(row, "tests")),
|
|
139
|
+
];
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
function stampSource(row, source) {
|
|
143
|
+
if (row === null || typeof row !== "object" || Array.isArray(row)) {
|
|
144
|
+
return row;
|
|
145
|
+
}
|
|
146
|
+
return { ...row, source };
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Project the raw `gradeChecks` return onto the record schema: `fullMarks`
|
|
151
|
+
* is derivable and dropped, `score` is omitted on binary tasks (`null`),
|
|
152
|
+
* `malformed` is omitted when clean.
|
|
153
|
+
* @param {GradeResult} raw
|
|
154
|
+
* @returns {{verdict: "pass"|"fail", gatesPass: boolean, score?: number, malformed?: number}}
|
|
155
|
+
*/
|
|
156
|
+
export function normalizeGrade({ verdict, gatesPass, score, malformed }) {
|
|
157
|
+
return {
|
|
158
|
+
verdict,
|
|
159
|
+
gatesPass,
|
|
160
|
+
...(score !== null && { score }),
|
|
161
|
+
...(malformed > 0 && { malformed }),
|
|
162
|
+
};
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
/**
|
|
166
|
+
* Classify one row per the role order in the module contract.
|
|
167
|
+
* @param {unknown} row
|
|
168
|
+
* @returns {"gate" | "diagnostic" | "scored" | "malformed"}
|
|
169
|
+
*/
|
|
170
|
+
function classifyRow(row) {
|
|
171
|
+
if (row === null || typeof row !== "object" || Array.isArray(row)) {
|
|
172
|
+
return "malformed";
|
|
173
|
+
}
|
|
174
|
+
if ("gate" in row) return classifyGateRow(row);
|
|
175
|
+
if ("weight" in row) return classifyWeightedRow(row);
|
|
176
|
+
return typeof row.pass === "boolean" ? "scored" : "malformed";
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
/**
|
|
180
|
+
* A row carrying a `gate` key: valid only as `gate: true` with a boolean
|
|
181
|
+
* `pass` and no `weight` key — any co-occurring weight is malformed so a
|
|
182
|
+
* stray weight can never silently disarm a gate.
|
|
183
|
+
* @param {object} row
|
|
184
|
+
* @returns {"gate" | "malformed"}
|
|
185
|
+
*/
|
|
186
|
+
function classifyGateRow(row) {
|
|
187
|
+
if ("weight" in row) return "malformed";
|
|
188
|
+
return row.gate === true && typeof row.pass === "boolean"
|
|
189
|
+
? "gate"
|
|
190
|
+
: "malformed";
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/**
|
|
194
|
+
* A gate-less row carrying a `weight` key: exactly 0 is a diagnostic, a
|
|
195
|
+
* finite positive weight with a boolean `pass` is scored, anything else is
|
|
196
|
+
* malformed.
|
|
197
|
+
* @param {object} row
|
|
198
|
+
* @returns {"diagnostic" | "scored" | "malformed"}
|
|
199
|
+
*/
|
|
200
|
+
function classifyWeightedRow(row) {
|
|
201
|
+
if (row.weight === 0) return "diagnostic";
|
|
202
|
+
return isValidWeight(row.weight) && typeof row.pass === "boolean"
|
|
203
|
+
? "scored"
|
|
204
|
+
: "malformed";
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
/**
|
|
208
|
+
* A malformed row fails at its own weight when it carries a valid positive
|
|
209
|
+
* one, else at unit weight 1.
|
|
210
|
+
* @param {unknown} row
|
|
211
|
+
* @returns {number}
|
|
212
|
+
*/
|
|
213
|
+
function malformedWeight(row) {
|
|
214
|
+
if (row !== null && typeof row === "object" && isValidWeight(row.weight)) {
|
|
215
|
+
return row.weight;
|
|
216
|
+
}
|
|
217
|
+
return 1;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
function isValidWeight(w) {
|
|
221
|
+
return typeof w === "number" && Number.isFinite(w) && w > 0;
|
|
222
|
+
}
|