@gaunt-sloth/batch 2.0.0-beta.8 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/bin.d.ts +2 -2
- package/dist/bin.js +2 -2
- package/dist/classificationReport.d.ts +2 -2
- package/dist/classificationReport.js +2 -2
- package/dist/classificationTypes.d.ts +7 -7
- package/dist/classificationTypes.js +1 -1
- package/dist/evalCompare.d.ts +128 -1
- package/dist/evalCompare.js +253 -2
- package/dist/evalCompare.js.map +1 -1
- package/dist/evalOutput.d.ts +1 -1
- package/dist/evalOutput.js +1 -1
- package/dist/evalRunner.d.ts +15 -3
- package/dist/evalRunner.js +82 -5
- package/dist/evalRunner.js.map +1 -1
- package/dist/evalSuite.d.ts +5 -1
- package/dist/evalSuite.js +218 -7
- package/dist/evalSuite.js.map +1 -1
- package/dist/evalTypes.d.ts +79 -18
- package/dist/evalTypes.js +2 -2
- package/dist/evalTypes.js.map +1 -1
- package/dist/index.d.ts +8 -2
- package/dist/index.js +9 -1
- package/dist/index.js.map +1 -1
- package/dist/judge.d.ts +2 -2
- package/dist/judge.js +10 -14
- package/dist/judge.js.map +1 -1
- package/dist/output.d.ts +1 -1
- package/dist/output.js +1 -1
- package/dist/parseOver.d.ts +1 -1
- package/dist/parseOver.js +1 -1
- package/dist/pipelineCli.d.ts +3 -3
- package/dist/pipelineCli.js +3 -3
- package/dist/raterPromptArm.d.ts +140 -0
- package/dist/raterPromptArm.js +306 -0
- package/dist/raterPromptArm.js.map +1 -0
- package/dist/raterTarget.d.ts +34 -12
- package/dist/raterTarget.js +124 -33
- package/dist/raterTarget.js.map +1 -1
- package/dist/reporters/registry.d.ts +1 -1
- package/dist/reporters/registry.js +1 -1
- package/dist/reporters/registry.js.map +1 -1
- package/dist/reporters/reporterTypes.d.ts +3 -3
- package/dist/reporters/textReporter.js +16 -0
- package/dist/reporters/textReporter.js.map +1 -1
- package/dist/toolCoverage.d.ts +244 -0
- package/dist/toolCoverage.js +414 -0
- package/dist/toolCoverage.js.map +1 -0
- package/dist/toolCoverageRender.d.ts +31 -0
- package/dist/toolCoverageRender.js +71 -0
- package/dist/toolCoverageRender.js.map +1 -0
- package/dist/toolResultChecks.d.ts +8 -2
- package/dist/toolResultChecks.js +45 -3
- package/dist/toolResultChecks.js.map +1 -1
- package/dist/types.d.ts +38 -3
- package/dist/types.js +0 -9
- package/dist/types.js.map +1 -1
- package/dist/workflow/runWorkflow.d.ts +4 -4
- package/dist/workflow/runWorkflow.js +3 -3
- package/package.json +9 -8
package/dist/evalTypes.d.ts
CHANGED
|
@@ -1,13 +1,15 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* @packageDocumentation
|
|
3
3
|
* BATCH-2 — the shapes for `gth eval`: a parsed suite/case, deterministic-check results, the
|
|
4
|
-
* judge's verdict, and per-case/suite outcomes. Deliberately separate from
|
|
4
|
+
* judge's verdict, and per-case/suite outcomes. Deliberately separate from `types.js`
|
|
5
5
|
* (BATCH-1's cell/outcome shapes), which documents itself as scoped to "cells and outcomes" only
|
|
6
6
|
* — eval's shapes layer on top of (not into) that file.
|
|
7
7
|
*/
|
|
8
8
|
import type { ApprovalRung } from '@gaunt-sloth/core/config/shell-policy.js';
|
|
9
9
|
import type { PreflightFloorKind } from '@gaunt-sloth/core/core/shell/raterVocabulary.js';
|
|
10
10
|
import type { ToolResultRecord } from '#src/types.js';
|
|
11
|
+
import type { RaterPromptArm } from '#src/raterPromptArm.js';
|
|
12
|
+
import type { AdvertisedToolInventory, ToolCoverageReport, ToolCoverageSpec } from '#src/toolCoverage.js';
|
|
11
13
|
import type { EvalCaseClassification, EvalClassificationReport, EvalClassificationSpec, EvalMetricSpec } from '#src/classificationTypes.js';
|
|
12
14
|
/** The 0-10 judge scale's default pass threshold, matching `review`'s own (unexported)
|
|
13
15
|
* `DEFAULT_PASS_THRESHOLD` in `packages/review/src/middleware/reviewRateMiddleware.ts` — same
|
|
@@ -71,7 +73,7 @@ declare const PREFLIGHT_MECHANISM_OF: {
|
|
|
71
73
|
* The floor is spelled here because it is not a preflight — it is a lexical scan core keeps no list
|
|
72
74
|
* of, consulted by the approvals gate before any rating (every rung but `bypass`) and again by the
|
|
73
75
|
* shell tool at execution time (every rung). The preflights are NOT spelled: they come from
|
|
74
|
-
*
|
|
76
|
+
* `PREFLIGHT_MECHANISM_OF`, which is core's own `PREFLIGHT_FLOOR_KINDS` with our `forced_by`
|
|
75
77
|
* spelling attached.
|
|
76
78
|
*
|
|
77
79
|
* ## Why a case asserts a mechanism at all (the I1 finding)
|
|
@@ -158,7 +160,7 @@ export declare const PREFLIGHT_MECHANISMS: ("open-world-preflight" | "script-env
|
|
|
158
160
|
export type PreflightMechanism = (typeof PREFLIGHT_MECHANISM_OF)[PreflightFloorKind];
|
|
159
161
|
/**
|
|
160
162
|
* The `forced_by` spelling of one of core's preflight arms — the typed lookup into
|
|
161
|
-
*
|
|
163
|
+
* `PREFLIGHT_MECHANISM_OF`, which stays private so this is the only way in.
|
|
162
164
|
*
|
|
163
165
|
* TOTAL over core's `PreflightFloorKind` and therefore never `undefined`, which is the point: a
|
|
164
166
|
* caller cannot be handed an arm it has no name for, because such an arm would not have compiled.
|
|
@@ -178,7 +180,7 @@ export declare function mechanismNeedsPermissiveRating(mechanism: ForcedByMechan
|
|
|
178
180
|
export interface GthAgentTarget {
|
|
179
181
|
type: 'gth-agent';
|
|
180
182
|
/** Suite-level profile hint. Only `undefined`/`'default'` is accepted (see
|
|
181
|
-
*
|
|
183
|
+
* `evalSuite.js`'s `parseEvalSuite`) — per-case/per-identity profile switching is
|
|
182
184
|
* `identities` scope, not this. */
|
|
183
185
|
profile?: string;
|
|
184
186
|
}
|
|
@@ -240,7 +242,7 @@ export interface AgUiAgentTarget {
|
|
|
240
242
|
* §8 hardline floor — decides some commands outright, with no model in the loop.
|
|
241
243
|
*
|
|
242
244
|
* Unlike the agent targets this one is NOT driven by a `RunCellFn`: it supplies the
|
|
243
|
-
* {@link RunClassifyFn} seam instead (`buildRaterClassifier` in
|
|
245
|
+
* {@link RunClassifyFn} seam instead (`buildRaterClassifier` in `raterTarget.js`), which
|
|
244
246
|
* the command passes to `runEvalSuite` as `options.classify`.
|
|
245
247
|
*/
|
|
246
248
|
export interface RaterTarget {
|
|
@@ -252,7 +254,7 @@ export interface RaterTarget {
|
|
|
252
254
|
* It is the suite's declaration, not the last word: a run whose config declares `approvals`
|
|
253
255
|
* overrides it, which is what makes the `rung × model` sweep work through the existing `config:`
|
|
254
256
|
* axis (a sweep cell cannot reach a `target` field). The override is announced, never silent —
|
|
255
|
-
* see
|
|
257
|
+
* see `raterTarget.js`.
|
|
256
258
|
*/
|
|
257
259
|
rung: ApprovalRung;
|
|
258
260
|
}
|
|
@@ -263,7 +265,7 @@ export interface RaterTarget {
|
|
|
263
265
|
* target), so the target only changes which runner the command builds. */
|
|
264
266
|
export type EvalTarget = GthAgentTarget | AdkAgentTarget | AgUiAgentTarget | RaterTarget;
|
|
265
267
|
/** One `json_path` assertion (BATCH-10): resolve `path` against the answer-parsed-as-JSON and check
|
|
266
|
-
* it. Exactly one of `equals`/`contains` is set (enforced in
|
|
268
|
+
* it. Exactly one of `equals`/`contains` is set (enforced in `evalSuite.js`'s parse):
|
|
267
269
|
* - `equals` — the resolved value must deep-equal this (any JSON value, incl. `null`).
|
|
268
270
|
* - `contains` — the resolved value must be a string containing this substring. */
|
|
269
271
|
export interface JsonPathCheck {
|
|
@@ -275,7 +277,7 @@ export interface JsonPathCheck {
|
|
|
275
277
|
* One `tool_result_json_path` assertion (BATCH-21): select tool results by name `tool` (exact or
|
|
276
278
|
* glob, the same matcher `must_call` uses), parse each matching result's payload as JSON, and
|
|
277
279
|
* resolve `path` against it (the same minimal dot/`[index]` path `json_path` uses). At most one of
|
|
278
|
-
* `equals`/`contains` may be set (enforced in
|
|
280
|
+
* `equals`/`contains` may be set (enforced in `evalSuite.js`'s parse):
|
|
279
281
|
* - neither — pure existence check: the path must resolve in a matching result's payload;
|
|
280
282
|
* - `equals` — the resolved value must deep-equal this (any JSON value, incl. `null`);
|
|
281
283
|
* - `contains` — the resolved value must be a string containing this substring.
|
|
@@ -359,6 +361,31 @@ export interface EvalExpectation {
|
|
|
359
361
|
* touching the first.
|
|
360
362
|
*/
|
|
361
363
|
expectAction?: string;
|
|
364
|
+
/**
|
|
365
|
+
* BATCH-45 — **assert that a MODEL actually rendered a verdict for this round**, i.e. that
|
|
366
|
+
* {@link ClassifyOutcome.modelLabel} is present. `true` is the only legal value; `undefined` =
|
|
367
|
+
* this block makes no such claim. `rater` target only.
|
|
368
|
+
*
|
|
369
|
+
* It is NOT an assertion about the rung, and NOT an assertion about WHAT the model said. Those
|
|
370
|
+
* are {@link expectLabel} and {@link expectAction}, and the whole point of this key is that it
|
|
371
|
+
* says neither: a case can pin "a rater ruled here" without predicting which verdict three
|
|
372
|
+
* different raters would render, which is a prediction no measurement backs.
|
|
373
|
+
*
|
|
374
|
+
* **Why the action column cannot do this job.** When the gate never obtains a rating (timeout,
|
|
375
|
+
* throw, unparseable answer) it fails closed, and since EXT-171 that ESCALATES rather than
|
|
376
|
+
* negotiating. A genuine `catastrophic` verdict escalates too. So `expect_action: escalate` is
|
|
377
|
+
* satisfied identically by a rater that judged the command unnegotiable and by a rater that
|
|
378
|
+
* never answered at all — and a case asserting only that passes on a run where nothing was
|
|
379
|
+
* measured. BATCH-45 found ten such cells live in this repo's own approvals corpus.
|
|
380
|
+
*
|
|
381
|
+
* **Why `modelLabel` and not the rationale text.** `modelLabel` is suppressed from the call's own
|
|
382
|
+
* capture (`rating.failClosed`), never from a reading of the verdict's prose. Core's
|
|
383
|
+
* `isFailClosed` answers a question about the reason TEXT, and the rating prompt tells a rater to
|
|
384
|
+
* answer `destructive` and say it could not assess a command it is unsure of — so that predicate
|
|
385
|
+
* scores an obedient judgement as a gate failure. EXT-171 and `raterHealth` both say never to key
|
|
386
|
+
* on verdict text.
|
|
387
|
+
*/
|
|
388
|
+
expectRated?: boolean;
|
|
362
389
|
}
|
|
363
390
|
/**
|
|
364
391
|
* One conversational turn (BATCH-12): the `user` message to send, and the {@link EvalExpectation}
|
|
@@ -378,7 +405,7 @@ export interface EvalTurn {
|
|
|
378
405
|
*
|
|
379
406
|
* **Carrying it is not the same as showing it to the rater.** §5.1 admits the justification from
|
|
380
407
|
* round 2 onward, so the round-1 rating withholds it; the target hands the whole context to core's
|
|
381
|
-
* {@link
|
|
408
|
+
* {@link @gaunt-sloth/core!core/shell/negotiation.ShellNegotiationState | ShellNegotiationState}, which is
|
|
382
409
|
* the single implementation of that rule.
|
|
383
410
|
*/
|
|
384
411
|
justification?: string;
|
|
@@ -410,7 +437,7 @@ export interface EvalTurn {
|
|
|
410
437
|
}
|
|
411
438
|
/**
|
|
412
439
|
* One turn's raw run outcome inside a multi-turn conversation (BATCH-12 Task 2). Structurally the
|
|
413
|
-
* per-turn analogue of BATCH-1's {@link
|
|
440
|
+
* per-turn analogue of BATCH-1's {@link "types.js"!CellRunOutcome | CellRunOutcome}: the turn's answer plus the
|
|
414
441
|
* tools/tokens captured FOR THAT TURN — a per-invoke GS2-16 delta (each `processMessages` call
|
|
415
442
|
* resets the tally), NOT the cumulative conversation total. `ok:false` with `error` set means that
|
|
416
443
|
* turn's SUT invocation failed (no answer to grade). The conversational runner returns one of these
|
|
@@ -425,12 +452,18 @@ export interface TurnRunOutcome {
|
|
|
425
452
|
/** BATCH-21 — this turn's per-tool-call result records (parallel to {@link tools}; a per-turn
|
|
426
453
|
* delta like everything else here). Only the in-process `gth-agent` runner populates it. */
|
|
427
454
|
toolResults?: ToolResultRecord[];
|
|
455
|
+
/**
|
|
456
|
+
* BATCH-32 — the advertised-tool inventory (the coverage denominator). **The one field here that
|
|
457
|
+
* is NOT a per-turn delta**: a conversation builds its agent once, so the inventory is fixed for
|
|
458
|
+
* the whole conversation and every turn repeats it.
|
|
459
|
+
*/
|
|
460
|
+
advertisedTools?: AdvertisedToolInventory;
|
|
428
461
|
error?: string;
|
|
429
462
|
}
|
|
430
463
|
/**
|
|
431
464
|
* Injectable "run one whole scripted conversation" function (BATCH-12 Task 2) — the multi-turn
|
|
432
|
-
* analogue of BATCH-1's {@link
|
|
433
|
-
*
|
|
465
|
+
* analogue of BATCH-1's {@link "types.js"!RunCellFn | RunCellFn}, and the seam that lets
|
|
466
|
+
* `evalRunner.js`'s multi-turn path be unit tested without any real LLM/MCP. Given the
|
|
434
467
|
* ordered user messages of ONE (case × identity) conversation, it builds the agent + resolves tools
|
|
435
468
|
* ONCE, runs each turn against the accumulated message history, and returns one {@link TurnRunOutcome}
|
|
436
469
|
* per turn (per-turn answer + per-turn tool delta), cleaning up once. The production wiring
|
|
@@ -447,7 +480,7 @@ export type RunConversationFn = (userMessages: string[]) => Promise<TurnRunOutco
|
|
|
447
480
|
* a round-1 rating, and a reset makes a later round a round-1 rating again — so the target hands the
|
|
448
481
|
* accumulated state to `ShellNegotiationState.contextFor` rather than deciding here. A shape that
|
|
449
482
|
* carried "what the rater sees" would be this package holding a second opinion about §5.1, which is
|
|
450
|
-
* the one thing
|
|
483
|
+
* the one thing `raterTarget.js` exists not to do.
|
|
451
484
|
*/
|
|
452
485
|
export interface ClassifyRound {
|
|
453
486
|
/** The command the agent proposes in this round. */
|
|
@@ -465,7 +498,7 @@ export interface ClassifyRound {
|
|
|
465
498
|
* ## The Half-B seam, now filled
|
|
466
499
|
*
|
|
467
500
|
* Half A shipped this shape, the runner plumbing and the tests (against an injected fake); Half B
|
|
468
|
-
* supplies the one function — `buildRaterClassifier` in
|
|
501
|
+
* supplies the one function — `buildRaterClassifier` in `raterTarget.js`, the
|
|
469
502
|
* {@link RaterTarget}'s implementation, which drives the approvals rating prompt + decision mapping
|
|
470
503
|
* at a declared rung. The shape did not have to move to fit it.
|
|
471
504
|
*
|
|
@@ -641,7 +674,7 @@ export interface EvalCase {
|
|
|
641
674
|
*/
|
|
642
675
|
modelFree: boolean;
|
|
643
676
|
}
|
|
644
|
-
/** A fully parsed and validated suite — see
|
|
677
|
+
/** A fully parsed and validated suite — see `evalSuite.js`'s `parseEvalSuite`. */
|
|
645
678
|
export interface EvalSuite {
|
|
646
679
|
target: EvalTarget;
|
|
647
680
|
/**
|
|
@@ -669,11 +702,21 @@ export interface EvalSuite {
|
|
|
669
702
|
* label enum has nothing to read).
|
|
670
703
|
*/
|
|
671
704
|
metrics: EvalMetricSpec[];
|
|
705
|
+
/**
|
|
706
|
+
* BATCH-32 — the suite's `tool_coverage:` declaration: waivers, an optional floor, and required
|
|
707
|
+
* tools. Absent = coverage is still REPORTED (that is the whole point of the node: a run that
|
|
708
|
+
* exercises 3 of 41 tools must say so without being asked), but nothing gates on it.
|
|
709
|
+
*
|
|
710
|
+
* A top-level key of its own rather than a `metrics:` entry: a metric is a predicate over LABELLED
|
|
711
|
+
* cases and its machinery assumes a `classification:` block, whereas coverage applies to suites
|
|
712
|
+
* with no labels at all — most of them.
|
|
713
|
+
*/
|
|
714
|
+
toolCoverage?: ToolCoverageSpec;
|
|
672
715
|
/**
|
|
673
716
|
* BATCH-25 — the config sweep: named cells the WHOLE suite is run once per, so one corpus
|
|
674
717
|
* produces one comparison table instead of N unrelated runs. Absent = a single run.
|
|
675
718
|
*
|
|
676
|
-
* Consumed by the CLI, not by {@link
|
|
719
|
+
* Consumed by the CLI, not by {@link "evalRunner.js"!runEvalSuite | runEvalSuite}: a sweep is a run-level
|
|
677
720
|
* concept (same suite, different config) exactly like BATCH-19's multi-suite loop, so the runner
|
|
678
721
|
* stays about grading and #405's identity matrix is untouched.
|
|
679
722
|
*/
|
|
@@ -698,6 +741,17 @@ export interface EvalSweepValue {
|
|
|
698
741
|
model?: string;
|
|
699
742
|
/** Plain-data config overrides deep-merged onto the resolved config (never `llm`). */
|
|
700
743
|
config?: Record<string, unknown>;
|
|
744
|
+
/**
|
|
745
|
+
* [[BATCH-31]] — the cell's PROMPT ARM: which of gth's own preflight notes this cell's ratings go
|
|
746
|
+
* out without, so a note-on / note-off A/B is one suite file rather than two. `rater` targets
|
|
747
|
+
* only; `{ omit: [] }` is the baseline arm and is required rather than implied.
|
|
748
|
+
*
|
|
749
|
+
* **A third override kind, not a third spelling of `config:`.** It is deliberately not a config
|
|
750
|
+
* key and never merges into one: a config key is exactly what a session could also be given, and
|
|
751
|
+
* an omission a session can reach is a switch that suppresses safety context in the live approvals
|
|
752
|
+
* gate. See `raterPromptArm.ts` for the full argument and for how that is enforced.
|
|
753
|
+
*/
|
|
754
|
+
notes?: RaterPromptArm;
|
|
701
755
|
}
|
|
702
756
|
/**
|
|
703
757
|
* BATCH-25 — the sweep: named axes whose CARTESIAN PRODUCT is the set of runs. The approvals
|
|
@@ -719,7 +773,7 @@ export interface DeterministicCheckResult {
|
|
|
719
773
|
failures: string[];
|
|
720
774
|
}
|
|
721
775
|
/** The judge's structured verdict on one case's answer — matches `review`'s `RateSchema` shape
|
|
722
|
-
* (0-10 `rate` + a reason string) for UX consistency, see
|
|
776
|
+
* (0-10 `rate` + a reason string) for UX consistency, see `judge.js`. */
|
|
723
777
|
export interface JudgeVerdict {
|
|
724
778
|
rate: number;
|
|
725
779
|
reason: string;
|
|
@@ -737,7 +791,7 @@ export interface JudgeOutcome {
|
|
|
737
791
|
error?: string;
|
|
738
792
|
}
|
|
739
793
|
/** Injectable "grade one answer against one rubric" function — the seam that lets
|
|
740
|
-
*
|
|
794
|
+
* `evalRunner.js`'s `runEvalSuite` be fully unit tested without any real LLM call, mirror
|
|
741
795
|
* of BATCH-1's `RunCellFn`. The production wiring (`evalCommand.ts`) adapts `judgeEvalCase` to
|
|
742
796
|
* this shape; tests inject a fake that resolves/fails as needed. */
|
|
743
797
|
export type JudgeFn = (answer: string, rubric: string) => Promise<JudgeOutcome>;
|
|
@@ -830,5 +884,12 @@ export interface EvalSuiteSummary {
|
|
|
830
884
|
/** BATCH-25 — the confusion matrices + declared metrics. Omitted entirely for a suite with no
|
|
831
885
|
* `classification:` block, so a pre-BATCH-25 `results.json` is byte-for-byte unchanged. */
|
|
832
886
|
classification?: EvalClassificationReport;
|
|
887
|
+
/**
|
|
888
|
+
* BATCH-32 — which of the agent's advertised tools this suite exercised. Omitted entirely when no
|
|
889
|
+
* cell reported an inventory: an external target, or a run whose SUT never initialised. That
|
|
890
|
+
* absence is the honest answer, and it is why this is optional rather than a zeroed block — a
|
|
891
|
+
* `0/0` coverage figure looks like a measurement and is not one.
|
|
892
|
+
*/
|
|
893
|
+
toolCoverage?: ToolCoverageReport;
|
|
833
894
|
}
|
|
834
895
|
export {};
|
package/dist/evalTypes.js
CHANGED
|
@@ -60,7 +60,7 @@ const PREFLIGHT_MECHANISM_OF = {
|
|
|
60
60
|
* The floor is spelled here because it is not a preflight — it is a lexical scan core keeps no list
|
|
61
61
|
* of, consulted by the approvals gate before any rating (every rung but `bypass`) and again by the
|
|
62
62
|
* shell tool at execution time (every rung). The preflights are NOT spelled: they come from
|
|
63
|
-
*
|
|
63
|
+
* `PREFLIGHT_MECHANISM_OF`, which is core's own `PREFLIGHT_FLOOR_KINDS` with our `forced_by`
|
|
64
64
|
* spelling attached.
|
|
65
65
|
*
|
|
66
66
|
* ## Why a case asserts a mechanism at all (the I1 finding)
|
|
@@ -132,7 +132,7 @@ export const FORCED_BY_ASSERTIONS = {
|
|
|
132
132
|
export const PREFLIGHT_MECHANISMS = Object.values(PREFLIGHT_MECHANISM_OF);
|
|
133
133
|
/**
|
|
134
134
|
* The `forced_by` spelling of one of core's preflight arms — the typed lookup into
|
|
135
|
-
*
|
|
135
|
+
* `PREFLIGHT_MECHANISM_OF`, which stays private so this is the only way in.
|
|
136
136
|
*
|
|
137
137
|
* TOTAL over core's `PreflightFloorKind` and therefore never `undefined`, which is the point: a
|
|
138
138
|
* caller cannot be handed an arm it has no name for, because such an arm would not have compiled.
|
package/dist/evalTypes.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"evalTypes.js","sourceRoot":"","sources":["../src/evalTypes.ts"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"evalTypes.js","sourceRoot":"","sources":["../src/evalTypes.ts"],"names":[],"mappings":"AA6BA;;;;mGAImG;AACnG,MAAM,CAAC,MAAM,2BAA2B,GAAG,CAAC,CAAC;AAE7C;;;;;;;;;;;;;;;;;;;;;;;;GAwBG;AACH,MAAM,CAAC,MAAM,uBAAuB,GAAG,yBAAyB,CAAC;AAEjE;;;;;;;;;;;;;;;;;;GAkBG;AACH,MAAM,sBAAsB,GAAG;IAC7B,iBAAiB,EAAE,2BAA2B;IAC9C,YAAY,EAAE,sBAAsB;CAC2B,CAAC;AAElE;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG;IAClC,gBAAgB;IAChB,GAAG,MAAM,CAAC,MAAM,CAAC,sBAAsB,CAAC;CAChC,CAAC;AAiBX,mGAAmG;AACnG,MAAM,CAAC,MAAM,gBAAgB,GAAG,WAAW,CAAC;AAE5C;;;;GAIG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAsC;IACrE,gBAAgB,EAAE,uBAAuB;IACzC,2BAA2B,EAAE,GAAG,gBAAgB,6BAA6B;IAC7E,sBAAsB,EAAE,GAAG,gBAAgB,wBAAwB;CACpE,CAAC;AAEF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAsCG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG,MAAM,CAAC,MAAM,CAAC,sBAAsB,CAAC,CAAC;AAW1E;;;;;;GAMG;AACH,MAAM,UAAU,qBAAqB,CAAC,IAAwB;IAC5D,OAAO,sBAAsB,CAAC,IAAI,CAAC,CAAC;AACtC,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,8BAA8B,CAAC,SAAwC;IACrF,OAAO,CACL,SAAS,KAAK,SAAS;QACtB,oBAAqD,CAAC,QAAQ,CAAC,SAAS,CAAC,CAC3E,CAAC;AACJ,CAAC"}
|
package/dist/index.d.ts
CHANGED
|
@@ -18,15 +18,21 @@ export { extractClassificationValue, buildConfusionMatrix, collectTags, readRaw,
|
|
|
18
18
|
export { buildClassificationReport } from '#src/classificationReport.js';
|
|
19
19
|
export { computeMetric, parseMetricPredicate, evaluatePredicate, evaluatePredicates, formatTally, } from '#src/metrics.js';
|
|
20
20
|
export { renderClassificationReport, renderConfusionMatrix, renderMetric, } from '#src/classificationRender.js';
|
|
21
|
+
export { computeToolCoverage, aggregateToolCoverage, gradeRunToolCoverage, hasRunToolCoverageSpec, WAIVER_SHARE_WARN_THRESHOLD, } from '#src/toolCoverage.js';
|
|
22
|
+
export type { AdvertisedToolInventory, AdvertisedToolRecord, RunToolCoverageGrade, RunToolCoverageSource, RunToolCoverageSpec, ToolCoverageInput, ToolCoverageReport, ToolCoverageServerReport, ToolCoverageSpec, } from '#src/toolCoverage.js';
|
|
23
|
+
export { renderToolCoverage } from '#src/toolCoverageRender.js';
|
|
24
|
+
export type { RenderToolCoverageOptions } from '#src/toolCoverageRender.js';
|
|
21
25
|
export { buildBlindExport, diffRelabel, renderRelabelDiff } from '#src/blindExport.js';
|
|
22
26
|
export type { BlindExport, BlindExportCase, RelabelDiff, RelabelEntry } from '#src/blindExport.js';
|
|
23
|
-
export { expandSweep, deepMerge, renderComparison, diffRuns, renderRunDiff, } from '#src/evalCompare.js';
|
|
24
|
-
export type { ComparisonColumn, RunDiff, RunDiffEntry, SweepCell } from '#src/evalCompare.js';
|
|
27
|
+
export { expandSweep, deepMerge, renderComparison, diffRuns, renderRunDiff, parseJudgeDriftFilter, DEFAULT_JUDGE_DRIFT_FILTER, DEFAULT_JUDGE_DRIFT_TOLERANCE, MAX_JUDGE_DRIFT_TOLERANCE, } from '#src/evalCompare.js';
|
|
28
|
+
export type { ComparisonColumn, JudgeDriftEntry, JudgeDriftFilter, JudgeDriftMean, RunDiff, RunDiffEntry, SweepCell, } from '#src/evalCompare.js';
|
|
25
29
|
export { UNRECOGNIZED_LABEL, NO_EXPECTATION } from '#src/classificationTypes.js';
|
|
26
30
|
export type { ClassificationExtractor, ClassifiedCell, EvalCaseClassification, EvalClassificationReport, EvalClassificationSpec, EvalConfusionMatrix, EvalMetricCoverage, EvalMetricResult, EvalMetricSpec, EvalMetricTally, MetricField, MetricPredicate, } from '#src/classificationTypes.js';
|
|
27
31
|
export type { ClassifyOutcome, ClassifyRequest, ClassifyRound, EvalSweep, EvalSweepValue, RaterTarget, RunClassifyFn, } from '#src/evalTypes.js';
|
|
28
32
|
export { buildRaterClassifier, HARDLINE_NO_RATING_REASON, HARDLINE_REFUSAL_MARKER, NEGOTIATION_BOUND_MARKER, NO_RATING_CALL_MARKER, } from '#src/raterTarget.js';
|
|
29
33
|
export type { RaterClassifierOptions } from '#src/raterTarget.js';
|
|
34
|
+
export { RATER_PROMPT_NOTE_NAMES } from '#src/raterPromptArm.js';
|
|
35
|
+
export type { RaterPromptArm, RaterPromptNoteName } from '#src/raterPromptArm.js';
|
|
30
36
|
export { resolveReporters, availableReporterNames } from '#src/reporters/registry.js';
|
|
31
37
|
export { driveReporters } from '#src/reporters/drive.js';
|
|
32
38
|
export { createTextReporter } from '#src/reporters/textReporter.js';
|
package/dist/index.js
CHANGED
|
@@ -17,12 +17,20 @@ export { extractClassificationValue, buildConfusionMatrix, collectTags, readRaw,
|
|
|
17
17
|
export { buildClassificationReport } from '#src/classificationReport.js';
|
|
18
18
|
export { computeMetric, parseMetricPredicate, evaluatePredicate, evaluatePredicates, formatTally, } from '#src/metrics.js';
|
|
19
19
|
export { renderClassificationReport, renderConfusionMatrix, renderMetric, } from '#src/classificationRender.js';
|
|
20
|
+
// BATCH-32 — tool coverage: which of the agent's advertised tools a suite exercised.
|
|
21
|
+
export { computeToolCoverage, aggregateToolCoverage, gradeRunToolCoverage, hasRunToolCoverageSpec, WAIVER_SHARE_WARN_THRESHOLD, } from '#src/toolCoverage.js';
|
|
22
|
+
export { renderToolCoverage } from '#src/toolCoverageRender.js';
|
|
20
23
|
export { buildBlindExport, diffRelabel, renderRelabelDiff } from '#src/blindExport.js';
|
|
21
|
-
export { expandSweep, deepMerge, renderComparison, diffRuns, renderRunDiff, } from '#src/evalCompare.js';
|
|
24
|
+
export { expandSweep, deepMerge, renderComparison, diffRuns, renderRunDiff, parseJudgeDriftFilter, DEFAULT_JUDGE_DRIFT_FILTER, DEFAULT_JUDGE_DRIFT_TOLERANCE, MAX_JUDGE_DRIFT_TOLERANCE, } from '#src/evalCompare.js';
|
|
22
25
|
export { UNRECOGNIZED_LABEL, NO_EXPECTATION } from '#src/classificationTypes.js';
|
|
23
26
|
// BATCH-25 Half B — the `rater` target: the one implementation of the `RunClassifyFn` seam, which
|
|
24
27
|
// drives gth's own approvals rater over a corpus of commands.
|
|
25
28
|
export { buildRaterClassifier, HARDLINE_NO_RATING_REASON, HARDLINE_REFUSAL_MARKER, NEGOTIATION_BOUND_MARKER, NO_RATING_CALL_MARKER, } from '#src/raterTarget.js';
|
|
29
|
+
// [[BATCH-31]] — the prompt arm: the note-on / note-off A/B a rater suite can express. Only the
|
|
30
|
+
// NAMES and the type are exported. `armRaterModel` deliberately is not — nothing outside this
|
|
31
|
+
// package needs to build one, and the narrower the surface the smaller the chance of a caller
|
|
32
|
+
// wiring an omission somewhere the argument in `raterPromptArm.ts` does not cover.
|
|
33
|
+
export { RATER_PROMPT_NOTE_NAMES } from '#src/raterPromptArm.js';
|
|
26
34
|
// BATCH-19 — the `gth eval` reporter facility (A1 seam). These are the public plugin contract an
|
|
27
35
|
// out-of-core `@gaunt-sloth/eval-reporter-*` package implements, exported from the package root so a
|
|
28
36
|
// reporter package can type its factory against ONE import.
|
package/dist/index.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,WAAW,EAAE,MAAM,gBAAgB,CAAC;AAC7C,OAAO,EAAE,eAAe,EAAE,MAAM,qBAAqB,CAAC;AACtD,OAAO,EAAE,aAAa,EAAE,MAAM,mBAAmB,CAAC;AAClD,OAAO,EAAE,cAAc,EAAE,iBAAiB,EAAE,MAAM,qBAAqB,CAAC;AACxE,OAAO,EAAE,gBAAgB,EAAE,MAAM,gBAAgB,CAAC;AAWlD,OAAO,EAAE,wBAAwB,EAAE,MAAM,eAAe,CAAC;AAEzD,gGAAgG;AAChG,OAAO,EAAE,cAAc,EAAE,MAAM,mBAAmB,CAAC;AACnD,OAAO,EAAE,sBAAsB,EAAE,MAAM,6BAA6B,CAAC;AACrE,OAAO,EACL,aAAa,EACb,kBAAkB,EAClB,iBAAiB,EACjB,6BAA6B,GAC9B,MAAM,eAAe,CAAC;AAEvB,OAAO,EAAE,YAAY,EAAE,MAAM,oBAAoB,CAAC;AAClD,OAAO,EAAE,eAAe,EAAE,MAAM,oBAAoB,CAAC;AAqBrD,OAAO,EAAE,2BAA2B,EAAE,MAAM,mBAAmB,CAAC;AAEhE,mGAAmG;AACnG,8FAA8F;AAC9F,OAAO,EACL,0BAA0B,EAC1B,oBAAoB,EACpB,WAAW,EACX,OAAO,GACR,MAAM,wBAAwB,CAAC;AAChC,OAAO,EAAE,yBAAyB,EAAE,MAAM,8BAA8B,CAAC;AACzE,OAAO,EACL,aAAa,EACb,oBAAoB,EACpB,iBAAiB,EACjB,kBAAkB,EAClB,WAAW,GACZ,MAAM,iBAAiB,CAAC;AACzB,OAAO,EACL,0BAA0B,EAC1B,qBAAqB,EACrB,YAAY,GACb,MAAM,8BAA8B,CAAC;AACtC,OAAO,EAAE,gBAAgB,EAAE,WAAW,EAAE,iBAAiB,EAAE,MAAM,qBAAqB,CAAC;AAEvF,OAAO,EACL,WAAW,EACX,SAAS,EACT,gBAAgB,EAChB,QAAQ,EACR,aAAa,
|
|
1
|
+
{"version":3,"file":"index.js","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,WAAW,EAAE,MAAM,gBAAgB,CAAC;AAC7C,OAAO,EAAE,eAAe,EAAE,MAAM,qBAAqB,CAAC;AACtD,OAAO,EAAE,aAAa,EAAE,MAAM,mBAAmB,CAAC;AAClD,OAAO,EAAE,cAAc,EAAE,iBAAiB,EAAE,MAAM,qBAAqB,CAAC;AACxE,OAAO,EAAE,gBAAgB,EAAE,MAAM,gBAAgB,CAAC;AAWlD,OAAO,EAAE,wBAAwB,EAAE,MAAM,eAAe,CAAC;AAEzD,gGAAgG;AAChG,OAAO,EAAE,cAAc,EAAE,MAAM,mBAAmB,CAAC;AACnD,OAAO,EAAE,sBAAsB,EAAE,MAAM,6BAA6B,CAAC;AACrE,OAAO,EACL,aAAa,EACb,kBAAkB,EAClB,iBAAiB,EACjB,6BAA6B,GAC9B,MAAM,eAAe,CAAC;AAEvB,OAAO,EAAE,YAAY,EAAE,MAAM,oBAAoB,CAAC;AAClD,OAAO,EAAE,eAAe,EAAE,MAAM,oBAAoB,CAAC;AAqBrD,OAAO,EAAE,2BAA2B,EAAE,MAAM,mBAAmB,CAAC;AAEhE,mGAAmG;AACnG,8FAA8F;AAC9F,OAAO,EACL,0BAA0B,EAC1B,oBAAoB,EACpB,WAAW,EACX,OAAO,GACR,MAAM,wBAAwB,CAAC;AAChC,OAAO,EAAE,yBAAyB,EAAE,MAAM,8BAA8B,CAAC;AACzE,OAAO,EACL,aAAa,EACb,oBAAoB,EACpB,iBAAiB,EACjB,kBAAkB,EAClB,WAAW,GACZ,MAAM,iBAAiB,CAAC;AACzB,OAAO,EACL,0BAA0B,EAC1B,qBAAqB,EACrB,YAAY,GACb,MAAM,8BAA8B,CAAC;AACtC,qFAAqF;AACrF,OAAO,EACL,mBAAmB,EACnB,qBAAqB,EACrB,oBAAoB,EACpB,sBAAsB,EACtB,2BAA2B,GAC5B,MAAM,sBAAsB,CAAC;AAY9B,OAAO,EAAE,kBAAkB,EAAE,MAAM,4BAA4B,CAAC;AAEhE,OAAO,EAAE,gBAAgB,EAAE,WAAW,EAAE,iBAAiB,EAAE,MAAM,qBAAqB,CAAC;AAEvF,OAAO,EACL,WAAW,EACX,SAAS,EACT,gBAAgB,EAChB,QAAQ,EACR,aAAa,EACb,qBAAqB,EACrB,0BAA0B,EAC1B,6BAA6B,EAC7B,yBAAyB,GAC1B,MAAM,qBAAqB,CAAC;AAU7B,OAAO,EAAE,kBAAkB,EAAE,cAAc,EAAE,MAAM,6BAA6B,CAAC;AAyBjF,kGAAkG;AAClG,8DAA8D;AAC9D,OAAO,EACL,oBAAoB,EACpB,yBAAyB,EACzB,uBAAuB,EACvB,wBAAwB,EACxB,qBAAqB,GACtB,MAAM,qBAAqB,CAAC;AAG7B,gGAAgG;AAChG,8FAA8F;AAC9F,8FAA8F;AAC9F,mFAAmF;AACnF,OAAO,EAAE,uBAAuB,EAAE,MAAM,wBAAwB,CAAC;AAGjE,iGAAiG;AACjG,qGAAqG;AACrG,4DAA4D;AAC5D,OAAO,EAAE,gBAAgB,EAAE,sBAAsB,EAAE,MAAM,4BAA4B,CAAC;AACtF,OAAO,EAAE,cAAc,EAAE,MAAM,yBAAyB,CAAC;AACzD,OAAO,EAAE,kBAAkB,EAAE,MAAM,gCAAgC,CAAC;AAQpE,kGAAkG;AAClG,6CAA6C;AAC7C,OAAO,EAAE,WAAW,EAAE,MAAM,8BAA8B,CAAC;AAC3D,OAAO,EAAE,4BAA4B,EAAE,MAAM,eAAe,CAAC"}
|
package/dist/judge.d.ts
CHANGED
|
@@ -1,6 +1,4 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* @module judge
|
|
3
|
-
*
|
|
4
2
|
* BATCH-2 — LLM-as-judge grading for `gth eval`. Adapts the *mechanism* of EXT-10's shell-safety
|
|
5
3
|
* judge (`packages/core/src/core/shell/judge.ts`): `model.withStructuredOutput(zodSchema)` for a
|
|
6
4
|
* single non-agentic structured call, raced against a timeout. The failure policy differs on
|
|
@@ -18,6 +16,8 @@
|
|
|
18
16
|
* `judgeShellCommand`'s own default. A separate `--judge <profile>` model is BATCH-2's own
|
|
19
17
|
* "Not in scope" list (identity-matrix/pluggable-target work); grading with the SUT's own model
|
|
20
18
|
* config is a known, real simplification for this first slice.
|
|
19
|
+
*
|
|
20
|
+
* @module
|
|
21
21
|
*/
|
|
22
22
|
import type { BaseChatModel } from '@langchain/core/language_models/chat_models';
|
|
23
23
|
import * as z from 'zod';
|
package/dist/judge.js
CHANGED
|
@@ -1,6 +1,4 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* @module judge
|
|
3
|
-
*
|
|
4
2
|
* BATCH-2 — LLM-as-judge grading for `gth eval`. Adapts the *mechanism* of EXT-10's shell-safety
|
|
5
3
|
* judge (`packages/core/src/core/shell/judge.ts`): `model.withStructuredOutput(zodSchema)` for a
|
|
6
4
|
* single non-agentic structured call, raced against a timeout. The failure policy differs on
|
|
@@ -18,8 +16,11 @@
|
|
|
18
16
|
* `judgeShellCommand`'s own default. A separate `--judge <profile>` model is BATCH-2's own
|
|
19
17
|
* "Not in scope" list (identity-matrix/pluggable-target work); grading with the SUT's own model
|
|
20
18
|
* config is a known, real simplification for this first slice.
|
|
19
|
+
*
|
|
20
|
+
* @module
|
|
21
21
|
*/
|
|
22
22
|
import { HumanMessage, SystemMessage } from '@langchain/core/messages';
|
|
23
|
+
import { CALL_TIMED_OUT, withCallDeadline } from '@gaunt-sloth/core/runtime/abortableCall.js';
|
|
23
24
|
import { structuredOutputBoundary } from '@gaunt-sloth/core/runtime/structuredOutput.js';
|
|
24
25
|
import * as z from 'zod';
|
|
25
26
|
/** Structured verdict the judge model must return — 0-10 `rate` + a `reason`, matching `review`'s
|
|
@@ -85,20 +86,19 @@ export async function judgeEvalCase(answer, rubric, model, options) {
|
|
|
85
86
|
return { attempted: true, ok: false, error: 'No usable judge model configured.' };
|
|
86
87
|
}
|
|
87
88
|
const { system, user } = buildJudgeMessages(answer, rubric);
|
|
88
|
-
let timer;
|
|
89
89
|
try {
|
|
90
90
|
// EXT-88 — routed through the shared boundary like every other structured-output call. The
|
|
91
91
|
// rubric verdict has no optional field today, so the boundary hands back this very schema by
|
|
92
92
|
// identity; going through it is what stops a later optional field re-introducing the defect.
|
|
93
93
|
const boundary = structuredOutputBoundary(EvalVerdictSchema);
|
|
94
94
|
const structured = model.withStructuredOutput(boundary.wireSchema);
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
const raced = await
|
|
101
|
-
if (raced ===
|
|
95
|
+
// [[EXT-179]] — the budget aborts the judge call rather than abandoning it. This matters most
|
|
96
|
+
// in a SWEEP: a suite grades many cases, and every stalled judge used to leave its own request
|
|
97
|
+
// in flight, so the leaked sockets accumulated across the run and kept the process alive after
|
|
98
|
+
// the report had been written. The timeout text is unchanged and the case still fails as a
|
|
99
|
+
// case — an aborted judgement is a judgement not obtained, never an auto-pass.
|
|
100
|
+
const raced = await withCallDeadline(timeoutMs, (signal) => structured.invoke([new SystemMessage(system), new HumanMessage(user)], { signal }));
|
|
101
|
+
if (raced === CALL_TIMED_OUT) {
|
|
102
102
|
return { attempted: true, ok: false, error: `Judge timed out after ${timeoutMs}ms.` };
|
|
103
103
|
}
|
|
104
104
|
// withStructuredOutput already coerces to the schema, but re-validate defensively: a fake or
|
|
@@ -116,9 +116,5 @@ export async function judgeEvalCase(answer, rubric, model, options) {
|
|
|
116
116
|
error: error instanceof Error ? error.message : String(error),
|
|
117
117
|
};
|
|
118
118
|
}
|
|
119
|
-
finally {
|
|
120
|
-
if (timer)
|
|
121
|
-
clearTimeout(timer);
|
|
122
|
-
}
|
|
123
119
|
}
|
|
124
120
|
//# sourceMappingURL=judge.js.map
|
package/dist/judge.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"judge.js","sourceRoot":"","sources":["../src/judge.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAGH,OAAO,EAAE,YAAY,EAAE,aAAa,EAAE,MAAM,0BAA0B,CAAC;AACvE,OAAO,EAAE,wBAAwB,EAAE,MAAM,+CAA+C,CAAC;AACzF,OAAO,KAAK,CAAC,MAAM,KAAK,CAAC;AAIzB;;oGAEoG;AACpG,MAAM,CAAC,MAAM,iBAAiB,GAAG,CAAC,CAAC,MAAM,CAAC;IACxC,IAAI,EAAE,CAAC;SACJ,MAAM,EAAE;SACR,GAAG,CAAC,CAAC,CAAC;SACN,GAAG,CAAC,EAAE,CAAC;SACP,QAAQ,CAAC,yDAAyD,CAAC;IACtE,MAAM,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,QAAQ,CAAC,2CAA2C,CAAC;CACzE,CAAC,CAAC;AAIH;;oGAEoG;AACpG,MAAM,CAAC,MAAM,6BAA6B,GAAG,MAAM,CAAC;AAEpD,MAAM,wBAAwB,GAAG;IAC/B,mCAAmC;IACnC,gGAAgG;IAChG,cAAc;IACd,EAAE;IACF,+FAA+F;IAC/F,gGAAgG;IAChG,mDAAmD;IACnD,6FAA6F;IAC7F,+BAA+B;CAChC,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AAEb;;;;GAIG;AACH,MAAM,UAAU,kBAAkB,CAChC,MAAc,EACd,MAAc;IAEd,MAAM,SAAS,GAAG;QAChB,SAAS;QACT,MAAM;QACN,EAAE;QACF,qBAAqB;QACrB,UAAU;QACV,MAAM,IAAI,gBAAgB;QAC1B,WAAW;KACZ,CAAC;IACF,OAAO,EAAE,MAAM,EAAE,wBAAwB,EAAE,IAAI,EAAE,SAAS,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;AAC1E,CAAC;AAED;;;;;;;;;;;;;GAaG;AACH,MAAM,CAAC,KAAK,UAAU,aAAa,CACjC,MAAc,EACd,MAAc,EACd,KAAgC,EAChC,OAAgC;IAEhC,MAAM,SAAS,GAAG,OAAO,EAAE,SAAS,IAAI,6BAA6B,CAAC;IAEtE,IAAI,CAAC,KAAK,IAAI,OAAO,KAAK,CAAC,oBAAoB,KAAK,UAAU,EAAE,CAAC;QAC/D,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,mCAAmC,EAAE,CAAC;IACpF,CAAC;IAED,MAAM,EAAE,MAAM,EAAE,IAAI,EAAE,GAAG,kBAAkB,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IAC5D,IAAI,
|
|
1
|
+
{"version":3,"file":"judge.js","sourceRoot":"","sources":["../src/judge.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAGH,OAAO,EAAE,YAAY,EAAE,aAAa,EAAE,MAAM,0BAA0B,CAAC;AACvE,OAAO,EAAE,cAAc,EAAE,gBAAgB,EAAE,MAAM,4CAA4C,CAAC;AAC9F,OAAO,EAAE,wBAAwB,EAAE,MAAM,+CAA+C,CAAC;AACzF,OAAO,KAAK,CAAC,MAAM,KAAK,CAAC;AAIzB;;oGAEoG;AACpG,MAAM,CAAC,MAAM,iBAAiB,GAAG,CAAC,CAAC,MAAM,CAAC;IACxC,IAAI,EAAE,CAAC;SACJ,MAAM,EAAE;SACR,GAAG,CAAC,CAAC,CAAC;SACN,GAAG,CAAC,EAAE,CAAC;SACP,QAAQ,CAAC,yDAAyD,CAAC;IACtE,MAAM,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,QAAQ,CAAC,2CAA2C,CAAC;CACzE,CAAC,CAAC;AAIH;;oGAEoG;AACpG,MAAM,CAAC,MAAM,6BAA6B,GAAG,MAAM,CAAC;AAEpD,MAAM,wBAAwB,GAAG;IAC/B,mCAAmC;IACnC,gGAAgG;IAChG,cAAc;IACd,EAAE;IACF,+FAA+F;IAC/F,gGAAgG;IAChG,mDAAmD;IACnD,6FAA6F;IAC7F,+BAA+B;CAChC,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AAEb;;;;GAIG;AACH,MAAM,UAAU,kBAAkB,CAChC,MAAc,EACd,MAAc;IAEd,MAAM,SAAS,GAAG;QAChB,SAAS;QACT,MAAM;QACN,EAAE;QACF,qBAAqB;QACrB,UAAU;QACV,MAAM,IAAI,gBAAgB;QAC1B,WAAW;KACZ,CAAC;IACF,OAAO,EAAE,MAAM,EAAE,wBAAwB,EAAE,IAAI,EAAE,SAAS,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;AAC1E,CAAC;AAED;;;;;;;;;;;;;GAaG;AACH,MAAM,CAAC,KAAK,UAAU,aAAa,CACjC,MAAc,EACd,MAAc,EACd,KAAgC,EAChC,OAAgC;IAEhC,MAAM,SAAS,GAAG,OAAO,EAAE,SAAS,IAAI,6BAA6B,CAAC;IAEtE,IAAI,CAAC,KAAK,IAAI,OAAO,KAAK,CAAC,oBAAoB,KAAK,UAAU,EAAE,CAAC;QAC/D,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,mCAAmC,EAAE,CAAC;IACpF,CAAC;IAED,MAAM,EAAE,MAAM,EAAE,IAAI,EAAE,GAAG,kBAAkB,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IAC5D,IAAI,CAAC;QACH,2FAA2F;QAC3F,6FAA6F;QAC7F,6FAA6F;QAC7F,MAAM,QAAQ,GAAG,wBAAwB,CAAC,iBAAiB,CAAC,CAAC;QAC7D,MAAM,UAAU,GAAG,KAAK,CAAC,oBAAoB,CAAC,QAAQ,CAAC,UAAU,CAAC,CAAC;QAEnE,8FAA8F;QAC9F,+FAA+F;QAC/F,+FAA+F;QAC/F,2FAA2F;QAC3F,+EAA+E;QAC/E,MAAM,KAAK,GAAG,MAAM,gBAAgB,CAAC,SAAS,EAAE,CAAC,MAAM,EAAE,EAAE,CACzD,UAAU,CAAC,MAAM,CAAC,CAAC,IAAI,aAAa,CAAC,MAAM,CAAC,EAAE,IAAI,YAAY,CAAC,IAAI,CAAC,CAAC,EAAE,EAAE,MAAM,EAAE,CAAC,CACnF,CAAC;QACF,IAAI,KAAK,KAAK,cAAc,EAAE,CAAC;YAC7B,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,yBAAyB,SAAS,KAAK,EAAE,CAAC;QACxF,CAAC;QAED,6FAA6F;QAC7F,0DAA0D;QAC1D,MAAM,MAAM,GAAG,QAAQ,CAAC,SAAS,CAAC,KAAK,CAAC,CAAC;QACzC,IAAI,CAAC,MAAM,CAAC,OAAO,EAAE,CAAC;YACpB,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,oCAAoC,EAAE,CAAC;QACrF,CAAC;QACD,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,IAAI,EAAE,OAAO,EAAE,MAAM,CAAC,IAAI,EAAE,CAAC;IAC7D,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QACf,OAAO;YACL,SAAS,EAAE,IAAI;YACf,EAAE,EAAE,KAAK;YACT,KAAK,EAAE,KAAK,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC;SAC9D,CAAC;IACJ,CAAC;AACH,CAAC"}
|
package/dist/output.d.ts
CHANGED
|
@@ -4,7 +4,7 @@ import type { BatchSummary, CellResult } from '#src/types.js';
|
|
|
4
4
|
* (pass/fail counts + a per-cell one-liner — a lightweight flake report) into `outputDir`.
|
|
5
5
|
* Creates `outputDir` (and any missing parents) if it doesn't exist.
|
|
6
6
|
*
|
|
7
|
-
* Pure I/O, deliberately separate from {@link runBatchMatrix}: the runner never touches the
|
|
7
|
+
* Pure I/O, deliberately separate from {@link @gaunt-sloth/batch!"BatchRunner.js".runBatchMatrix | runBatchMatrix}: the runner never touches the
|
|
8
8
|
* filesystem, so unit tests can exercise matrix/concurrency/retry logic without a tmp dir, and this
|
|
9
9
|
* function can be tested in isolation with a fixed set of results.
|
|
10
10
|
*
|
package/dist/output.js
CHANGED
|
@@ -6,7 +6,7 @@ import { buildBatchSummary } from '#src/BatchRunner.js';
|
|
|
6
6
|
* (pass/fail counts + a per-cell one-liner — a lightweight flake report) into `outputDir`.
|
|
7
7
|
* Creates `outputDir` (and any missing parents) if it doesn't exist.
|
|
8
8
|
*
|
|
9
|
-
* Pure I/O, deliberately separate from {@link runBatchMatrix}: the runner never touches the
|
|
9
|
+
* Pure I/O, deliberately separate from {@link @gaunt-sloth/batch!"BatchRunner.js".runBatchMatrix | runBatchMatrix}: the runner never touches the
|
|
10
10
|
* filesystem, so unit tests can exercise matrix/concurrency/retry logic without a tmp dir, and this
|
|
11
11
|
* function can be tested in isolation with a fixed set of results.
|
|
12
12
|
*
|
package/dist/parseOver.d.ts
CHANGED
|
@@ -4,7 +4,7 @@ import type { MatrixRow } from '#src/types.js';
|
|
|
4
4
|
* `.jsonl`/`.ndjson` → one JSON object per line; anything else (including `.csv`) → CSV.
|
|
5
5
|
*
|
|
6
6
|
* This is content binding only (BATCH-1 scope): every row becomes an object of string fields that
|
|
7
|
-
* {@link bindCellContent} interpolates into the script. A glob-of-binary-files path binding is out
|
|
7
|
+
* {@link @gaunt-sloth/batch!"interpolate.js".bindCellContent | bindCellContent} interpolates into the script. A glob-of-binary-files path binding is out
|
|
8
8
|
* of scope for this task.
|
|
9
9
|
*
|
|
10
10
|
* Throws a descriptive `Error` on malformed input — the harness-level failure the CLI surface doc
|
package/dist/parseOver.js
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
* `.jsonl`/`.ndjson` → one JSON object per line; anything else (including `.csv`) → CSV.
|
|
4
4
|
*
|
|
5
5
|
* This is content binding only (BATCH-1 scope): every row becomes an object of string fields that
|
|
6
|
-
* {@link bindCellContent} interpolates into the script. A glob-of-binary-files path binding is out
|
|
6
|
+
* {@link @gaunt-sloth/batch!"interpolate.js".bindCellContent | bindCellContent} interpolates into the script. A glob-of-binary-files path binding is out
|
|
7
7
|
* of scope for this task.
|
|
8
8
|
*
|
|
9
9
|
* Throws a descriptive `Error` on malformed input — the harness-level failure the CLI surface doc
|
package/dist/pipelineCli.d.ts
CHANGED
|
@@ -1,6 +1,4 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* @module pipelineCli
|
|
3
|
-
*
|
|
4
2
|
* BATCH-9 — the standalone `gth-batch` pipeline runner. A thin entry point that runs the BATCH-1
|
|
5
3
|
* matrix runtime ({@link buildMatrix} + {@link runBatchMatrix}) directly from a shell pipeline,
|
|
6
4
|
* without pulling in the whole `gaunt-sloth` app. It takes a prompt-executable script + `--over`
|
|
@@ -16,10 +14,12 @@
|
|
|
16
14
|
* the pipeline shell already handles) and output is JSONL on stdout (not a directory of files).
|
|
17
15
|
*
|
|
18
16
|
* stdout discipline: the run itself is noisy (the runtime's `display()`/`ProgressIndicator`/token
|
|
19
|
-
* streaming all target `process.stdout`). The bin entry ({@link
|
|
17
|
+
* streaming all target `process.stdout`). The bin entry ({@link @gaunt-sloth/batch!"bin.js" | ./bin.ts}) redirects
|
|
20
18
|
* `process.stdout.write` to stderr for the duration and this module writes the machine JSONL
|
|
21
19
|
* straight to fd 1 (`fs.writeSync`), so stdout stays a clean data channel — the same "protocol
|
|
22
20
|
* channel" discipline `packages/app/cli.js` uses for ACP.
|
|
21
|
+
*
|
|
22
|
+
* @module
|
|
23
23
|
*/
|
|
24
24
|
import type { CommandLineConfigOverrides, GthConfig } from '@gaunt-sloth/core/config.js';
|
|
25
25
|
import type { BatchSummary, MatrixRow, RunCellFn } from '#src/types.js';
|
package/dist/pipelineCli.js
CHANGED
|
@@ -1,6 +1,4 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* @module pipelineCli
|
|
3
|
-
*
|
|
4
2
|
* BATCH-9 — the standalone `gth-batch` pipeline runner. A thin entry point that runs the BATCH-1
|
|
5
3
|
* matrix runtime ({@link buildMatrix} + {@link runBatchMatrix}) directly from a shell pipeline,
|
|
6
4
|
* without pulling in the whole `gaunt-sloth` app. It takes a prompt-executable script + `--over`
|
|
@@ -16,10 +14,12 @@
|
|
|
16
14
|
* the pipeline shell already handles) and output is JSONL on stdout (not a directory of files).
|
|
17
15
|
*
|
|
18
16
|
* stdout discipline: the run itself is noisy (the runtime's `display()`/`ProgressIndicator`/token
|
|
19
|
-
* streaming all target `process.stdout`). The bin entry ({@link
|
|
17
|
+
* streaming all target `process.stdout`). The bin entry ({@link @gaunt-sloth/batch!"bin.js" | ./bin.ts}) redirects
|
|
20
18
|
* `process.stdout.write` to stderr for the duration and this module writes the machine JSONL
|
|
21
19
|
* straight to fd 1 (`fs.writeSync`), so stdout stays a clean data channel — the same "protocol
|
|
22
20
|
* channel" discipline `packages/app/cli.js` uses for ACP.
|
|
21
|
+
*
|
|
22
|
+
* @module
|
|
23
23
|
*/
|
|
24
24
|
import { writeSync } from 'node:fs';
|
|
25
25
|
import { readFileSync } from 'node:fs';
|