@gaunt-sloth/batch 2.0.0-beta.8 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/dist/bin.d.ts +2 -2
  2. package/dist/bin.js +2 -2
  3. package/dist/classificationReport.d.ts +2 -2
  4. package/dist/classificationReport.js +2 -2
  5. package/dist/classificationTypes.d.ts +7 -7
  6. package/dist/classificationTypes.js +1 -1
  7. package/dist/evalCompare.d.ts +128 -1
  8. package/dist/evalCompare.js +253 -2
  9. package/dist/evalCompare.js.map +1 -1
  10. package/dist/evalOutput.d.ts +1 -1
  11. package/dist/evalOutput.js +1 -1
  12. package/dist/evalRunner.d.ts +15 -3
  13. package/dist/evalRunner.js +82 -5
  14. package/dist/evalRunner.js.map +1 -1
  15. package/dist/evalSuite.d.ts +5 -1
  16. package/dist/evalSuite.js +218 -7
  17. package/dist/evalSuite.js.map +1 -1
  18. package/dist/evalTypes.d.ts +79 -18
  19. package/dist/evalTypes.js +2 -2
  20. package/dist/evalTypes.js.map +1 -1
  21. package/dist/index.d.ts +8 -2
  22. package/dist/index.js +9 -1
  23. package/dist/index.js.map +1 -1
  24. package/dist/judge.d.ts +2 -2
  25. package/dist/judge.js +10 -14
  26. package/dist/judge.js.map +1 -1
  27. package/dist/output.d.ts +1 -1
  28. package/dist/output.js +1 -1
  29. package/dist/parseOver.d.ts +1 -1
  30. package/dist/parseOver.js +1 -1
  31. package/dist/pipelineCli.d.ts +3 -3
  32. package/dist/pipelineCli.js +3 -3
  33. package/dist/raterPromptArm.d.ts +140 -0
  34. package/dist/raterPromptArm.js +306 -0
  35. package/dist/raterPromptArm.js.map +1 -0
  36. package/dist/raterTarget.d.ts +34 -12
  37. package/dist/raterTarget.js +124 -33
  38. package/dist/raterTarget.js.map +1 -1
  39. package/dist/reporters/registry.d.ts +1 -1
  40. package/dist/reporters/registry.js +1 -1
  41. package/dist/reporters/registry.js.map +1 -1
  42. package/dist/reporters/reporterTypes.d.ts +3 -3
  43. package/dist/reporters/textReporter.js +16 -0
  44. package/dist/reporters/textReporter.js.map +1 -1
  45. package/dist/toolCoverage.d.ts +244 -0
  46. package/dist/toolCoverage.js +414 -0
  47. package/dist/toolCoverage.js.map +1 -0
  48. package/dist/toolCoverageRender.d.ts +31 -0
  49. package/dist/toolCoverageRender.js +71 -0
  50. package/dist/toolCoverageRender.js.map +1 -0
  51. package/dist/toolResultChecks.d.ts +8 -2
  52. package/dist/toolResultChecks.js +45 -3
  53. package/dist/toolResultChecks.js.map +1 -1
  54. package/dist/types.d.ts +38 -3
  55. package/dist/types.js +0 -9
  56. package/dist/types.js.map +1 -1
  57. package/dist/workflow/runWorkflow.d.ts +4 -4
  58. package/dist/workflow/runWorkflow.js +3 -3
  59. package/package.json +9 -8
@@ -1,13 +1,15 @@
1
1
  /**
2
2
  * @packageDocumentation
3
3
  * BATCH-2 — the shapes for `gth eval`: a parsed suite/case, deterministic-check results, the
4
- * judge's verdict, and per-case/suite outcomes. Deliberately separate from {@link ../types.js}
4
+ * judge's verdict, and per-case/suite outcomes. Deliberately separate from `types.js`
5
5
  * (BATCH-1's cell/outcome shapes), which documents itself as scoped to "cells and outcomes" only
6
6
  * — eval's shapes layer on top of (not into) that file.
7
7
  */
8
8
  import type { ApprovalRung } from '@gaunt-sloth/core/config/shell-policy.js';
9
9
  import type { PreflightFloorKind } from '@gaunt-sloth/core/core/shell/raterVocabulary.js';
10
10
  import type { ToolResultRecord } from '#src/types.js';
11
+ import type { RaterPromptArm } from '#src/raterPromptArm.js';
12
+ import type { AdvertisedToolInventory, ToolCoverageReport, ToolCoverageSpec } from '#src/toolCoverage.js';
11
13
  import type { EvalCaseClassification, EvalClassificationReport, EvalClassificationSpec, EvalMetricSpec } from '#src/classificationTypes.js';
12
14
  /** The 0-10 judge scale's default pass threshold, matching `review`'s own (unexported)
13
15
  * `DEFAULT_PASS_THRESHOLD` in `packages/review/src/middleware/reviewRateMiddleware.ts` — same
@@ -71,7 +73,7 @@ declare const PREFLIGHT_MECHANISM_OF: {
71
73
  * The floor is spelled here because it is not a preflight — it is a lexical scan core keeps no list
72
74
  * of, consulted by the approvals gate before any rating (every rung but `bypass`) and again by the
73
75
  * shell tool at execution time (every rung). The preflights are NOT spelled: they come from
74
- * {@link PREFLIGHT_MECHANISM_OF}, which is core's own `PREFLIGHT_FLOOR_KINDS` with our `forced_by`
76
+ * `PREFLIGHT_MECHANISM_OF`, which is core's own `PREFLIGHT_FLOOR_KINDS` with our `forced_by`
75
77
  * spelling attached.
76
78
  *
77
79
  * ## Why a case asserts a mechanism at all (the I1 finding)
@@ -158,7 +160,7 @@ export declare const PREFLIGHT_MECHANISMS: ("open-world-preflight" | "script-env
158
160
  export type PreflightMechanism = (typeof PREFLIGHT_MECHANISM_OF)[PreflightFloorKind];
159
161
  /**
160
162
  * The `forced_by` spelling of one of core's preflight arms — the typed lookup into
161
- * {@link PREFLIGHT_MECHANISM_OF}, which stays private so this is the only way in.
163
+ * `PREFLIGHT_MECHANISM_OF`, which stays private so this is the only way in.
162
164
  *
163
165
  * TOTAL over core's `PreflightFloorKind` and therefore never `undefined`, which is the point: a
164
166
  * caller cannot be handed an arm it has no name for, because such an arm would not have compiled.
@@ -178,7 +180,7 @@ export declare function mechanismNeedsPermissiveRating(mechanism: ForcedByMechan
178
180
  export interface GthAgentTarget {
179
181
  type: 'gth-agent';
180
182
  /** Suite-level profile hint. Only `undefined`/`'default'` is accepted (see
181
- * {@link ../evalSuite.js}'s `parseEvalSuite`) — per-case/per-identity profile switching is
183
+ * `evalSuite.js`'s `parseEvalSuite`) — per-case/per-identity profile switching is
182
184
  * `identities` scope, not this. */
183
185
  profile?: string;
184
186
  }
@@ -240,7 +242,7 @@ export interface AgUiAgentTarget {
240
242
  * §8 hardline floor — decides some commands outright, with no model in the loop.
241
243
  *
242
244
  * Unlike the agent targets this one is NOT driven by a `RunCellFn`: it supplies the
243
- * {@link RunClassifyFn} seam instead (`buildRaterClassifier` in {@link ../raterTarget.js}), which
245
+ * {@link RunClassifyFn} seam instead (`buildRaterClassifier` in `raterTarget.js`), which
244
246
  * the command passes to `runEvalSuite` as `options.classify`.
245
247
  */
246
248
  export interface RaterTarget {
@@ -252,7 +254,7 @@ export interface RaterTarget {
252
254
  * It is the suite's declaration, not the last word: a run whose config declares `approvals`
253
255
  * overrides it, which is what makes the `rung × model` sweep work through the existing `config:`
254
256
  * axis (a sweep cell cannot reach a `target` field). The override is announced, never silent —
255
- * see {@link ../raterTarget.js}.
257
+ * see `raterTarget.js`.
256
258
  */
257
259
  rung: ApprovalRung;
258
260
  }
@@ -263,7 +265,7 @@ export interface RaterTarget {
263
265
  * target), so the target only changes which runner the command builds. */
264
266
  export type EvalTarget = GthAgentTarget | AdkAgentTarget | AgUiAgentTarget | RaterTarget;
265
267
  /** One `json_path` assertion (BATCH-10): resolve `path` against the answer-parsed-as-JSON and check
266
- * it. Exactly one of `equals`/`contains` is set (enforced in {@link ../evalSuite.js}'s parse):
268
+ * it. Exactly one of `equals`/`contains` is set (enforced in `evalSuite.js`'s parse):
267
269
  * - `equals` — the resolved value must deep-equal this (any JSON value, incl. `null`).
268
270
  * - `contains` — the resolved value must be a string containing this substring. */
269
271
  export interface JsonPathCheck {
@@ -275,7 +277,7 @@ export interface JsonPathCheck {
275
277
  * One `tool_result_json_path` assertion (BATCH-21): select tool results by name `tool` (exact or
276
278
  * glob, the same matcher `must_call` uses), parse each matching result's payload as JSON, and
277
279
  * resolve `path` against it (the same minimal dot/`[index]` path `json_path` uses). At most one of
278
- * `equals`/`contains` may be set (enforced in {@link ../evalSuite.js}'s parse):
280
+ * `equals`/`contains` may be set (enforced in `evalSuite.js`'s parse):
279
281
  * - neither — pure existence check: the path must resolve in a matching result's payload;
280
282
  * - `equals` — the resolved value must deep-equal this (any JSON value, incl. `null`);
281
283
  * - `contains` — the resolved value must be a string containing this substring.
@@ -359,6 +361,31 @@ export interface EvalExpectation {
359
361
  * touching the first.
360
362
  */
361
363
  expectAction?: string;
364
+ /**
365
+ * BATCH-45 — **assert that a MODEL actually rendered a verdict for this round**, i.e. that
366
+ * {@link ClassifyOutcome.modelLabel} is present. `true` is the only legal value; `undefined` =
367
+ * this block makes no such claim. `rater` target only.
368
+ *
369
+ * It is NOT an assertion about the rung, and NOT an assertion about WHAT the model said. Those
370
+ * are {@link expectLabel} and {@link expectAction}, and the whole point of this key is that it
371
+ * says neither: a case can pin "a rater ruled here" without predicting which verdict three
372
+ * different raters would render, which is a prediction no measurement backs.
373
+ *
374
+ * **Why the action column cannot do this job.** When the gate never obtains a rating (timeout,
375
+ * throw, unparseable answer) it fails closed, and since EXT-171 that ESCALATES rather than
376
+ * negotiating. A genuine `catastrophic` verdict escalates too. So `expect_action: escalate` is
377
+ * satisfied identically by a rater that judged the command unnegotiable and by a rater that
378
+ * never answered at all — and a case asserting only that passes on a run where nothing was
379
+ * measured. BATCH-45 found ten such cells live in this repo's own approvals corpus.
380
+ *
381
+ * **Why `modelLabel` and not the rationale text.** `modelLabel` is suppressed from the call's own
382
+ * capture (`rating.failClosed`), never from a reading of the verdict's prose. Core's
383
+ * `isFailClosed` answers a question about the reason TEXT, and the rating prompt tells a rater to
384
+ * answer `destructive` and say it could not assess a command it is unsure of — so that predicate
385
+ * scores an obedient judgement as a gate failure. EXT-171 and `raterHealth` both say never to key
386
+ * on verdict text.
387
+ */
388
+ expectRated?: boolean;
362
389
  }
363
390
  /**
364
391
  * One conversational turn (BATCH-12): the `user` message to send, and the {@link EvalExpectation}
@@ -378,7 +405,7 @@ export interface EvalTurn {
378
405
  *
379
406
  * **Carrying it is not the same as showing it to the rater.** §5.1 admits the justification from
380
407
  * round 2 onward, so the round-1 rating withholds it; the target hands the whole context to core's
381
- * {@link import('@gaunt-sloth/core/core/shell/negotiation.js').ShellNegotiationState}, which is
408
+ * {@link @gaunt-sloth/core!core/shell/negotiation.ShellNegotiationState | ShellNegotiationState}, which is
382
409
  * the single implementation of that rule.
383
410
  */
384
411
  justification?: string;
@@ -410,7 +437,7 @@ export interface EvalTurn {
410
437
  }
411
438
  /**
412
439
  * One turn's raw run outcome inside a multi-turn conversation (BATCH-12 Task 2). Structurally the
413
- * per-turn analogue of BATCH-1's {@link ../types.js CellRunOutcome}: the turn's answer plus the
440
+ * per-turn analogue of BATCH-1's {@link "types.js"!CellRunOutcome | CellRunOutcome}: the turn's answer plus the
414
441
  * tools/tokens captured FOR THAT TURN — a per-invoke GS2-16 delta (each `processMessages` call
415
442
  * resets the tally), NOT the cumulative conversation total. `ok:false` with `error` set means that
416
443
  * turn's SUT invocation failed (no answer to grade). The conversational runner returns one of these
@@ -425,12 +452,18 @@ export interface TurnRunOutcome {
425
452
  /** BATCH-21 — this turn's per-tool-call result records (parallel to {@link tools}; a per-turn
426
453
  * delta like everything else here). Only the in-process `gth-agent` runner populates it. */
427
454
  toolResults?: ToolResultRecord[];
455
+ /**
456
+ * BATCH-32 — the advertised-tool inventory (the coverage denominator). **The one field here that
457
+ * is NOT a per-turn delta**: a conversation builds its agent once, so the inventory is fixed for
458
+ * the whole conversation and every turn repeats it.
459
+ */
460
+ advertisedTools?: AdvertisedToolInventory;
428
461
  error?: string;
429
462
  }
430
463
  /**
431
464
  * Injectable "run one whole scripted conversation" function (BATCH-12 Task 2) — the multi-turn
432
- * analogue of BATCH-1's {@link ../types.js RunCellFn}, and the seam that lets
433
- * {@link ../evalRunner.js}'s multi-turn path be unit tested without any real LLM/MCP. Given the
465
+ * analogue of BATCH-1's {@link "types.js"!RunCellFn | RunCellFn}, and the seam that lets
466
+ * `evalRunner.js`'s multi-turn path be unit tested without any real LLM/MCP. Given the
434
467
  * ordered user messages of ONE (case × identity) conversation, it builds the agent + resolves tools
435
468
  * ONCE, runs each turn against the accumulated message history, and returns one {@link TurnRunOutcome}
436
469
  * per turn (per-turn answer + per-turn tool delta), cleaning up once. The production wiring
@@ -447,7 +480,7 @@ export type RunConversationFn = (userMessages: string[]) => Promise<TurnRunOutco
447
480
  * a round-1 rating, and a reset makes a later round a round-1 rating again — so the target hands the
448
481
  * accumulated state to `ShellNegotiationState.contextFor` rather than deciding here. A shape that
449
482
  * carried "what the rater sees" would be this package holding a second opinion about §5.1, which is
450
- * the one thing {@link ../raterTarget.js} exists not to do.
483
+ * the one thing `raterTarget.js` exists not to do.
451
484
  */
452
485
  export interface ClassifyRound {
453
486
  /** The command the agent proposes in this round. */
@@ -465,7 +498,7 @@ export interface ClassifyRound {
465
498
  * ## The Half-B seam, now filled
466
499
  *
467
500
  * Half A shipped this shape, the runner plumbing and the tests (against an injected fake); Half B
468
- * supplies the one function — `buildRaterClassifier` in {@link ../raterTarget.js}, the
501
+ * supplies the one function — `buildRaterClassifier` in `raterTarget.js`, the
469
502
  * {@link RaterTarget}'s implementation, which drives the approvals rating prompt + decision mapping
470
503
  * at a declared rung. The shape did not have to move to fit it.
471
504
  *
@@ -641,7 +674,7 @@ export interface EvalCase {
641
674
  */
642
675
  modelFree: boolean;
643
676
  }
644
- /** A fully parsed and validated suite — see {@link ../evalSuite.js}'s `parseEvalSuite`. */
677
+ /** A fully parsed and validated suite — see `evalSuite.js`'s `parseEvalSuite`. */
645
678
  export interface EvalSuite {
646
679
  target: EvalTarget;
647
680
  /**
@@ -669,11 +702,21 @@ export interface EvalSuite {
669
702
  * label enum has nothing to read).
670
703
  */
671
704
  metrics: EvalMetricSpec[];
705
+ /**
706
+ * BATCH-32 — the suite's `tool_coverage:` declaration: waivers, an optional floor, and required
707
+ * tools. Absent = coverage is still REPORTED (that is the whole point of the node: a run that
708
+ * exercises 3 of 41 tools must say so without being asked), but nothing gates on it.
709
+ *
710
+ * A top-level key of its own rather than a `metrics:` entry: a metric is a predicate over LABELLED
711
+ * cases and its machinery assumes a `classification:` block, whereas coverage applies to suites
712
+ * with no labels at all — most of them.
713
+ */
714
+ toolCoverage?: ToolCoverageSpec;
672
715
  /**
673
716
  * BATCH-25 — the config sweep: named cells the WHOLE suite is run once per, so one corpus
674
717
  * produces one comparison table instead of N unrelated runs. Absent = a single run.
675
718
  *
676
- * Consumed by the CLI, not by {@link ../evalRunner.js runEvalSuite}: a sweep is a run-level
719
+ * Consumed by the CLI, not by {@link "evalRunner.js"!runEvalSuite | runEvalSuite}: a sweep is a run-level
677
720
  * concept (same suite, different config) exactly like BATCH-19's multi-suite loop, so the runner
678
721
  * stays about grading and #405's identity matrix is untouched.
679
722
  */
@@ -698,6 +741,17 @@ export interface EvalSweepValue {
698
741
  model?: string;
699
742
  /** Plain-data config overrides deep-merged onto the resolved config (never `llm`). */
700
743
  config?: Record<string, unknown>;
744
+ /**
745
+ * [[BATCH-31]] — the cell's PROMPT ARM: which of gth's own preflight notes this cell's ratings go
746
+ * out without, so a note-on / note-off A/B is one suite file rather than two. `rater` targets
747
+ * only; `{ omit: [] }` is the baseline arm and is required rather than implied.
748
+ *
749
+ * **A third override kind, not a third spelling of `config:`.** It is deliberately not a config
750
+ * key and never merges into one: a config key is exactly what a session could also be given, and
751
+ * an omission a session can reach is a switch that suppresses safety context in the live approvals
752
+ * gate. See `raterPromptArm.ts` for the full argument and for how that is enforced.
753
+ */
754
+ notes?: RaterPromptArm;
701
755
  }
702
756
  /**
703
757
  * BATCH-25 — the sweep: named axes whose CARTESIAN PRODUCT is the set of runs. The approvals
@@ -719,7 +773,7 @@ export interface DeterministicCheckResult {
719
773
  failures: string[];
720
774
  }
721
775
  /** The judge's structured verdict on one case's answer — matches `review`'s `RateSchema` shape
722
- * (0-10 `rate` + a reason string) for UX consistency, see {@link ../judge.js}. */
776
+ * (0-10 `rate` + a reason string) for UX consistency, see `judge.js`. */
723
777
  export interface JudgeVerdict {
724
778
  rate: number;
725
779
  reason: string;
@@ -737,7 +791,7 @@ export interface JudgeOutcome {
737
791
  error?: string;
738
792
  }
739
793
  /** Injectable "grade one answer against one rubric" function — the seam that lets
740
- * {@link ../evalRunner.js}'s `runEvalSuite` be fully unit tested without any real LLM call, mirror
794
+ * `evalRunner.js`'s `runEvalSuite` be fully unit tested without any real LLM call, mirror
741
795
  * of BATCH-1's `RunCellFn`. The production wiring (`evalCommand.ts`) adapts `judgeEvalCase` to
742
796
  * this shape; tests inject a fake that resolves/fails as needed. */
743
797
  export type JudgeFn = (answer: string, rubric: string) => Promise<JudgeOutcome>;
@@ -830,5 +884,12 @@ export interface EvalSuiteSummary {
830
884
  /** BATCH-25 — the confusion matrices + declared metrics. Omitted entirely for a suite with no
831
885
  * `classification:` block, so a pre-BATCH-25 `results.json` is byte-for-byte unchanged. */
832
886
  classification?: EvalClassificationReport;
887
+ /**
888
+ * BATCH-32 — which of the agent's advertised tools this suite exercised. Omitted entirely when no
889
+ * cell reported an inventory: an external target, or a run whose SUT never initialised. That
890
+ * absence is the honest answer, and it is why this is optional rather than a zeroed block — a
891
+ * `0/0` coverage figure looks like a measurement and is not one.
892
+ */
893
+ toolCoverage?: ToolCoverageReport;
833
894
  }
834
895
  export {};
package/dist/evalTypes.js CHANGED
@@ -60,7 +60,7 @@ const PREFLIGHT_MECHANISM_OF = {
60
60
  * The floor is spelled here because it is not a preflight — it is a lexical scan core keeps no list
61
61
  * of, consulted by the approvals gate before any rating (every rung but `bypass`) and again by the
62
62
  * shell tool at execution time (every rung). The preflights are NOT spelled: they come from
63
- * {@link PREFLIGHT_MECHANISM_OF}, which is core's own `PREFLIGHT_FLOOR_KINDS` with our `forced_by`
63
+ * `PREFLIGHT_MECHANISM_OF`, which is core's own `PREFLIGHT_FLOOR_KINDS` with our `forced_by`
64
64
  * spelling attached.
65
65
  *
66
66
  * ## Why a case asserts a mechanism at all (the I1 finding)
@@ -132,7 +132,7 @@ export const FORCED_BY_ASSERTIONS = {
132
132
  export const PREFLIGHT_MECHANISMS = Object.values(PREFLIGHT_MECHANISM_OF);
133
133
  /**
134
134
  * The `forced_by` spelling of one of core's preflight arms — the typed lookup into
135
- * {@link PREFLIGHT_MECHANISM_OF}, which stays private so this is the only way in.
135
+ * `PREFLIGHT_MECHANISM_OF`, which stays private so this is the only way in.
136
136
  *
137
137
  * TOTAL over core's `PreflightFloorKind` and therefore never `undefined`, which is the point: a
138
138
  * caller cannot be handed an arm it has no name for, because such an arm would not have compiled.
@@ -1 +1 @@
1
- {"version":3,"file":"evalTypes.js","sourceRoot":"","sources":["../src/evalTypes.ts"],"names":[],"mappings":"AAqBA;;;;mGAImG;AACnG,MAAM,CAAC,MAAM,2BAA2B,GAAG,CAAC,CAAC;AAE7C;;;;;;;;;;;;;;;;;;;;;;;;GAwBG;AACH,MAAM,CAAC,MAAM,uBAAuB,GAAG,yBAAyB,CAAC;AAEjE;;;;;;;;;;;;;;;;;;GAkBG;AACH,MAAM,sBAAsB,GAAG;IAC7B,iBAAiB,EAAE,2BAA2B;IAC9C,YAAY,EAAE,sBAAsB;CAC2B,CAAC;AAElE;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG;IAClC,gBAAgB;IAChB,GAAG,MAAM,CAAC,MAAM,CAAC,sBAAsB,CAAC;CAChC,CAAC;AAiBX,mGAAmG;AACnG,MAAM,CAAC,MAAM,gBAAgB,GAAG,WAAW,CAAC;AAE5C;;;;GAIG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAsC;IACrE,gBAAgB,EAAE,uBAAuB;IACzC,2BAA2B,EAAE,GAAG,gBAAgB,6BAA6B;IAC7E,sBAAsB,EAAE,GAAG,gBAAgB,wBAAwB;CACpE,CAAC;AAEF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAsCG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG,MAAM,CAAC,MAAM,CAAC,sBAAsB,CAAC,CAAC;AAW1E;;;;;;GAMG;AACH,MAAM,UAAU,qBAAqB,CAAC,IAAwB;IAC5D,OAAO,sBAAsB,CAAC,IAAI,CAAC,CAAC;AACtC,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,8BAA8B,CAAC,SAAwC;IACrF,OAAO,CACL,SAAS,KAAK,SAAS;QACtB,oBAAqD,CAAC,QAAQ,CAAC,SAAS,CAAC,CAC3E,CAAC;AACJ,CAAC"}
1
+ {"version":3,"file":"evalTypes.js","sourceRoot":"","sources":["../src/evalTypes.ts"],"names":[],"mappings":"AA6BA;;;;mGAImG;AACnG,MAAM,CAAC,MAAM,2BAA2B,GAAG,CAAC,CAAC;AAE7C;;;;;;;;;;;;;;;;;;;;;;;;GAwBG;AACH,MAAM,CAAC,MAAM,uBAAuB,GAAG,yBAAyB,CAAC;AAEjE;;;;;;;;;;;;;;;;;;GAkBG;AACH,MAAM,sBAAsB,GAAG;IAC7B,iBAAiB,EAAE,2BAA2B;IAC9C,YAAY,EAAE,sBAAsB;CAC2B,CAAC;AAElE;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG;IAClC,gBAAgB;IAChB,GAAG,MAAM,CAAC,MAAM,CAAC,sBAAsB,CAAC;CAChC,CAAC;AAiBX,mGAAmG;AACnG,MAAM,CAAC,MAAM,gBAAgB,GAAG,WAAW,CAAC;AAE5C;;;;GAIG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAsC;IACrE,gBAAgB,EAAE,uBAAuB;IACzC,2BAA2B,EAAE,GAAG,gBAAgB,6BAA6B;IAC7E,sBAAsB,EAAE,GAAG,gBAAgB,wBAAwB;CACpE,CAAC;AAEF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAsCG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG,MAAM,CAAC,MAAM,CAAC,sBAAsB,CAAC,CAAC;AAW1E;;;;;;GAMG;AACH,MAAM,UAAU,qBAAqB,CAAC,IAAwB;IAC5D,OAAO,sBAAsB,CAAC,IAAI,CAAC,CAAC;AACtC,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,8BAA8B,CAAC,SAAwC;IACrF,OAAO,CACL,SAAS,KAAK,SAAS;QACtB,oBAAqD,CAAC,QAAQ,CAAC,SAAS,CAAC,CAC3E,CAAC;AACJ,CAAC"}
package/dist/index.d.ts CHANGED
@@ -18,15 +18,21 @@ export { extractClassificationValue, buildConfusionMatrix, collectTags, readRaw,
18
18
  export { buildClassificationReport } from '#src/classificationReport.js';
19
19
  export { computeMetric, parseMetricPredicate, evaluatePredicate, evaluatePredicates, formatTally, } from '#src/metrics.js';
20
20
  export { renderClassificationReport, renderConfusionMatrix, renderMetric, } from '#src/classificationRender.js';
21
+ export { computeToolCoverage, aggregateToolCoverage, gradeRunToolCoverage, hasRunToolCoverageSpec, WAIVER_SHARE_WARN_THRESHOLD, } from '#src/toolCoverage.js';
22
+ export type { AdvertisedToolInventory, AdvertisedToolRecord, RunToolCoverageGrade, RunToolCoverageSource, RunToolCoverageSpec, ToolCoverageInput, ToolCoverageReport, ToolCoverageServerReport, ToolCoverageSpec, } from '#src/toolCoverage.js';
23
+ export { renderToolCoverage } from '#src/toolCoverageRender.js';
24
+ export type { RenderToolCoverageOptions } from '#src/toolCoverageRender.js';
21
25
  export { buildBlindExport, diffRelabel, renderRelabelDiff } from '#src/blindExport.js';
22
26
  export type { BlindExport, BlindExportCase, RelabelDiff, RelabelEntry } from '#src/blindExport.js';
23
- export { expandSweep, deepMerge, renderComparison, diffRuns, renderRunDiff, } from '#src/evalCompare.js';
24
- export type { ComparisonColumn, RunDiff, RunDiffEntry, SweepCell } from '#src/evalCompare.js';
27
+ export { expandSweep, deepMerge, renderComparison, diffRuns, renderRunDiff, parseJudgeDriftFilter, DEFAULT_JUDGE_DRIFT_FILTER, DEFAULT_JUDGE_DRIFT_TOLERANCE, MAX_JUDGE_DRIFT_TOLERANCE, } from '#src/evalCompare.js';
28
+ export type { ComparisonColumn, JudgeDriftEntry, JudgeDriftFilter, JudgeDriftMean, RunDiff, RunDiffEntry, SweepCell, } from '#src/evalCompare.js';
25
29
  export { UNRECOGNIZED_LABEL, NO_EXPECTATION } from '#src/classificationTypes.js';
26
30
  export type { ClassificationExtractor, ClassifiedCell, EvalCaseClassification, EvalClassificationReport, EvalClassificationSpec, EvalConfusionMatrix, EvalMetricCoverage, EvalMetricResult, EvalMetricSpec, EvalMetricTally, MetricField, MetricPredicate, } from '#src/classificationTypes.js';
27
31
  export type { ClassifyOutcome, ClassifyRequest, ClassifyRound, EvalSweep, EvalSweepValue, RaterTarget, RunClassifyFn, } from '#src/evalTypes.js';
28
32
  export { buildRaterClassifier, HARDLINE_NO_RATING_REASON, HARDLINE_REFUSAL_MARKER, NEGOTIATION_BOUND_MARKER, NO_RATING_CALL_MARKER, } from '#src/raterTarget.js';
29
33
  export type { RaterClassifierOptions } from '#src/raterTarget.js';
34
+ export { RATER_PROMPT_NOTE_NAMES } from '#src/raterPromptArm.js';
35
+ export type { RaterPromptArm, RaterPromptNoteName } from '#src/raterPromptArm.js';
30
36
  export { resolveReporters, availableReporterNames } from '#src/reporters/registry.js';
31
37
  export { driveReporters } from '#src/reporters/drive.js';
32
38
  export { createTextReporter } from '#src/reporters/textReporter.js';
package/dist/index.js CHANGED
@@ -17,12 +17,20 @@ export { extractClassificationValue, buildConfusionMatrix, collectTags, readRaw,
17
17
  export { buildClassificationReport } from '#src/classificationReport.js';
18
18
  export { computeMetric, parseMetricPredicate, evaluatePredicate, evaluatePredicates, formatTally, } from '#src/metrics.js';
19
19
  export { renderClassificationReport, renderConfusionMatrix, renderMetric, } from '#src/classificationRender.js';
20
+ // BATCH-32 — tool coverage: which of the agent's advertised tools a suite exercised.
21
+ export { computeToolCoverage, aggregateToolCoverage, gradeRunToolCoverage, hasRunToolCoverageSpec, WAIVER_SHARE_WARN_THRESHOLD, } from '#src/toolCoverage.js';
22
+ export { renderToolCoverage } from '#src/toolCoverageRender.js';
20
23
  export { buildBlindExport, diffRelabel, renderRelabelDiff } from '#src/blindExport.js';
21
- export { expandSweep, deepMerge, renderComparison, diffRuns, renderRunDiff, } from '#src/evalCompare.js';
24
+ export { expandSweep, deepMerge, renderComparison, diffRuns, renderRunDiff, parseJudgeDriftFilter, DEFAULT_JUDGE_DRIFT_FILTER, DEFAULT_JUDGE_DRIFT_TOLERANCE, MAX_JUDGE_DRIFT_TOLERANCE, } from '#src/evalCompare.js';
22
25
  export { UNRECOGNIZED_LABEL, NO_EXPECTATION } from '#src/classificationTypes.js';
23
26
  // BATCH-25 Half B — the `rater` target: the one implementation of the `RunClassifyFn` seam, which
24
27
  // drives gth's own approvals rater over a corpus of commands.
25
28
  export { buildRaterClassifier, HARDLINE_NO_RATING_REASON, HARDLINE_REFUSAL_MARKER, NEGOTIATION_BOUND_MARKER, NO_RATING_CALL_MARKER, } from '#src/raterTarget.js';
29
+ // [[BATCH-31]] — the prompt arm: the note-on / note-off A/B a rater suite can express. Only the
30
+ // NAMES and the type are exported. `armRaterModel` deliberately is not — nothing outside this
31
+ // package needs to build one, and the narrower the surface the smaller the chance of a caller
32
+ // wiring an omission somewhere the argument in `raterPromptArm.ts` does not cover.
33
+ export { RATER_PROMPT_NOTE_NAMES } from '#src/raterPromptArm.js';
26
34
  // BATCH-19 — the `gth eval` reporter facility (A1 seam). These are the public plugin contract an
27
35
  // out-of-core `@gaunt-sloth/eval-reporter-*` package implements, exported from the package root so a
28
36
  // reporter package can type its factory against ONE import.
package/dist/index.js.map CHANGED
@@ -1 +1 @@
1
- {"version":3,"file":"index.js","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,WAAW,EAAE,MAAM,gBAAgB,CAAC;AAC7C,OAAO,EAAE,eAAe,EAAE,MAAM,qBAAqB,CAAC;AACtD,OAAO,EAAE,aAAa,EAAE,MAAM,mBAAmB,CAAC;AAClD,OAAO,EAAE,cAAc,EAAE,iBAAiB,EAAE,MAAM,qBAAqB,CAAC;AACxE,OAAO,EAAE,gBAAgB,EAAE,MAAM,gBAAgB,CAAC;AAWlD,OAAO,EAAE,wBAAwB,EAAE,MAAM,eAAe,CAAC;AAEzD,gGAAgG;AAChG,OAAO,EAAE,cAAc,EAAE,MAAM,mBAAmB,CAAC;AACnD,OAAO,EAAE,sBAAsB,EAAE,MAAM,6BAA6B,CAAC;AACrE,OAAO,EACL,aAAa,EACb,kBAAkB,EAClB,iBAAiB,EACjB,6BAA6B,GAC9B,MAAM,eAAe,CAAC;AAEvB,OAAO,EAAE,YAAY,EAAE,MAAM,oBAAoB,CAAC;AAClD,OAAO,EAAE,eAAe,EAAE,MAAM,oBAAoB,CAAC;AAqBrD,OAAO,EAAE,2BAA2B,EAAE,MAAM,mBAAmB,CAAC;AAEhE,mGAAmG;AACnG,8FAA8F;AAC9F,OAAO,EACL,0BAA0B,EAC1B,oBAAoB,EACpB,WAAW,EACX,OAAO,GACR,MAAM,wBAAwB,CAAC;AAChC,OAAO,EAAE,yBAAyB,EAAE,MAAM,8BAA8B,CAAC;AACzE,OAAO,EACL,aAAa,EACb,oBAAoB,EACpB,iBAAiB,EACjB,kBAAkB,EAClB,WAAW,GACZ,MAAM,iBAAiB,CAAC;AACzB,OAAO,EACL,0BAA0B,EAC1B,qBAAqB,EACrB,YAAY,GACb,MAAM,8BAA8B,CAAC;AACtC,OAAO,EAAE,gBAAgB,EAAE,WAAW,EAAE,iBAAiB,EAAE,MAAM,qBAAqB,CAAC;AAEvF,OAAO,EACL,WAAW,EACX,SAAS,EACT,gBAAgB,EAChB,QAAQ,EACR,aAAa,GACd,MAAM,qBAAqB,CAAC;AAE7B,OAAO,EAAE,kBAAkB,EAAE,cAAc,EAAE,MAAM,6BAA6B,CAAC;AAyBjF,kGAAkG;AAClG,8DAA8D;AAC9D,OAAO,EACL,oBAAoB,EACpB,yBAAyB,EACzB,uBAAuB,EACvB,wBAAwB,EACxB,qBAAqB,GACtB,MAAM,qBAAqB,CAAC;AAG7B,iGAAiG;AACjG,qGAAqG;AACrG,4DAA4D;AAC5D,OAAO,EAAE,gBAAgB,EAAE,sBAAsB,EAAE,MAAM,4BAA4B,CAAC;AACtF,OAAO,EAAE,cAAc,EAAE,MAAM,yBAAyB,CAAC;AACzD,OAAO,EAAE,kBAAkB,EAAE,MAAM,gCAAgC,CAAC;AAQpE,kGAAkG;AAClG,6CAA6C;AAC7C,OAAO,EAAE,WAAW,EAAE,MAAM,8BAA8B,CAAC;AAC3D,OAAO,EAAE,4BAA4B,EAAE,MAAM,eAAe,CAAC"}
1
+ {"version":3,"file":"index.js","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,WAAW,EAAE,MAAM,gBAAgB,CAAC;AAC7C,OAAO,EAAE,eAAe,EAAE,MAAM,qBAAqB,CAAC;AACtD,OAAO,EAAE,aAAa,EAAE,MAAM,mBAAmB,CAAC;AAClD,OAAO,EAAE,cAAc,EAAE,iBAAiB,EAAE,MAAM,qBAAqB,CAAC;AACxE,OAAO,EAAE,gBAAgB,EAAE,MAAM,gBAAgB,CAAC;AAWlD,OAAO,EAAE,wBAAwB,EAAE,MAAM,eAAe,CAAC;AAEzD,gGAAgG;AAChG,OAAO,EAAE,cAAc,EAAE,MAAM,mBAAmB,CAAC;AACnD,OAAO,EAAE,sBAAsB,EAAE,MAAM,6BAA6B,CAAC;AACrE,OAAO,EACL,aAAa,EACb,kBAAkB,EAClB,iBAAiB,EACjB,6BAA6B,GAC9B,MAAM,eAAe,CAAC;AAEvB,OAAO,EAAE,YAAY,EAAE,MAAM,oBAAoB,CAAC;AAClD,OAAO,EAAE,eAAe,EAAE,MAAM,oBAAoB,CAAC;AAqBrD,OAAO,EAAE,2BAA2B,EAAE,MAAM,mBAAmB,CAAC;AAEhE,mGAAmG;AACnG,8FAA8F;AAC9F,OAAO,EACL,0BAA0B,EAC1B,oBAAoB,EACpB,WAAW,EACX,OAAO,GACR,MAAM,wBAAwB,CAAC;AAChC,OAAO,EAAE,yBAAyB,EAAE,MAAM,8BAA8B,CAAC;AACzE,OAAO,EACL,aAAa,EACb,oBAAoB,EACpB,iBAAiB,EACjB,kBAAkB,EAClB,WAAW,GACZ,MAAM,iBAAiB,CAAC;AACzB,OAAO,EACL,0BAA0B,EAC1B,qBAAqB,EACrB,YAAY,GACb,MAAM,8BAA8B,CAAC;AACtC,qFAAqF;AACrF,OAAO,EACL,mBAAmB,EACnB,qBAAqB,EACrB,oBAAoB,EACpB,sBAAsB,EACtB,2BAA2B,GAC5B,MAAM,sBAAsB,CAAC;AAY9B,OAAO,EAAE,kBAAkB,EAAE,MAAM,4BAA4B,CAAC;AAEhE,OAAO,EAAE,gBAAgB,EAAE,WAAW,EAAE,iBAAiB,EAAE,MAAM,qBAAqB,CAAC;AAEvF,OAAO,EACL,WAAW,EACX,SAAS,EACT,gBAAgB,EAChB,QAAQ,EACR,aAAa,EACb,qBAAqB,EACrB,0BAA0B,EAC1B,6BAA6B,EAC7B,yBAAyB,GAC1B,MAAM,qBAAqB,CAAC;AAU7B,OAAO,EAAE,kBAAkB,EAAE,cAAc,EAAE,MAAM,6BAA6B,CAAC;AAyBjF,kGAAkG;AAClG,8DAA8D;AAC9D,OAAO,EACL,oBAAoB,EACpB,yBAAyB,EACzB,uBAAuB,EACvB,wBAAwB,EACxB,qBAAqB,GACtB,MAAM,qBAAqB,CAAC;AAG7B,gGAAgG;AAChG,8FAA8F;AAC9F,8FAA8F;AAC9F,mFAAmF;AACnF,OAAO,EAAE,uBAAuB,EAAE,MAAM,wBAAwB,CAAC;AAGjE,iGAAiG;AACjG,qGAAqG;AACrG,4DAA4D;AAC5D,OAAO,EAAE,gBAAgB,EAAE,sBAAsB,EAAE,MAAM,4BAA4B,CAAC;AACtF,OAAO,EAAE,cAAc,EAAE,MAAM,yBAAyB,CAAC;AACzD,OAAO,EAAE,kBAAkB,EAAE,MAAM,gCAAgC,CAAC;AAQpE,kGAAkG;AAClG,6CAA6C;AAC7C,OAAO,EAAE,WAAW,EAAE,MAAM,8BAA8B,CAAC;AAC3D,OAAO,EAAE,4BAA4B,EAAE,MAAM,eAAe,CAAC"}
package/dist/judge.d.ts CHANGED
@@ -1,6 +1,4 @@
1
1
  /**
2
- * @module judge
3
- *
4
2
  * BATCH-2 — LLM-as-judge grading for `gth eval`. Adapts the *mechanism* of EXT-10's shell-safety
5
3
  * judge (`packages/core/src/core/shell/judge.ts`): `model.withStructuredOutput(zodSchema)` for a
6
4
  * single non-agentic structured call, raced against a timeout. The failure policy differs on
@@ -18,6 +16,8 @@
18
16
  * `judgeShellCommand`'s own default. A separate `--judge <profile>` model is BATCH-2's own
19
17
  * "Not in scope" list (identity-matrix/pluggable-target work); grading with the SUT's own model
20
18
  * config is a known, real simplification for this first slice.
19
+ *
20
+ * @module
21
21
  */
22
22
  import type { BaseChatModel } from '@langchain/core/language_models/chat_models';
23
23
  import * as z from 'zod';
package/dist/judge.js CHANGED
@@ -1,6 +1,4 @@
1
1
  /**
2
- * @module judge
3
- *
4
2
  * BATCH-2 — LLM-as-judge grading for `gth eval`. Adapts the *mechanism* of EXT-10's shell-safety
5
3
  * judge (`packages/core/src/core/shell/judge.ts`): `model.withStructuredOutput(zodSchema)` for a
6
4
  * single non-agentic structured call, raced against a timeout. The failure policy differs on
@@ -18,8 +16,11 @@
18
16
  * `judgeShellCommand`'s own default. A separate `--judge <profile>` model is BATCH-2's own
19
17
  * "Not in scope" list (identity-matrix/pluggable-target work); grading with the SUT's own model
20
18
  * config is a known, real simplification for this first slice.
19
+ *
20
+ * @module
21
21
  */
22
22
  import { HumanMessage, SystemMessage } from '@langchain/core/messages';
23
+ import { CALL_TIMED_OUT, withCallDeadline } from '@gaunt-sloth/core/runtime/abortableCall.js';
23
24
  import { structuredOutputBoundary } from '@gaunt-sloth/core/runtime/structuredOutput.js';
24
25
  import * as z from 'zod';
25
26
  /** Structured verdict the judge model must return — 0-10 `rate` + a `reason`, matching `review`'s
@@ -85,20 +86,19 @@ export async function judgeEvalCase(answer, rubric, model, options) {
85
86
  return { attempted: true, ok: false, error: 'No usable judge model configured.' };
86
87
  }
87
88
  const { system, user } = buildJudgeMessages(answer, rubric);
88
- let timer;
89
89
  try {
90
90
  // EXT-88 — routed through the shared boundary like every other structured-output call. The
91
91
  // rubric verdict has no optional field today, so the boundary hands back this very schema by
92
92
  // identity; going through it is what stops a later optional field re-introducing the defect.
93
93
  const boundary = structuredOutputBoundary(EvalVerdictSchema);
94
94
  const structured = model.withStructuredOutput(boundary.wireSchema);
95
- const judgePromise = structured.invoke([new SystemMessage(system), new HumanMessage(user)]);
96
- const TIMEOUT = Symbol('eval-judge-timeout');
97
- const timeoutPromise = new Promise((resolve) => {
98
- timer = setTimeout(() => resolve(TIMEOUT), timeoutMs);
99
- });
100
- const raced = await Promise.race([judgePromise, timeoutPromise]);
101
- if (raced === TIMEOUT) {
95
+ // [[EXT-179]] — the budget aborts the judge call rather than abandoning it. This matters most
96
+ // in a SWEEP: a suite grades many cases, and every stalled judge used to leave its own request
97
+ // in flight, so the leaked sockets accumulated across the run and kept the process alive after
98
+ // the report had been written. The timeout text is unchanged and the case still fails as a
99
+ // case — an aborted judgement is a judgement not obtained, never an auto-pass.
100
+ const raced = await withCallDeadline(timeoutMs, (signal) => structured.invoke([new SystemMessage(system), new HumanMessage(user)], { signal }));
101
+ if (raced === CALL_TIMED_OUT) {
102
102
  return { attempted: true, ok: false, error: `Judge timed out after ${timeoutMs}ms.` };
103
103
  }
104
104
  // withStructuredOutput already coerces to the schema, but re-validate defensively: a fake or
@@ -116,9 +116,5 @@ export async function judgeEvalCase(answer, rubric, model, options) {
116
116
  error: error instanceof Error ? error.message : String(error),
117
117
  };
118
118
  }
119
- finally {
120
- if (timer)
121
- clearTimeout(timer);
122
- }
123
119
  }
124
120
  //# sourceMappingURL=judge.js.map
package/dist/judge.js.map CHANGED
@@ -1 +1 @@
1
- {"version":3,"file":"judge.js","sourceRoot":"","sources":["../src/judge.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAGH,OAAO,EAAE,YAAY,EAAE,aAAa,EAAE,MAAM,0BAA0B,CAAC;AACvE,OAAO,EAAE,wBAAwB,EAAE,MAAM,+CAA+C,CAAC;AACzF,OAAO,KAAK,CAAC,MAAM,KAAK,CAAC;AAIzB;;oGAEoG;AACpG,MAAM,CAAC,MAAM,iBAAiB,GAAG,CAAC,CAAC,MAAM,CAAC;IACxC,IAAI,EAAE,CAAC;SACJ,MAAM,EAAE;SACR,GAAG,CAAC,CAAC,CAAC;SACN,GAAG,CAAC,EAAE,CAAC;SACP,QAAQ,CAAC,yDAAyD,CAAC;IACtE,MAAM,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,QAAQ,CAAC,2CAA2C,CAAC;CACzE,CAAC,CAAC;AAIH;;oGAEoG;AACpG,MAAM,CAAC,MAAM,6BAA6B,GAAG,MAAM,CAAC;AAEpD,MAAM,wBAAwB,GAAG;IAC/B,mCAAmC;IACnC,gGAAgG;IAChG,cAAc;IACd,EAAE;IACF,+FAA+F;IAC/F,gGAAgG;IAChG,mDAAmD;IACnD,6FAA6F;IAC7F,+BAA+B;CAChC,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AAEb;;;;GAIG;AACH,MAAM,UAAU,kBAAkB,CAChC,MAAc,EACd,MAAc;IAEd,MAAM,SAAS,GAAG;QAChB,SAAS;QACT,MAAM;QACN,EAAE;QACF,qBAAqB;QACrB,UAAU;QACV,MAAM,IAAI,gBAAgB;QAC1B,WAAW;KACZ,CAAC;IACF,OAAO,EAAE,MAAM,EAAE,wBAAwB,EAAE,IAAI,EAAE,SAAS,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;AAC1E,CAAC;AAED;;;;;;;;;;;;;GAaG;AACH,MAAM,CAAC,KAAK,UAAU,aAAa,CACjC,MAAc,EACd,MAAc,EACd,KAAgC,EAChC,OAAgC;IAEhC,MAAM,SAAS,GAAG,OAAO,EAAE,SAAS,IAAI,6BAA6B,CAAC;IAEtE,IAAI,CAAC,KAAK,IAAI,OAAO,KAAK,CAAC,oBAAoB,KAAK,UAAU,EAAE,CAAC;QAC/D,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,mCAAmC,EAAE,CAAC;IACpF,CAAC;IAED,MAAM,EAAE,MAAM,EAAE,IAAI,EAAE,GAAG,kBAAkB,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IAC5D,IAAI,KAAgD,CAAC;IACrD,IAAI,CAAC;QACH,2FAA2F;QAC3F,6FAA6F;QAC7F,6FAA6F;QAC7F,MAAM,QAAQ,GAAG,wBAAwB,CAAC,iBAAiB,CAAC,CAAC;QAC7D,MAAM,UAAU,GAAG,KAAK,CAAC,oBAAoB,CAAC,QAAQ,CAAC,UAAU,CAAC,CAAC;QACnE,MAAM,YAAY,GAAG,UAAU,CAAC,MAAM,CAAC,CAAC,IAAI,aAAa,CAAC,MAAM,CAAC,EAAE,IAAI,YAAY,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC;QAE5F,MAAM,OAAO,GAAG,MAAM,CAAC,oBAAoB,CAAC,CAAC;QAC7C,MAAM,cAAc,GAAG,IAAI,OAAO,CAAiB,CAAC,OAAO,EAAE,EAAE;YAC7D,KAAK,GAAG,UAAU,CAAC,GAAG,EAAE,CAAC,OAAO,CAAC,OAAO,CAAC,EAAE,SAAS,CAAC,CAAC;QACxD,CAAC,CAAC,CAAC;QAEH,MAAM,KAAK,GAAG,MAAM,OAAO,CAAC,IAAI,CAAC,CAAC,YAAY,EAAE,cAAc,CAAC,CAAC,CAAC;QACjE,IAAI,KAAK,KAAK,OAAO,EAAE,CAAC;YACtB,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,yBAAyB,SAAS,KAAK,EAAE,CAAC;QACxF,CAAC;QAED,6FAA6F;QAC7F,0DAA0D;QAC1D,MAAM,MAAM,GAAG,QAAQ,CAAC,SAAS,CAAC,KAAK,CAAC,CAAC;QACzC,IAAI,CAAC,MAAM,CAAC,OAAO,EAAE,CAAC;YACpB,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,oCAAoC,EAAE,CAAC;QACrF,CAAC;QACD,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,IAAI,EAAE,OAAO,EAAE,MAAM,CAAC,IAAI,EAAE,CAAC;IAC7D,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QACf,OAAO;YACL,SAAS,EAAE,IAAI;YACf,EAAE,EAAE,KAAK;YACT,KAAK,EAAE,KAAK,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC;SAC9D,CAAC;IACJ,CAAC;YAAS,CAAC;QACT,IAAI,KAAK;YAAE,YAAY,CAAC,KAAK,CAAC,CAAC;IACjC,CAAC;AACH,CAAC"}
1
+ {"version":3,"file":"judge.js","sourceRoot":"","sources":["../src/judge.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAGH,OAAO,EAAE,YAAY,EAAE,aAAa,EAAE,MAAM,0BAA0B,CAAC;AACvE,OAAO,EAAE,cAAc,EAAE,gBAAgB,EAAE,MAAM,4CAA4C,CAAC;AAC9F,OAAO,EAAE,wBAAwB,EAAE,MAAM,+CAA+C,CAAC;AACzF,OAAO,KAAK,CAAC,MAAM,KAAK,CAAC;AAIzB;;oGAEoG;AACpG,MAAM,CAAC,MAAM,iBAAiB,GAAG,CAAC,CAAC,MAAM,CAAC;IACxC,IAAI,EAAE,CAAC;SACJ,MAAM,EAAE;SACR,GAAG,CAAC,CAAC,CAAC;SACN,GAAG,CAAC,EAAE,CAAC;SACP,QAAQ,CAAC,yDAAyD,CAAC;IACtE,MAAM,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,QAAQ,CAAC,2CAA2C,CAAC;CACzE,CAAC,CAAC;AAIH;;oGAEoG;AACpG,MAAM,CAAC,MAAM,6BAA6B,GAAG,MAAM,CAAC;AAEpD,MAAM,wBAAwB,GAAG;IAC/B,mCAAmC;IACnC,gGAAgG;IAChG,cAAc;IACd,EAAE;IACF,+FAA+F;IAC/F,gGAAgG;IAChG,mDAAmD;IACnD,6FAA6F;IAC7F,+BAA+B;CAChC,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AAEb;;;;GAIG;AACH,MAAM,UAAU,kBAAkB,CAChC,MAAc,EACd,MAAc;IAEd,MAAM,SAAS,GAAG;QAChB,SAAS;QACT,MAAM;QACN,EAAE;QACF,qBAAqB;QACrB,UAAU;QACV,MAAM,IAAI,gBAAgB;QAC1B,WAAW;KACZ,CAAC;IACF,OAAO,EAAE,MAAM,EAAE,wBAAwB,EAAE,IAAI,EAAE,SAAS,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;AAC1E,CAAC;AAED;;;;;;;;;;;;;GAaG;AACH,MAAM,CAAC,KAAK,UAAU,aAAa,CACjC,MAAc,EACd,MAAc,EACd,KAAgC,EAChC,OAAgC;IAEhC,MAAM,SAAS,GAAG,OAAO,EAAE,SAAS,IAAI,6BAA6B,CAAC;IAEtE,IAAI,CAAC,KAAK,IAAI,OAAO,KAAK,CAAC,oBAAoB,KAAK,UAAU,EAAE,CAAC;QAC/D,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,mCAAmC,EAAE,CAAC;IACpF,CAAC;IAED,MAAM,EAAE,MAAM,EAAE,IAAI,EAAE,GAAG,kBAAkB,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IAC5D,IAAI,CAAC;QACH,2FAA2F;QAC3F,6FAA6F;QAC7F,6FAA6F;QAC7F,MAAM,QAAQ,GAAG,wBAAwB,CAAC,iBAAiB,CAAC,CAAC;QAC7D,MAAM,UAAU,GAAG,KAAK,CAAC,oBAAoB,CAAC,QAAQ,CAAC,UAAU,CAAC,CAAC;QAEnE,8FAA8F;QAC9F,+FAA+F;QAC/F,+FAA+F;QAC/F,2FAA2F;QAC3F,+EAA+E;QAC/E,MAAM,KAAK,GAAG,MAAM,gBAAgB,CAAC,SAAS,EAAE,CAAC,MAAM,EAAE,EAAE,CACzD,UAAU,CAAC,MAAM,CAAC,CAAC,IAAI,aAAa,CAAC,MAAM,CAAC,EAAE,IAAI,YAAY,CAAC,IAAI,CAAC,CAAC,EAAE,EAAE,MAAM,EAAE,CAAC,CACnF,CAAC;QACF,IAAI,KAAK,KAAK,cAAc,EAAE,CAAC;YAC7B,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,yBAAyB,SAAS,KAAK,EAAE,CAAC;QACxF,CAAC;QAED,6FAA6F;QAC7F,0DAA0D;QAC1D,MAAM,MAAM,GAAG,QAAQ,CAAC,SAAS,CAAC,KAAK,CAAC,CAAC;QACzC,IAAI,CAAC,MAAM,CAAC,OAAO,EAAE,CAAC;YACpB,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,oCAAoC,EAAE,CAAC;QACrF,CAAC;QACD,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,IAAI,EAAE,OAAO,EAAE,MAAM,CAAC,IAAI,EAAE,CAAC;IAC7D,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QACf,OAAO;YACL,SAAS,EAAE,IAAI;YACf,EAAE,EAAE,KAAK;YACT,KAAK,EAAE,KAAK,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC;SAC9D,CAAC;IACJ,CAAC;AACH,CAAC"}
package/dist/output.d.ts CHANGED
@@ -4,7 +4,7 @@ import type { BatchSummary, CellResult } from '#src/types.js';
4
4
  * (pass/fail counts + a per-cell one-liner — a lightweight flake report) into `outputDir`.
5
5
  * Creates `outputDir` (and any missing parents) if it doesn't exist.
6
6
  *
7
- * Pure I/O, deliberately separate from {@link runBatchMatrix}: the runner never touches the
7
+ * Pure I/O, deliberately separate from {@link @gaunt-sloth/batch!"BatchRunner.js".runBatchMatrix | runBatchMatrix}: the runner never touches the
8
8
  * filesystem, so unit tests can exercise matrix/concurrency/retry logic without a tmp dir, and this
9
9
  * function can be tested in isolation with a fixed set of results.
10
10
  *
package/dist/output.js CHANGED
@@ -6,7 +6,7 @@ import { buildBatchSummary } from '#src/BatchRunner.js';
6
6
  * (pass/fail counts + a per-cell one-liner — a lightweight flake report) into `outputDir`.
7
7
  * Creates `outputDir` (and any missing parents) if it doesn't exist.
8
8
  *
9
- * Pure I/O, deliberately separate from {@link runBatchMatrix}: the runner never touches the
9
+ * Pure I/O, deliberately separate from {@link @gaunt-sloth/batch!"BatchRunner.js".runBatchMatrix | runBatchMatrix}: the runner never touches the
10
10
  * filesystem, so unit tests can exercise matrix/concurrency/retry logic without a tmp dir, and this
11
11
  * function can be tested in isolation with a fixed set of results.
12
12
  *
@@ -4,7 +4,7 @@ import type { MatrixRow } from '#src/types.js';
4
4
  * `.jsonl`/`.ndjson` → one JSON object per line; anything else (including `.csv`) → CSV.
5
5
  *
6
6
  * This is content binding only (BATCH-1 scope): every row becomes an object of string fields that
7
- * {@link bindCellContent} interpolates into the script. A glob-of-binary-files path binding is out
7
+ * {@link @gaunt-sloth/batch!"interpolate.js".bindCellContent | bindCellContent} interpolates into the script. A glob-of-binary-files path binding is out
8
8
  * of scope for this task.
9
9
  *
10
10
  * Throws a descriptive `Error` on malformed input — the harness-level failure the CLI surface doc
package/dist/parseOver.js CHANGED
@@ -3,7 +3,7 @@
3
3
  * `.jsonl`/`.ndjson` → one JSON object per line; anything else (including `.csv`) → CSV.
4
4
  *
5
5
  * This is content binding only (BATCH-1 scope): every row becomes an object of string fields that
6
- * {@link bindCellContent} interpolates into the script. A glob-of-binary-files path binding is out
6
+ * {@link @gaunt-sloth/batch!"interpolate.js".bindCellContent | bindCellContent} interpolates into the script. A glob-of-binary-files path binding is out
7
7
  * of scope for this task.
8
8
  *
9
9
  * Throws a descriptive `Error` on malformed input — the harness-level failure the CLI surface doc
@@ -1,6 +1,4 @@
1
1
  /**
2
- * @module pipelineCli
3
- *
4
2
  * BATCH-9 — the standalone `gth-batch` pipeline runner. A thin entry point that runs the BATCH-1
5
3
  * matrix runtime ({@link buildMatrix} + {@link runBatchMatrix}) directly from a shell pipeline,
6
4
  * without pulling in the whole `gaunt-sloth` app. It takes a prompt-executable script + `--over`
@@ -16,10 +14,12 @@
16
14
  * the pipeline shell already handles) and output is JSONL on stdout (not a directory of files).
17
15
  *
18
16
  * stdout discipline: the run itself is noisy (the runtime's `display()`/`ProgressIndicator`/token
19
- * streaming all target `process.stdout`). The bin entry ({@link file://./bin.ts}) redirects
17
+ * streaming all target `process.stdout`). The bin entry ({@link @gaunt-sloth/batch!"bin.js" | ./bin.ts}) redirects
20
18
  * `process.stdout.write` to stderr for the duration and this module writes the machine JSONL
21
19
  * straight to fd 1 (`fs.writeSync`), so stdout stays a clean data channel — the same "protocol
22
20
  * channel" discipline `packages/app/cli.js` uses for ACP.
21
+ *
22
+ * @module
23
23
  */
24
24
  import type { CommandLineConfigOverrides, GthConfig } from '@gaunt-sloth/core/config.js';
25
25
  import type { BatchSummary, MatrixRow, RunCellFn } from '#src/types.js';
@@ -1,6 +1,4 @@
1
1
  /**
2
- * @module pipelineCli
3
- *
4
2
  * BATCH-9 — the standalone `gth-batch` pipeline runner. A thin entry point that runs the BATCH-1
5
3
  * matrix runtime ({@link buildMatrix} + {@link runBatchMatrix}) directly from a shell pipeline,
6
4
  * without pulling in the whole `gaunt-sloth` app. It takes a prompt-executable script + `--over`
@@ -16,10 +14,12 @@
16
14
  * the pipeline shell already handles) and output is JSONL on stdout (not a directory of files).
17
15
  *
18
16
  * stdout discipline: the run itself is noisy (the runtime's `display()`/`ProgressIndicator`/token
19
- * streaming all target `process.stdout`). The bin entry ({@link file://./bin.ts}) redirects
17
+ * streaming all target `process.stdout`). The bin entry ({@link @gaunt-sloth/batch!"bin.js" | ./bin.ts}) redirects
20
18
  * `process.stdout.write` to stderr for the duration and this module writes the machine JSONL
21
19
  * straight to fd 1 (`fs.writeSync`), so stdout stays a clean data channel — the same "protocol
22
20
  * channel" discipline `packages/app/cli.js` uses for ACP.
21
+ *
22
+ * @module
23
23
  */
24
24
  import { writeSync } from 'node:fs';
25
25
  import { readFileSync } from 'node:fs';