@gaunt-sloth/batch 2.0.0-beta.12 → 2.0.0-beta.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/dist/bin.d.ts +2 -2
  2. package/dist/bin.js +2 -2
  3. package/dist/classificationReport.d.ts +2 -2
  4. package/dist/classificationReport.js +2 -2
  5. package/dist/classificationTypes.d.ts +7 -7
  6. package/dist/classificationTypes.js +1 -1
  7. package/dist/evalCompare.d.ts +11 -0
  8. package/dist/evalCompare.js +7 -0
  9. package/dist/evalCompare.js.map +1 -1
  10. package/dist/evalOutput.d.ts +1 -1
  11. package/dist/evalOutput.js +1 -1
  12. package/dist/evalRunner.d.ts +3 -3
  13. package/dist/evalRunner.js +25 -4
  14. package/dist/evalRunner.js.map +1 -1
  15. package/dist/evalSuite.d.ts +5 -1
  16. package/dist/evalSuite.js +152 -6
  17. package/dist/evalSuite.js.map +1 -1
  18. package/dist/evalTypes.d.ts +55 -18
  19. package/dist/evalTypes.js +2 -2
  20. package/dist/evalTypes.js.map +1 -1
  21. package/dist/index.d.ts +2 -0
  22. package/dist/index.js +5 -0
  23. package/dist/index.js.map +1 -1
  24. package/dist/judge.d.ts +2 -2
  25. package/dist/judge.js +10 -14
  26. package/dist/judge.js.map +1 -1
  27. package/dist/output.d.ts +1 -1
  28. package/dist/output.js +1 -1
  29. package/dist/parseOver.d.ts +1 -1
  30. package/dist/parseOver.js +1 -1
  31. package/dist/pipelineCli.d.ts +3 -3
  32. package/dist/pipelineCli.js +3 -3
  33. package/dist/raterPromptArm.d.ts +140 -0
  34. package/dist/raterPromptArm.js +306 -0
  35. package/dist/raterPromptArm.js.map +1 -0
  36. package/dist/raterTarget.d.ts +34 -12
  37. package/dist/raterTarget.js +87 -27
  38. package/dist/raterTarget.js.map +1 -1
  39. package/dist/reporters/registry.d.ts +1 -1
  40. package/dist/reporters/registry.js +1 -1
  41. package/dist/reporters/registry.js.map +1 -1
  42. package/dist/reporters/reporterTypes.d.ts +3 -3
  43. package/dist/types.d.ts +2 -2
  44. package/dist/workflow/runWorkflow.d.ts +4 -4
  45. package/dist/workflow/runWorkflow.js +3 -3
  46. package/package.json +3 -3
@@ -1,13 +1,14 @@
1
1
  /**
2
2
  * @packageDocumentation
3
3
  * BATCH-2 — the shapes for `gth eval`: a parsed suite/case, deterministic-check results, the
4
- * judge's verdict, and per-case/suite outcomes. Deliberately separate from {@link ../types.js}
4
+ * judge's verdict, and per-case/suite outcomes. Deliberately separate from `types.js`
5
5
  * (BATCH-1's cell/outcome shapes), which documents itself as scoped to "cells and outcomes" only
6
6
  * — eval's shapes layer on top of (not into) that file.
7
7
  */
8
8
  import type { ApprovalRung } from '@gaunt-sloth/core/config/shell-policy.js';
9
9
  import type { PreflightFloorKind } from '@gaunt-sloth/core/core/shell/raterVocabulary.js';
10
10
  import type { ToolResultRecord } from '#src/types.js';
11
+ import type { RaterPromptArm } from '#src/raterPromptArm.js';
11
12
  import type { AdvertisedToolInventory, ToolCoverageReport, ToolCoverageSpec } from '#src/toolCoverage.js';
12
13
  import type { EvalCaseClassification, EvalClassificationReport, EvalClassificationSpec, EvalMetricSpec } from '#src/classificationTypes.js';
13
14
  /** The 0-10 judge scale's default pass threshold, matching `review`'s own (unexported)
@@ -72,7 +73,7 @@ declare const PREFLIGHT_MECHANISM_OF: {
72
73
  * The floor is spelled here because it is not a preflight — it is a lexical scan core keeps no list
73
74
  * of, consulted by the approvals gate before any rating (every rung but `bypass`) and again by the
74
75
  * shell tool at execution time (every rung). The preflights are NOT spelled: they come from
75
- * {@link PREFLIGHT_MECHANISM_OF}, which is core's own `PREFLIGHT_FLOOR_KINDS` with our `forced_by`
76
+ * `PREFLIGHT_MECHANISM_OF`, which is core's own `PREFLIGHT_FLOOR_KINDS` with our `forced_by`
76
77
  * spelling attached.
77
78
  *
78
79
  * ## Why a case asserts a mechanism at all (the I1 finding)
@@ -159,7 +160,7 @@ export declare const PREFLIGHT_MECHANISMS: ("open-world-preflight" | "script-env
159
160
  export type PreflightMechanism = (typeof PREFLIGHT_MECHANISM_OF)[PreflightFloorKind];
160
161
  /**
161
162
  * The `forced_by` spelling of one of core's preflight arms — the typed lookup into
162
- * {@link PREFLIGHT_MECHANISM_OF}, which stays private so this is the only way in.
163
+ * `PREFLIGHT_MECHANISM_OF`, which stays private so this is the only way in.
163
164
  *
164
165
  * TOTAL over core's `PreflightFloorKind` and therefore never `undefined`, which is the point: a
165
166
  * caller cannot be handed an arm it has no name for, because such an arm would not have compiled.
@@ -179,7 +180,7 @@ export declare function mechanismNeedsPermissiveRating(mechanism: ForcedByMechan
179
180
  export interface GthAgentTarget {
180
181
  type: 'gth-agent';
181
182
  /** Suite-level profile hint. Only `undefined`/`'default'` is accepted (see
182
- * {@link ../evalSuite.js}'s `parseEvalSuite`) — per-case/per-identity profile switching is
183
+ * `evalSuite.js`'s `parseEvalSuite`) — per-case/per-identity profile switching is
183
184
  * `identities` scope, not this. */
184
185
  profile?: string;
185
186
  }
@@ -241,7 +242,7 @@ export interface AgUiAgentTarget {
241
242
  * §8 hardline floor — decides some commands outright, with no model in the loop.
242
243
  *
243
244
  * Unlike the agent targets this one is NOT driven by a `RunCellFn`: it supplies the
244
- * {@link RunClassifyFn} seam instead (`buildRaterClassifier` in {@link ../raterTarget.js}), which
245
+ * {@link RunClassifyFn} seam instead (`buildRaterClassifier` in `raterTarget.js`), which
245
246
  * the command passes to `runEvalSuite` as `options.classify`.
246
247
  */
247
248
  export interface RaterTarget {
@@ -253,7 +254,7 @@ export interface RaterTarget {
253
254
  * It is the suite's declaration, not the last word: a run whose config declares `approvals`
254
255
  * overrides it, which is what makes the `rung × model` sweep work through the existing `config:`
255
256
  * axis (a sweep cell cannot reach a `target` field). The override is announced, never silent —
256
- * see {@link ../raterTarget.js}.
257
+ * see `raterTarget.js`.
257
258
  */
258
259
  rung: ApprovalRung;
259
260
  }
@@ -264,7 +265,7 @@ export interface RaterTarget {
264
265
  * target), so the target only changes which runner the command builds. */
265
266
  export type EvalTarget = GthAgentTarget | AdkAgentTarget | AgUiAgentTarget | RaterTarget;
266
267
  /** One `json_path` assertion (BATCH-10): resolve `path` against the answer-parsed-as-JSON and check
267
- * it. Exactly one of `equals`/`contains` is set (enforced in {@link ../evalSuite.js}'s parse):
268
+ * it. Exactly one of `equals`/`contains` is set (enforced in `evalSuite.js`'s parse):
268
269
  * - `equals` — the resolved value must deep-equal this (any JSON value, incl. `null`).
269
270
  * - `contains` — the resolved value must be a string containing this substring. */
270
271
  export interface JsonPathCheck {
@@ -276,7 +277,7 @@ export interface JsonPathCheck {
276
277
  * One `tool_result_json_path` assertion (BATCH-21): select tool results by name `tool` (exact or
277
278
  * glob, the same matcher `must_call` uses), parse each matching result's payload as JSON, and
278
279
  * resolve `path` against it (the same minimal dot/`[index]` path `json_path` uses). At most one of
279
- * `equals`/`contains` may be set (enforced in {@link ../evalSuite.js}'s parse):
280
+ * `equals`/`contains` may be set (enforced in `evalSuite.js`'s parse):
280
281
  * - neither — pure existence check: the path must resolve in a matching result's payload;
281
282
  * - `equals` — the resolved value must deep-equal this (any JSON value, incl. `null`);
282
283
  * - `contains` — the resolved value must be a string containing this substring.
@@ -360,6 +361,31 @@ export interface EvalExpectation {
360
361
  * touching the first.
361
362
  */
362
363
  expectAction?: string;
364
+ /**
365
+ * BATCH-45 — **assert that a MODEL actually rendered a verdict for this round**, i.e. that
366
+ * {@link ClassifyOutcome.modelLabel} is present. `true` is the only legal value; `undefined` =
367
+ * this block makes no such claim. `rater` target only.
368
+ *
369
+ * It is NOT an assertion about the rung, and NOT an assertion about WHAT the model said. Those
370
+ * are {@link expectLabel} and {@link expectAction}, and the whole point of this key is that it
371
+ * says neither: a case can pin "a rater ruled here" without predicting which verdict three
372
+ * different raters would render, which is a prediction no measurement backs.
373
+ *
374
+ * **Why the action column cannot do this job.** When the gate never obtains a rating (timeout,
375
+ * throw, unparseable answer) it fails closed, and since EXT-171 that ESCALATES rather than
376
+ * negotiating. A genuine `catastrophic` verdict escalates too. So `expect_action: escalate` is
377
+ * satisfied identically by a rater that judged the command unnegotiable and by a rater that
378
+ * never answered at all — and a case asserting only that passes on a run where nothing was
379
+ * measured. BATCH-45 found ten such cells live in this repo's own approvals corpus.
380
+ *
381
+ * **Why `modelLabel` and not the rationale text.** `modelLabel` is suppressed from the call's own
382
+ * capture (`rating.failClosed`), never from a reading of the verdict's prose. Core's
383
+ * `isFailClosed` answers a question about the reason TEXT, and the rating prompt tells a rater to
384
+ * answer `destructive` and say it could not assess a command it is unsure of — so that predicate
385
+ * scores an obedient judgement as a gate failure. EXT-171 and `raterHealth` both say never to key
386
+ * on verdict text.
387
+ */
388
+ expectRated?: boolean;
363
389
  }
364
390
  /**
365
391
  * One conversational turn (BATCH-12): the `user` message to send, and the {@link EvalExpectation}
@@ -379,7 +405,7 @@ export interface EvalTurn {
379
405
  *
380
406
  * **Carrying it is not the same as showing it to the rater.** §5.1 admits the justification from
381
407
  * round 2 onward, so the round-1 rating withholds it; the target hands the whole context to core's
382
- * {@link import('@gaunt-sloth/core/core/shell/negotiation.js').ShellNegotiationState}, which is
408
+ * {@link @gaunt-sloth/core!core/shell/negotiation.ShellNegotiationState | ShellNegotiationState}, which is
383
409
  * the single implementation of that rule.
384
410
  */
385
411
  justification?: string;
@@ -411,7 +437,7 @@ export interface EvalTurn {
411
437
  }
412
438
  /**
413
439
  * One turn's raw run outcome inside a multi-turn conversation (BATCH-12 Task 2). Structurally the
414
- * per-turn analogue of BATCH-1's {@link ../types.js CellRunOutcome}: the turn's answer plus the
440
+ * per-turn analogue of BATCH-1's {@link "types.js"!CellRunOutcome | CellRunOutcome}: the turn's answer plus the
415
441
  * tools/tokens captured FOR THAT TURN — a per-invoke GS2-16 delta (each `processMessages` call
416
442
  * resets the tally), NOT the cumulative conversation total. `ok:false` with `error` set means that
417
443
  * turn's SUT invocation failed (no answer to grade). The conversational runner returns one of these
@@ -436,8 +462,8 @@ export interface TurnRunOutcome {
436
462
  }
437
463
  /**
438
464
  * Injectable "run one whole scripted conversation" function (BATCH-12 Task 2) — the multi-turn
439
- * analogue of BATCH-1's {@link ../types.js RunCellFn}, and the seam that lets
440
- * {@link ../evalRunner.js}'s multi-turn path be unit tested without any real LLM/MCP. Given the
465
+ * analogue of BATCH-1's {@link "types.js"!RunCellFn | RunCellFn}, and the seam that lets
466
+ * `evalRunner.js`'s multi-turn path be unit tested without any real LLM/MCP. Given the
441
467
  * ordered user messages of ONE (case × identity) conversation, it builds the agent + resolves tools
442
468
  * ONCE, runs each turn against the accumulated message history, and returns one {@link TurnRunOutcome}
443
469
  * per turn (per-turn answer + per-turn tool delta), cleaning up once. The production wiring
@@ -454,7 +480,7 @@ export type RunConversationFn = (userMessages: string[]) => Promise<TurnRunOutco
454
480
  * a round-1 rating, and a reset makes a later round a round-1 rating again — so the target hands the
455
481
  * accumulated state to `ShellNegotiationState.contextFor` rather than deciding here. A shape that
456
482
  * carried "what the rater sees" would be this package holding a second opinion about §5.1, which is
457
- * the one thing {@link ../raterTarget.js} exists not to do.
483
+ * the one thing `raterTarget.js` exists not to do.
458
484
  */
459
485
  export interface ClassifyRound {
460
486
  /** The command the agent proposes in this round. */
@@ -472,7 +498,7 @@ export interface ClassifyRound {
472
498
  * ## The Half-B seam, now filled
473
499
  *
474
500
  * Half A shipped this shape, the runner plumbing and the tests (against an injected fake); Half B
475
- * supplies the one function — `buildRaterClassifier` in {@link ../raterTarget.js}, the
501
+ * supplies the one function — `buildRaterClassifier` in `raterTarget.js`, the
476
502
  * {@link RaterTarget}'s implementation, which drives the approvals rating prompt + decision mapping
477
503
  * at a declared rung. The shape did not have to move to fit it.
478
504
  *
@@ -648,7 +674,7 @@ export interface EvalCase {
648
674
  */
649
675
  modelFree: boolean;
650
676
  }
651
- /** A fully parsed and validated suite — see {@link ../evalSuite.js}'s `parseEvalSuite`. */
677
+ /** A fully parsed and validated suite — see `evalSuite.js`'s `parseEvalSuite`. */
652
678
  export interface EvalSuite {
653
679
  target: EvalTarget;
654
680
  /**
@@ -690,7 +716,7 @@ export interface EvalSuite {
690
716
  * BATCH-25 — the config sweep: named cells the WHOLE suite is run once per, so one corpus
691
717
  * produces one comparison table instead of N unrelated runs. Absent = a single run.
692
718
  *
693
- * Consumed by the CLI, not by {@link ../evalRunner.js runEvalSuite}: a sweep is a run-level
719
+ * Consumed by the CLI, not by {@link "evalRunner.js"!runEvalSuite | runEvalSuite}: a sweep is a run-level
694
720
  * concept (same suite, different config) exactly like BATCH-19's multi-suite loop, so the runner
695
721
  * stays about grading and #405's identity matrix is untouched.
696
722
  */
@@ -715,6 +741,17 @@ export interface EvalSweepValue {
715
741
  model?: string;
716
742
  /** Plain-data config overrides deep-merged onto the resolved config (never `llm`). */
717
743
  config?: Record<string, unknown>;
744
+ /**
745
+ * [[BATCH-31]] — the cell's PROMPT ARM: which of gth's own preflight notes this cell's ratings go
746
+ * out without, so a note-on / note-off A/B is one suite file rather than two. `rater` targets
747
+ * only; `{ omit: [] }` is the baseline arm and is required rather than implied.
748
+ *
749
+ * **A third override kind, not a third spelling of `config:`.** It is deliberately not a config
750
+ * key and never merges into one: a config key is exactly what a session could also be given, and
751
+ * an omission a session can reach is a switch that suppresses safety context in the live approvals
752
+ * gate. See `raterPromptArm.ts` for the full argument and for how that is enforced.
753
+ */
754
+ notes?: RaterPromptArm;
718
755
  }
719
756
  /**
720
757
  * BATCH-25 — the sweep: named axes whose CARTESIAN PRODUCT is the set of runs. The approvals
@@ -736,7 +773,7 @@ export interface DeterministicCheckResult {
736
773
  failures: string[];
737
774
  }
738
775
  /** The judge's structured verdict on one case's answer — matches `review`'s `RateSchema` shape
739
- * (0-10 `rate` + a reason string) for UX consistency, see {@link ../judge.js}. */
776
+ * (0-10 `rate` + a reason string) for UX consistency, see `judge.js`. */
740
777
  export interface JudgeVerdict {
741
778
  rate: number;
742
779
  reason: string;
@@ -754,7 +791,7 @@ export interface JudgeOutcome {
754
791
  error?: string;
755
792
  }
756
793
  /** Injectable "grade one answer against one rubric" function — the seam that lets
757
- * {@link ../evalRunner.js}'s `runEvalSuite` be fully unit tested without any real LLM call, mirror
794
+ * `evalRunner.js`'s `runEvalSuite` be fully unit tested without any real LLM call, mirror
758
795
  * of BATCH-1's `RunCellFn`. The production wiring (`evalCommand.ts`) adapts `judgeEvalCase` to
759
796
  * this shape; tests inject a fake that resolves/fails as needed. */
760
797
  export type JudgeFn = (answer: string, rubric: string) => Promise<JudgeOutcome>;
package/dist/evalTypes.js CHANGED
@@ -60,7 +60,7 @@ const PREFLIGHT_MECHANISM_OF = {
60
60
  * The floor is spelled here because it is not a preflight — it is a lexical scan core keeps no list
61
61
  * of, consulted by the approvals gate before any rating (every rung but `bypass`) and again by the
62
62
  * shell tool at execution time (every rung). The preflights are NOT spelled: they come from
63
- * {@link PREFLIGHT_MECHANISM_OF}, which is core's own `PREFLIGHT_FLOOR_KINDS` with our `forced_by`
63
+ * `PREFLIGHT_MECHANISM_OF`, which is core's own `PREFLIGHT_FLOOR_KINDS` with our `forced_by`
64
64
  * spelling attached.
65
65
  *
66
66
  * ## Why a case asserts a mechanism at all (the I1 finding)
@@ -132,7 +132,7 @@ export const FORCED_BY_ASSERTIONS = {
132
132
  export const PREFLIGHT_MECHANISMS = Object.values(PREFLIGHT_MECHANISM_OF);
133
133
  /**
134
134
  * The `forced_by` spelling of one of core's preflight arms — the typed lookup into
135
- * {@link PREFLIGHT_MECHANISM_OF}, which stays private so this is the only way in.
135
+ * `PREFLIGHT_MECHANISM_OF`, which stays private so this is the only way in.
136
136
  *
137
137
  * TOTAL over core's `PreflightFloorKind` and therefore never `undefined`, which is the point: a
138
138
  * caller cannot be handed an arm it has no name for, because such an arm would not have compiled.
@@ -1 +1 @@
1
- {"version":3,"file":"evalTypes.js","sourceRoot":"","sources":["../src/evalTypes.ts"],"names":[],"mappings":"AA0BA;;;;mGAImG;AACnG,MAAM,CAAC,MAAM,2BAA2B,GAAG,CAAC,CAAC;AAE7C;;;;;;;;;;;;;;;;;;;;;;;;GAwBG;AACH,MAAM,CAAC,MAAM,uBAAuB,GAAG,yBAAyB,CAAC;AAEjE;;;;;;;;;;;;;;;;;;GAkBG;AACH,MAAM,sBAAsB,GAAG;IAC7B,iBAAiB,EAAE,2BAA2B;IAC9C,YAAY,EAAE,sBAAsB;CAC2B,CAAC;AAElE;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG;IAClC,gBAAgB;IAChB,GAAG,MAAM,CAAC,MAAM,CAAC,sBAAsB,CAAC;CAChC,CAAC;AAiBX,mGAAmG;AACnG,MAAM,CAAC,MAAM,gBAAgB,GAAG,WAAW,CAAC;AAE5C;;;;GAIG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAsC;IACrE,gBAAgB,EAAE,uBAAuB;IACzC,2BAA2B,EAAE,GAAG,gBAAgB,6BAA6B;IAC7E,sBAAsB,EAAE,GAAG,gBAAgB,wBAAwB;CACpE,CAAC;AAEF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAsCG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG,MAAM,CAAC,MAAM,CAAC,sBAAsB,CAAC,CAAC;AAW1E;;;;;;GAMG;AACH,MAAM,UAAU,qBAAqB,CAAC,IAAwB;IAC5D,OAAO,sBAAsB,CAAC,IAAI,CAAC,CAAC;AACtC,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,8BAA8B,CAAC,SAAwC;IACrF,OAAO,CACL,SAAS,KAAK,SAAS;QACtB,oBAAqD,CAAC,QAAQ,CAAC,SAAS,CAAC,CAC3E,CAAC;AACJ,CAAC"}
1
+ {"version":3,"file":"evalTypes.js","sourceRoot":"","sources":["../src/evalTypes.ts"],"names":[],"mappings":"AA6BA;;;;mGAImG;AACnG,MAAM,CAAC,MAAM,2BAA2B,GAAG,CAAC,CAAC;AAE7C;;;;;;;;;;;;;;;;;;;;;;;;GAwBG;AACH,MAAM,CAAC,MAAM,uBAAuB,GAAG,yBAAyB,CAAC;AAEjE;;;;;;;;;;;;;;;;;;GAkBG;AACH,MAAM,sBAAsB,GAAG;IAC7B,iBAAiB,EAAE,2BAA2B;IAC9C,YAAY,EAAE,sBAAsB;CAC2B,CAAC;AAElE;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG;IAClC,gBAAgB;IAChB,GAAG,MAAM,CAAC,MAAM,CAAC,sBAAsB,CAAC;CAChC,CAAC;AAiBX,mGAAmG;AACnG,MAAM,CAAC,MAAM,gBAAgB,GAAG,WAAW,CAAC;AAE5C;;;;GAIG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAsC;IACrE,gBAAgB,EAAE,uBAAuB;IACzC,2BAA2B,EAAE,GAAG,gBAAgB,6BAA6B;IAC7E,sBAAsB,EAAE,GAAG,gBAAgB,wBAAwB;CACpE,CAAC;AAEF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAsCG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG,MAAM,CAAC,MAAM,CAAC,sBAAsB,CAAC,CAAC;AAW1E;;;;;;GAMG;AACH,MAAM,UAAU,qBAAqB,CAAC,IAAwB;IAC5D,OAAO,sBAAsB,CAAC,IAAI,CAAC,CAAC;AACtC,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,8BAA8B,CAAC,SAAwC;IACrF,OAAO,CACL,SAAS,KAAK,SAAS;QACtB,oBAAqD,CAAC,QAAQ,CAAC,SAAS,CAAC,CAC3E,CAAC;AACJ,CAAC"}
package/dist/index.d.ts CHANGED
@@ -31,6 +31,8 @@ export type { ClassificationExtractor, ClassifiedCell, EvalCaseClassification, E
31
31
  export type { ClassifyOutcome, ClassifyRequest, ClassifyRound, EvalSweep, EvalSweepValue, RaterTarget, RunClassifyFn, } from '#src/evalTypes.js';
32
32
  export { buildRaterClassifier, HARDLINE_NO_RATING_REASON, HARDLINE_REFUSAL_MARKER, NEGOTIATION_BOUND_MARKER, NO_RATING_CALL_MARKER, } from '#src/raterTarget.js';
33
33
  export type { RaterClassifierOptions } from '#src/raterTarget.js';
34
+ export { RATER_PROMPT_NOTE_NAMES } from '#src/raterPromptArm.js';
35
+ export type { RaterPromptArm, RaterPromptNoteName } from '#src/raterPromptArm.js';
34
36
  export { resolveReporters, availableReporterNames } from '#src/reporters/registry.js';
35
37
  export { driveReporters } from '#src/reporters/drive.js';
36
38
  export { createTextReporter } from '#src/reporters/textReporter.js';
package/dist/index.js CHANGED
@@ -26,6 +26,11 @@ export { UNRECOGNIZED_LABEL, NO_EXPECTATION } from '#src/classificationTypes.js'
26
26
  // BATCH-25 Half B — the `rater` target: the one implementation of the `RunClassifyFn` seam, which
27
27
  // drives gth's own approvals rater over a corpus of commands.
28
28
  export { buildRaterClassifier, HARDLINE_NO_RATING_REASON, HARDLINE_REFUSAL_MARKER, NEGOTIATION_BOUND_MARKER, NO_RATING_CALL_MARKER, } from '#src/raterTarget.js';
29
+ // [[BATCH-31]] — the prompt arm: the note-on / note-off A/B a rater suite can express. Only the
30
+ // NAMES and the type are exported. `armRaterModel` deliberately is not — nothing outside this
31
+ // package needs to build one, and the narrower the surface the smaller the chance of a caller
32
+ // wiring an omission somewhere the argument in `raterPromptArm.ts` does not cover.
33
+ export { RATER_PROMPT_NOTE_NAMES } from '#src/raterPromptArm.js';
29
34
  // BATCH-19 — the `gth eval` reporter facility (A1 seam). These are the public plugin contract an
30
35
  // out-of-core `@gaunt-sloth/eval-reporter-*` package implements, exported from the package root so a
31
36
  // reporter package can type its factory against ONE import.
package/dist/index.js.map CHANGED
@@ -1 +1 @@
1
- {"version":3,"file":"index.js","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,WAAW,EAAE,MAAM,gBAAgB,CAAC;AAC7C,OAAO,EAAE,eAAe,EAAE,MAAM,qBAAqB,CAAC;AACtD,OAAO,EAAE,aAAa,EAAE,MAAM,mBAAmB,CAAC;AAClD,OAAO,EAAE,cAAc,EAAE,iBAAiB,EAAE,MAAM,qBAAqB,CAAC;AACxE,OAAO,EAAE,gBAAgB,EAAE,MAAM,gBAAgB,CAAC;AAWlD,OAAO,EAAE,wBAAwB,EAAE,MAAM,eAAe,CAAC;AAEzD,gGAAgG;AAChG,OAAO,EAAE,cAAc,EAAE,MAAM,mBAAmB,CAAC;AACnD,OAAO,EAAE,sBAAsB,EAAE,MAAM,6BAA6B,CAAC;AACrE,OAAO,EACL,aAAa,EACb,kBAAkB,EAClB,iBAAiB,EACjB,6BAA6B,GAC9B,MAAM,eAAe,CAAC;AAEvB,OAAO,EAAE,YAAY,EAAE,MAAM,oBAAoB,CAAC;AAClD,OAAO,EAAE,eAAe,EAAE,MAAM,oBAAoB,CAAC;AAqBrD,OAAO,EAAE,2BAA2B,EAAE,MAAM,mBAAmB,CAAC;AAEhE,mGAAmG;AACnG,8FAA8F;AAC9F,OAAO,EACL,0BAA0B,EAC1B,oBAAoB,EACpB,WAAW,EACX,OAAO,GACR,MAAM,wBAAwB,CAAC;AAChC,OAAO,EAAE,yBAAyB,EAAE,MAAM,8BAA8B,CAAC;AACzE,OAAO,EACL,aAAa,EACb,oBAAoB,EACpB,iBAAiB,EACjB,kBAAkB,EAClB,WAAW,GACZ,MAAM,iBAAiB,CAAC;AACzB,OAAO,EACL,0BAA0B,EAC1B,qBAAqB,EACrB,YAAY,GACb,MAAM,8BAA8B,CAAC;AACtC,qFAAqF;AACrF,OAAO,EACL,mBAAmB,EACnB,qBAAqB,EACrB,2BAA2B,GAC5B,MAAM,sBAAsB,CAAC;AAS9B,OAAO,EAAE,kBAAkB,EAAE,MAAM,4BAA4B,CAAC;AAEhE,OAAO,EAAE,gBAAgB,EAAE,WAAW,EAAE,iBAAiB,EAAE,MAAM,qBAAqB,CAAC;AAEvF,OAAO,EACL,WAAW,EACX,SAAS,EACT,gBAAgB,EAChB,QAAQ,EACR,aAAa,EACb,qBAAqB,EACrB,0BAA0B,EAC1B,6BAA6B,EAC7B,yBAAyB,GAC1B,MAAM,qBAAqB,CAAC;AAU7B,OAAO,EAAE,kBAAkB,EAAE,cAAc,EAAE,MAAM,6BAA6B,CAAC;AAyBjF,kGAAkG;AAClG,8DAA8D;AAC9D,OAAO,EACL,oBAAoB,EACpB,yBAAyB,EACzB,uBAAuB,EACvB,wBAAwB,EACxB,qBAAqB,GACtB,MAAM,qBAAqB,CAAC;AAG7B,iGAAiG;AACjG,qGAAqG;AACrG,4DAA4D;AAC5D,OAAO,EAAE,gBAAgB,EAAE,sBAAsB,EAAE,MAAM,4BAA4B,CAAC;AACtF,OAAO,EAAE,cAAc,EAAE,MAAM,yBAAyB,CAAC;AACzD,OAAO,EAAE,kBAAkB,EAAE,MAAM,gCAAgC,CAAC;AAQpE,kGAAkG;AAClG,6CAA6C;AAC7C,OAAO,EAAE,WAAW,EAAE,MAAM,8BAA8B,CAAC;AAC3D,OAAO,EAAE,4BAA4B,EAAE,MAAM,eAAe,CAAC"}
1
+ {"version":3,"file":"index.js","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,WAAW,EAAE,MAAM,gBAAgB,CAAC;AAC7C,OAAO,EAAE,eAAe,EAAE,MAAM,qBAAqB,CAAC;AACtD,OAAO,EAAE,aAAa,EAAE,MAAM,mBAAmB,CAAC;AAClD,OAAO,EAAE,cAAc,EAAE,iBAAiB,EAAE,MAAM,qBAAqB,CAAC;AACxE,OAAO,EAAE,gBAAgB,EAAE,MAAM,gBAAgB,CAAC;AAWlD,OAAO,EAAE,wBAAwB,EAAE,MAAM,eAAe,CAAC;AAEzD,gGAAgG;AAChG,OAAO,EAAE,cAAc,EAAE,MAAM,mBAAmB,CAAC;AACnD,OAAO,EAAE,sBAAsB,EAAE,MAAM,6BAA6B,CAAC;AACrE,OAAO,EACL,aAAa,EACb,kBAAkB,EAClB,iBAAiB,EACjB,6BAA6B,GAC9B,MAAM,eAAe,CAAC;AAEvB,OAAO,EAAE,YAAY,EAAE,MAAM,oBAAoB,CAAC;AAClD,OAAO,EAAE,eAAe,EAAE,MAAM,oBAAoB,CAAC;AAqBrD,OAAO,EAAE,2BAA2B,EAAE,MAAM,mBAAmB,CAAC;AAEhE,mGAAmG;AACnG,8FAA8F;AAC9F,OAAO,EACL,0BAA0B,EAC1B,oBAAoB,EACpB,WAAW,EACX,OAAO,GACR,MAAM,wBAAwB,CAAC;AAChC,OAAO,EAAE,yBAAyB,EAAE,MAAM,8BAA8B,CAAC;AACzE,OAAO,EACL,aAAa,EACb,oBAAoB,EACpB,iBAAiB,EACjB,kBAAkB,EAClB,WAAW,GACZ,MAAM,iBAAiB,CAAC;AACzB,OAAO,EACL,0BAA0B,EAC1B,qBAAqB,EACrB,YAAY,GACb,MAAM,8BAA8B,CAAC;AACtC,qFAAqF;AACrF,OAAO,EACL,mBAAmB,EACnB,qBAAqB,EACrB,2BAA2B,GAC5B,MAAM,sBAAsB,CAAC;AAS9B,OAAO,EAAE,kBAAkB,EAAE,MAAM,4BAA4B,CAAC;AAEhE,OAAO,EAAE,gBAAgB,EAAE,WAAW,EAAE,iBAAiB,EAAE,MAAM,qBAAqB,CAAC;AAEvF,OAAO,EACL,WAAW,EACX,SAAS,EACT,gBAAgB,EAChB,QAAQ,EACR,aAAa,EACb,qBAAqB,EACrB,0BAA0B,EAC1B,6BAA6B,EAC7B,yBAAyB,GAC1B,MAAM,qBAAqB,CAAC;AAU7B,OAAO,EAAE,kBAAkB,EAAE,cAAc,EAAE,MAAM,6BAA6B,CAAC;AAyBjF,kGAAkG;AAClG,8DAA8D;AAC9D,OAAO,EACL,oBAAoB,EACpB,yBAAyB,EACzB,uBAAuB,EACvB,wBAAwB,EACxB,qBAAqB,GACtB,MAAM,qBAAqB,CAAC;AAG7B,gGAAgG;AAChG,8FAA8F;AAC9F,8FAA8F;AAC9F,mFAAmF;AACnF,OAAO,EAAE,uBAAuB,EAAE,MAAM,wBAAwB,CAAC;AAGjE,iGAAiG;AACjG,qGAAqG;AACrG,4DAA4D;AAC5D,OAAO,EAAE,gBAAgB,EAAE,sBAAsB,EAAE,MAAM,4BAA4B,CAAC;AACtF,OAAO,EAAE,cAAc,EAAE,MAAM,yBAAyB,CAAC;AACzD,OAAO,EAAE,kBAAkB,EAAE,MAAM,gCAAgC,CAAC;AAQpE,kGAAkG;AAClG,6CAA6C;AAC7C,OAAO,EAAE,WAAW,EAAE,MAAM,8BAA8B,CAAC;AAC3D,OAAO,EAAE,4BAA4B,EAAE,MAAM,eAAe,CAAC"}
package/dist/judge.d.ts CHANGED
@@ -1,6 +1,4 @@
1
1
  /**
2
- * @module judge
3
- *
4
2
  * BATCH-2 — LLM-as-judge grading for `gth eval`. Adapts the *mechanism* of EXT-10's shell-safety
5
3
  * judge (`packages/core/src/core/shell/judge.ts`): `model.withStructuredOutput(zodSchema)` for a
6
4
  * single non-agentic structured call, raced against a timeout. The failure policy differs on
@@ -18,6 +16,8 @@
18
16
  * `judgeShellCommand`'s own default. A separate `--judge <profile>` model is BATCH-2's own
19
17
  * "Not in scope" list (identity-matrix/pluggable-target work); grading with the SUT's own model
20
18
  * config is a known, real simplification for this first slice.
19
+ *
20
+ * @module
21
21
  */
22
22
  import type { BaseChatModel } from '@langchain/core/language_models/chat_models';
23
23
  import * as z from 'zod';
package/dist/judge.js CHANGED
@@ -1,6 +1,4 @@
1
1
  /**
2
- * @module judge
3
- *
4
2
  * BATCH-2 — LLM-as-judge grading for `gth eval`. Adapts the *mechanism* of EXT-10's shell-safety
5
3
  * judge (`packages/core/src/core/shell/judge.ts`): `model.withStructuredOutput(zodSchema)` for a
6
4
  * single non-agentic structured call, raced against a timeout. The failure policy differs on
@@ -18,8 +16,11 @@
18
16
  * `judgeShellCommand`'s own default. A separate `--judge <profile>` model is BATCH-2's own
19
17
  * "Not in scope" list (identity-matrix/pluggable-target work); grading with the SUT's own model
20
18
  * config is a known, real simplification for this first slice.
19
+ *
20
+ * @module
21
21
  */
22
22
  import { HumanMessage, SystemMessage } from '@langchain/core/messages';
23
+ import { CALL_TIMED_OUT, withCallDeadline } from '@gaunt-sloth/core/runtime/abortableCall.js';
23
24
  import { structuredOutputBoundary } from '@gaunt-sloth/core/runtime/structuredOutput.js';
24
25
  import * as z from 'zod';
25
26
  /** Structured verdict the judge model must return — 0-10 `rate` + a `reason`, matching `review`'s
@@ -85,20 +86,19 @@ export async function judgeEvalCase(answer, rubric, model, options) {
85
86
  return { attempted: true, ok: false, error: 'No usable judge model configured.' };
86
87
  }
87
88
  const { system, user } = buildJudgeMessages(answer, rubric);
88
- let timer;
89
89
  try {
90
90
  // EXT-88 — routed through the shared boundary like every other structured-output call. The
91
91
  // rubric verdict has no optional field today, so the boundary hands back this very schema by
92
92
  // identity; going through it is what stops a later optional field re-introducing the defect.
93
93
  const boundary = structuredOutputBoundary(EvalVerdictSchema);
94
94
  const structured = model.withStructuredOutput(boundary.wireSchema);
95
- const judgePromise = structured.invoke([new SystemMessage(system), new HumanMessage(user)]);
96
- const TIMEOUT = Symbol('eval-judge-timeout');
97
- const timeoutPromise = new Promise((resolve) => {
98
- timer = setTimeout(() => resolve(TIMEOUT), timeoutMs);
99
- });
100
- const raced = await Promise.race([judgePromise, timeoutPromise]);
101
- if (raced === TIMEOUT) {
95
+ // [[EXT-179]] — the budget aborts the judge call rather than abandoning it. This matters most
96
+ // in a SWEEP: a suite grades many cases, and every stalled judge used to leave its own request
97
+ // in flight, so the leaked sockets accumulated across the run and kept the process alive after
98
+ // the report had been written. The timeout text is unchanged and the case still fails as a
99
+ // case — an aborted judgement is a judgement not obtained, never an auto-pass.
100
+ const raced = await withCallDeadline(timeoutMs, (signal) => structured.invoke([new SystemMessage(system), new HumanMessage(user)], { signal }));
101
+ if (raced === CALL_TIMED_OUT) {
102
102
  return { attempted: true, ok: false, error: `Judge timed out after ${timeoutMs}ms.` };
103
103
  }
104
104
  // withStructuredOutput already coerces to the schema, but re-validate defensively: a fake or
@@ -116,9 +116,5 @@ export async function judgeEvalCase(answer, rubric, model, options) {
116
116
  error: error instanceof Error ? error.message : String(error),
117
117
  };
118
118
  }
119
- finally {
120
- if (timer)
121
- clearTimeout(timer);
122
- }
123
119
  }
124
120
  //# sourceMappingURL=judge.js.map
package/dist/judge.js.map CHANGED
@@ -1 +1 @@
1
- {"version":3,"file":"judge.js","sourceRoot":"","sources":["../src/judge.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAGH,OAAO,EAAE,YAAY,EAAE,aAAa,EAAE,MAAM,0BAA0B,CAAC;AACvE,OAAO,EAAE,wBAAwB,EAAE,MAAM,+CAA+C,CAAC;AACzF,OAAO,KAAK,CAAC,MAAM,KAAK,CAAC;AAIzB;;oGAEoG;AACpG,MAAM,CAAC,MAAM,iBAAiB,GAAG,CAAC,CAAC,MAAM,CAAC;IACxC,IAAI,EAAE,CAAC;SACJ,MAAM,EAAE;SACR,GAAG,CAAC,CAAC,CAAC;SACN,GAAG,CAAC,EAAE,CAAC;SACP,QAAQ,CAAC,yDAAyD,CAAC;IACtE,MAAM,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,QAAQ,CAAC,2CAA2C,CAAC;CACzE,CAAC,CAAC;AAIH;;oGAEoG;AACpG,MAAM,CAAC,MAAM,6BAA6B,GAAG,MAAM,CAAC;AAEpD,MAAM,wBAAwB,GAAG;IAC/B,mCAAmC;IACnC,gGAAgG;IAChG,cAAc;IACd,EAAE;IACF,+FAA+F;IAC/F,gGAAgG;IAChG,mDAAmD;IACnD,6FAA6F;IAC7F,+BAA+B;CAChC,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AAEb;;;;GAIG;AACH,MAAM,UAAU,kBAAkB,CAChC,MAAc,EACd,MAAc;IAEd,MAAM,SAAS,GAAG;QAChB,SAAS;QACT,MAAM;QACN,EAAE;QACF,qBAAqB;QACrB,UAAU;QACV,MAAM,IAAI,gBAAgB;QAC1B,WAAW;KACZ,CAAC;IACF,OAAO,EAAE,MAAM,EAAE,wBAAwB,EAAE,IAAI,EAAE,SAAS,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;AAC1E,CAAC;AAED;;;;;;;;;;;;;GAaG;AACH,MAAM,CAAC,KAAK,UAAU,aAAa,CACjC,MAAc,EACd,MAAc,EACd,KAAgC,EAChC,OAAgC;IAEhC,MAAM,SAAS,GAAG,OAAO,EAAE,SAAS,IAAI,6BAA6B,CAAC;IAEtE,IAAI,CAAC,KAAK,IAAI,OAAO,KAAK,CAAC,oBAAoB,KAAK,UAAU,EAAE,CAAC;QAC/D,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,mCAAmC,EAAE,CAAC;IACpF,CAAC;IAED,MAAM,EAAE,MAAM,EAAE,IAAI,EAAE,GAAG,kBAAkB,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IAC5D,IAAI,KAAgD,CAAC;IACrD,IAAI,CAAC;QACH,2FAA2F;QAC3F,6FAA6F;QAC7F,6FAA6F;QAC7F,MAAM,QAAQ,GAAG,wBAAwB,CAAC,iBAAiB,CAAC,CAAC;QAC7D,MAAM,UAAU,GAAG,KAAK,CAAC,oBAAoB,CAAC,QAAQ,CAAC,UAAU,CAAC,CAAC;QACnE,MAAM,YAAY,GAAG,UAAU,CAAC,MAAM,CAAC,CAAC,IAAI,aAAa,CAAC,MAAM,CAAC,EAAE,IAAI,YAAY,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC;QAE5F,MAAM,OAAO,GAAG,MAAM,CAAC,oBAAoB,CAAC,CAAC;QAC7C,MAAM,cAAc,GAAG,IAAI,OAAO,CAAiB,CAAC,OAAO,EAAE,EAAE;YAC7D,KAAK,GAAG,UAAU,CAAC,GAAG,EAAE,CAAC,OAAO,CAAC,OAAO,CAAC,EAAE,SAAS,CAAC,CAAC;QACxD,CAAC,CAAC,CAAC;QAEH,MAAM,KAAK,GAAG,MAAM,OAAO,CAAC,IAAI,CAAC,CAAC,YAAY,EAAE,cAAc,CAAC,CAAC,CAAC;QACjE,IAAI,KAAK,KAAK,OAAO,EAAE,CAAC;YACtB,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,yBAAyB,SAAS,KAAK,EAAE,CAAC;QACxF,CAAC;QAED,6FAA6F;QAC7F,0DAA0D;QAC1D,MAAM,MAAM,GAAG,QAAQ,CAAC,SAAS,CAAC,KAAK,CAAC,CAAC;QACzC,IAAI,CAAC,MAAM,CAAC,OAAO,EAAE,CAAC;YACpB,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,oCAAoC,EAAE,CAAC;QACrF,CAAC;QACD,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,IAAI,EAAE,OAAO,EAAE,MAAM,CAAC,IAAI,EAAE,CAAC;IAC7D,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QACf,OAAO;YACL,SAAS,EAAE,IAAI;YACf,EAAE,EAAE,KAAK;YACT,KAAK,EAAE,KAAK,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC;SAC9D,CAAC;IACJ,CAAC;YAAS,CAAC;QACT,IAAI,KAAK;YAAE,YAAY,CAAC,KAAK,CAAC,CAAC;IACjC,CAAC;AACH,CAAC"}
1
+ {"version":3,"file":"judge.js","sourceRoot":"","sources":["../src/judge.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAGH,OAAO,EAAE,YAAY,EAAE,aAAa,EAAE,MAAM,0BAA0B,CAAC;AACvE,OAAO,EAAE,cAAc,EAAE,gBAAgB,EAAE,MAAM,4CAA4C,CAAC;AAC9F,OAAO,EAAE,wBAAwB,EAAE,MAAM,+CAA+C,CAAC;AACzF,OAAO,KAAK,CAAC,MAAM,KAAK,CAAC;AAIzB;;oGAEoG;AACpG,MAAM,CAAC,MAAM,iBAAiB,GAAG,CAAC,CAAC,MAAM,CAAC;IACxC,IAAI,EAAE,CAAC;SACJ,MAAM,EAAE;SACR,GAAG,CAAC,CAAC,CAAC;SACN,GAAG,CAAC,EAAE,CAAC;SACP,QAAQ,CAAC,yDAAyD,CAAC;IACtE,MAAM,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,QAAQ,CAAC,2CAA2C,CAAC;CACzE,CAAC,CAAC;AAIH;;oGAEoG;AACpG,MAAM,CAAC,MAAM,6BAA6B,GAAG,MAAM,CAAC;AAEpD,MAAM,wBAAwB,GAAG;IAC/B,mCAAmC;IACnC,gGAAgG;IAChG,cAAc;IACd,EAAE;IACF,+FAA+F;IAC/F,gGAAgG;IAChG,mDAAmD;IACnD,6FAA6F;IAC7F,+BAA+B;CAChC,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AAEb;;;;GAIG;AACH,MAAM,UAAU,kBAAkB,CAChC,MAAc,EACd,MAAc;IAEd,MAAM,SAAS,GAAG;QAChB,SAAS;QACT,MAAM;QACN,EAAE;QACF,qBAAqB;QACrB,UAAU;QACV,MAAM,IAAI,gBAAgB;QAC1B,WAAW;KACZ,CAAC;IACF,OAAO,EAAE,MAAM,EAAE,wBAAwB,EAAE,IAAI,EAAE,SAAS,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;AAC1E,CAAC;AAED;;;;;;;;;;;;;GAaG;AACH,MAAM,CAAC,KAAK,UAAU,aAAa,CACjC,MAAc,EACd,MAAc,EACd,KAAgC,EAChC,OAAgC;IAEhC,MAAM,SAAS,GAAG,OAAO,EAAE,SAAS,IAAI,6BAA6B,CAAC;IAEtE,IAAI,CAAC,KAAK,IAAI,OAAO,KAAK,CAAC,oBAAoB,KAAK,UAAU,EAAE,CAAC;QAC/D,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,mCAAmC,EAAE,CAAC;IACpF,CAAC;IAED,MAAM,EAAE,MAAM,EAAE,IAAI,EAAE,GAAG,kBAAkB,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IAC5D,IAAI,CAAC;QACH,2FAA2F;QAC3F,6FAA6F;QAC7F,6FAA6F;QAC7F,MAAM,QAAQ,GAAG,wBAAwB,CAAC,iBAAiB,CAAC,CAAC;QAC7D,MAAM,UAAU,GAAG,KAAK,CAAC,oBAAoB,CAAC,QAAQ,CAAC,UAAU,CAAC,CAAC;QAEnE,8FAA8F;QAC9F,+FAA+F;QAC/F,+FAA+F;QAC/F,2FAA2F;QAC3F,+EAA+E;QAC/E,MAAM,KAAK,GAAG,MAAM,gBAAgB,CAAC,SAAS,EAAE,CAAC,MAAM,EAAE,EAAE,CACzD,UAAU,CAAC,MAAM,CAAC,CAAC,IAAI,aAAa,CAAC,MAAM,CAAC,EAAE,IAAI,YAAY,CAAC,IAAI,CAAC,CAAC,EAAE,EAAE,MAAM,EAAE,CAAC,CACnF,CAAC;QACF,IAAI,KAAK,KAAK,cAAc,EAAE,CAAC;YAC7B,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,yBAAyB,SAAS,KAAK,EAAE,CAAC;QACxF,CAAC;QAED,6FAA6F;QAC7F,0DAA0D;QAC1D,MAAM,MAAM,GAAG,QAAQ,CAAC,SAAS,CAAC,KAAK,CAAC,CAAC;QACzC,IAAI,CAAC,MAAM,CAAC,OAAO,EAAE,CAAC;YACpB,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,oCAAoC,EAAE,CAAC;QACrF,CAAC;QACD,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,IAAI,EAAE,OAAO,EAAE,MAAM,CAAC,IAAI,EAAE,CAAC;IAC7D,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QACf,OAAO;YACL,SAAS,EAAE,IAAI;YACf,EAAE,EAAE,KAAK;YACT,KAAK,EAAE,KAAK,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC;SAC9D,CAAC;IACJ,CAAC;AACH,CAAC"}
package/dist/output.d.ts CHANGED
@@ -4,7 +4,7 @@ import type { BatchSummary, CellResult } from '#src/types.js';
4
4
  * (pass/fail counts + a per-cell one-liner — a lightweight flake report) into `outputDir`.
5
5
  * Creates `outputDir` (and any missing parents) if it doesn't exist.
6
6
  *
7
- * Pure I/O, deliberately separate from {@link runBatchMatrix}: the runner never touches the
7
+ * Pure I/O, deliberately separate from {@link @gaunt-sloth/batch!"BatchRunner.js".runBatchMatrix | runBatchMatrix}: the runner never touches the
8
8
  * filesystem, so unit tests can exercise matrix/concurrency/retry logic without a tmp dir, and this
9
9
  * function can be tested in isolation with a fixed set of results.
10
10
  *
package/dist/output.js CHANGED
@@ -6,7 +6,7 @@ import { buildBatchSummary } from '#src/BatchRunner.js';
6
6
  * (pass/fail counts + a per-cell one-liner — a lightweight flake report) into `outputDir`.
7
7
  * Creates `outputDir` (and any missing parents) if it doesn't exist.
8
8
  *
9
- * Pure I/O, deliberately separate from {@link runBatchMatrix}: the runner never touches the
9
+ * Pure I/O, deliberately separate from {@link @gaunt-sloth/batch!"BatchRunner.js".runBatchMatrix | runBatchMatrix}: the runner never touches the
10
10
  * filesystem, so unit tests can exercise matrix/concurrency/retry logic without a tmp dir, and this
11
11
  * function can be tested in isolation with a fixed set of results.
12
12
  *
@@ -4,7 +4,7 @@ import type { MatrixRow } from '#src/types.js';
4
4
  * `.jsonl`/`.ndjson` → one JSON object per line; anything else (including `.csv`) → CSV.
5
5
  *
6
6
  * This is content binding only (BATCH-1 scope): every row becomes an object of string fields that
7
- * {@link bindCellContent} interpolates into the script. A glob-of-binary-files path binding is out
7
+ * {@link @gaunt-sloth/batch!"interpolate.js".bindCellContent | bindCellContent} interpolates into the script. A glob-of-binary-files path binding is out
8
8
  * of scope for this task.
9
9
  *
10
10
  * Throws a descriptive `Error` on malformed input — the harness-level failure the CLI surface doc
package/dist/parseOver.js CHANGED
@@ -3,7 +3,7 @@
3
3
  * `.jsonl`/`.ndjson` → one JSON object per line; anything else (including `.csv`) → CSV.
4
4
  *
5
5
  * This is content binding only (BATCH-1 scope): every row becomes an object of string fields that
6
- * {@link bindCellContent} interpolates into the script. A glob-of-binary-files path binding is out
6
+ * {@link @gaunt-sloth/batch!"interpolate.js".bindCellContent | bindCellContent} interpolates into the script. A glob-of-binary-files path binding is out
7
7
  * of scope for this task.
8
8
  *
9
9
  * Throws a descriptive `Error` on malformed input — the harness-level failure the CLI surface doc
@@ -1,6 +1,4 @@
1
1
  /**
2
- * @module pipelineCli
3
- *
4
2
  * BATCH-9 — the standalone `gth-batch` pipeline runner. A thin entry point that runs the BATCH-1
5
3
  * matrix runtime ({@link buildMatrix} + {@link runBatchMatrix}) directly from a shell pipeline,
6
4
  * without pulling in the whole `gaunt-sloth` app. It takes a prompt-executable script + `--over`
@@ -16,10 +14,12 @@
16
14
  * the pipeline shell already handles) and output is JSONL on stdout (not a directory of files).
17
15
  *
18
16
  * stdout discipline: the run itself is noisy (the runtime's `display()`/`ProgressIndicator`/token
19
- * streaming all target `process.stdout`). The bin entry ({@link file://./bin.ts}) redirects
17
+ * streaming all target `process.stdout`). The bin entry ({@link @gaunt-sloth/batch!"bin.js" | ./bin.ts}) redirects
20
18
  * `process.stdout.write` to stderr for the duration and this module writes the machine JSONL
21
19
  * straight to fd 1 (`fs.writeSync`), so stdout stays a clean data channel — the same "protocol
22
20
  * channel" discipline `packages/app/cli.js` uses for ACP.
21
+ *
22
+ * @module
23
23
  */
24
24
  import type { CommandLineConfigOverrides, GthConfig } from '@gaunt-sloth/core/config.js';
25
25
  import type { BatchSummary, MatrixRow, RunCellFn } from '#src/types.js';
@@ -1,6 +1,4 @@
1
1
  /**
2
- * @module pipelineCli
3
- *
4
2
  * BATCH-9 — the standalone `gth-batch` pipeline runner. A thin entry point that runs the BATCH-1
5
3
  * matrix runtime ({@link buildMatrix} + {@link runBatchMatrix}) directly from a shell pipeline,
6
4
  * without pulling in the whole `gaunt-sloth` app. It takes a prompt-executable script + `--over`
@@ -16,10 +14,12 @@
16
14
  * the pipeline shell already handles) and output is JSONL on stdout (not a directory of files).
17
15
  *
18
16
  * stdout discipline: the run itself is noisy (the runtime's `display()`/`ProgressIndicator`/token
19
- * streaming all target `process.stdout`). The bin entry ({@link file://./bin.ts}) redirects
17
+ * streaming all target `process.stdout`). The bin entry ({@link @gaunt-sloth/batch!"bin.js" | ./bin.ts}) redirects
20
18
  * `process.stdout.write` to stderr for the duration and this module writes the machine JSONL
21
19
  * straight to fd 1 (`fs.writeSync`), so stdout stays a clean data channel — the same "protocol
22
20
  * channel" discipline `packages/app/cli.js` uses for ACP.
21
+ *
22
+ * @module
23
23
  */
24
24
  import { writeSync } from 'node:fs';
25
25
  import { readFileSync } from 'node:fs';
@@ -0,0 +1,140 @@
1
+ import type { BaseChatModel } from '@langchain/core/language_models/chat_models';
2
+ import type { RaterCallCapture } from '@gaunt-sloth/core/core/shell/approvalCapture.js';
3
+ import { buildComposedOpenWorldNote } from '@gaunt-sloth/core/core/shell/openWorld.js';
4
+ import { buildParserPreflightNote } from '@gaunt-sloth/core/core/shell/abstention.js';
5
+ /**
6
+ * The notes an arm can omit: a stable suite-facing token → **core's own builder for that note's
7
+ * exact text**.
8
+ *
9
+ * **Every entry is core's function, never a copy of its prose, and that is the entry requirement.**
10
+ * The off arm is produced by deleting a block from the built prompt, so the text used to find the
11
+ * block has to be the same bytes the builder put there. A second copy in this package would pass
12
+ * review, drift on the first wording change in core, then find nothing to delete — and an arm that
13
+ * deletes nothing sends the on-arm prompt under the off arm's name, so the comparison reports the
14
+ * note having no effect. {@link reconcileArmedCapture} raises on exactly that, but a registry built
15
+ * from copies would be relying on a runtime check to catch a defect the design need not have.
16
+ *
17
+ * **Two of `buildRaterPrompt`'s four preflight notes are therefore absent**: the script-env-leak
18
+ * note and the open-world floor note are composed inline in that function and core exports no
19
+ * builder for either. They become expressible the day core extracts one — which is a change to core
20
+ * with its own justification, not something to smuggle in here as a copied string.
21
+ */
22
+ export declare const RATER_PROMPT_NOTES: {
23
+ /** [[EXT-81]]'s note: the data flow across the parts of a command our parser could not resolve
24
+ * as a whole. The note the A/B that motivated this node is about. */
25
+ readonly 'composed-open-world': typeof buildComposedOpenWorldNote;
26
+ /** The parser's own abstention note: what about the command's shape could not be resolved. */
27
+ readonly parser: typeof buildParserPreflightNote;
28
+ };
29
+ /** A note name a suite may omit — the keys of {@link RATER_PROMPT_NOTES}. */
30
+ export type RaterPromptNoteName = keyof typeof RATER_PROMPT_NOTES;
31
+ /** Every note name a suite may write, in declaration order, for error messages and validation. */
32
+ export declare const RATER_PROMPT_NOTE_NAMES: RaterPromptNoteName[];
33
+ /**
34
+ * One sweep cell's prompt arm: which of our own preflight notes this cell's ratings go out without.
35
+ *
36
+ * `omit: []` is the meaningful, and required, spelling of the baseline arm. A sweep value that
37
+ * declared nothing at all would be indistinguishable from an authoring slip, and the whole point of
38
+ * an A/B is that both arms say what they are.
39
+ */
40
+ export interface RaterPromptArm {
41
+ readonly omit: readonly RaterPromptNoteName[];
42
+ }
43
+ /** The prompt as it actually left for the model. */
44
+ export interface ArmedPrompt {
45
+ system: string;
46
+ user: string;
47
+ }
48
+ /**
49
+ * What one armed rating call did — recorded by the decorator, adjudicated afterwards by
50
+ * {@link reconcileArmedCapture}.
51
+ */
52
+ export interface RaterPromptArmOutcome {
53
+ /** Whether the decorated model was invoked at all. `false` means `rateShellCommand` returned
54
+ * before the send site — it found no usable model — so nothing was measured. */
55
+ invoked: boolean;
56
+ /** Notes whose block was found and removed. */
57
+ removed: RaterPromptNoteName[];
58
+ /** Notes that do not apply to this command, so there was no block to remove. This is the
59
+ * byte-identical case and it is silent by design: a command carrying no composed note must
60
+ * produce the same prompt under both arms. */
61
+ absent: RaterPromptNoteName[];
62
+ /** Notes whose block core says exists on this command and the arm could not find in the prompt —
63
+ * a leak. Raised on, never tolerated. */
64
+ leaked: RaterPromptNoteName[];
65
+ /** The prompt as sent, once the arm had finished with it. */
66
+ sent?: ArmedPrompt;
67
+ }
68
+ /**
69
+ * Delete the arm's notes from a built rater user message.
70
+ *
71
+ * Pure, and keyed on the command: each note's block is `'\n\n' + <core's text for this command>`,
72
+ * because `buildRaterPrompt` pushes every note into its line buffer as `('', note)` and joins on
73
+ * `'\n'`. A note that does not apply to the command yields `null` and is recorded as `absent`.
74
+ */
75
+ export declare function stripRaterPromptNotes(user: string, command: string, arm: RaterPromptArm): {
76
+ user: string;
77
+ removed: RaterPromptNoteName[];
78
+ absent: RaterPromptNoteName[];
79
+ leaked: RaterPromptNoteName[];
80
+ };
81
+ /**
82
+ * Wrap a rating model so this call's prompt goes out without the arm's notes.
83
+ *
84
+ * **Why the MODEL and not the prompt builder.** The prompt is built inside `rateShellCommand`, and
85
+ * every seam that could change it there would be a switch in the session's gate. The model is the
86
+ * one thing the gate takes from its caller, so decorating it is the only place a measurement can
87
+ * change the outgoing prompt without core growing a parameter for it. The `eval` rater target is
88
+ * also the only caller that constructs such a model.
89
+ *
90
+ * **A Proxy rather than a hand-written delegate**, because `rateShellCommand` reads more off a model
91
+ * than `withStructuredOutput` — `raterModelLabel` calls `_llmType()` and reads `model`/`modelName`/
92
+ * `modelId` — and a delegate that enumerated today's reads would silently answer `undefined` for
93
+ * tomorrow's. Every trap forwards with the TARGET as the receiver and binds methods to the target,
94
+ * so a provider class keeping state behind private `#fields` is not handed a `this` it cannot use.
95
+ *
96
+ * **A shape it does not recognise is a leak, not a pass-through.** If the messages are not the
97
+ * `[SystemMessage, HumanMessage]` pair with string content that `rateShellCommand` sends, the call
98
+ * is forwarded unchanged and every requested note is recorded as leaked, so the run fails loudly
99
+ * instead of quietly measuring the on-arm prompt under the off arm's name.
100
+ *
101
+ * @param model The resolved rating model.
102
+ * @param command The command this call rates — the notes are a function of it.
103
+ * @param arm The notes to omit.
104
+ * @param outcome The record this call writes into; one per rating call, so concurrent cases cannot
105
+ * share it.
106
+ */
107
+ export declare function armRaterModel(model: BaseChatModel, command: string, arm: RaterPromptArm, outcome: RaterPromptArmOutcome): BaseChatModel;
108
+ /**
109
+ * A fresh outcome plus the model that writes into it — **one per rating call**, because the
110
+ * classifier is reused across cases the suite runner may run concurrently and a shared record would
111
+ * let one case's arm be adjudicated on another's call.
112
+ *
113
+ * `model` is `undefined` only in the state `buildRaterClassifier` refuses to build: an arm declared
114
+ * with no model to decorate. It is admitted here rather than guarded a second time, so that state
115
+ * lands on {@link reconcileArmedCapture}'s "never reached a rating call" arm — one rule, one place —
116
+ * instead of on a copy of the rule that could drift from it.
117
+ */
118
+ export declare function armFor(model: BaseChatModel | undefined, command: string, arm: RaterPromptArm): {
119
+ model: BaseChatModel | undefined;
120
+ outcome: RaterPromptArmOutcome;
121
+ };
122
+ /**
123
+ * Adjudicate one armed rating call, **after it has returned and outside `rateShellCommand`'s
124
+ * fail-closed `try`**, and repair the call's diagnostic record.
125
+ *
126
+ * Two jobs, and both are about not lying:
127
+ *
128
+ * - **Raise on an arm that did not do what it said.** A leaked note, or a rating call the decorated
129
+ * model never saw, means this cell's numbers are the other arm's numbers under this arm's name.
130
+ * Silently reporting them is strictly worse than failing the run, because the comparison table
131
+ * would then say the note changed nothing.
132
+ * - **Make `capture.prompt` the prompt that was SENT.** `rateShellCommand` fills the capture from
133
+ * `buildRaterPrompt`'s output at the send site and its contract is that nothing downstream may
134
+ * leave the archive disagreeing with what the model saw. The arm edits the message below that
135
+ * point, so the arm is what owes the correction — and for a facility whose whole subject is prompt
136
+ * content, a diagnostic record of the wrong arm's prompt would be the worst possible artifact.
137
+ *
138
+ * @param capture The record `rateShellCommand` handed back through `onCapture`, when it made one.
139
+ */
140
+ export declare function reconcileArmedCapture(command: string, arm: RaterPromptArm, outcome: RaterPromptArmOutcome, capture: RaterCallCapture | undefined): void;