@gaunt-sloth/batch 2.0.0-beta.12 → 2.0.0-beta.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/bin.d.ts +2 -2
- package/dist/bin.js +2 -2
- package/dist/classificationReport.d.ts +2 -2
- package/dist/classificationReport.js +2 -2
- package/dist/classificationTypes.d.ts +7 -7
- package/dist/classificationTypes.js +1 -1
- package/dist/evalCompare.d.ts +11 -0
- package/dist/evalCompare.js +7 -0
- package/dist/evalCompare.js.map +1 -1
- package/dist/evalOutput.d.ts +1 -1
- package/dist/evalOutput.js +1 -1
- package/dist/evalRunner.d.ts +3 -3
- package/dist/evalRunner.js +25 -4
- package/dist/evalRunner.js.map +1 -1
- package/dist/evalSuite.d.ts +5 -1
- package/dist/evalSuite.js +152 -6
- package/dist/evalSuite.js.map +1 -1
- package/dist/evalTypes.d.ts +55 -18
- package/dist/evalTypes.js +2 -2
- package/dist/evalTypes.js.map +1 -1
- package/dist/index.d.ts +2 -0
- package/dist/index.js +5 -0
- package/dist/index.js.map +1 -1
- package/dist/judge.d.ts +2 -2
- package/dist/judge.js +10 -14
- package/dist/judge.js.map +1 -1
- package/dist/output.d.ts +1 -1
- package/dist/output.js +1 -1
- package/dist/parseOver.d.ts +1 -1
- package/dist/parseOver.js +1 -1
- package/dist/pipelineCli.d.ts +3 -3
- package/dist/pipelineCli.js +3 -3
- package/dist/raterPromptArm.d.ts +140 -0
- package/dist/raterPromptArm.js +306 -0
- package/dist/raterPromptArm.js.map +1 -0
- package/dist/raterTarget.d.ts +34 -12
- package/dist/raterTarget.js +87 -27
- package/dist/raterTarget.js.map +1 -1
- package/dist/reporters/registry.d.ts +1 -1
- package/dist/reporters/registry.js +1 -1
- package/dist/reporters/registry.js.map +1 -1
- package/dist/reporters/reporterTypes.d.ts +3 -3
- package/dist/types.d.ts +2 -2
- package/dist/workflow/runWorkflow.d.ts +4 -4
- package/dist/workflow/runWorkflow.js +3 -3
- package/package.json +5 -4
package/dist/evalTypes.d.ts
CHANGED
|
@@ -1,13 +1,14 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* @packageDocumentation
|
|
3
3
|
* BATCH-2 — the shapes for `gth eval`: a parsed suite/case, deterministic-check results, the
|
|
4
|
-
* judge's verdict, and per-case/suite outcomes. Deliberately separate from
|
|
4
|
+
* judge's verdict, and per-case/suite outcomes. Deliberately separate from `types.js`
|
|
5
5
|
* (BATCH-1's cell/outcome shapes), which documents itself as scoped to "cells and outcomes" only
|
|
6
6
|
* — eval's shapes layer on top of (not into) that file.
|
|
7
7
|
*/
|
|
8
8
|
import type { ApprovalRung } from '@gaunt-sloth/core/config/shell-policy.js';
|
|
9
9
|
import type { PreflightFloorKind } from '@gaunt-sloth/core/core/shell/raterVocabulary.js';
|
|
10
10
|
import type { ToolResultRecord } from '#src/types.js';
|
|
11
|
+
import type { RaterPromptArm } from '#src/raterPromptArm.js';
|
|
11
12
|
import type { AdvertisedToolInventory, ToolCoverageReport, ToolCoverageSpec } from '#src/toolCoverage.js';
|
|
12
13
|
import type { EvalCaseClassification, EvalClassificationReport, EvalClassificationSpec, EvalMetricSpec } from '#src/classificationTypes.js';
|
|
13
14
|
/** The 0-10 judge scale's default pass threshold, matching `review`'s own (unexported)
|
|
@@ -72,7 +73,7 @@ declare const PREFLIGHT_MECHANISM_OF: {
|
|
|
72
73
|
* The floor is spelled here because it is not a preflight — it is a lexical scan core keeps no list
|
|
73
74
|
* of, consulted by the approvals gate before any rating (every rung but `bypass`) and again by the
|
|
74
75
|
* shell tool at execution time (every rung). The preflights are NOT spelled: they come from
|
|
75
|
-
*
|
|
76
|
+
* `PREFLIGHT_MECHANISM_OF`, which is core's own `PREFLIGHT_FLOOR_KINDS` with our `forced_by`
|
|
76
77
|
* spelling attached.
|
|
77
78
|
*
|
|
78
79
|
* ## Why a case asserts a mechanism at all (the I1 finding)
|
|
@@ -159,7 +160,7 @@ export declare const PREFLIGHT_MECHANISMS: ("open-world-preflight" | "script-env
|
|
|
159
160
|
export type PreflightMechanism = (typeof PREFLIGHT_MECHANISM_OF)[PreflightFloorKind];
|
|
160
161
|
/**
|
|
161
162
|
* The `forced_by` spelling of one of core's preflight arms — the typed lookup into
|
|
162
|
-
*
|
|
163
|
+
* `PREFLIGHT_MECHANISM_OF`, which stays private so this is the only way in.
|
|
163
164
|
*
|
|
164
165
|
* TOTAL over core's `PreflightFloorKind` and therefore never `undefined`, which is the point: a
|
|
165
166
|
* caller cannot be handed an arm it has no name for, because such an arm would not have compiled.
|
|
@@ -179,7 +180,7 @@ export declare function mechanismNeedsPermissiveRating(mechanism: ForcedByMechan
|
|
|
179
180
|
export interface GthAgentTarget {
|
|
180
181
|
type: 'gth-agent';
|
|
181
182
|
/** Suite-level profile hint. Only `undefined`/`'default'` is accepted (see
|
|
182
|
-
*
|
|
183
|
+
* `evalSuite.js`'s `parseEvalSuite`) — per-case/per-identity profile switching is
|
|
183
184
|
* `identities` scope, not this. */
|
|
184
185
|
profile?: string;
|
|
185
186
|
}
|
|
@@ -241,7 +242,7 @@ export interface AgUiAgentTarget {
|
|
|
241
242
|
* §8 hardline floor — decides some commands outright, with no model in the loop.
|
|
242
243
|
*
|
|
243
244
|
* Unlike the agent targets this one is NOT driven by a `RunCellFn`: it supplies the
|
|
244
|
-
* {@link RunClassifyFn} seam instead (`buildRaterClassifier` in
|
|
245
|
+
* {@link RunClassifyFn} seam instead (`buildRaterClassifier` in `raterTarget.js`), which
|
|
245
246
|
* the command passes to `runEvalSuite` as `options.classify`.
|
|
246
247
|
*/
|
|
247
248
|
export interface RaterTarget {
|
|
@@ -253,7 +254,7 @@ export interface RaterTarget {
|
|
|
253
254
|
* It is the suite's declaration, not the last word: a run whose config declares `approvals`
|
|
254
255
|
* overrides it, which is what makes the `rung × model` sweep work through the existing `config:`
|
|
255
256
|
* axis (a sweep cell cannot reach a `target` field). The override is announced, never silent —
|
|
256
|
-
* see
|
|
257
|
+
* see `raterTarget.js`.
|
|
257
258
|
*/
|
|
258
259
|
rung: ApprovalRung;
|
|
259
260
|
}
|
|
@@ -264,7 +265,7 @@ export interface RaterTarget {
|
|
|
264
265
|
* target), so the target only changes which runner the command builds. */
|
|
265
266
|
export type EvalTarget = GthAgentTarget | AdkAgentTarget | AgUiAgentTarget | RaterTarget;
|
|
266
267
|
/** One `json_path` assertion (BATCH-10): resolve `path` against the answer-parsed-as-JSON and check
|
|
267
|
-
* it. Exactly one of `equals`/`contains` is set (enforced in
|
|
268
|
+
* it. Exactly one of `equals`/`contains` is set (enforced in `evalSuite.js`'s parse):
|
|
268
269
|
* - `equals` — the resolved value must deep-equal this (any JSON value, incl. `null`).
|
|
269
270
|
* - `contains` — the resolved value must be a string containing this substring. */
|
|
270
271
|
export interface JsonPathCheck {
|
|
@@ -276,7 +277,7 @@ export interface JsonPathCheck {
|
|
|
276
277
|
* One `tool_result_json_path` assertion (BATCH-21): select tool results by name `tool` (exact or
|
|
277
278
|
* glob, the same matcher `must_call` uses), parse each matching result's payload as JSON, and
|
|
278
279
|
* resolve `path` against it (the same minimal dot/`[index]` path `json_path` uses). At most one of
|
|
279
|
-
* `equals`/`contains` may be set (enforced in
|
|
280
|
+
* `equals`/`contains` may be set (enforced in `evalSuite.js`'s parse):
|
|
280
281
|
* - neither — pure existence check: the path must resolve in a matching result's payload;
|
|
281
282
|
* - `equals` — the resolved value must deep-equal this (any JSON value, incl. `null`);
|
|
282
283
|
* - `contains` — the resolved value must be a string containing this substring.
|
|
@@ -360,6 +361,31 @@ export interface EvalExpectation {
|
|
|
360
361
|
* touching the first.
|
|
361
362
|
*/
|
|
362
363
|
expectAction?: string;
|
|
364
|
+
/**
|
|
365
|
+
* BATCH-45 — **assert that a MODEL actually rendered a verdict for this round**, i.e. that
|
|
366
|
+
* {@link ClassifyOutcome.modelLabel} is present. `true` is the only legal value; `undefined` =
|
|
367
|
+
* this block makes no such claim. `rater` target only.
|
|
368
|
+
*
|
|
369
|
+
* It is NOT an assertion about the rung, and NOT an assertion about WHAT the model said. Those
|
|
370
|
+
* are {@link expectLabel} and {@link expectAction}, and the whole point of this key is that it
|
|
371
|
+
* says neither: a case can pin "a rater ruled here" without predicting which verdict three
|
|
372
|
+
* different raters would render, which is a prediction no measurement backs.
|
|
373
|
+
*
|
|
374
|
+
* **Why the action column cannot do this job.** When the gate never obtains a rating (timeout,
|
|
375
|
+
* throw, unparseable answer) it fails closed, and since EXT-171 that ESCALATES rather than
|
|
376
|
+
* negotiating. A genuine `catastrophic` verdict escalates too. So `expect_action: escalate` is
|
|
377
|
+
* satisfied identically by a rater that judged the command unnegotiable and by a rater that
|
|
378
|
+
* never answered at all — and a case asserting only that passes on a run where nothing was
|
|
379
|
+
* measured. BATCH-45 found ten such cells live in this repo's own approvals corpus.
|
|
380
|
+
*
|
|
381
|
+
* **Why `modelLabel` and not the rationale text.** `modelLabel` is suppressed from the call's own
|
|
382
|
+
* capture (`rating.failClosed`), never from a reading of the verdict's prose. Core's
|
|
383
|
+
* `isFailClosed` answers a question about the reason TEXT, and the rating prompt tells a rater to
|
|
384
|
+
* answer `destructive` and say it could not assess a command it is unsure of — so that predicate
|
|
385
|
+
* scores an obedient judgement as a gate failure. EXT-171 and `raterHealth` both say never to key
|
|
386
|
+
* on verdict text.
|
|
387
|
+
*/
|
|
388
|
+
expectRated?: boolean;
|
|
363
389
|
}
|
|
364
390
|
/**
|
|
365
391
|
* One conversational turn (BATCH-12): the `user` message to send, and the {@link EvalExpectation}
|
|
@@ -379,7 +405,7 @@ export interface EvalTurn {
|
|
|
379
405
|
*
|
|
380
406
|
* **Carrying it is not the same as showing it to the rater.** §5.1 admits the justification from
|
|
381
407
|
* round 2 onward, so the round-1 rating withholds it; the target hands the whole context to core's
|
|
382
|
-
* {@link
|
|
408
|
+
* {@link @gaunt-sloth/core!core/shell/negotiation.ShellNegotiationState | ShellNegotiationState}, which is
|
|
383
409
|
* the single implementation of that rule.
|
|
384
410
|
*/
|
|
385
411
|
justification?: string;
|
|
@@ -411,7 +437,7 @@ export interface EvalTurn {
|
|
|
411
437
|
}
|
|
412
438
|
/**
|
|
413
439
|
* One turn's raw run outcome inside a multi-turn conversation (BATCH-12 Task 2). Structurally the
|
|
414
|
-
* per-turn analogue of BATCH-1's {@link
|
|
440
|
+
* per-turn analogue of BATCH-1's {@link "types.js"!CellRunOutcome | CellRunOutcome}: the turn's answer plus the
|
|
415
441
|
* tools/tokens captured FOR THAT TURN — a per-invoke GS2-16 delta (each `processMessages` call
|
|
416
442
|
* resets the tally), NOT the cumulative conversation total. `ok:false` with `error` set means that
|
|
417
443
|
* turn's SUT invocation failed (no answer to grade). The conversational runner returns one of these
|
|
@@ -436,8 +462,8 @@ export interface TurnRunOutcome {
|
|
|
436
462
|
}
|
|
437
463
|
/**
|
|
438
464
|
* Injectable "run one whole scripted conversation" function (BATCH-12 Task 2) — the multi-turn
|
|
439
|
-
* analogue of BATCH-1's {@link
|
|
440
|
-
*
|
|
465
|
+
* analogue of BATCH-1's {@link "types.js"!RunCellFn | RunCellFn}, and the seam that lets
|
|
466
|
+
* `evalRunner.js`'s multi-turn path be unit tested without any real LLM/MCP. Given the
|
|
441
467
|
* ordered user messages of ONE (case × identity) conversation, it builds the agent + resolves tools
|
|
442
468
|
* ONCE, runs each turn against the accumulated message history, and returns one {@link TurnRunOutcome}
|
|
443
469
|
* per turn (per-turn answer + per-turn tool delta), cleaning up once. The production wiring
|
|
@@ -454,7 +480,7 @@ export type RunConversationFn = (userMessages: string[]) => Promise<TurnRunOutco
|
|
|
454
480
|
* a round-1 rating, and a reset makes a later round a round-1 rating again — so the target hands the
|
|
455
481
|
* accumulated state to `ShellNegotiationState.contextFor` rather than deciding here. A shape that
|
|
456
482
|
* carried "what the rater sees" would be this package holding a second opinion about §5.1, which is
|
|
457
|
-
* the one thing
|
|
483
|
+
* the one thing `raterTarget.js` exists not to do.
|
|
458
484
|
*/
|
|
459
485
|
export interface ClassifyRound {
|
|
460
486
|
/** The command the agent proposes in this round. */
|
|
@@ -472,7 +498,7 @@ export interface ClassifyRound {
|
|
|
472
498
|
* ## The Half-B seam, now filled
|
|
473
499
|
*
|
|
474
500
|
* Half A shipped this shape, the runner plumbing and the tests (against an injected fake); Half B
|
|
475
|
-
* supplies the one function — `buildRaterClassifier` in
|
|
501
|
+
* supplies the one function — `buildRaterClassifier` in `raterTarget.js`, the
|
|
476
502
|
* {@link RaterTarget}'s implementation, which drives the approvals rating prompt + decision mapping
|
|
477
503
|
* at a declared rung. The shape did not have to move to fit it.
|
|
478
504
|
*
|
|
@@ -648,7 +674,7 @@ export interface EvalCase {
|
|
|
648
674
|
*/
|
|
649
675
|
modelFree: boolean;
|
|
650
676
|
}
|
|
651
|
-
/** A fully parsed and validated suite — see
|
|
677
|
+
/** A fully parsed and validated suite — see `evalSuite.js`'s `parseEvalSuite`. */
|
|
652
678
|
export interface EvalSuite {
|
|
653
679
|
target: EvalTarget;
|
|
654
680
|
/**
|
|
@@ -690,7 +716,7 @@ export interface EvalSuite {
|
|
|
690
716
|
* BATCH-25 — the config sweep: named cells the WHOLE suite is run once per, so one corpus
|
|
691
717
|
* produces one comparison table instead of N unrelated runs. Absent = a single run.
|
|
692
718
|
*
|
|
693
|
-
* Consumed by the CLI, not by {@link
|
|
719
|
+
* Consumed by the CLI, not by {@link "evalRunner.js"!runEvalSuite | runEvalSuite}: a sweep is a run-level
|
|
694
720
|
* concept (same suite, different config) exactly like BATCH-19's multi-suite loop, so the runner
|
|
695
721
|
* stays about grading and #405's identity matrix is untouched.
|
|
696
722
|
*/
|
|
@@ -715,6 +741,17 @@ export interface EvalSweepValue {
|
|
|
715
741
|
model?: string;
|
|
716
742
|
/** Plain-data config overrides deep-merged onto the resolved config (never `llm`). */
|
|
717
743
|
config?: Record<string, unknown>;
|
|
744
|
+
/**
|
|
745
|
+
* [[BATCH-31]] — the cell's PROMPT ARM: which of gth's own preflight notes this cell's ratings go
|
|
746
|
+
* out without, so a note-on / note-off A/B is one suite file rather than two. `rater` targets
|
|
747
|
+
* only; `{ omit: [] }` is the baseline arm and is required rather than implied.
|
|
748
|
+
*
|
|
749
|
+
* **A third override kind, not a third spelling of `config:`.** It is deliberately not a config
|
|
750
|
+
* key and never merges into one: a config key is exactly what a session could also be given, and
|
|
751
|
+
* an omission a session can reach is a switch that suppresses safety context in the live approvals
|
|
752
|
+
* gate. See `raterPromptArm.ts` for the full argument and for how that is enforced.
|
|
753
|
+
*/
|
|
754
|
+
notes?: RaterPromptArm;
|
|
718
755
|
}
|
|
719
756
|
/**
|
|
720
757
|
* BATCH-25 — the sweep: named axes whose CARTESIAN PRODUCT is the set of runs. The approvals
|
|
@@ -736,7 +773,7 @@ export interface DeterministicCheckResult {
|
|
|
736
773
|
failures: string[];
|
|
737
774
|
}
|
|
738
775
|
/** The judge's structured verdict on one case's answer — matches `review`'s `RateSchema` shape
|
|
739
|
-
* (0-10 `rate` + a reason string) for UX consistency, see
|
|
776
|
+
* (0-10 `rate` + a reason string) for UX consistency, see `judge.js`. */
|
|
740
777
|
export interface JudgeVerdict {
|
|
741
778
|
rate: number;
|
|
742
779
|
reason: string;
|
|
@@ -754,7 +791,7 @@ export interface JudgeOutcome {
|
|
|
754
791
|
error?: string;
|
|
755
792
|
}
|
|
756
793
|
/** Injectable "grade one answer against one rubric" function — the seam that lets
|
|
757
|
-
*
|
|
794
|
+
* `evalRunner.js`'s `runEvalSuite` be fully unit tested without any real LLM call, mirror
|
|
758
795
|
* of BATCH-1's `RunCellFn`. The production wiring (`evalCommand.ts`) adapts `judgeEvalCase` to
|
|
759
796
|
* this shape; tests inject a fake that resolves/fails as needed. */
|
|
760
797
|
export type JudgeFn = (answer: string, rubric: string) => Promise<JudgeOutcome>;
|
package/dist/evalTypes.js
CHANGED
|
@@ -60,7 +60,7 @@ const PREFLIGHT_MECHANISM_OF = {
|
|
|
60
60
|
* The floor is spelled here because it is not a preflight — it is a lexical scan core keeps no list
|
|
61
61
|
* of, consulted by the approvals gate before any rating (every rung but `bypass`) and again by the
|
|
62
62
|
* shell tool at execution time (every rung). The preflights are NOT spelled: they come from
|
|
63
|
-
*
|
|
63
|
+
* `PREFLIGHT_MECHANISM_OF`, which is core's own `PREFLIGHT_FLOOR_KINDS` with our `forced_by`
|
|
64
64
|
* spelling attached.
|
|
65
65
|
*
|
|
66
66
|
* ## Why a case asserts a mechanism at all (the I1 finding)
|
|
@@ -132,7 +132,7 @@ export const FORCED_BY_ASSERTIONS = {
|
|
|
132
132
|
export const PREFLIGHT_MECHANISMS = Object.values(PREFLIGHT_MECHANISM_OF);
|
|
133
133
|
/**
|
|
134
134
|
* The `forced_by` spelling of one of core's preflight arms — the typed lookup into
|
|
135
|
-
*
|
|
135
|
+
* `PREFLIGHT_MECHANISM_OF`, which stays private so this is the only way in.
|
|
136
136
|
*
|
|
137
137
|
* TOTAL over core's `PreflightFloorKind` and therefore never `undefined`, which is the point: a
|
|
138
138
|
* caller cannot be handed an arm it has no name for, because such an arm would not have compiled.
|
package/dist/evalTypes.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"evalTypes.js","sourceRoot":"","sources":["../src/evalTypes.ts"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"evalTypes.js","sourceRoot":"","sources":["../src/evalTypes.ts"],"names":[],"mappings":"AA6BA;;;;mGAImG;AACnG,MAAM,CAAC,MAAM,2BAA2B,GAAG,CAAC,CAAC;AAE7C;;;;;;;;;;;;;;;;;;;;;;;;GAwBG;AACH,MAAM,CAAC,MAAM,uBAAuB,GAAG,yBAAyB,CAAC;AAEjE;;;;;;;;;;;;;;;;;;GAkBG;AACH,MAAM,sBAAsB,GAAG;IAC7B,iBAAiB,EAAE,2BAA2B;IAC9C,YAAY,EAAE,sBAAsB;CAC2B,CAAC;AAElE;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG;IAClC,gBAAgB;IAChB,GAAG,MAAM,CAAC,MAAM,CAAC,sBAAsB,CAAC;CAChC,CAAC;AAiBX,mGAAmG;AACnG,MAAM,CAAC,MAAM,gBAAgB,GAAG,WAAW,CAAC;AAE5C;;;;GAIG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAsC;IACrE,gBAAgB,EAAE,uBAAuB;IACzC,2BAA2B,EAAE,GAAG,gBAAgB,6BAA6B;IAC7E,sBAAsB,EAAE,GAAG,gBAAgB,wBAAwB;CACpE,CAAC;AAEF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAsCG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG,MAAM,CAAC,MAAM,CAAC,sBAAsB,CAAC,CAAC;AAW1E;;;;;;GAMG;AACH,MAAM,UAAU,qBAAqB,CAAC,IAAwB;IAC5D,OAAO,sBAAsB,CAAC,IAAI,CAAC,CAAC;AACtC,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,8BAA8B,CAAC,SAAwC;IACrF,OAAO,CACL,SAAS,KAAK,SAAS;QACtB,oBAAqD,CAAC,QAAQ,CAAC,SAAS,CAAC,CAC3E,CAAC;AACJ,CAAC"}
|
package/dist/index.d.ts
CHANGED
|
@@ -31,6 +31,8 @@ export type { ClassificationExtractor, ClassifiedCell, EvalCaseClassification, E
|
|
|
31
31
|
export type { ClassifyOutcome, ClassifyRequest, ClassifyRound, EvalSweep, EvalSweepValue, RaterTarget, RunClassifyFn, } from '#src/evalTypes.js';
|
|
32
32
|
export { buildRaterClassifier, HARDLINE_NO_RATING_REASON, HARDLINE_REFUSAL_MARKER, NEGOTIATION_BOUND_MARKER, NO_RATING_CALL_MARKER, } from '#src/raterTarget.js';
|
|
33
33
|
export type { RaterClassifierOptions } from '#src/raterTarget.js';
|
|
34
|
+
export { RATER_PROMPT_NOTE_NAMES } from '#src/raterPromptArm.js';
|
|
35
|
+
export type { RaterPromptArm, RaterPromptNoteName } from '#src/raterPromptArm.js';
|
|
34
36
|
export { resolveReporters, availableReporterNames } from '#src/reporters/registry.js';
|
|
35
37
|
export { driveReporters } from '#src/reporters/drive.js';
|
|
36
38
|
export { createTextReporter } from '#src/reporters/textReporter.js';
|
package/dist/index.js
CHANGED
|
@@ -26,6 +26,11 @@ export { UNRECOGNIZED_LABEL, NO_EXPECTATION } from '#src/classificationTypes.js'
|
|
|
26
26
|
// BATCH-25 Half B — the `rater` target: the one implementation of the `RunClassifyFn` seam, which
|
|
27
27
|
// drives gth's own approvals rater over a corpus of commands.
|
|
28
28
|
export { buildRaterClassifier, HARDLINE_NO_RATING_REASON, HARDLINE_REFUSAL_MARKER, NEGOTIATION_BOUND_MARKER, NO_RATING_CALL_MARKER, } from '#src/raterTarget.js';
|
|
29
|
+
// [[BATCH-31]] — the prompt arm: the note-on / note-off A/B a rater suite can express. Only the
|
|
30
|
+
// NAMES and the type are exported. `armRaterModel` deliberately is not — nothing outside this
|
|
31
|
+
// package needs to build one, and the narrower the surface the smaller the chance of a caller
|
|
32
|
+
// wiring an omission somewhere the argument in `raterPromptArm.ts` does not cover.
|
|
33
|
+
export { RATER_PROMPT_NOTE_NAMES } from '#src/raterPromptArm.js';
|
|
29
34
|
// BATCH-19 — the `gth eval` reporter facility (A1 seam). These are the public plugin contract an
|
|
30
35
|
// out-of-core `@gaunt-sloth/eval-reporter-*` package implements, exported from the package root so a
|
|
31
36
|
// reporter package can type its factory against ONE import.
|
package/dist/index.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,WAAW,EAAE,MAAM,gBAAgB,CAAC;AAC7C,OAAO,EAAE,eAAe,EAAE,MAAM,qBAAqB,CAAC;AACtD,OAAO,EAAE,aAAa,EAAE,MAAM,mBAAmB,CAAC;AAClD,OAAO,EAAE,cAAc,EAAE,iBAAiB,EAAE,MAAM,qBAAqB,CAAC;AACxE,OAAO,EAAE,gBAAgB,EAAE,MAAM,gBAAgB,CAAC;AAWlD,OAAO,EAAE,wBAAwB,EAAE,MAAM,eAAe,CAAC;AAEzD,gGAAgG;AAChG,OAAO,EAAE,cAAc,EAAE,MAAM,mBAAmB,CAAC;AACnD,OAAO,EAAE,sBAAsB,EAAE,MAAM,6BAA6B,CAAC;AACrE,OAAO,EACL,aAAa,EACb,kBAAkB,EAClB,iBAAiB,EACjB,6BAA6B,GAC9B,MAAM,eAAe,CAAC;AAEvB,OAAO,EAAE,YAAY,EAAE,MAAM,oBAAoB,CAAC;AAClD,OAAO,EAAE,eAAe,EAAE,MAAM,oBAAoB,CAAC;AAqBrD,OAAO,EAAE,2BAA2B,EAAE,MAAM,mBAAmB,CAAC;AAEhE,mGAAmG;AACnG,8FAA8F;AAC9F,OAAO,EACL,0BAA0B,EAC1B,oBAAoB,EACpB,WAAW,EACX,OAAO,GACR,MAAM,wBAAwB,CAAC;AAChC,OAAO,EAAE,yBAAyB,EAAE,MAAM,8BAA8B,CAAC;AACzE,OAAO,EACL,aAAa,EACb,oBAAoB,EACpB,iBAAiB,EACjB,kBAAkB,EAClB,WAAW,GACZ,MAAM,iBAAiB,CAAC;AACzB,OAAO,EACL,0BAA0B,EAC1B,qBAAqB,EACrB,YAAY,GACb,MAAM,8BAA8B,CAAC;AACtC,qFAAqF;AACrF,OAAO,EACL,mBAAmB,EACnB,qBAAqB,EACrB,2BAA2B,GAC5B,MAAM,sBAAsB,CAAC;AAS9B,OAAO,EAAE,kBAAkB,EAAE,MAAM,4BAA4B,CAAC;AAEhE,OAAO,EAAE,gBAAgB,EAAE,WAAW,EAAE,iBAAiB,EAAE,MAAM,qBAAqB,CAAC;AAEvF,OAAO,EACL,WAAW,EACX,SAAS,EACT,gBAAgB,EAChB,QAAQ,EACR,aAAa,EACb,qBAAqB,EACrB,0BAA0B,EAC1B,6BAA6B,EAC7B,yBAAyB,GAC1B,MAAM,qBAAqB,CAAC;AAU7B,OAAO,EAAE,kBAAkB,EAAE,cAAc,EAAE,MAAM,6BAA6B,CAAC;AAyBjF,kGAAkG;AAClG,8DAA8D;AAC9D,OAAO,EACL,oBAAoB,EACpB,yBAAyB,EACzB,uBAAuB,EACvB,wBAAwB,EACxB,qBAAqB,GACtB,MAAM,qBAAqB,CAAC;AAG7B,iGAAiG;AACjG,qGAAqG;AACrG,4DAA4D;AAC5D,OAAO,EAAE,gBAAgB,EAAE,sBAAsB,EAAE,MAAM,4BAA4B,CAAC;AACtF,OAAO,EAAE,cAAc,EAAE,MAAM,yBAAyB,CAAC;AACzD,OAAO,EAAE,kBAAkB,EAAE,MAAM,gCAAgC,CAAC;AAQpE,kGAAkG;AAClG,6CAA6C;AAC7C,OAAO,EAAE,WAAW,EAAE,MAAM,8BAA8B,CAAC;AAC3D,OAAO,EAAE,4BAA4B,EAAE,MAAM,eAAe,CAAC"}
|
|
1
|
+
{"version":3,"file":"index.js","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,WAAW,EAAE,MAAM,gBAAgB,CAAC;AAC7C,OAAO,EAAE,eAAe,EAAE,MAAM,qBAAqB,CAAC;AACtD,OAAO,EAAE,aAAa,EAAE,MAAM,mBAAmB,CAAC;AAClD,OAAO,EAAE,cAAc,EAAE,iBAAiB,EAAE,MAAM,qBAAqB,CAAC;AACxE,OAAO,EAAE,gBAAgB,EAAE,MAAM,gBAAgB,CAAC;AAWlD,OAAO,EAAE,wBAAwB,EAAE,MAAM,eAAe,CAAC;AAEzD,gGAAgG;AAChG,OAAO,EAAE,cAAc,EAAE,MAAM,mBAAmB,CAAC;AACnD,OAAO,EAAE,sBAAsB,EAAE,MAAM,6BAA6B,CAAC;AACrE,OAAO,EACL,aAAa,EACb,kBAAkB,EAClB,iBAAiB,EACjB,6BAA6B,GAC9B,MAAM,eAAe,CAAC;AAEvB,OAAO,EAAE,YAAY,EAAE,MAAM,oBAAoB,CAAC;AAClD,OAAO,EAAE,eAAe,EAAE,MAAM,oBAAoB,CAAC;AAqBrD,OAAO,EAAE,2BAA2B,EAAE,MAAM,mBAAmB,CAAC;AAEhE,mGAAmG;AACnG,8FAA8F;AAC9F,OAAO,EACL,0BAA0B,EAC1B,oBAAoB,EACpB,WAAW,EACX,OAAO,GACR,MAAM,wBAAwB,CAAC;AAChC,OAAO,EAAE,yBAAyB,EAAE,MAAM,8BAA8B,CAAC;AACzE,OAAO,EACL,aAAa,EACb,oBAAoB,EACpB,iBAAiB,EACjB,kBAAkB,EAClB,WAAW,GACZ,MAAM,iBAAiB,CAAC;AACzB,OAAO,EACL,0BAA0B,EAC1B,qBAAqB,EACrB,YAAY,GACb,MAAM,8BAA8B,CAAC;AACtC,qFAAqF;AACrF,OAAO,EACL,mBAAmB,EACnB,qBAAqB,EACrB,2BAA2B,GAC5B,MAAM,sBAAsB,CAAC;AAS9B,OAAO,EAAE,kBAAkB,EAAE,MAAM,4BAA4B,CAAC;AAEhE,OAAO,EAAE,gBAAgB,EAAE,WAAW,EAAE,iBAAiB,EAAE,MAAM,qBAAqB,CAAC;AAEvF,OAAO,EACL,WAAW,EACX,SAAS,EACT,gBAAgB,EAChB,QAAQ,EACR,aAAa,EACb,qBAAqB,EACrB,0BAA0B,EAC1B,6BAA6B,EAC7B,yBAAyB,GAC1B,MAAM,qBAAqB,CAAC;AAU7B,OAAO,EAAE,kBAAkB,EAAE,cAAc,EAAE,MAAM,6BAA6B,CAAC;AAyBjF,kGAAkG;AAClG,8DAA8D;AAC9D,OAAO,EACL,oBAAoB,EACpB,yBAAyB,EACzB,uBAAuB,EACvB,wBAAwB,EACxB,qBAAqB,GACtB,MAAM,qBAAqB,CAAC;AAG7B,gGAAgG;AAChG,8FAA8F;AAC9F,8FAA8F;AAC9F,mFAAmF;AACnF,OAAO,EAAE,uBAAuB,EAAE,MAAM,wBAAwB,CAAC;AAGjE,iGAAiG;AACjG,qGAAqG;AACrG,4DAA4D;AAC5D,OAAO,EAAE,gBAAgB,EAAE,sBAAsB,EAAE,MAAM,4BAA4B,CAAC;AACtF,OAAO,EAAE,cAAc,EAAE,MAAM,yBAAyB,CAAC;AACzD,OAAO,EAAE,kBAAkB,EAAE,MAAM,gCAAgC,CAAC;AAQpE,kGAAkG;AAClG,6CAA6C;AAC7C,OAAO,EAAE,WAAW,EAAE,MAAM,8BAA8B,CAAC;AAC3D,OAAO,EAAE,4BAA4B,EAAE,MAAM,eAAe,CAAC"}
|
package/dist/judge.d.ts
CHANGED
|
@@ -1,6 +1,4 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* @module judge
|
|
3
|
-
*
|
|
4
2
|
* BATCH-2 — LLM-as-judge grading for `gth eval`. Adapts the *mechanism* of EXT-10's shell-safety
|
|
5
3
|
* judge (`packages/core/src/core/shell/judge.ts`): `model.withStructuredOutput(zodSchema)` for a
|
|
6
4
|
* single non-agentic structured call, raced against a timeout. The failure policy differs on
|
|
@@ -18,6 +16,8 @@
|
|
|
18
16
|
* `judgeShellCommand`'s own default. A separate `--judge <profile>` model is BATCH-2's own
|
|
19
17
|
* "Not in scope" list (identity-matrix/pluggable-target work); grading with the SUT's own model
|
|
20
18
|
* config is a known, real simplification for this first slice.
|
|
19
|
+
*
|
|
20
|
+
* @module
|
|
21
21
|
*/
|
|
22
22
|
import type { BaseChatModel } from '@langchain/core/language_models/chat_models';
|
|
23
23
|
import * as z from 'zod';
|
package/dist/judge.js
CHANGED
|
@@ -1,6 +1,4 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* @module judge
|
|
3
|
-
*
|
|
4
2
|
* BATCH-2 — LLM-as-judge grading for `gth eval`. Adapts the *mechanism* of EXT-10's shell-safety
|
|
5
3
|
* judge (`packages/core/src/core/shell/judge.ts`): `model.withStructuredOutput(zodSchema)` for a
|
|
6
4
|
* single non-agentic structured call, raced against a timeout. The failure policy differs on
|
|
@@ -18,8 +16,11 @@
|
|
|
18
16
|
* `judgeShellCommand`'s own default. A separate `--judge <profile>` model is BATCH-2's own
|
|
19
17
|
* "Not in scope" list (identity-matrix/pluggable-target work); grading with the SUT's own model
|
|
20
18
|
* config is a known, real simplification for this first slice.
|
|
19
|
+
*
|
|
20
|
+
* @module
|
|
21
21
|
*/
|
|
22
22
|
import { HumanMessage, SystemMessage } from '@langchain/core/messages';
|
|
23
|
+
import { CALL_TIMED_OUT, withCallDeadline } from '@gaunt-sloth/core/runtime/abortableCall.js';
|
|
23
24
|
import { structuredOutputBoundary } from '@gaunt-sloth/core/runtime/structuredOutput.js';
|
|
24
25
|
import * as z from 'zod';
|
|
25
26
|
/** Structured verdict the judge model must return — 0-10 `rate` + a `reason`, matching `review`'s
|
|
@@ -85,20 +86,19 @@ export async function judgeEvalCase(answer, rubric, model, options) {
|
|
|
85
86
|
return { attempted: true, ok: false, error: 'No usable judge model configured.' };
|
|
86
87
|
}
|
|
87
88
|
const { system, user } = buildJudgeMessages(answer, rubric);
|
|
88
|
-
let timer;
|
|
89
89
|
try {
|
|
90
90
|
// EXT-88 — routed through the shared boundary like every other structured-output call. The
|
|
91
91
|
// rubric verdict has no optional field today, so the boundary hands back this very schema by
|
|
92
92
|
// identity; going through it is what stops a later optional field re-introducing the defect.
|
|
93
93
|
const boundary = structuredOutputBoundary(EvalVerdictSchema);
|
|
94
94
|
const structured = model.withStructuredOutput(boundary.wireSchema);
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
const raced = await
|
|
101
|
-
if (raced ===
|
|
95
|
+
// [[EXT-179]] — the budget aborts the judge call rather than abandoning it. This matters most
|
|
96
|
+
// in a SWEEP: a suite grades many cases, and every stalled judge used to leave its own request
|
|
97
|
+
// in flight, so the leaked sockets accumulated across the run and kept the process alive after
|
|
98
|
+
// the report had been written. The timeout text is unchanged and the case still fails as a
|
|
99
|
+
// case — an aborted judgement is a judgement not obtained, never an auto-pass.
|
|
100
|
+
const raced = await withCallDeadline(timeoutMs, (signal) => structured.invoke([new SystemMessage(system), new HumanMessage(user)], { signal }));
|
|
101
|
+
if (raced === CALL_TIMED_OUT) {
|
|
102
102
|
return { attempted: true, ok: false, error: `Judge timed out after ${timeoutMs}ms.` };
|
|
103
103
|
}
|
|
104
104
|
// withStructuredOutput already coerces to the schema, but re-validate defensively: a fake or
|
|
@@ -116,9 +116,5 @@ export async function judgeEvalCase(answer, rubric, model, options) {
|
|
|
116
116
|
error: error instanceof Error ? error.message : String(error),
|
|
117
117
|
};
|
|
118
118
|
}
|
|
119
|
-
finally {
|
|
120
|
-
if (timer)
|
|
121
|
-
clearTimeout(timer);
|
|
122
|
-
}
|
|
123
119
|
}
|
|
124
120
|
//# sourceMappingURL=judge.js.map
|
package/dist/judge.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"judge.js","sourceRoot":"","sources":["../src/judge.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAGH,OAAO,EAAE,YAAY,EAAE,aAAa,EAAE,MAAM,0BAA0B,CAAC;AACvE,OAAO,EAAE,wBAAwB,EAAE,MAAM,+CAA+C,CAAC;AACzF,OAAO,KAAK,CAAC,MAAM,KAAK,CAAC;AAIzB;;oGAEoG;AACpG,MAAM,CAAC,MAAM,iBAAiB,GAAG,CAAC,CAAC,MAAM,CAAC;IACxC,IAAI,EAAE,CAAC;SACJ,MAAM,EAAE;SACR,GAAG,CAAC,CAAC,CAAC;SACN,GAAG,CAAC,EAAE,CAAC;SACP,QAAQ,CAAC,yDAAyD,CAAC;IACtE,MAAM,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,QAAQ,CAAC,2CAA2C,CAAC;CACzE,CAAC,CAAC;AAIH;;oGAEoG;AACpG,MAAM,CAAC,MAAM,6BAA6B,GAAG,MAAM,CAAC;AAEpD,MAAM,wBAAwB,GAAG;IAC/B,mCAAmC;IACnC,gGAAgG;IAChG,cAAc;IACd,EAAE;IACF,+FAA+F;IAC/F,gGAAgG;IAChG,mDAAmD;IACnD,6FAA6F;IAC7F,+BAA+B;CAChC,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AAEb;;;;GAIG;AACH,MAAM,UAAU,kBAAkB,CAChC,MAAc,EACd,MAAc;IAEd,MAAM,SAAS,GAAG;QAChB,SAAS;QACT,MAAM;QACN,EAAE;QACF,qBAAqB;QACrB,UAAU;QACV,MAAM,IAAI,gBAAgB;QAC1B,WAAW;KACZ,CAAC;IACF,OAAO,EAAE,MAAM,EAAE,wBAAwB,EAAE,IAAI,EAAE,SAAS,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;AAC1E,CAAC;AAED;;;;;;;;;;;;;GAaG;AACH,MAAM,CAAC,KAAK,UAAU,aAAa,CACjC,MAAc,EACd,MAAc,EACd,KAAgC,EAChC,OAAgC;IAEhC,MAAM,SAAS,GAAG,OAAO,EAAE,SAAS,IAAI,6BAA6B,CAAC;IAEtE,IAAI,CAAC,KAAK,IAAI,OAAO,KAAK,CAAC,oBAAoB,KAAK,UAAU,EAAE,CAAC;QAC/D,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,mCAAmC,EAAE,CAAC;IACpF,CAAC;IAED,MAAM,EAAE,MAAM,EAAE,IAAI,EAAE,GAAG,kBAAkB,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IAC5D,IAAI,
|
|
1
|
+
{"version":3,"file":"judge.js","sourceRoot":"","sources":["../src/judge.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAGH,OAAO,EAAE,YAAY,EAAE,aAAa,EAAE,MAAM,0BAA0B,CAAC;AACvE,OAAO,EAAE,cAAc,EAAE,gBAAgB,EAAE,MAAM,4CAA4C,CAAC;AAC9F,OAAO,EAAE,wBAAwB,EAAE,MAAM,+CAA+C,CAAC;AACzF,OAAO,KAAK,CAAC,MAAM,KAAK,CAAC;AAIzB;;oGAEoG;AACpG,MAAM,CAAC,MAAM,iBAAiB,GAAG,CAAC,CAAC,MAAM,CAAC;IACxC,IAAI,EAAE,CAAC;SACJ,MAAM,EAAE;SACR,GAAG,CAAC,CAAC,CAAC;SACN,GAAG,CAAC,EAAE,CAAC;SACP,QAAQ,CAAC,yDAAyD,CAAC;IACtE,MAAM,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,QAAQ,CAAC,2CAA2C,CAAC;CACzE,CAAC,CAAC;AAIH;;oGAEoG;AACpG,MAAM,CAAC,MAAM,6BAA6B,GAAG,MAAM,CAAC;AAEpD,MAAM,wBAAwB,GAAG;IAC/B,mCAAmC;IACnC,gGAAgG;IAChG,cAAc;IACd,EAAE;IACF,+FAA+F;IAC/F,gGAAgG;IAChG,mDAAmD;IACnD,6FAA6F;IAC7F,+BAA+B;CAChC,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AAEb;;;;GAIG;AACH,MAAM,UAAU,kBAAkB,CAChC,MAAc,EACd,MAAc;IAEd,MAAM,SAAS,GAAG;QAChB,SAAS;QACT,MAAM;QACN,EAAE;QACF,qBAAqB;QACrB,UAAU;QACV,MAAM,IAAI,gBAAgB;QAC1B,WAAW;KACZ,CAAC;IACF,OAAO,EAAE,MAAM,EAAE,wBAAwB,EAAE,IAAI,EAAE,SAAS,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;AAC1E,CAAC;AAED;;;;;;;;;;;;;GAaG;AACH,MAAM,CAAC,KAAK,UAAU,aAAa,CACjC,MAAc,EACd,MAAc,EACd,KAAgC,EAChC,OAAgC;IAEhC,MAAM,SAAS,GAAG,OAAO,EAAE,SAAS,IAAI,6BAA6B,CAAC;IAEtE,IAAI,CAAC,KAAK,IAAI,OAAO,KAAK,CAAC,oBAAoB,KAAK,UAAU,EAAE,CAAC;QAC/D,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,mCAAmC,EAAE,CAAC;IACpF,CAAC;IAED,MAAM,EAAE,MAAM,EAAE,IAAI,EAAE,GAAG,kBAAkB,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IAC5D,IAAI,CAAC;QACH,2FAA2F;QAC3F,6FAA6F;QAC7F,6FAA6F;QAC7F,MAAM,QAAQ,GAAG,wBAAwB,CAAC,iBAAiB,CAAC,CAAC;QAC7D,MAAM,UAAU,GAAG,KAAK,CAAC,oBAAoB,CAAC,QAAQ,CAAC,UAAU,CAAC,CAAC;QAEnE,8FAA8F;QAC9F,+FAA+F;QAC/F,+FAA+F;QAC/F,2FAA2F;QAC3F,+EAA+E;QAC/E,MAAM,KAAK,GAAG,MAAM,gBAAgB,CAAC,SAAS,EAAE,CAAC,MAAM,EAAE,EAAE,CACzD,UAAU,CAAC,MAAM,CAAC,CAAC,IAAI,aAAa,CAAC,MAAM,CAAC,EAAE,IAAI,YAAY,CAAC,IAAI,CAAC,CAAC,EAAE,EAAE,MAAM,EAAE,CAAC,CACnF,CAAC;QACF,IAAI,KAAK,KAAK,cAAc,EAAE,CAAC;YAC7B,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,yBAAyB,SAAS,KAAK,EAAE,CAAC;QACxF,CAAC;QAED,6FAA6F;QAC7F,0DAA0D;QAC1D,MAAM,MAAM,GAAG,QAAQ,CAAC,SAAS,CAAC,KAAK,CAAC,CAAC;QACzC,IAAI,CAAC,MAAM,CAAC,OAAO,EAAE,CAAC;YACpB,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,oCAAoC,EAAE,CAAC;QACrF,CAAC;QACD,OAAO,EAAE,SAAS,EAAE,IAAI,EAAE,EAAE,EAAE,IAAI,EAAE,OAAO,EAAE,MAAM,CAAC,IAAI,EAAE,CAAC;IAC7D,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QACf,OAAO;YACL,SAAS,EAAE,IAAI;YACf,EAAE,EAAE,KAAK;YACT,KAAK,EAAE,KAAK,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC;SAC9D,CAAC;IACJ,CAAC;AACH,CAAC"}
|
package/dist/output.d.ts
CHANGED
|
@@ -4,7 +4,7 @@ import type { BatchSummary, CellResult } from '#src/types.js';
|
|
|
4
4
|
* (pass/fail counts + a per-cell one-liner — a lightweight flake report) into `outputDir`.
|
|
5
5
|
* Creates `outputDir` (and any missing parents) if it doesn't exist.
|
|
6
6
|
*
|
|
7
|
-
* Pure I/O, deliberately separate from {@link runBatchMatrix}: the runner never touches the
|
|
7
|
+
* Pure I/O, deliberately separate from {@link @gaunt-sloth/batch!"BatchRunner.js".runBatchMatrix | runBatchMatrix}: the runner never touches the
|
|
8
8
|
* filesystem, so unit tests can exercise matrix/concurrency/retry logic without a tmp dir, and this
|
|
9
9
|
* function can be tested in isolation with a fixed set of results.
|
|
10
10
|
*
|
package/dist/output.js
CHANGED
|
@@ -6,7 +6,7 @@ import { buildBatchSummary } from '#src/BatchRunner.js';
|
|
|
6
6
|
* (pass/fail counts + a per-cell one-liner — a lightweight flake report) into `outputDir`.
|
|
7
7
|
* Creates `outputDir` (and any missing parents) if it doesn't exist.
|
|
8
8
|
*
|
|
9
|
-
* Pure I/O, deliberately separate from {@link runBatchMatrix}: the runner never touches the
|
|
9
|
+
* Pure I/O, deliberately separate from {@link @gaunt-sloth/batch!"BatchRunner.js".runBatchMatrix | runBatchMatrix}: the runner never touches the
|
|
10
10
|
* filesystem, so unit tests can exercise matrix/concurrency/retry logic without a tmp dir, and this
|
|
11
11
|
* function can be tested in isolation with a fixed set of results.
|
|
12
12
|
*
|
package/dist/parseOver.d.ts
CHANGED
|
@@ -4,7 +4,7 @@ import type { MatrixRow } from '#src/types.js';
|
|
|
4
4
|
* `.jsonl`/`.ndjson` → one JSON object per line; anything else (including `.csv`) → CSV.
|
|
5
5
|
*
|
|
6
6
|
* This is content binding only (BATCH-1 scope): every row becomes an object of string fields that
|
|
7
|
-
* {@link bindCellContent} interpolates into the script. A glob-of-binary-files path binding is out
|
|
7
|
+
* {@link @gaunt-sloth/batch!"interpolate.js".bindCellContent | bindCellContent} interpolates into the script. A glob-of-binary-files path binding is out
|
|
8
8
|
* of scope for this task.
|
|
9
9
|
*
|
|
10
10
|
* Throws a descriptive `Error` on malformed input — the harness-level failure the CLI surface doc
|
package/dist/parseOver.js
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
* `.jsonl`/`.ndjson` → one JSON object per line; anything else (including `.csv`) → CSV.
|
|
4
4
|
*
|
|
5
5
|
* This is content binding only (BATCH-1 scope): every row becomes an object of string fields that
|
|
6
|
-
* {@link bindCellContent} interpolates into the script. A glob-of-binary-files path binding is out
|
|
6
|
+
* {@link @gaunt-sloth/batch!"interpolate.js".bindCellContent | bindCellContent} interpolates into the script. A glob-of-binary-files path binding is out
|
|
7
7
|
* of scope for this task.
|
|
8
8
|
*
|
|
9
9
|
* Throws a descriptive `Error` on malformed input — the harness-level failure the CLI surface doc
|
package/dist/pipelineCli.d.ts
CHANGED
|
@@ -1,6 +1,4 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* @module pipelineCli
|
|
3
|
-
*
|
|
4
2
|
* BATCH-9 — the standalone `gth-batch` pipeline runner. A thin entry point that runs the BATCH-1
|
|
5
3
|
* matrix runtime ({@link buildMatrix} + {@link runBatchMatrix}) directly from a shell pipeline,
|
|
6
4
|
* without pulling in the whole `gaunt-sloth` app. It takes a prompt-executable script + `--over`
|
|
@@ -16,10 +14,12 @@
|
|
|
16
14
|
* the pipeline shell already handles) and output is JSONL on stdout (not a directory of files).
|
|
17
15
|
*
|
|
18
16
|
* stdout discipline: the run itself is noisy (the runtime's `display()`/`ProgressIndicator`/token
|
|
19
|
-
* streaming all target `process.stdout`). The bin entry ({@link
|
|
17
|
+
* streaming all target `process.stdout`). The bin entry ({@link @gaunt-sloth/batch!"bin.js" | ./bin.ts}) redirects
|
|
20
18
|
* `process.stdout.write` to stderr for the duration and this module writes the machine JSONL
|
|
21
19
|
* straight to fd 1 (`fs.writeSync`), so stdout stays a clean data channel — the same "protocol
|
|
22
20
|
* channel" discipline `packages/app/cli.js` uses for ACP.
|
|
21
|
+
*
|
|
22
|
+
* @module
|
|
23
23
|
*/
|
|
24
24
|
import type { CommandLineConfigOverrides, GthConfig } from '@gaunt-sloth/core/config.js';
|
|
25
25
|
import type { BatchSummary, MatrixRow, RunCellFn } from '#src/types.js';
|
package/dist/pipelineCli.js
CHANGED
|
@@ -1,6 +1,4 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* @module pipelineCli
|
|
3
|
-
*
|
|
4
2
|
* BATCH-9 — the standalone `gth-batch` pipeline runner. A thin entry point that runs the BATCH-1
|
|
5
3
|
* matrix runtime ({@link buildMatrix} + {@link runBatchMatrix}) directly from a shell pipeline,
|
|
6
4
|
* without pulling in the whole `gaunt-sloth` app. It takes a prompt-executable script + `--over`
|
|
@@ -16,10 +14,12 @@
|
|
|
16
14
|
* the pipeline shell already handles) and output is JSONL on stdout (not a directory of files).
|
|
17
15
|
*
|
|
18
16
|
* stdout discipline: the run itself is noisy (the runtime's `display()`/`ProgressIndicator`/token
|
|
19
|
-
* streaming all target `process.stdout`). The bin entry ({@link
|
|
17
|
+
* streaming all target `process.stdout`). The bin entry ({@link @gaunt-sloth/batch!"bin.js" | ./bin.ts}) redirects
|
|
20
18
|
* `process.stdout.write` to stderr for the duration and this module writes the machine JSONL
|
|
21
19
|
* straight to fd 1 (`fs.writeSync`), so stdout stays a clean data channel — the same "protocol
|
|
22
20
|
* channel" discipline `packages/app/cli.js` uses for ACP.
|
|
21
|
+
*
|
|
22
|
+
* @module
|
|
23
23
|
*/
|
|
24
24
|
import { writeSync } from 'node:fs';
|
|
25
25
|
import { readFileSync } from 'node:fs';
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
import type { BaseChatModel } from '@langchain/core/language_models/chat_models';
|
|
2
|
+
import type { RaterCallCapture } from '@gaunt-sloth/core/core/shell/approvalCapture.js';
|
|
3
|
+
import { buildComposedOpenWorldNote } from '@gaunt-sloth/core/core/shell/openWorld.js';
|
|
4
|
+
import { buildParserPreflightNote } from '@gaunt-sloth/core/core/shell/abstention.js';
|
|
5
|
+
/**
|
|
6
|
+
* The notes an arm can omit: a stable suite-facing token → **core's own builder for that note's
|
|
7
|
+
* exact text**.
|
|
8
|
+
*
|
|
9
|
+
* **Every entry is core's function, never a copy of its prose, and that is the entry requirement.**
|
|
10
|
+
* The off arm is produced by deleting a block from the built prompt, so the text used to find the
|
|
11
|
+
* block has to be the same bytes the builder put there. A second copy in this package would pass
|
|
12
|
+
* review, drift on the first wording change in core, then find nothing to delete — and an arm that
|
|
13
|
+
* deletes nothing sends the on-arm prompt under the off arm's name, so the comparison reports the
|
|
14
|
+
* note having no effect. {@link reconcileArmedCapture} raises on exactly that, but a registry built
|
|
15
|
+
* from copies would be relying on a runtime check to catch a defect the design need not have.
|
|
16
|
+
*
|
|
17
|
+
* **Two of `buildRaterPrompt`'s four preflight notes are therefore absent**: the script-env-leak
|
|
18
|
+
* note and the open-world floor note are composed inline in that function and core exports no
|
|
19
|
+
* builder for either. They become expressible the day core extracts one — which is a change to core
|
|
20
|
+
* with its own justification, not something to smuggle in here as a copied string.
|
|
21
|
+
*/
|
|
22
|
+
export declare const RATER_PROMPT_NOTES: {
|
|
23
|
+
/** [[EXT-81]]'s note: the data flow across the parts of a command our parser could not resolve
|
|
24
|
+
* as a whole. The note the A/B that motivated this node is about. */
|
|
25
|
+
readonly 'composed-open-world': typeof buildComposedOpenWorldNote;
|
|
26
|
+
/** The parser's own abstention note: what about the command's shape could not be resolved. */
|
|
27
|
+
readonly parser: typeof buildParserPreflightNote;
|
|
28
|
+
};
|
|
29
|
+
/** A note name a suite may omit — the keys of {@link RATER_PROMPT_NOTES}. */
|
|
30
|
+
export type RaterPromptNoteName = keyof typeof RATER_PROMPT_NOTES;
|
|
31
|
+
/** Every note name a suite may write, in declaration order, for error messages and validation. */
|
|
32
|
+
export declare const RATER_PROMPT_NOTE_NAMES: RaterPromptNoteName[];
|
|
33
|
+
/**
|
|
34
|
+
* One sweep cell's prompt arm: which of our own preflight notes this cell's ratings go out without.
|
|
35
|
+
*
|
|
36
|
+
* `omit: []` is the meaningful, and required, spelling of the baseline arm. A sweep value that
|
|
37
|
+
* declared nothing at all would be indistinguishable from an authoring slip, and the whole point of
|
|
38
|
+
* an A/B is that both arms say what they are.
|
|
39
|
+
*/
|
|
40
|
+
export interface RaterPromptArm {
|
|
41
|
+
readonly omit: readonly RaterPromptNoteName[];
|
|
42
|
+
}
|
|
43
|
+
/** The prompt as it actually left for the model. */
|
|
44
|
+
export interface ArmedPrompt {
|
|
45
|
+
system: string;
|
|
46
|
+
user: string;
|
|
47
|
+
}
|
|
48
|
+
/**
|
|
49
|
+
* What one armed rating call did — recorded by the decorator, adjudicated afterwards by
|
|
50
|
+
* {@link reconcileArmedCapture}.
|
|
51
|
+
*/
|
|
52
|
+
export interface RaterPromptArmOutcome {
|
|
53
|
+
/** Whether the decorated model was invoked at all. `false` means `rateShellCommand` returned
|
|
54
|
+
* before the send site — it found no usable model — so nothing was measured. */
|
|
55
|
+
invoked: boolean;
|
|
56
|
+
/** Notes whose block was found and removed. */
|
|
57
|
+
removed: RaterPromptNoteName[];
|
|
58
|
+
/** Notes that do not apply to this command, so there was no block to remove. This is the
|
|
59
|
+
* byte-identical case and it is silent by design: a command carrying no composed note must
|
|
60
|
+
* produce the same prompt under both arms. */
|
|
61
|
+
absent: RaterPromptNoteName[];
|
|
62
|
+
/** Notes whose block core says exists on this command and the arm could not find in the prompt —
|
|
63
|
+
* a leak. Raised on, never tolerated. */
|
|
64
|
+
leaked: RaterPromptNoteName[];
|
|
65
|
+
/** The prompt as sent, once the arm had finished with it. */
|
|
66
|
+
sent?: ArmedPrompt;
|
|
67
|
+
}
|
|
68
|
+
/**
|
|
69
|
+
* Delete the arm's notes from a built rater user message.
|
|
70
|
+
*
|
|
71
|
+
* Pure, and keyed on the command: each note's block is `'\n\n' + <core's text for this command>`,
|
|
72
|
+
* because `buildRaterPrompt` pushes every note into its line buffer as `('', note)` and joins on
|
|
73
|
+
* `'\n'`. A note that does not apply to the command yields `null` and is recorded as `absent`.
|
|
74
|
+
*/
|
|
75
|
+
export declare function stripRaterPromptNotes(user: string, command: string, arm: RaterPromptArm): {
|
|
76
|
+
user: string;
|
|
77
|
+
removed: RaterPromptNoteName[];
|
|
78
|
+
absent: RaterPromptNoteName[];
|
|
79
|
+
leaked: RaterPromptNoteName[];
|
|
80
|
+
};
|
|
81
|
+
/**
|
|
82
|
+
* Wrap a rating model so this call's prompt goes out without the arm's notes.
|
|
83
|
+
*
|
|
84
|
+
* **Why the MODEL and not the prompt builder.** The prompt is built inside `rateShellCommand`, and
|
|
85
|
+
* every seam that could change it there would be a switch in the session's gate. The model is the
|
|
86
|
+
* one thing the gate takes from its caller, so decorating it is the only place a measurement can
|
|
87
|
+
* change the outgoing prompt without core growing a parameter for it. The `eval` rater target is
|
|
88
|
+
* also the only caller that constructs such a model.
|
|
89
|
+
*
|
|
90
|
+
* **A Proxy rather than a hand-written delegate**, because `rateShellCommand` reads more off a model
|
|
91
|
+
* than `withStructuredOutput` — `raterModelLabel` calls `_llmType()` and reads `model`/`modelName`/
|
|
92
|
+
* `modelId` — and a delegate that enumerated today's reads would silently answer `undefined` for
|
|
93
|
+
* tomorrow's. Every trap forwards with the TARGET as the receiver and binds methods to the target,
|
|
94
|
+
* so a provider class keeping state behind private `#fields` is not handed a `this` it cannot use.
|
|
95
|
+
*
|
|
96
|
+
* **A shape it does not recognise is a leak, not a pass-through.** If the messages are not the
|
|
97
|
+
* `[SystemMessage, HumanMessage]` pair with string content that `rateShellCommand` sends, the call
|
|
98
|
+
* is forwarded unchanged and every requested note is recorded as leaked, so the run fails loudly
|
|
99
|
+
* instead of quietly measuring the on-arm prompt under the off arm's name.
|
|
100
|
+
*
|
|
101
|
+
* @param model The resolved rating model.
|
|
102
|
+
* @param command The command this call rates — the notes are a function of it.
|
|
103
|
+
* @param arm The notes to omit.
|
|
104
|
+
* @param outcome The record this call writes into; one per rating call, so concurrent cases cannot
|
|
105
|
+
* share it.
|
|
106
|
+
*/
|
|
107
|
+
export declare function armRaterModel(model: BaseChatModel, command: string, arm: RaterPromptArm, outcome: RaterPromptArmOutcome): BaseChatModel;
|
|
108
|
+
/**
|
|
109
|
+
* A fresh outcome plus the model that writes into it — **one per rating call**, because the
|
|
110
|
+
* classifier is reused across cases the suite runner may run concurrently and a shared record would
|
|
111
|
+
* let one case's arm be adjudicated on another's call.
|
|
112
|
+
*
|
|
113
|
+
* `model` is `undefined` only in the state `buildRaterClassifier` refuses to build: an arm declared
|
|
114
|
+
* with no model to decorate. It is admitted here rather than guarded a second time, so that state
|
|
115
|
+
* lands on {@link reconcileArmedCapture}'s "never reached a rating call" arm — one rule, one place —
|
|
116
|
+
* instead of on a copy of the rule that could drift from it.
|
|
117
|
+
*/
|
|
118
|
+
export declare function armFor(model: BaseChatModel | undefined, command: string, arm: RaterPromptArm): {
|
|
119
|
+
model: BaseChatModel | undefined;
|
|
120
|
+
outcome: RaterPromptArmOutcome;
|
|
121
|
+
};
|
|
122
|
+
/**
|
|
123
|
+
* Adjudicate one armed rating call, **after it has returned and outside `rateShellCommand`'s
|
|
124
|
+
* fail-closed `try`**, and repair the call's diagnostic record.
|
|
125
|
+
*
|
|
126
|
+
* Two jobs, and both are about not lying:
|
|
127
|
+
*
|
|
128
|
+
* - **Raise on an arm that did not do what it said.** A leaked note, or a rating call the decorated
|
|
129
|
+
* model never saw, means this cell's numbers are the other arm's numbers under this arm's name.
|
|
130
|
+
* Silently reporting them is strictly worse than failing the run, because the comparison table
|
|
131
|
+
* would then say the note changed nothing.
|
|
132
|
+
* - **Make `capture.prompt` the prompt that was SENT.** `rateShellCommand` fills the capture from
|
|
133
|
+
* `buildRaterPrompt`'s output at the send site and its contract is that nothing downstream may
|
|
134
|
+
* leave the archive disagreeing with what the model saw. The arm edits the message below that
|
|
135
|
+
* point, so the arm is what owes the correction — and for a facility whose whole subject is prompt
|
|
136
|
+
* content, a diagnostic record of the wrong arm's prompt would be the worst possible artifact.
|
|
137
|
+
*
|
|
138
|
+
* @param capture The record `rateShellCommand` handed back through `onCapture`, when it made one.
|
|
139
|
+
*/
|
|
140
|
+
export declare function reconcileArmedCapture(command: string, arm: RaterPromptArm, outcome: RaterPromptArmOutcome, capture: RaterCallCapture | undefined): void;
|