@gaunt-sloth/batch 2.0.0-beta.8 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/dist/bin.d.ts +2 -2
  2. package/dist/bin.js +2 -2
  3. package/dist/classificationReport.d.ts +2 -2
  4. package/dist/classificationReport.js +2 -2
  5. package/dist/classificationTypes.d.ts +7 -7
  6. package/dist/classificationTypes.js +1 -1
  7. package/dist/evalCompare.d.ts +128 -1
  8. package/dist/evalCompare.js +253 -2
  9. package/dist/evalCompare.js.map +1 -1
  10. package/dist/evalOutput.d.ts +1 -1
  11. package/dist/evalOutput.js +1 -1
  12. package/dist/evalRunner.d.ts +15 -3
  13. package/dist/evalRunner.js +82 -5
  14. package/dist/evalRunner.js.map +1 -1
  15. package/dist/evalSuite.d.ts +5 -1
  16. package/dist/evalSuite.js +218 -7
  17. package/dist/evalSuite.js.map +1 -1
  18. package/dist/evalTypes.d.ts +79 -18
  19. package/dist/evalTypes.js +2 -2
  20. package/dist/evalTypes.js.map +1 -1
  21. package/dist/index.d.ts +8 -2
  22. package/dist/index.js +9 -1
  23. package/dist/index.js.map +1 -1
  24. package/dist/judge.d.ts +2 -2
  25. package/dist/judge.js +10 -14
  26. package/dist/judge.js.map +1 -1
  27. package/dist/output.d.ts +1 -1
  28. package/dist/output.js +1 -1
  29. package/dist/parseOver.d.ts +1 -1
  30. package/dist/parseOver.js +1 -1
  31. package/dist/pipelineCli.d.ts +3 -3
  32. package/dist/pipelineCli.js +3 -3
  33. package/dist/raterPromptArm.d.ts +140 -0
  34. package/dist/raterPromptArm.js +306 -0
  35. package/dist/raterPromptArm.js.map +1 -0
  36. package/dist/raterTarget.d.ts +34 -12
  37. package/dist/raterTarget.js +124 -33
  38. package/dist/raterTarget.js.map +1 -1
  39. package/dist/reporters/registry.d.ts +1 -1
  40. package/dist/reporters/registry.js +1 -1
  41. package/dist/reporters/registry.js.map +1 -1
  42. package/dist/reporters/reporterTypes.d.ts +3 -3
  43. package/dist/reporters/textReporter.js +16 -0
  44. package/dist/reporters/textReporter.js.map +1 -1
  45. package/dist/toolCoverage.d.ts +244 -0
  46. package/dist/toolCoverage.js +414 -0
  47. package/dist/toolCoverage.js.map +1 -0
  48. package/dist/toolCoverageRender.d.ts +31 -0
  49. package/dist/toolCoverageRender.js +71 -0
  50. package/dist/toolCoverageRender.js.map +1 -0
  51. package/dist/toolResultChecks.d.ts +8 -2
  52. package/dist/toolResultChecks.js +45 -3
  53. package/dist/toolResultChecks.js.map +1 -1
  54. package/dist/types.d.ts +38 -3
  55. package/dist/types.js +0 -9
  56. package/dist/types.js.map +1 -1
  57. package/dist/workflow/runWorkflow.d.ts +4 -4
  58. package/dist/workflow/runWorkflow.js +3 -3
  59. package/package.json +9 -8
@@ -0,0 +1,140 @@
1
+ import type { BaseChatModel } from '@langchain/core/language_models/chat_models';
2
+ import type { RaterCallCapture } from '@gaunt-sloth/core/core/shell/approvalCapture.js';
3
+ import { buildComposedOpenWorldNote } from '@gaunt-sloth/core/core/shell/openWorld.js';
4
+ import { buildParserPreflightNote } from '@gaunt-sloth/core/core/shell/abstention.js';
5
+ /**
6
+ * The notes an arm can omit: a stable suite-facing token → **core's own builder for that note's
7
+ * exact text**.
8
+ *
9
+ * **Every entry is core's function, never a copy of its prose, and that is the entry requirement.**
10
+ * The off arm is produced by deleting a block from the built prompt, so the text used to find the
11
+ * block has to be the same bytes the builder put there. A second copy in this package would pass
12
+ * review, drift on the first wording change in core, then find nothing to delete — and an arm that
13
+ * deletes nothing sends the on-arm prompt under the off arm's name, so the comparison reports the
14
+ * note having no effect. {@link reconcileArmedCapture} raises on exactly that, but a registry built
15
+ * from copies would be relying on a runtime check to catch a defect the design need not have.
16
+ *
17
+ * **Two of `buildRaterPrompt`'s four preflight notes are therefore absent**: the script-env-leak
18
+ * note and the open-world floor note are composed inline in that function and core exports no
19
+ * builder for either. They become expressible the day core extracts one — which is a change to core
20
+ * with its own justification, not something to smuggle in here as a copied string.
21
+ */
22
+ export declare const RATER_PROMPT_NOTES: {
23
+ /** [[EXT-81]]'s note: the data flow across the parts of a command our parser could not resolve
24
+ * as a whole. The note the A/B that motivated this node is about. */
25
+ readonly 'composed-open-world': typeof buildComposedOpenWorldNote;
26
+ /** The parser's own abstention note: what about the command's shape could not be resolved. */
27
+ readonly parser: typeof buildParserPreflightNote;
28
+ };
29
+ /** A note name a suite may omit — the keys of {@link RATER_PROMPT_NOTES}. */
30
+ export type RaterPromptNoteName = keyof typeof RATER_PROMPT_NOTES;
31
+ /** Every note name a suite may write, in declaration order, for error messages and validation. */
32
+ export declare const RATER_PROMPT_NOTE_NAMES: RaterPromptNoteName[];
33
+ /**
34
+ * One sweep cell's prompt arm: which of our own preflight notes this cell's ratings go out without.
35
+ *
36
+ * `omit: []` is the meaningful, and required, spelling of the baseline arm. A sweep value that
37
+ * declared nothing at all would be indistinguishable from an authoring slip, and the whole point of
38
+ * an A/B is that both arms say what they are.
39
+ */
40
+ export interface RaterPromptArm {
41
+ readonly omit: readonly RaterPromptNoteName[];
42
+ }
43
+ /** The prompt as it actually left for the model. */
44
+ export interface ArmedPrompt {
45
+ system: string;
46
+ user: string;
47
+ }
48
+ /**
49
+ * What one armed rating call did — recorded by the decorator, adjudicated afterwards by
50
+ * {@link reconcileArmedCapture}.
51
+ */
52
+ export interface RaterPromptArmOutcome {
53
+ /** Whether the decorated model was invoked at all. `false` means `rateShellCommand` returned
54
+ * before the send site — it found no usable model — so nothing was measured. */
55
+ invoked: boolean;
56
+ /** Notes whose block was found and removed. */
57
+ removed: RaterPromptNoteName[];
58
+ /** Notes that do not apply to this command, so there was no block to remove. This is the
59
+ * byte-identical case and it is silent by design: a command carrying no composed note must
60
+ * produce the same prompt under both arms. */
61
+ absent: RaterPromptNoteName[];
62
+ /** Notes whose block core says exists on this command and the arm could not find in the prompt —
63
+ * a leak. Raised on, never tolerated. */
64
+ leaked: RaterPromptNoteName[];
65
+ /** The prompt as sent, once the arm had finished with it. */
66
+ sent?: ArmedPrompt;
67
+ }
68
+ /**
69
+ * Delete the arm's notes from a built rater user message.
70
+ *
71
+ * Pure, and keyed on the command: each note's block is `'\n\n' + <core's text for this command>`,
72
+ * because `buildRaterPrompt` pushes every note into its line buffer as `('', note)` and joins on
73
+ * `'\n'`. A note that does not apply to the command yields `null` and is recorded as `absent`.
74
+ */
75
+ export declare function stripRaterPromptNotes(user: string, command: string, arm: RaterPromptArm): {
76
+ user: string;
77
+ removed: RaterPromptNoteName[];
78
+ absent: RaterPromptNoteName[];
79
+ leaked: RaterPromptNoteName[];
80
+ };
81
+ /**
82
+ * Wrap a rating model so this call's prompt goes out without the arm's notes.
83
+ *
84
+ * **Why the MODEL and not the prompt builder.** The prompt is built inside `rateShellCommand`, and
85
+ * every seam that could change it there would be a switch in the session's gate. The model is the
86
+ * one thing the gate takes from its caller, so decorating it is the only place a measurement can
87
+ * change the outgoing prompt without core growing a parameter for it. The `eval` rater target is
88
+ * also the only caller that constructs such a model.
89
+ *
90
+ * **A Proxy rather than a hand-written delegate**, because `rateShellCommand` reads more off a model
91
+ * than `withStructuredOutput` — `raterModelLabel` calls `_llmType()` and reads `model`/`modelName`/
92
+ * `modelId` — and a delegate that enumerated today's reads would silently answer `undefined` for
93
+ * tomorrow's. Every trap forwards with the TARGET as the receiver and binds methods to the target,
94
+ * so a provider class keeping state behind private `#fields` is not handed a `this` it cannot use.
95
+ *
96
+ * **A shape it does not recognise is a leak, not a pass-through.** If the messages are not the
97
+ * `[SystemMessage, HumanMessage]` pair with string content that `rateShellCommand` sends, the call
98
+ * is forwarded unchanged and every requested note is recorded as leaked, so the run fails loudly
99
+ * instead of quietly measuring the on-arm prompt under the off arm's name.
100
+ *
101
+ * @param model The resolved rating model.
102
+ * @param command The command this call rates — the notes are a function of it.
103
+ * @param arm The notes to omit.
104
+ * @param outcome The record this call writes into; one per rating call, so concurrent cases cannot
105
+ * share it.
106
+ */
107
+ export declare function armRaterModel(model: BaseChatModel, command: string, arm: RaterPromptArm, outcome: RaterPromptArmOutcome): BaseChatModel;
108
+ /**
109
+ * A fresh outcome plus the model that writes into it — **one per rating call**, because the
110
+ * classifier is reused across cases the suite runner may run concurrently and a shared record would
111
+ * let one case's arm be adjudicated on another's call.
112
+ *
113
+ * `model` is `undefined` only in the state `buildRaterClassifier` refuses to build: an arm declared
114
+ * with no model to decorate. It is admitted here rather than guarded a second time, so that state
115
+ * lands on {@link reconcileArmedCapture}'s "never reached a rating call" arm — one rule, one place —
116
+ * instead of on a copy of the rule that could drift from it.
117
+ */
118
+ export declare function armFor(model: BaseChatModel | undefined, command: string, arm: RaterPromptArm): {
119
+ model: BaseChatModel | undefined;
120
+ outcome: RaterPromptArmOutcome;
121
+ };
122
+ /**
123
+ * Adjudicate one armed rating call, **after it has returned and outside `rateShellCommand`'s
124
+ * fail-closed `try`**, and repair the call's diagnostic record.
125
+ *
126
+ * Two jobs, and both are about not lying:
127
+ *
128
+ * - **Raise on an arm that did not do what it said.** A leaked note, or a rating call the decorated
129
+ * model never saw, means this cell's numbers are the other arm's numbers under this arm's name.
130
+ * Silently reporting them is strictly worse than failing the run, because the comparison table
131
+ * would then say the note changed nothing.
132
+ * - **Make `capture.prompt` the prompt that was SENT.** `rateShellCommand` fills the capture from
133
+ * `buildRaterPrompt`'s output at the send site and its contract is that nothing downstream may
134
+ * leave the archive disagreeing with what the model saw. The arm edits the message below that
135
+ * point, so the arm is what owes the correction — and for a facility whose whole subject is prompt
136
+ * content, a diagnostic record of the wrong arm's prompt would be the worst possible artifact.
137
+ *
138
+ * @param capture The record `rateShellCommand` handed back through `onCapture`, when it made one.
139
+ */
140
+ export declare function reconcileArmedCapture(command: string, arm: RaterPromptArm, outcome: RaterPromptArmOutcome, capture: RaterCallCapture | undefined): void;
@@ -0,0 +1,306 @@
1
+ /**
2
+ * @packageDocumentation
3
+ * [[BATCH-31]] — **prompt-content A/B for the approvals rater: the same corpus rated twice, once
4
+ * with one of our own preflight notes in the prompt and once without.**
5
+ *
6
+ * [[EXT-81]] needed exactly that comparison and could not express it as a suite. A sweep can move
7
+ * the rung and the model; nothing could say *omit this note*. It was measured instead by a
8
+ * standalone harness
9
+ * (`docs/test-sessions/ext-81-composed-note-sweep-2026-08-04/harness/composed-open-world-note.mjs`)
10
+ * that built the shipping prompt and **deleted the note block out of it** — and that harness, not a
11
+ * new prompt builder, is what this module generalises.
12
+ *
13
+ * ## What "production" means here, because the whole argument below hangs on it
14
+ *
15
+ * **Production is the SESSION's approvals gate**: `GthAgentRunner`'s rating call, the one that
16
+ * decides whether a shell command a live agent proposed runs, prompts, or halts. That is the gate
17
+ * whose safety context must never be suppressible. `gth eval` is a measuring instrument — it rates
18
+ * a corpus and prints a table, and gates nothing — so making an omission expressible *there* is the
19
+ * point of this node, not a hazard.
20
+ *
21
+ * The forbidden design is therefore precise: **a switch inside the gate** — a parameter on
22
+ * `rateShellCommand` or `buildRaterPrompt` that suppresses a note. [[EXT-81]]'s implementer
23
+ * identified that shape and declined to build it. Nothing here adds one; **core is untouched by
24
+ * this node.**
25
+ *
26
+ * ## The shape taken: a third sweep-override kind with its own route
27
+ *
28
+ * A sweep value already carries two override kinds, and they reach the run by two different routes:
29
+ * `model:` through `initConfig({ model })`, and `config:` by a deep merge onto the resolved
30
+ * `GthConfig`. This adds a third, `notes:`, whose route is:
31
+ *
32
+ * ```text
33
+ * suite sweep value notes: { omit: [<note>] }
34
+ * -> EvalSweepValue.notes (parsed and validated against the registry below)
35
+ * -> SweepCell.notes (expandSweep; never merged into the cell's `config`)
36
+ * -> RaterClassifierOptions.notes
37
+ * -> armRaterModel(...) <- a decorator around the RATING MODEL, in this package
38
+ * -> the note block is deleted from the outgoing user message
39
+ * ```
40
+ *
41
+ * The one-file A/B EXT-81 wanted is then a one-axis sweep: `{ name: on, notes: { omit: [] } }`
42
+ * against `{ name: off, notes: { omit: [composed-open-world] } }`.
43
+ *
44
+ * ## How production-unreachability is ENFORCED, not merely intended
45
+ *
46
+ * Four independent legs, each of which a test can fail:
47
+ *
48
+ * 1. **The omission never becomes config.** `notes:` is a sibling of `config:`, not a key inside it,
49
+ * and `initConfigForCell` deep-merges only `cell.config`. So the value has no representation in
50
+ * `GthConfig`, and therefore none in `.gsloth.config.json`, in an env var, or in a CLI flag —
51
+ * the three carriers the node names. Pinned by a spec that plants the omission in a config, in
52
+ * an `approvals` block, and on the rating call's own options object, drives the SESSION gate over
53
+ * a noted command, and asserts the note is still in the prompt that was sent.
54
+ * 2. **The omission is not data at all — it is a function.** What actually removes the note is a
55
+ * `BaseChatModel` decorator constructed here at classifier-build time. A config file, an
56
+ * environment variable and a command-line flag can carry a string; none of them can carry a model
57
+ * decorator. There is no serialisable spelling of this facility for a session to deserialise.
58
+ * 3. **The session's gate cannot import this module.** `@gaunt-sloth/batch` depends on
59
+ * `@gaunt-sloth/core`, where `GthAgentRunner` and `rateShellCommand` live. The reverse edge is a
60
+ * dependency cycle, so core cannot reach `armRaterModel` even deliberately. This is enforced by
61
+ * the package graph rather than by a convention someone has to remember.
62
+ * 4. **The declaration is rejected on every target that runs an agent.** `notes:` parses only on a
63
+ * `rater` suite, which runs no agent and executes no tool — so the surface does not exist on the
64
+ * suite kinds that drive real sessions.
65
+ *
66
+ * **What none of that proves, stated because the honest limit matters:** leg 3 shows core cannot
67
+ * construct this arm; it does not show core will never grow a suppression switch of its own. Only a
68
+ * *wired* switch is a production path, so the discriminating test is the one in leg 1 — drive the
69
+ * runner's gate and watch the note survive — and the mutation that proves it real is to make the
70
+ * runner pass a suppression flag and see that spec go red. A test that merely asserted something
71
+ * about this package's barrel exports would pass while proving nothing about the hazard.
72
+ *
73
+ * ## Where the route is held, end to end
74
+ *
75
+ * Three specs, one per link, because a facility can be perfectly built and still measure nothing:
76
+ * `packages/batch/spec/raterPromptArm.spec.ts` (parse, expand, arm, reconcile),
77
+ * `packages/core/spec/raterPromptNotesNotSuppressible.spec.ts` (leg 1 above — the session's gate),
78
+ * and `packages/app/spec/evalCommandRaterArm.spec.ts`, which runs the A/B suite through the `eval`
79
+ * command itself. The last one exists because the hand-off in `evalCommand.ts` is the single line
80
+ * that connects the other two: cut it and every arm silently sends the same prompt, the comparison
81
+ * table honestly reports the note changing nothing, and both of the other specs stay green.
82
+ *
83
+ * ## Shapes rejected
84
+ *
85
+ * - **A `notes:` option on `rateShellCommand` / `buildRaterPrompt`.** This is the naive fix and it
86
+ * is the one the node forbids: a switch in the gate that suppresses safety context for one
87
+ * measurement's benefit. Defaults and documentation do not redeem it — it puts the code path in
88
+ * the session's own rating call, where the next caller to grow an option bag can reach it.
89
+ * - **A sweep axis that overrides `target:` fields.** The node names this as a candidate, and it was
90
+ * rejected because it would falsify a constraint the existing sweep depends on. `resolveRung`'s
91
+ * docblock states that a cell can override `config:` and `model:` but cannot reach a `target`
92
+ * field, and that this is *deliberate* and is what makes `config: { approvals: auto }` the way an
93
+ * axis moves the rung. An axis able to rewrite `target` would give the rung two spellings with a
94
+ * silent precedence between them. Adding a third route leaves that constraint exactly as true as
95
+ * it was.
96
+ * - **A `target.notes:` field with no sweep reach.** Expressible, but a `target` is suite-level, so
97
+ * the A/B would again be two hand-written suite files — the thing this node exists to remove.
98
+ * - **Reproducing a note's text in this package.** See {@link RATER_PROMPT_NOTES}: a second copy of
99
+ * the prose would drift, and a drifted copy deletes nothing, so both arms would send the same
100
+ * prompt and the measurement would report "the note has no effect". That is the failure mode that
101
+ * passes its own test, so only notes core exports a byte-exact builder for are expressible.
102
+ *
103
+ * ## Why the decorator never throws, and the caller adjudicates instead
104
+ *
105
+ * The strip happens inside `structured.invoke`, which `rateShellCommand` runs inside the `try` whose
106
+ * whole job is to turn a throw into the fail-closed `destructive` verdict. A decorator that threw on
107
+ * a leak would therefore be recorded as a rating, and the arm's own failure would read as the
108
+ * rater's judgement. So the decorator only ever RECORDS what it did, and
109
+ * {@link reconcileArmedCapture} — called by the target after the rating call has returned, outside
110
+ * that `try` — is what raises.
111
+ */
112
+ import { HumanMessage, SystemMessage } from '@langchain/core/messages';
113
+ import { buildComposedOpenWorldNote } from '@gaunt-sloth/core/core/shell/openWorld.js';
114
+ import { buildParserPreflightNote } from '@gaunt-sloth/core/core/shell/abstention.js';
115
+ /**
116
+ * The notes an arm can omit: a stable suite-facing token → **core's own builder for that note's
117
+ * exact text**.
118
+ *
119
+ * **Every entry is core's function, never a copy of its prose, and that is the entry requirement.**
120
+ * The off arm is produced by deleting a block from the built prompt, so the text used to find the
121
+ * block has to be the same bytes the builder put there. A second copy in this package would pass
122
+ * review, drift on the first wording change in core, then find nothing to delete — and an arm that
123
+ * deletes nothing sends the on-arm prompt under the off arm's name, so the comparison reports the
124
+ * note having no effect. {@link reconcileArmedCapture} raises on exactly that, but a registry built
125
+ * from copies would be relying on a runtime check to catch a defect the design need not have.
126
+ *
127
+ * **Two of `buildRaterPrompt`'s four preflight notes are therefore absent**: the script-env-leak
128
+ * note and the open-world floor note are composed inline in that function and core exports no
129
+ * builder for either. They become expressible the day core extracts one — which is a change to core
130
+ * with its own justification, not something to smuggle in here as a copied string.
131
+ */
132
+ export const RATER_PROMPT_NOTES = {
133
+ /** [[EXT-81]]'s note: the data flow across the parts of a command our parser could not resolve
134
+ * as a whole. The note the A/B that motivated this node is about. */
135
+ 'composed-open-world': buildComposedOpenWorldNote,
136
+ /** The parser's own abstention note: what about the command's shape could not be resolved. */
137
+ parser: buildParserPreflightNote,
138
+ };
139
+ /** Every note name a suite may write, in declaration order, for error messages and validation. */
140
+ export const RATER_PROMPT_NOTE_NAMES = Object.keys(RATER_PROMPT_NOTES);
141
+ /** A fresh, empty outcome for one rating call. */
142
+ function emptyOutcome() {
143
+ return { invoked: false, removed: [], absent: [], leaked: [] };
144
+ }
145
+ /**
146
+ * Delete the arm's notes from a built rater user message.
147
+ *
148
+ * Pure, and keyed on the command: each note's block is `'\n\n' + <core's text for this command>`,
149
+ * because `buildRaterPrompt` pushes every note into its line buffer as `('', note)` and joins on
150
+ * `'\n'`. A note that does not apply to the command yields `null` and is recorded as `absent`.
151
+ */
152
+ export function stripRaterPromptNotes(user, command, arm) {
153
+ let out = user;
154
+ const removed = [];
155
+ const absent = [];
156
+ const leaked = [];
157
+ for (const name of arm.omit) {
158
+ const note = RATER_PROMPT_NOTES[name](command);
159
+ if (note === null) {
160
+ absent.push(name);
161
+ continue;
162
+ }
163
+ const block = `\n\n${note}`;
164
+ const at = out.indexOf(block);
165
+ if (at === -1) {
166
+ leaked.push(name);
167
+ continue;
168
+ }
169
+ out = out.slice(0, at) + out.slice(at + block.length);
170
+ removed.push(name);
171
+ }
172
+ return { user: out, removed, absent, leaked };
173
+ }
174
+ /**
175
+ * Wrap a rating model so this call's prompt goes out without the arm's notes.
176
+ *
177
+ * **Why the MODEL and not the prompt builder.** The prompt is built inside `rateShellCommand`, and
178
+ * every seam that could change it there would be a switch in the session's gate. The model is the
179
+ * one thing the gate takes from its caller, so decorating it is the only place a measurement can
180
+ * change the outgoing prompt without core growing a parameter for it. The `eval` rater target is
181
+ * also the only caller that constructs such a model.
182
+ *
183
+ * **A Proxy rather than a hand-written delegate**, because `rateShellCommand` reads more off a model
184
+ * than `withStructuredOutput` — `raterModelLabel` calls `_llmType()` and reads `model`/`modelName`/
185
+ * `modelId` — and a delegate that enumerated today's reads would silently answer `undefined` for
186
+ * tomorrow's. Every trap forwards with the TARGET as the receiver and binds methods to the target,
187
+ * so a provider class keeping state behind private `#fields` is not handed a `this` it cannot use.
188
+ *
189
+ * **A shape it does not recognise is a leak, not a pass-through.** If the messages are not the
190
+ * `[SystemMessage, HumanMessage]` pair with string content that `rateShellCommand` sends, the call
191
+ * is forwarded unchanged and every requested note is recorded as leaked, so the run fails loudly
192
+ * instead of quietly measuring the on-arm prompt under the off arm's name.
193
+ *
194
+ * @param model The resolved rating model.
195
+ * @param command The command this call rates — the notes are a function of it.
196
+ * @param arm The notes to omit.
197
+ * @param outcome The record this call writes into; one per rating call, so concurrent cases cannot
198
+ * share it.
199
+ */
200
+ export function armRaterModel(model, command, arm, outcome) {
201
+ const forward = (target, prop) => {
202
+ const value = Reflect.get(target, prop, target);
203
+ return typeof value === 'function'
204
+ ? value.bind(target)
205
+ : value;
206
+ };
207
+ const armInvoke = (messages) => {
208
+ outcome.invoked = true;
209
+ if (!Array.isArray(messages) || messages.length !== 2) {
210
+ outcome.leaked.push(...arm.omit);
211
+ return messages;
212
+ }
213
+ const [system, user] = messages;
214
+ if (typeof system?.content !== 'string' || typeof user?.content !== 'string') {
215
+ outcome.leaked.push(...arm.omit);
216
+ return messages;
217
+ }
218
+ const stripped = stripRaterPromptNotes(user.content, command, arm);
219
+ outcome.removed.push(...stripped.removed);
220
+ outcome.absent.push(...stripped.absent);
221
+ outcome.leaked.push(...stripped.leaked);
222
+ outcome.sent = { system: system.content, user: stripped.user };
223
+ return [new SystemMessage(system.content), new HumanMessage(stripped.user)];
224
+ };
225
+ return new Proxy(model, {
226
+ get(target, prop) {
227
+ if (prop !== 'withStructuredOutput')
228
+ return forward(target, prop);
229
+ const original = Reflect.get(target, prop, target);
230
+ if (typeof original !== 'function')
231
+ return original;
232
+ return (...args) => {
233
+ const runnable = original.apply(target, args);
234
+ if (runnable === null || typeof runnable !== 'object')
235
+ return runnable;
236
+ return new Proxy(runnable, {
237
+ get(runnableTarget, runnableProp) {
238
+ if (runnableProp !== 'invoke')
239
+ return forward(runnableTarget, runnableProp);
240
+ const invoke = Reflect.get(runnableTarget, runnableProp, runnableTarget);
241
+ if (typeof invoke !== 'function')
242
+ return invoke;
243
+ return (messages, ...rest) => invoke.apply(runnableTarget, [
244
+ armInvoke(messages),
245
+ ...rest,
246
+ ]);
247
+ },
248
+ });
249
+ };
250
+ },
251
+ });
252
+ }
253
+ /**
254
+ * A fresh outcome plus the model that writes into it — **one per rating call**, because the
255
+ * classifier is reused across cases the suite runner may run concurrently and a shared record would
256
+ * let one case's arm be adjudicated on another's call.
257
+ *
258
+ * `model` is `undefined` only in the state `buildRaterClassifier` refuses to build: an arm declared
259
+ * with no model to decorate. It is admitted here rather than guarded a second time, so that state
260
+ * lands on {@link reconcileArmedCapture}'s "never reached a rating call" arm — one rule, one place —
261
+ * instead of on a copy of the rule that could drift from it.
262
+ */
263
+ export function armFor(model, command, arm) {
264
+ const outcome = emptyOutcome();
265
+ return {
266
+ model: model === undefined ? undefined : armRaterModel(model, command, arm, outcome),
267
+ outcome,
268
+ };
269
+ }
270
+ /**
271
+ * Adjudicate one armed rating call, **after it has returned and outside `rateShellCommand`'s
272
+ * fail-closed `try`**, and repair the call's diagnostic record.
273
+ *
274
+ * Two jobs, and both are about not lying:
275
+ *
276
+ * - **Raise on an arm that did not do what it said.** A leaked note, or a rating call the decorated
277
+ * model never saw, means this cell's numbers are the other arm's numbers under this arm's name.
278
+ * Silently reporting them is strictly worse than failing the run, because the comparison table
279
+ * would then say the note changed nothing.
280
+ * - **Make `capture.prompt` the prompt that was SENT.** `rateShellCommand` fills the capture from
281
+ * `buildRaterPrompt`'s output at the send site and its contract is that nothing downstream may
282
+ * leave the archive disagreeing with what the model saw. The arm edits the message below that
283
+ * point, so the arm is what owes the correction — and for a facility whose whole subject is prompt
284
+ * content, a diagnostic record of the wrong arm's prompt would be the worst possible artifact.
285
+ *
286
+ * @param capture The record `rateShellCommand` handed back through `onCapture`, when it made one.
287
+ */
288
+ export function reconcileArmedCapture(command, arm, outcome, capture) {
289
+ if (!outcome.invoked) {
290
+ throw new Error(`eval: the prompt arm omitting [${arm.omit.join(', ')}] never reached a rating call for ` +
291
+ `"${command}" — the gate found no usable rating model, so this cell would report the ` +
292
+ "other arm's prompt under this arm's name.");
293
+ }
294
+ if (outcome.leaked.length > 0) {
295
+ // This sentence names a DRIFT between core and the arm, and it is allowed to, because the one
296
+ // authoring slip that could otherwise reach it — the same note named twice, where the second
297
+ // pass finds the block the first already removed — is refused when the suite is parsed.
298
+ throw new Error(`eval: the prompt arm could not remove [${outcome.leaked.join(', ')}] from the rating ` +
299
+ `prompt for "${command}" — core builds that note for this command but it was not found in ` +
300
+ 'the prompt. The arm and the prompt builder have diverged; this cell would have rated the ' +
301
+ 'un-omitted prompt.');
302
+ }
303
+ if (capture && outcome.sent)
304
+ capture.prompt = outcome.sent;
305
+ }
306
+ //# sourceMappingURL=raterPromptArm.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"raterPromptArm.js","sourceRoot":"","sources":["../src/raterPromptArm.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA8GG;AACH,OAAO,EAAE,YAAY,EAAE,aAAa,EAAE,MAAM,0BAA0B,CAAC;AAGvE,OAAO,EAAE,0BAA0B,EAAE,MAAM,2CAA2C,CAAC;AACvF,OAAO,EAAE,wBAAwB,EAAE,MAAM,4CAA4C,CAAC;AAEtF;;;;;;;;;;;;;;;;GAgBG;AACH,MAAM,CAAC,MAAM,kBAAkB,GAAG;IAChC;yEACqE;IACrE,qBAAqB,EAAE,0BAA0B;IACjD,8FAA8F;IAC9F,MAAM,EAAE,wBAAwB;CACqC,CAAC;AAKxE,kGAAkG;AAClG,MAAM,CAAC,MAAM,uBAAuB,GAAG,MAAM,CAAC,IAAI,CAAC,kBAAkB,CAA0B,CAAC;AAwChG,kDAAkD;AAClD,SAAS,YAAY;IACnB,OAAO,EAAE,OAAO,EAAE,KAAK,EAAE,OAAO,EAAE,EAAE,EAAE,MAAM,EAAE,EAAE,EAAE,MAAM,EAAE,EAAE,EAAE,CAAC;AACjE,CAAC;AAED;;;;;;GAMG;AACH,MAAM,UAAU,qBAAqB,CACnC,IAAY,EACZ,OAAe,EACf,GAAmB;IAOnB,IAAI,GAAG,GAAG,IAAI,CAAC;IACf,MAAM,OAAO,GAA0B,EAAE,CAAC;IAC1C,MAAM,MAAM,GAA0B,EAAE,CAAC;IACzC,MAAM,MAAM,GAA0B,EAAE,CAAC;IACzC,KAAK,MAAM,IAAI,IAAI,GAAG,CAAC,IAAI,EAAE,CAAC;QAC5B,MAAM,IAAI,GAAG,kBAAkB,CAAC,IAAI,CAAC,CAAC,OAAO,CAAC,CAAC;QAC/C,IAAI,IAAI,KAAK,IAAI,EAAE,CAAC;YAClB,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;YAClB,SAAS;QACX,CAAC;QACD,MAAM,KAAK,GAAG,OAAO,IAAI,EAAE,CAAC;QAC5B,MAAM,EAAE,GAAG,GAAG,CAAC,OAAO,CAAC,KAAK,CAAC,CAAC;QAC9B,IAAI,EAAE,KAAK,CAAC,CAAC,EAAE,CAAC;YACd,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;YAClB,SAAS;QACX,CAAC;QACD,GAAG,GAAG,GAAG,CAAC,KAAK,CAAC,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,CAAC,KAAK,CAAC,EAAE,GAAG,KAAK,CAAC,MAAM,CAAC,CAAC;QACtD,OAAO,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;IACrB,CAAC;IACD,OAAO,EAAE,IAAI,EAAE,GAAG,EAAE,OAAO,EAAE,MAAM,EAAE,MAAM,EAAE,CAAC;AAChD,CAAC;AAED;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AACH,MAAM,UAAU,aAAa,CAC3B,KAAoB,EACpB,OAAe,EACf,GAAmB,EACnB,OAA8B;IAE9B,MAAM,OAAO,GAAG,CAAC,MAAc,EAAE,IAAqB,EAAW,EAAE;QACjE,MAAM,KAAK,GAAG,OAAO,CAAC,GAAG,CAAC,MAAM,EAAE,IAAI,EAAE,MAAM,CAAC,CAAC;QAChD,OAAO,OAAO,KAAK,KAAK,UAAU;YAChC,CAAC,CAAE,KAAsC,CAAC,IAAI,CAAC,MAAM,CAAC;YACtD,CAAC,CAAC,KAAK,CAAC;IACZ,CAAC,CAAC;IAEF,MAAM,SAAS,GAAG,CAAC,QAAiB,EAAW,EAAE;QAC/C,OAAO,CAAC,OAAO,GAAG,IAAI,CAAC;QACvB,IAAI,CAAC,KAAK,CAAC,OAAO,CAAC,QAAQ,CAAC,IAAI,QAAQ,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;YACtD,OAAO,CAAC,MAAM,CAAC,IAAI,CAAC,GAAG,GAAG,CAAC,IAAI,CAAC,CAAC;YACjC,OAAO,QAAQ,CAAC;QAClB,CAAC;QACD,MAAM,CAAC,MAAM,EAAE,IAAI,CAAC,GAAG,QAAmC,CAAC;QAC3D,IAAI,OAAO,MAAM,EAAE,OAAO,KAAK,QAAQ,IAAI,OAAO,IAAI,EAAE,OAAO,KAAK,QAAQ,EAAE,CAAC;YAC7E,OAAO,CAAC,MAAM,CAAC,IAAI,CAAC,GAAG,GAAG,CAAC,IAAI,CAAC,CAAC;YACjC,OAAO,QAAQ,CAAC;QAClB,CAAC;QACD,MAAM,QAAQ,GAAG,qBAAqB,CAAC,IAAI,CAAC,OAAO,EAAE,OAAO,EAAE,GAAG,CAAC,CAAC;QACnE,OAAO,CAAC,OAAO,CAAC,IAAI,CAAC,GAAG,QAAQ,CAAC,OAAO,CAAC,CAAC;QAC1C,OAAO,CAAC,MAAM,CAAC,IAAI,CAAC,GAAG,QAAQ,CAAC,MAAM,CAAC,CAAC;QACxC,OAAO,CAAC,MAAM,CAAC,IAAI,CAAC,GAAG,QAAQ,CAAC,MAAM,CAAC,CAAC;QACxC,OAAO,CAAC,IAAI,GAAG,EAAE,MAAM,EAAE,MAAM,CAAC,OAAO,EAAE,IAAI,EAAE,QAAQ,CAAC,IAAI,EAAE,CAAC;QAC/D,OAAO,CAAC,IAAI,aAAa,CAAC,MAAM,CAAC,OAAO,CAAC,EAAE,IAAI,YAAY,CAAC,QAAQ,CAAC,IAAI,CAAC,CAAC,CAAC;IAC9E,CAAC,CAAC;IAEF,OAAO,IAAI,KAAK,CAAC,KAAK,EAAE;QACtB,GAAG,CAAC,MAAM,EAAE,IAAI;YACd,IAAI,IAAI,KAAK,sBAAsB;gBAAE,OAAO,OAAO,CAAC,MAAM,EAAE,IAAI,CAAC,CAAC;YAClE,MAAM,QAAQ,GAAG,OAAO,CAAC,GAAG,CAAC,MAAM,EAAE,IAAI,EAAE,MAAM,CAAC,CAAC;YACnD,IAAI,OAAO,QAAQ,KAAK,UAAU;gBAAE,OAAO,QAAQ,CAAC;YACpD,OAAO,CAAC,GAAG,IAAe,EAAW,EAAE;gBACrC,MAAM,QAAQ,GAAI,QAAyC,CAAC,KAAK,CAAC,MAAM,EAAE,IAAI,CAAC,CAAC;gBAChF,IAAI,QAAQ,KAAK,IAAI,IAAI,OAAO,QAAQ,KAAK,QAAQ;oBAAE,OAAO,QAAQ,CAAC;gBACvE,OAAO,IAAI,KAAK,CAAC,QAAkB,EAAE;oBACnC,GAAG,CAAC,cAAc,EAAE,YAAY;wBAC9B,IAAI,YAAY,KAAK,QAAQ;4BAAE,OAAO,OAAO,CAAC,cAAc,EAAE,YAAY,CAAC,CAAC;wBAC5E,MAAM,MAAM,GAAG,OAAO,CAAC,GAAG,CAAC,cAAc,EAAE,YAAY,EAAE,cAAc,CAAC,CAAC;wBACzE,IAAI,OAAO,MAAM,KAAK,UAAU;4BAAE,OAAO,MAAM,CAAC;wBAChD,OAAO,CAAC,QAAiB,EAAE,GAAG,IAAe,EAAW,EAAE,CACvD,MAAuC,CAAC,KAAK,CAAC,cAAc,EAAE;4BAC7D,SAAS,CAAC,QAAQ,CAAC;4BACnB,GAAG,IAAI;yBACR,CAAC,CAAC;oBACP,CAAC;iBACF,CAAC,CAAC;YACL,CAAC,CAAC;QACJ,CAAC;KACF,CAAkB,CAAC;AACtB,CAAC;AAED;;;;;;;;;GASG;AACH,MAAM,UAAU,MAAM,CACpB,KAAgC,EAChC,OAAe,EACf,GAAmB;IAEnB,MAAM,OAAO,GAAG,YAAY,EAAE,CAAC;IAC/B,OAAO;QACL,KAAK,EAAE,KAAK,KAAK,SAAS,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,aAAa,CAAC,KAAK,EAAE,OAAO,EAAE,GAAG,EAAE,OAAO,CAAC;QACpF,OAAO;KACR,CAAC;AACJ,CAAC;AAED;;;;;;;;;;;;;;;;;GAiBG;AACH,MAAM,UAAU,qBAAqB,CACnC,OAAe,EACf,GAAmB,EACnB,OAA8B,EAC9B,OAAqC;IAErC,IAAI,CAAC,OAAO,CAAC,OAAO,EAAE,CAAC;QACrB,MAAM,IAAI,KAAK,CACb,kCAAkC,GAAG,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,oCAAoC;YACvF,IAAI,OAAO,2EAA2E;YACtF,2CAA2C,CAC9C,CAAC;IACJ,CAAC;IACD,IAAI,OAAO,CAAC,MAAM,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QAC9B,8FAA8F;QAC9F,6FAA6F;QAC7F,wFAAwF;QACxF,MAAM,IAAI,KAAK,CACb,0CAA0C,OAAO,CAAC,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,oBAAoB;YACrF,eAAe,OAAO,qEAAqE;YAC3F,2FAA2F;YAC3F,oBAAoB,CACvB,CAAC;IACJ,CAAC;IACD,IAAI,OAAO,IAAI,OAAO,CAAC,IAAI;QAAE,OAAO,CAAC,MAAM,GAAG,OAAO,CAAC,IAAI,CAAC;AAC7D,CAAC"}
@@ -3,7 +3,7 @@
3
3
  * BATCH-25 Half B — the `rater` classification target: grade gth's OWN approvals rater over a
4
4
  * corpus of shell commands.
5
5
  *
6
- * It is the {@link ../evalTypes.js RunClassifyFn} seam Half A defined, and nothing more. Every case
6
+ * It is the {@link "evalTypes.js"!RunClassifyFn | RunClassifyFn} seam Half A defined, and nothing more. Every case
7
7
  * is a sequence of commands; each round is put through the SAME three pieces the production gate
8
8
  * uses — `rateShellCommand` (the rating prompt), `mapVerdictToAction` (the rung-keyed decision
9
9
  * mapping) and `ShellNegotiationState` (§5's transcript and its two bounds) — and what they return
@@ -23,8 +23,9 @@
23
23
  * because the argument ran out is produced nowhere else, so `neg-01-escalate` — the same command
24
24
  * proposed three times, ending at a human — is not expressible without it.
25
25
  *
26
- * Both come from core's own {@link import('@gaunt-sloth/core/core/shell/negotiation.js')
27
- * ShellNegotiationState}, the class the production runner drives, rather than from counters of our
26
+ * Both come from core's own
27
+ * {@link @gaunt-sloth/core!core/shell/negotiation.ShellNegotiationState | ShellNegotiationState},
28
+ * the class the production runner drives, rather than from counters of our
28
29
  * own. Re-implementing "when is this round-1" or "when is the bound spent" here would be exactly the
29
30
  * second opinion rule 1 below forbids — and §5.6's warning is that an implementation clearing the
30
31
  * counter without the transcript *"looks correct and passes any obvious test"*.
@@ -39,15 +40,16 @@
39
40
  * eval facility has no second opinion about what a rating MEANS, and cannot drift from the thing
40
41
  * it measures. A suite's `classification.labels` is the only place the vocabulary is written
41
42
  * down, and it is authored, not compiled in.
42
- * 2. **A model-free decision reports no label.** See {@link classifyOneRound}.
43
+ * 2. **A model-free decision reports no label.** See `classifyOneRound`.
43
44
  */
44
45
  import type { BaseChatModel } from '@langchain/core/language_models/chat_models';
45
46
  import type { GthConfig } from '@gaunt-sloth/core/config.js';
47
+ import type { RaterPromptArm } from '#src/raterPromptArm.js';
46
48
  import { HARDLINE_REFUSAL_MARKER } from '#src/evalTypes.js';
47
49
  import type { ForcedByMechanism, PreflightMechanism, RaterTarget, RunClassifyFn } from '#src/evalTypes.js';
48
50
  /**
49
51
  * The marker a rationale carries when the §8 hardline floor refuses the command — see
50
- * {@link ../evalTypes.js HARDLINE_REFUSAL_MARKER} for why it is a rationale marker and not a label.
52
+ * {@link "evalTypes.js"!HARDLINE_REFUSAL_MARKER | HARDLINE_REFUSAL_MARKER} for why it is a rationale marker and not a label.
51
53
  * Re-exported here, beside the code that emits it:
52
54
  *
53
55
  * ```yaml
@@ -102,7 +104,7 @@ export declare const NEGOTIATION_BOUND_MARKER = "negotiation bound spent";
102
104
  *
103
105
  * ## What the probes are for
104
106
  *
105
- * Two things, neither of which is attribution — {@link forcedMechanism} gets the arm's NAME from
107
+ * Two things, neither of which is attribution — `forcedMechanism` gets the arm's NAME from
106
108
  * core directly:
107
109
  *
108
110
  * 1. **They derive the permissive rating.** A preflight raises only an outcome below core's
@@ -114,14 +116,14 @@ export declare const NEGOTIATION_BOUND_MARKER = "negotiation bound spent";
114
116
  * corpus. That belongs in a unit test here rather than in someone's eval report, which is what
115
117
  * {@link calibrateMechanisms} is for.
116
118
  *
117
- * The probe table is typed as a TOTAL record over {@link ../evalTypes.js PreflightMechanism}, which
119
+ * The probe table is typed as a TOTAL record over {@link "evalTypes.js"!PreflightMechanism | PreflightMechanism}, which
118
120
  * derives from core's own preflight list — so a preflight core gains does not compile here until
119
121
  * someone writes the command that observes it.
120
122
  *
121
123
  * A probe cannot be run with NO verdict: the preflights raise only an outcome below the
122
124
  * deterministic floor, and a missing verdict becomes core's fail-closed one, which is *at* the
123
125
  * floor. Every command then comes back with the identical placeholder — see
124
- * {@link ../evalTypes.js PREFLIGHT_MECHANISMS} for the whole rule.
126
+ * {@link "evalTypes.js"!PREFLIGHT_MECHANISMS | PREFLIGHT_MECHANISMS} for the whole rule.
125
127
  *
126
128
  * Exported for the unit suite, which exercises each mechanism on a command that is NOT its probe:
127
129
  * that test has to be able to prove the two differ, or it silently stops being that test.
@@ -135,7 +137,7 @@ export declare const MECHANISM_PROBES: Readonly<Record<PreflightMechanism, strin
135
137
  * fail across every corpus. That must be a red unit test rather than a surprise in someone's eval
136
138
  * report.
137
139
  *
138
- * Attribution itself is not at stake. {@link forcedMechanism} names the arm from core's
140
+ * Attribution itself is not at stake. `forcedMechanism` names the arm from core's
139
141
  * `preflightFloorFinding` rather than from this index, so two mechanisms that came to share a
140
142
  * sentence stay individually attributable — a shrunken map is the SIGNAL that something moved, not
141
143
  * the damage it causes.
@@ -143,7 +145,7 @@ export declare const MECHANISM_PROBES: Readonly<Record<PreflightMechanism, strin
143
145
  export declare function calibrateMechanisms(): Map<string, ForcedByMechanism>;
144
146
  /**
145
147
  * The rating a `forced_by: <preflight>` round is driven with, discovered from core rather than
146
- * spelled — see {@link calibrateGate}. Exported for the unit suite, which pins that it is an outcome
148
+ * spelled — see `calibrateGate`. Exported for the unit suite, which pins that it is an outcome
147
149
  * the gate APPROVES on a command with no preflight (i.e. genuinely permissive, so overriding it is a
148
150
  * real finding), without naming one.
149
151
  */
@@ -157,8 +159,28 @@ export interface RaterClassifierOptions {
157
159
  /** `$HOME`, for the prompt's home-path folding. Defaults to the process environment's, exactly as
158
160
  * production does. */
159
161
  home?: string;
160
- /** Rating-call timeout; defaults to core's `RATER_DEFAULT_TIMEOUT_MS`. */
162
+ /**
163
+ * Timeout for ONE model call of the gate — the rating call, and the alignment check behind it.
164
+ * Absent, each falls back to the run's `approvals.raterTimeoutMs` and then to core's
165
+ * `RATER_DEFAULT_TIMEOUT_MS`.
166
+ *
167
+ * It is one budget PER CALL and not one shared across both, because that is core's contract
168
+ * (`AlignmentCheckOptions.timeoutMs`) and production's behaviour: `GthAgentRunner` hands the same
169
+ * resolved value to each. A declined command at a negotiating rung therefore costs up to twice
170
+ * this in wall-clock, which is what the rationale's two fail-closed sentences let a reader see.
171
+ */
161
172
  timeoutMs?: number;
173
+ /**
174
+ * [[BATCH-31]] — this cell's PROMPT ARM: which of our own preflight notes its ratings go out
175
+ * without, so one suite can express a note-on / note-off comparison instead of two.
176
+ *
177
+ * It does NOT mirror an option `rateShellCommand` takes, and that is the design rather than an
178
+ * inconsistency with the options above: a suppression the gate itself understood would be a
179
+ * switch inside the gate. What arrives here is a declaration; what removes the note is a model
180
+ * decorator built in this package. See `raterPromptArm.ts` for the whole argument, including how
181
+ * the omission is kept out of every route a session can reach.
182
+ */
183
+ notes?: RaterPromptArm;
162
184
  }
163
185
  /**
164
186
  * Build the {@link RunClassifyFn} for a {@link RaterTarget} — the whole of Half B.
@@ -168,7 +190,7 @@ export interface RaterClassifierOptions {
168
190
  * config error into an N-times-repeated one and add a profile load to every case's latency.
169
191
  *
170
192
  * @param target The parsed `rater` target (its `rung` is the suite's declaration — see
171
- * {@link resolveRung}).
193
+ * `resolveRung`).
172
194
  * @param config The run's resolved config — the sweep cell's, when sweeping, so `model:` and
173
195
  * `config: { approvals: … }` axes both land here.
174
196
  * @param options Test/caller overrides; see {@link RaterClassifierOptions}.