@gaunt-sloth/batch 2.0.0-beta.9 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/dist/bin.d.ts +2 -2
  2. package/dist/bin.js +2 -2
  3. package/dist/classificationReport.d.ts +2 -2
  4. package/dist/classificationReport.js +2 -2
  5. package/dist/classificationTypes.d.ts +7 -7
  6. package/dist/classificationTypes.js +1 -1
  7. package/dist/evalCompare.d.ts +128 -1
  8. package/dist/evalCompare.js +253 -2
  9. package/dist/evalCompare.js.map +1 -1
  10. package/dist/evalOutput.d.ts +1 -1
  11. package/dist/evalOutput.js +1 -1
  12. package/dist/evalRunner.d.ts +15 -3
  13. package/dist/evalRunner.js +82 -5
  14. package/dist/evalRunner.js.map +1 -1
  15. package/dist/evalSuite.d.ts +5 -1
  16. package/dist/evalSuite.js +218 -7
  17. package/dist/evalSuite.js.map +1 -1
  18. package/dist/evalTypes.d.ts +79 -18
  19. package/dist/evalTypes.js +2 -2
  20. package/dist/evalTypes.js.map +1 -1
  21. package/dist/index.d.ts +8 -2
  22. package/dist/index.js +9 -1
  23. package/dist/index.js.map +1 -1
  24. package/dist/judge.d.ts +2 -2
  25. package/dist/judge.js +10 -14
  26. package/dist/judge.js.map +1 -1
  27. package/dist/output.d.ts +1 -1
  28. package/dist/output.js +1 -1
  29. package/dist/parseOver.d.ts +1 -1
  30. package/dist/parseOver.js +1 -1
  31. package/dist/pipelineCli.d.ts +3 -3
  32. package/dist/pipelineCli.js +3 -3
  33. package/dist/raterPromptArm.d.ts +140 -0
  34. package/dist/raterPromptArm.js +306 -0
  35. package/dist/raterPromptArm.js.map +1 -0
  36. package/dist/raterTarget.d.ts +34 -12
  37. package/dist/raterTarget.js +124 -33
  38. package/dist/raterTarget.js.map +1 -1
  39. package/dist/reporters/registry.d.ts +1 -1
  40. package/dist/reporters/registry.js +1 -1
  41. package/dist/reporters/registry.js.map +1 -1
  42. package/dist/reporters/reporterTypes.d.ts +3 -3
  43. package/dist/reporters/textReporter.js +16 -0
  44. package/dist/reporters/textReporter.js.map +1 -1
  45. package/dist/toolCoverage.d.ts +244 -0
  46. package/dist/toolCoverage.js +414 -0
  47. package/dist/toolCoverage.js.map +1 -0
  48. package/dist/toolCoverageRender.d.ts +31 -0
  49. package/dist/toolCoverageRender.js +71 -0
  50. package/dist/toolCoverageRender.js.map +1 -0
  51. package/dist/toolResultChecks.d.ts +8 -2
  52. package/dist/toolResultChecks.js +45 -3
  53. package/dist/toolResultChecks.js.map +1 -1
  54. package/dist/types.d.ts +38 -3
  55. package/dist/types.js +0 -9
  56. package/dist/types.js.map +1 -1
  57. package/dist/workflow/runWorkflow.d.ts +4 -4
  58. package/dist/workflow/runWorkflow.js +3 -3
  59. package/package.json +9 -8
package/dist/bin.d.ts CHANGED
@@ -1,7 +1,5 @@
1
1
  #!/usr/bin/env node
2
2
  /**
3
- * @module bin
4
- *
5
3
  * BATCH-9 — the `gth-batch` executable. A minimal shebang wrapper around {@link runBatchCli}
6
4
  * (pipelineCli.ts) whose only job beyond delegating is to keep **stdout a clean machine channel**: the
7
5
  * batch runtime's human/status/streaming output all lands on `process.stdout` (via
@@ -9,5 +7,7 @@
9
7
  * for the duration of the run. The JSONL cell records are written straight to fd 1 inside
10
8
  * `runBatchCli` (`fs.writeSync`), bypassing this redirect — the same "protocol channel" discipline
11
9
  * `packages/app/cli.js` uses for the ACP stdio channel.
10
+ *
11
+ * @module
12
12
  */
13
13
  export {};
package/dist/bin.js CHANGED
@@ -1,7 +1,5 @@
1
1
  #!/usr/bin/env node
2
2
  /**
3
- * @module bin
4
- *
5
3
  * BATCH-9 — the `gth-batch` executable. A minimal shebang wrapper around {@link runBatchCli}
6
4
  * (pipelineCli.ts) whose only job beyond delegating is to keep **stdout a clean machine channel**: the
7
5
  * batch runtime's human/status/streaming output all lands on `process.stdout` (via
@@ -9,6 +7,8 @@
9
7
  * for the duration of the run. The JSONL cell records are written straight to fd 1 inside
10
8
  * `runBatchCli` (`fs.writeSync`), bypassing this redirect — the same "protocol channel" discipline
11
9
  * `packages/app/cli.js` uses for the ACP stdio channel.
10
+ *
11
+ * @module
12
12
  */
13
13
  import { runBatchCli } from '#src/pipelineCli.js';
14
14
  const originalStdoutWrite = process.stdout.write.bind(process.stdout);
@@ -4,8 +4,8 @@ import type { ClassifiedCell, EvalClassificationReport, EvalClassificationSpec,
4
4
  * per tag), every declared metric (overall and per tag), the corpus-wide coverage, and the list of
5
5
  * `gate: fail` metrics that were breached.
6
6
  *
7
- * Kept separate from both the extractor ({@link ./classification.js}) and the metric engine
8
- * ({@link ./metrics.js}) so neither imports the other, and so the whole aggregation is exercisable
7
+ * Kept separate from both the extractor ({@link @gaunt-sloth/batch!"classification.js" | ./classification.js}) and the metric engine
8
+ * ({@link @gaunt-sloth/batch!"metrics.js" | ./metrics.js}) so neither imports the other, and so the whole aggregation is exercisable
9
9
  * from a plain array of {@link ClassifiedCell}s with no runner, no I/O, and no model.
10
10
  */
11
11
  export declare function buildClassificationReport(spec: EvalClassificationSpec, metricSpecs: EvalMetricSpec[], cells: ClassifiedCell[]): EvalClassificationReport;
@@ -6,8 +6,8 @@ import { UNRECOGNIZED_LABEL } from '#src/classificationTypes.js';
6
6
  * per tag), every declared metric (overall and per tag), the corpus-wide coverage, and the list of
7
7
  * `gate: fail` metrics that were breached.
8
8
  *
9
- * Kept separate from both the extractor ({@link ./classification.js}) and the metric engine
10
- * ({@link ./metrics.js}) so neither imports the other, and so the whole aggregation is exercisable
9
+ * Kept separate from both the extractor ({@link @gaunt-sloth/batch!"classification.js" | ./classification.js}) and the metric engine
10
+ * ({@link @gaunt-sloth/batch!"metrics.js" | ./metrics.js}) so neither imports the other, and so the whole aggregation is exercisable
11
11
  * from a plain array of {@link ClassifiedCell}s with no runner, no I/O, and no model.
12
12
  */
13
13
  export function buildClassificationReport(spec, metricSpecs, cells) {
@@ -3,7 +3,7 @@
3
3
  * BATCH-25 — the shapes for a CLASSIFIER eval: a suite-declared label/action enum, the extractors
4
4
  * that turn a SUT answer into a label, the confusion matrix, and declared aggregate metrics.
5
5
  *
6
- * Deliberately its own file rather than more weight in {@link ./evalTypes.js}: eval's per-case
6
+ * Deliberately its own file rather than more weight in `evalTypes.js`: eval's per-case
7
7
  * PASS/FAIL shapes and a classifier's corpus-wide *distribution* shapes answer different questions,
8
8
  * and only the latter needs the anti-blind-metric machinery below.
9
9
  *
@@ -91,7 +91,7 @@ export interface EvalClassificationSpec {
91
91
  *
92
92
  * There is deliberately **no `model.action`**: the deterministic layer overrides the label and
93
93
  * produces the action, so an "action the model chose" does not exist and a predicate on it could
94
- * never match. {@link ./metrics.js parseMetricPredicate} rejects it by name.
94
+ * never match. {@link "metrics.js"!parseMetricPredicate | parseMetricPredicate} rejects it by name.
95
95
  */
96
96
  export type MetricField = 'expected.label' | 'expected.action' | 'actual.label' | 'actual.action' | 'model.label';
97
97
  /** The literal `none` in a predicate — matches a cell where that field is absent. */
@@ -267,7 +267,7 @@ export interface EvalConfusionMatrix {
267
267
  }
268
268
  /**
269
269
  * The suite-level classification report — the classifier eval's answer, attached to
270
- * {@link ./evalTypes.js EvalSuiteSummary}. Absent entirely for a suite that declares no
270
+ * {@link "evalTypes.js"!EvalSuiteSummary | EvalSuiteSummary}. Absent entirely for a suite that declares no
271
271
  * `classification:` block, so a #405-era suite's `results.json` is unchanged.
272
272
  */
273
273
  export interface EvalClassificationReport {
@@ -295,7 +295,7 @@ export interface EvalClassificationReport {
295
295
  }
296
296
  /**
297
297
  * One graded cell reduced to what the matrices and the metrics need. Built by
298
- * {@link ../evalRunner.js} from the graded results; its own tiny shape so both consumers read the
298
+ * `evalRunner.js` from the graded results; its own tiny shape so both consumers read the
299
299
  * same thing and the aggregation is unit-testable without constructing whole `EvalCaseResult`s.
300
300
  */
301
301
  export interface ClassifiedCell {
@@ -307,13 +307,13 @@ export interface ClassifiedCell {
307
307
  actualAction?: string;
308
308
  /** BATCH-26 — the judgement the MODEL rendered, read by `model.label`. Equal to
309
309
  * {@link actualLabel} except where a deterministic step raised it; absent when nobody judged. See
310
- * {@link ./evalTypes.js ClassifyOutcome.modelLabel}. */
310
+ * {@link "evalTypes.js"!ClassifyOutcome.modelLabel | ClassifyOutcome.modelLabel}. */
311
311
  modelLabel?: string;
312
312
  /** `false` = the SUT did not run / the classifier failed, so this cell has no place in any matrix
313
313
  * or denominator. It is counted as `excluded` and reported, never as a wrong answer. */
314
314
  scored: boolean;
315
315
  }
316
- /** One cell's classification, as recorded on its {@link ./evalTypes.js EvalCaseResult}. */
316
+ /** One cell's classification, as recorded on its {@link "evalTypes.js"!EvalCaseResult | EvalCaseResult}. */
317
317
  export interface EvalCaseClassification {
318
318
  expectedLabel?: string;
319
319
  expectedAction?: string;
@@ -322,7 +322,7 @@ export interface EvalCaseClassification {
322
322
  /** BATCH-26 — the judgement the MODEL rendered, before any deterministic step overrode it. Equal
323
323
  * to {@link actualLabel} except where one did; absent when nobody judged. Written to the cell's
324
324
  * `results.json` so a floored case is diagnosable without re-running, and read by a `model.label`
325
- * metric. See {@link ./evalTypes.js ClassifyOutcome.modelLabel}. */
325
+ * metric. See {@link "evalTypes.js"!ClassifyOutcome.modelLabel | ClassifyOutcome.modelLabel}. */
326
326
  modelLabel?: string;
327
327
  /** The raw text the extractors read, kept so an `(unrecognized)` result is diagnosable without
328
328
  * re-running. Omitted when it is identical to the cell's `answer`. */
@@ -3,7 +3,7 @@
3
3
  * BATCH-25 — the shapes for a CLASSIFIER eval: a suite-declared label/action enum, the extractors
4
4
  * that turn a SUT answer into a label, the confusion matrix, and declared aggregate metrics.
5
5
  *
6
- * Deliberately its own file rather than more weight in {@link ./evalTypes.js}: eval's per-case
6
+ * Deliberately its own file rather than more weight in `evalTypes.js`: eval's per-case
7
7
  * PASS/FAIL shapes and a classifier's corpus-wide *distribution* shapes answer different questions,
8
8
  * and only the latter needs the anti-blind-metric machinery below.
9
9
  *
@@ -1,4 +1,5 @@
1
1
  import type { EvalSuite, EvalSuiteSummary, EvalSweep } from '#src/evalTypes.js';
2
+ import type { RaterPromptArm } from '#src/raterPromptArm.js';
2
3
  /**
3
4
  * BATCH-25 — the comparison layer: run the same corpus across a sweep of configurations and emit
4
5
  * ONE comparison table, and diff a run against a previous one.
@@ -30,6 +31,16 @@ export interface SweepCell {
30
31
  model?: string;
31
32
  /** The merged plain-data config overrides for this cell. */
32
33
  config: Record<string, unknown>;
34
+ /**
35
+ * [[BATCH-31]] — the cell's rater prompt arm, when an axis value declares one.
36
+ *
37
+ * **A sibling of {@link config}, never a key inside it, and that separation is the facility's
38
+ * first enforcement leg.** `initConfigForCell` deep-merges `config` onto the resolved `GthConfig`
39
+ * and nothing else; keeping the arm out of it is what makes the omission unable to become a
40
+ * config value — and therefore unable to be written in a config file, an env var or a CLI flag,
41
+ * the three carriers that would make it reachable from a live session's approvals gate.
42
+ */
43
+ notes?: RaterPromptArm;
33
44
  }
34
45
  /**
35
46
  * Expand a {@link EvalSweep} into its cartesian product of cells, in declared axis order.
@@ -71,6 +82,101 @@ export interface RunDiffEntry {
71
82
  before: string;
72
83
  after: string;
73
84
  }
85
+ /**
86
+ * BATCH-33 — how a run-over-run JUDGE-SCORE drift report is filtered.
87
+ *
88
+ * ## Why there is a filter at all, and why the raw form is not a member of this union
89
+ *
90
+ * A judge score is a model's 0-10 opinion, and it wobbles between identical runs. Reporting every
91
+ * per-case delta therefore prints a section on every run, most of it noise, and trains the reader to
92
+ * skip it — at which point the real slide is skipped along with it. That is strictly WORSE than
93
+ * reporting no drift at all: a section nobody reads still spends the run's credibility, which is the
94
+ * same failure BATCH-25's untrusted metric ran into.
95
+ *
96
+ * So the raw unfiltered form is deliberately not representable. There is no `all` member, and no
97
+ * value of the knobs below reaches one — {@link parseJudgeDriftFilter} rejects `min:0` and bounds
98
+ * the threshold-ward tolerance, because a knob that admits its own degenerate value ships the form
99
+ * this type exists to withhold.
100
+ *
101
+ * ## What each member is for
102
+ *
103
+ * - `threshold-ward` (**the default**) — report a case only when its score CROSSED the pass
104
+ * threshold, or moved DOWN to within `tolerance` points of it. Distance to the gate is the thing
105
+ * worth alerting on: a 10 → 8 that stays clear of a gate at 6 is a different event from a 7 → 6
106
+ * sitting on it, and only the second is about to cost a verdict. It is the default because it is
107
+ * the one filter whose false-positive rate on a stable suite is near zero: at the default
108
+ * tolerance a score must land at or below its own gate to report, so a corpus grading anywhere
109
+ * above its gate wobbles freely and prints nothing.
110
+ * - `min-points` — report any movement of at least N points, either direction. The blunt
111
+ * instrument: it sees large swings wherever they land, and pays for that by firing on a wobbly
112
+ * judge no matter how much headroom the case had.
113
+ * - `mean` — report only the suite-level mean. Averaging over the corpus cancels the wobble, so it
114
+ * is the quietest form of all and the one that cannot say WHICH case moved.
115
+ * - `off` — no drift report. Opting out is not the raw form; it prints less, not more.
116
+ */
117
+ export type JudgeDriftFilter = {
118
+ mode: 'threshold-ward';
119
+ tolerance: number;
120
+ } | {
121
+ mode: 'min-points';
122
+ points: number;
123
+ } | {
124
+ mode: 'mean';
125
+ } | {
126
+ mode: 'off';
127
+ };
128
+ /**
129
+ * How many points above the gate still counts as sitting ON it, when the caller names no tolerance.
130
+ *
131
+ * Zero, and the reason is a measurement rather than a preference. A local judge asked to re-rate one
132
+ * fixed answer against one fixed rubric twelve times returned rates spanning a full point (7 and 8,
133
+ * on the same input every time). So one point of movement carries no information: it is the width of
134
+ * the noise. A shoulder of 1 would report every cell that wobbled from 8 to 7 against the default
135
+ * gate of 6 — an ordinary judged suite grades right there, so the section would fire on a stable
136
+ * re-run, which is the one failure this node exists to avoid.
137
+ *
138
+ * At zero the rule still reports both movements the node names: a 7 → 6 lands ON the gate and
139
+ * reports, a 10 → 8 stays clear and does not. The shoulder was buying nothing those two clauses did
140
+ * not already cover, and was costing exactly the measured noise band. Widen it deliberately with
141
+ * `--drift threshold-ward:<n>` on a corpus whose judge is steadier than this.
142
+ */
143
+ export declare const DEFAULT_JUDGE_DRIFT_TOLERANCE = 0;
144
+ /**
145
+ * The widest tolerance a caller may ask for. The judge scale is 0-10 and a typical gate is 6, so a
146
+ * shoulder much wider than this covers the whole usable band and every downward move reports —
147
+ * the unfiltered form by another route, which is what the bound exists to refuse.
148
+ */
149
+ export declare const MAX_JUDGE_DRIFT_TOLERANCE = 3;
150
+ /** The filter {@link diffRuns} applies when the caller names none. */
151
+ export declare const DEFAULT_JUDGE_DRIFT_FILTER: JudgeDriftFilter;
152
+ /**
153
+ * Parse a `--drift` spec into a {@link JudgeDriftFilter}, throwing on anything unrecognised —
154
+ * including the degenerate values that would reconstitute the raw per-case report.
155
+ *
156
+ * Accepted: `threshold-ward` (or `threshold-ward:<tolerance>`), `min:<points>`, `mean`, `off`.
157
+ */
158
+ export declare function parseJudgeDriftFilter(spec: string): JudgeDriftFilter;
159
+ /** One case whose JUDGE SCORE moved between two runs, as kept by a {@link JudgeDriftFilter}. */
160
+ export interface JudgeDriftEntry {
161
+ id: string;
162
+ before: number;
163
+ after: number;
164
+ /** `after - before`. Negative is a slide toward the gate. */
165
+ delta: number;
166
+ /** The gate the movement is measured against — THIS run's, since that is the one now in force. */
167
+ passThreshold: number;
168
+ /** Why the filter kept it: the score crossed the gate, slid onto the gate's shoulder, or simply
169
+ * moved far enough for a `min-points` filter. */
170
+ reason: 'crossed' | 'near-threshold' | 'moved';
171
+ }
172
+ /** The suite-level judge-score mean — the `mean` filter's whole output. */
173
+ export interface JudgeDriftMean {
174
+ before: number;
175
+ after: number;
176
+ delta: number;
177
+ /** Cases carrying a rate on BOTH sides, i.e. the mean's denominator. */
178
+ cases: number;
179
+ }
74
180
  /** The run-over-run diff. */
75
181
  export interface RunDiff {
76
182
  /** Cases in both runs. */
@@ -88,6 +194,20 @@ export interface RunDiff {
88
194
  after: number | null;
89
195
  delta: number | null;
90
196
  }[];
197
+ /**
198
+ * BATCH-33 — per-case judge-score movements the active filter kept. The signal a VERDICT-only
199
+ * diff structurally cannot see: a judge-graded suite with no `classification:` block has neither
200
+ * `reclassified` nor `metricDeltas`, so before this its diff was strictly binary and a score
201
+ * sliding toward its gate reported "no change." until the run it finally broke.
202
+ *
203
+ * EMPTY under the `mean` and `off` filters, which report no per-case rows at all.
204
+ */
205
+ judgeDrift: JudgeDriftEntry[];
206
+ /** The suite-level mean, under the `mean` filter only. */
207
+ judgeDriftMean?: JudgeDriftMean;
208
+ /** The filter that produced the two fields above, echoed so the render can name it: a quiet drift
209
+ * section means nothing unless the reader can see WHICH filter was quiet. */
210
+ judgeDriftFilter: JudgeDriftFilter;
91
211
  /** Ids in only one of the two runs — reported, so a shrunken corpus cannot read as "no change". */
92
212
  onlyInBefore: string[];
93
213
  onlyInAfter: string[];
@@ -100,8 +220,15 @@ export interface RunDiff {
100
220
  * gate reads, verdict fixes are what a change claims to have done, and RECLASSIFICATIONS are what
101
221
  * a prompt edit actually moved — a case can keep its verdict while the label underneath it changes,
102
222
  * and that is exactly the drift a pass-rate comparison cannot see.
223
+ *
224
+ * A fourth, {@link RunDiff.judgeDrift}, covers the suites the other three cannot: `reclassified`
225
+ * and `metricDeltas` both need a `classification:` block, so an ordinary judge-graded corpus got a
226
+ * strictly binary diff and learned nothing until a score finally broke its gate. `filter` decides
227
+ * how much of that movement is worth printing — see {@link JudgeDriftFilter} for why it defaults to
228
+ * the threshold-ward one and why the unfiltered form is not on offer. **The default lives here, not
229
+ * only in the CLI**, so every caller gets the quiet behaviour without having to know to ask.
103
230
  */
104
- export declare function diffRuns(before: EvalSuiteSummary, after: EvalSuiteSummary): RunDiff;
231
+ export declare function diffRuns(before: EvalSuiteSummary, after: EvalSuiteSummary, filter?: JudgeDriftFilter): RunDiff;
105
232
  /** Render a {@link RunDiff} as plain lines. */
106
233
  export declare function renderRunDiff(diff: RunDiff): string[];
107
234
  /** Does this suite declare a sweep? Small helper so the command reads declaratively. */
@@ -20,17 +20,24 @@ export function expandSweep(sweep) {
20
20
  return cells.map((cell) => {
21
21
  let model;
22
22
  let config = {};
23
+ // [[BATCH-31]] — the arm follows `model`'s rule, not `config`'s: a later axis REPLACES it rather
24
+ // than merging. Two axes declaring arms are describing the same knob twice (the cell name shows
25
+ // it), and a union of their omissions would build a third arm neither axis wrote.
26
+ let notes;
23
27
  for (const part of cell.parts) {
24
28
  if (part.value.model !== undefined)
25
29
  model = part.value.model;
26
30
  if (part.value.config)
27
31
  config = deepMerge(config, part.value.config);
32
+ if (part.value.notes !== undefined)
33
+ notes = part.value.notes;
28
34
  }
29
35
  return {
30
36
  name: cell.parts.map((part) => `${part.axis}=${part.value.name}`).join(' · '),
31
37
  dirName: cell.parts.map((part) => `${part.axis}-${part.value.name}`).join('__'),
32
38
  model,
33
39
  config,
40
+ ...(notes !== undefined ? { notes } : {}),
34
41
  };
35
42
  });
36
43
  }
@@ -128,6 +135,137 @@ function cellMetricValue(column, metricName, tag) {
128
135
  return 'n/a';
129
136
  return formatTally(tally);
130
137
  }
138
+ /**
139
+ * How many points above the gate still counts as sitting ON it, when the caller names no tolerance.
140
+ *
141
+ * Zero, and the reason is a measurement rather than a preference. A local judge asked to re-rate one
142
+ * fixed answer against one fixed rubric twelve times returned rates spanning a full point (7 and 8,
143
+ * on the same input every time). So one point of movement carries no information: it is the width of
144
+ * the noise. A shoulder of 1 would report every cell that wobbled from 8 to 7 against the default
145
+ * gate of 6 — an ordinary judged suite grades right there, so the section would fire on a stable
146
+ * re-run, which is the one failure this node exists to avoid.
147
+ *
148
+ * At zero the rule still reports both movements the node names: a 7 → 6 lands ON the gate and
149
+ * reports, a 10 → 8 stays clear and does not. The shoulder was buying nothing those two clauses did
150
+ * not already cover, and was costing exactly the measured noise band. Widen it deliberately with
151
+ * `--drift threshold-ward:<n>` on a corpus whose judge is steadier than this.
152
+ */
153
+ export const DEFAULT_JUDGE_DRIFT_TOLERANCE = 0;
154
+ /**
155
+ * The widest tolerance a caller may ask for. The judge scale is 0-10 and a typical gate is 6, so a
156
+ * shoulder much wider than this covers the whole usable band and every downward move reports —
157
+ * the unfiltered form by another route, which is what the bound exists to refuse.
158
+ */
159
+ export const MAX_JUDGE_DRIFT_TOLERANCE = 3;
160
+ /** The filter {@link diffRuns} applies when the caller names none. */
161
+ export const DEFAULT_JUDGE_DRIFT_FILTER = {
162
+ mode: 'threshold-ward',
163
+ tolerance: DEFAULT_JUDGE_DRIFT_TOLERANCE,
164
+ };
165
+ /**
166
+ * Parse a `--drift` spec into a {@link JudgeDriftFilter}, throwing on anything unrecognised —
167
+ * including the degenerate values that would reconstitute the raw per-case report.
168
+ *
169
+ * Accepted: `threshold-ward` (or `threshold-ward:<tolerance>`), `min:<points>`, `mean`, `off`.
170
+ */
171
+ export function parseJudgeDriftFilter(spec) {
172
+ const parts = spec.trim().split(':');
173
+ if (parts.length > 2)
174
+ throw new Error(driftSpecError(spec));
175
+ const mode = parts[0].trim().toLowerCase();
176
+ const value = parts[1];
177
+ if (mode === 'off' || mode === 'mean') {
178
+ if (value !== undefined)
179
+ throw new Error(driftSpecError(spec));
180
+ return mode === 'off' ? { mode: 'off' } : { mode: 'mean' };
181
+ }
182
+ if (mode === 'min') {
183
+ const points = driftNumber(value, spec);
184
+ // 0 or less would keep every movement — the unfiltered report this filter exists to avoid.
185
+ if (points < 1 || points > 10) {
186
+ throw new Error(`--drift min:<points> takes 1-10, so "${spec}" is not accepted: it would report every ` +
187
+ 'judge-score movement, which is the unfiltered form that trains readers to skip the ' +
188
+ 'section.');
189
+ }
190
+ return { mode: 'min-points', points };
191
+ }
192
+ if (mode === 'threshold-ward' || mode === 'threshold') {
193
+ if (value === undefined)
194
+ return { mode: 'threshold-ward', tolerance: DEFAULT_JUDGE_DRIFT_TOLERANCE };
195
+ const tolerance = driftNumber(value, spec);
196
+ if (tolerance < 0 || tolerance > MAX_JUDGE_DRIFT_TOLERANCE) {
197
+ throw new Error(`--drift threshold-ward:<tolerance> takes 0-${MAX_JUDGE_DRIFT_TOLERANCE}, so "${spec}" is ` +
198
+ 'not accepted: a shoulder that wide reports nearly every downward move, which is the ' +
199
+ 'unfiltered form that trains readers to skip the section.');
200
+ }
201
+ return { mode: 'threshold-ward', tolerance };
202
+ }
203
+ throw new Error(driftSpecError(spec));
204
+ }
205
+ function driftSpecError(spec) {
206
+ return (`unrecognised --drift filter "${spec}". Use "threshold-ward" (the default, optionally ` +
207
+ `"threshold-ward:<0-${MAX_JUDGE_DRIFT_TOLERANCE}>"), "min:<1-10>", "mean", or "off".`);
208
+ }
209
+ function driftNumber(raw, spec) {
210
+ if (raw === undefined || raw.trim() === '')
211
+ throw new Error(driftSpecError(spec));
212
+ const value = Number(raw.trim());
213
+ if (!Number.isInteger(value))
214
+ throw new Error(driftSpecError(spec));
215
+ return value;
216
+ }
217
+ /**
218
+ * The judge rate representing one CELL, or `undefined` when no verdict was rendered for it.
219
+ *
220
+ * A single-turn cell carries its own {@link EvalCaseResult.judge}. A MULTI-TURN cell leaves that
221
+ * unset and grades per turn, so its representative rate is the LOWEST any turn scored: the cell
222
+ * passes iff every turn passes, which makes the weakest turn the one sitting nearest the gate and
223
+ * therefore the one this report is about. Reading only the top-level field would hand every
224
+ * multi-turn suite an empty drift section forever — the same silence this node exists to end.
225
+ */
226
+ function judgeRateOf(result) {
227
+ const direct = rateOfOutcome(result.judge);
228
+ if (direct !== undefined)
229
+ return direct;
230
+ const turnRates = (result.turns ?? [])
231
+ .map((turn) => rateOfOutcome(turn.judge))
232
+ .filter((rate) => rate !== undefined);
233
+ return turnRates.length === 0 ? undefined : Math.min(...turnRates);
234
+ }
235
+ /** A usable 0-10 rate, or `undefined`. A judge that errored, timed out, returned something
236
+ * unparseable or was never asked has no score to compare, and substituting one would report a slide
237
+ * nobody measured. */
238
+ function rateOfOutcome(judge) {
239
+ if (judge?.ok !== true)
240
+ return undefined;
241
+ const rate = judge.verdict?.rate;
242
+ return typeof rate === 'number' && Number.isFinite(rate) ? rate : undefined;
243
+ }
244
+ /**
245
+ * Does one movement survive the filter? Returns the reason it was kept, or `undefined`.
246
+ *
247
+ * The threshold-ward rule is two clauses and BOTH are load-bearing:
248
+ *
249
+ * - a CROSSING of the gate, either direction, always reports. That is the moment the score changed
250
+ * what it is worth, and its numbers are what explain the verdict flip printed beside it.
251
+ * - otherwise, only a DOWNWARD move landing within `tolerance` of the gate. Direction alone is not
252
+ * the rule — a 10 → 8 is downward and still clear of a gate at 6, and keeping it is exactly how
253
+ * the section fills with wobble. Proximity alone is not the rule either, or a case recovering
254
+ * 6 → 7 would report. It is the pair, and a reader tempted to simplify one clause away should
255
+ * expect the section to go noisy on the next stable re-run.
256
+ */
257
+ function driftReason(before, after, passThreshold, filter) {
258
+ if (filter.mode === 'min-points') {
259
+ return Math.abs(after - before) >= filter.points ? 'moved' : undefined;
260
+ }
261
+ if (filter.mode !== 'threshold-ward')
262
+ return undefined;
263
+ if (before >= passThreshold !== after >= passThreshold)
264
+ return 'crossed';
265
+ if (after < before && after <= passThreshold + filter.tolerance)
266
+ return 'near-threshold';
267
+ return undefined;
268
+ }
131
269
  /** The key a cell is diffed on across runs: id plus identity, since a matrix cell's id alone is
132
270
  * ambiguous. */
133
271
  function diffKey(result) {
@@ -140,19 +278,44 @@ function diffKey(result) {
140
278
  * gate reads, verdict fixes are what a change claims to have done, and RECLASSIFICATIONS are what
141
279
  * a prompt edit actually moved — a case can keep its verdict while the label underneath it changes,
142
280
  * and that is exactly the drift a pass-rate comparison cannot see.
281
+ *
282
+ * A fourth, {@link RunDiff.judgeDrift}, covers the suites the other three cannot: `reclassified`
283
+ * and `metricDeltas` both need a `classification:` block, so an ordinary judge-graded corpus got a
284
+ * strictly binary diff and learned nothing until a score finally broke its gate. `filter` decides
285
+ * how much of that movement is worth printing — see {@link JudgeDriftFilter} for why it defaults to
286
+ * the threshold-ward one and why the unfiltered form is not on offer. **The default lives here, not
287
+ * only in the CLI**, so every caller gets the quiet behaviour without having to know to ask.
143
288
  */
144
- export function diffRuns(before, after) {
289
+ export function diffRuns(before, after, filter = DEFAULT_JUDGE_DRIFT_FILTER) {
145
290
  const beforeByKey = new Map(before.cases.map((result) => [diffKey(result), result]));
146
291
  const afterByKey = new Map(after.cases.map((result) => [diffKey(result), result]));
147
292
  const regressed = [];
148
293
  const fixed = [];
149
294
  const reclassified = [];
295
+ const ratePairs = [];
296
+ let lostRates = 0;
297
+ let movedGates = 0;
150
298
  let compared = 0;
151
299
  for (const [key, afterCase] of afterByKey) {
152
300
  const beforeCase = beforeByKey.get(key);
153
301
  if (!beforeCase)
154
302
  continue;
155
303
  compared += 1;
304
+ const beforeRate = judgeRateOf(beforeCase);
305
+ const afterRate = judgeRateOf(afterCase);
306
+ if (beforeRate !== undefined && afterRate !== undefined) {
307
+ ratePairs.push({
308
+ key,
309
+ before: beforeRate,
310
+ after: afterRate,
311
+ passThreshold: afterCase.passThreshold,
312
+ });
313
+ if (beforeCase.passThreshold !== afterCase.passThreshold)
314
+ movedGates += 1;
315
+ }
316
+ else if (beforeRate !== undefined) {
317
+ lostRates += 1;
318
+ }
156
319
  if (beforeCase.verdict === 'PASS' && afterCase.verdict === 'FAIL') {
157
320
  regressed.push({ id: key, before: 'PASS', after: 'FAIL' });
158
321
  }
@@ -181,7 +344,48 @@ export function diffRuns(before, after) {
181
344
  delta: beforeValue === null || afterValue === null ? null : afterValue - beforeValue,
182
345
  });
183
346
  }
347
+ // BATCH-33 — the judge-score drift, as much of it as the filter keeps. `mean` reports only the
348
+ // suite aggregate (no per-case rows); every other filter reports rows and no aggregate.
349
+ const judgeDrift = [];
350
+ let judgeDriftMean;
351
+ if (filter.mode === 'mean') {
352
+ if (ratePairs.length > 0) {
353
+ const mean = (pick) => ratePairs.reduce((sum, pair) => sum + pick(pair), 0) / ratePairs.length;
354
+ const meanBefore = mean((pair) => pair.before);
355
+ const meanAfter = mean((pair) => pair.after);
356
+ judgeDriftMean = {
357
+ before: meanBefore,
358
+ after: meanAfter,
359
+ delta: meanAfter - meanBefore,
360
+ cases: ratePairs.length,
361
+ };
362
+ }
363
+ }
364
+ else {
365
+ for (const pair of ratePairs) {
366
+ const reason = driftReason(pair.before, pair.after, pair.passThreshold, filter);
367
+ if (reason === undefined)
368
+ continue;
369
+ judgeDrift.push({
370
+ id: pair.key,
371
+ before: pair.before,
372
+ after: pair.after,
373
+ delta: pair.after - pair.before,
374
+ passThreshold: pair.passThreshold,
375
+ reason,
376
+ });
377
+ }
378
+ }
184
379
  const warnings = [];
380
+ if (lostRates > 0 && filter.mode !== 'off') {
381
+ warnings.push(`${lostRates} case(s) were judged in the baseline and produced no judge score in this run, ` +
382
+ 'so their drift could not be computed. A quiet drift section here is not "the scores held".');
383
+ }
384
+ if (movedGates > 0 && filter.mode !== 'off') {
385
+ warnings.push(`${movedGates} case(s) changed their pass threshold between the two runs. Drift is measured ` +
386
+ "against THIS run's gate, so some of the distance reported here is the gate moving rather " +
387
+ 'than the score.');
388
+ }
185
389
  if (onlyInBefore.length > 0 || onlyInAfter.length > 0) {
186
390
  warnings.push(`the two runs do not cover the same cases: ${onlyInBefore.length} only in the baseline, ` +
187
391
  `${onlyInAfter.length} only in this run. The comparison covers ${compared} case(s); a ` +
@@ -193,6 +397,9 @@ export function diffRuns(before, after) {
193
397
  fixed,
194
398
  reclassified,
195
399
  metricDeltas,
400
+ judgeDrift,
401
+ judgeDriftMean,
402
+ judgeDriftFilter: filter,
196
403
  onlyInBefore,
197
404
  onlyInAfter,
198
405
  warnings,
@@ -221,6 +428,33 @@ export function renderRunDiff(diff) {
221
428
  list('REGRESSED', diff.regressed);
222
429
  list('fixed', diff.fixed);
223
430
  list('reclassified', diff.reclassified);
431
+ // BATCH-33 — judge drift under its own heading, so a judge-graded suite with no `classification:`
432
+ // block has something to read here at all. Nothing is printed when the filter kept nothing: a
433
+ // quiet section IS the report on a stable re-run, and adding a reassuring "no drift" line would
434
+ // put a row on every run, which is the noise the filter exists to prevent.
435
+ if (diff.judgeDrift.length > 0) {
436
+ lines.push(` JUDGE DRIFT — ${describeDriftFilter(diff.judgeDriftFilter)} (${diff.judgeDrift.length}):`);
437
+ for (const entry of diff.judgeDrift) {
438
+ const sign = entry.delta >= 0 ? '+' : '';
439
+ const margin = entry.after - entry.passThreshold;
440
+ const where = entry.reason === 'crossed'
441
+ ? `CROSSED the pass threshold ${entry.passThreshold}`
442
+ : entry.reason === 'near-threshold'
443
+ ? margin === 0
444
+ ? `now AT the pass threshold ${entry.passThreshold}`
445
+ : margin < 0
446
+ ? `now ${-margin} BELOW the pass threshold ${entry.passThreshold}`
447
+ : `now ${margin} above the pass threshold ${entry.passThreshold}`
448
+ : `pass threshold ${entry.passThreshold}`;
449
+ lines.push(` ${entry.id}: ${entry.before} → ${entry.after} (${sign}${entry.delta}) — ${where}`);
450
+ }
451
+ }
452
+ if (diff.judgeDriftMean) {
453
+ const mean = diff.judgeDriftMean;
454
+ const sign = mean.delta >= 0 ? '+' : '';
455
+ lines.push(` judge mean: ${mean.before.toFixed(2)} → ${mean.after.toFixed(2)} ` +
456
+ `(${sign}${mean.delta.toFixed(2)}) over ${mean.cases} case(s)`);
457
+ }
224
458
  if (diff.metricDeltas.length > 0) {
225
459
  lines.push(' metric deltas:');
226
460
  for (const delta of diff.metricDeltas) {
@@ -234,11 +468,28 @@ export function renderRunDiff(diff) {
234
468
  if (diff.regressed.length === 0 &&
235
469
  diff.fixed.length === 0 &&
236
470
  diff.reclassified.length === 0 &&
237
- diff.metricDeltas.every((delta) => delta.delta === 0)) {
471
+ diff.metricDeltas.every((delta) => delta.delta === 0) &&
472
+ // BATCH-33 — drift has to count here too, or a run that printed a JUDGE DRIFT section would
473
+ // print "no change." underneath it and contradict itself.
474
+ diff.judgeDrift.length === 0 &&
475
+ (diff.judgeDriftMean === undefined || diff.judgeDriftMean.delta === 0)) {
238
476
  lines.push(' no change.');
239
477
  }
240
478
  return lines;
241
479
  }
480
+ /** How the active drift filter is named in the report heading. */
481
+ function describeDriftFilter(filter) {
482
+ switch (filter.mode) {
483
+ case 'threshold-ward':
484
+ return `toward the pass threshold (tolerance ${filter.tolerance})`;
485
+ case 'min-points':
486
+ return `movements of ${filter.points}+ point(s)`;
487
+ case 'mean':
488
+ return 'suite mean';
489
+ case 'off':
490
+ return 'off';
491
+ }
492
+ }
242
493
  /** Does this suite declare a sweep? Small helper so the command reads declaratively. */
243
494
  export function suiteSweep(suite) {
244
495
  return suite.sweep;