driftproof 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/diff.js ADDED
@@ -0,0 +1,152 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ const { bandVerdict, round } = require('./stats');
5
+ const { EFFECT_FLOOR } = require('../config');
6
+
7
+ // Practical-significance gate applied ON TOP of band separation. bandVerdict()
8
+ // stays a pure geometry test (kept that way so its unit checks are unambiguous);
9
+ // this wrapper additionally requires |delta| >= EFFECT_FLOOR before a verdict is
10
+ // claimed. A separated-but-trivial move (below the judge's quantization floor)
11
+ // becomes "within noise (below effect floor)".
12
+ const WITHIN_NOISE_FLOOR = 'within noise (below effect floor)';
13
+ function verdictWithFloor(before, after, delta) {
14
+ const raw = bandVerdict(before.mean, before.stddev, after.mean, after.stddev);
15
+ if ((raw === 'regression' || raw === 'improvement') && Math.abs(delta) < EFFECT_FLOOR) {
16
+ return WITHIN_NOISE_FLOOR;
17
+ }
18
+ return raw;
19
+ }
20
+ function isWithinNoise(v) { return v === 'within noise' || v === WITHIN_NOISE_FLOOR; }
21
+
22
+ // Build a drift report (markdown) between two receipts.
23
+ //
24
+ // CREDIBILITY CORE — the anti-false-positive rule: a per-case regression (or
25
+ // improvement) is claimed ONLY when (1) the two confidence bands do NOT overlap
26
+ // AND (2) the mean moved by at least EFFECT_FLOOR (practical-significance floor,
27
+ // see config.js). When the bands overlap OR the move is below the floor, the
28
+ // change is reported as "within noise" and is NOT counted as a regression. A tool
29
+ // that cries wolf is worse than useless, so band separation PLUS a real-sized
30
+ // delta — not either alone — is what triggers a verdict.
31
+
32
+ // Map case-id → { mean, stddev } for a receipt's with_skill cases. Falls back to
33
+ // score/0 for v0.1 receipts that have no per-case band.
34
+ function withSkillBands(receipt) {
35
+ const out = {};
36
+ for (const c of receipt.results.cases) {
37
+ if (c.mode !== 'with_skill') continue;
38
+ out[c.id] = { mean: c.mean != null ? c.mean : c.score, stddev: c.stddev || 0 };
39
+ }
40
+ return out;
41
+ }
42
+
43
+ // Aggregate with_skill band for a receipt (mean_score ± stddev). v0.1 receipts
44
+ // have no aggregate stddev → 0 (bands collapse to points; every move looks real,
45
+ // which is exactly the v0.1 weakness v0.2 fixes).
46
+ function aggWithBand(receipt) {
47
+ const a = receipt.results.aggregates.with_skill;
48
+ return { mean: a.mean_score, stddev: a.stddev || 0 };
49
+ }
50
+
51
+ function fmt(n) { return n == null ? 'n/a' : (n >= 0 ? '+' : '') + n.toFixed(3); }
52
+ function pct(n) { return n == null ? 'n/a' : n.toFixed(3); }
53
+ function bandStr(x) { return `${x.mean.toFixed(3)} ± ${x.stddev.toFixed(3)}`; }
54
+ function short(h) { return h ? String(h).slice(0, 12) : 'n/a'; }
55
+
56
+ // The headline is a SUMMARY of the per-case band-overlap verdicts — the credibility
57
+ // core lives per case (judge-sample bands), and the headline just aggregates it.
58
+ // It deliberately does NOT run a separate band test on the aggregate mean: the
59
+ // aggregate band is suite dispersion, and a separate test there would either cry
60
+ // wolf (if too tight) or mask real per-case drift (if too wide).
61
+ function headlineVerdict(perCase) {
62
+ const reg = perCase.filter((r) => r.verdict === 'regression').length;
63
+ const imp = perCase.filter((r) => r.verdict === 'improvement').length;
64
+ const s = (n) => (n === 1 ? '' : 's');
65
+ if (reg && imp) return `MIXED — ${reg} case regression${s(reg)} and ${imp} improvement${s(imp)} on non-overlapping bands.`;
66
+ if (reg) return `DRIFT — ${reg} case${s(reg)} regressed (bands do not overlap); the skill is measurably weaker on ${reg} case${s(reg)}.`;
67
+ if (imp) return `IMPROVED — ${imp} case${s(imp)} improved (bands do not overlap); none regressed.`;
68
+ return 'WITHIN NOISE — no case moved beyond its confidence band; the skill holds up.';
69
+ }
70
+
71
+ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B' } = {}) {
72
+ const aB = withSkillBands(a);
73
+ const bB = withSkillBands(b);
74
+ const ids = [...new Set([...Object.keys(aB), ...Object.keys(bB)])];
75
+
76
+ const perCase = ids.map((id) => {
77
+ const before = aB[id] || null;
78
+ const after = bB[id] || null;
79
+ const delta = (before && after) ? round(after.mean - before.mean) : null;
80
+ const verdict = (before && after) ? verdictWithFloor(before, after, delta) : 'n/a';
81
+ return { id, before, after, delta, verdict };
82
+ });
83
+ // Sort worst-first: regressions, then by delta.
84
+ const order = { regression: 0, 'within noise': 1, [WITHIN_NOISE_FLOOR]: 1, improvement: 2, 'n/a': 3 };
85
+ perCase.sort((x, y) => (order[x.verdict] - order[y.verdict]) || ((x.delta || 0) - (y.delta || 0)));
86
+
87
+ const aAgg = aggWithBand(a);
88
+ const bAgg = aggWithBand(b);
89
+ const headlineDelta = round(bAgg.mean - aAgg.mean);
90
+ const regressions = perCase.filter((r) => r.verdict === 'regression');
91
+
92
+ const L = [];
93
+ L.push(`# Drift report`);
94
+ L.push('');
95
+ L.push(`**Skill:** ${a.skill.name} \`${a.skill.version}\``);
96
+ L.push('');
97
+ L.push(`| | ${labelA} | ${labelB} |`);
98
+ L.push(`|---|---|---|`);
99
+ L.push(`| model | \`${a.run.model_id}\` | \`${b.run.model_id}\` |`);
100
+ L.push(`| run date (UTC) | ${a.run.date_utc} | ${b.run.date_utc} |`);
101
+ L.push(`| surface | ${a.run.surface} | ${b.run.surface} |`);
102
+ L.push(`| judge samples/case | ${(a.run.judge || {}).samples || 1} | ${(b.run.judge || {}).samples || 1} |`);
103
+ L.push(`| skill content_hash | \`${short(a.skill.content_hash)}\` | \`${short(b.skill.content_hash)}\` |`);
104
+ L.push(`| suite_hash | \`${short(a.suite.suite_hash)}\` | \`${short(b.suite.suite_hash)}\` |`);
105
+ L.push(`| with_skill (mean ± band) | ${bandStr(aAgg)} | ${bandStr(bAgg)} |`);
106
+ L.push(`| baseline score | ${pct(a.comparison.baseline_score)} | ${pct(b.comparison.baseline_score)} |`);
107
+ L.push(`| skill lift (Δ) | ${fmt(a.comparison.delta)} | ${fmt(b.comparison.delta)} |`);
108
+ L.push('');
109
+
110
+ const warnings = [];
111
+ if (a.skill.content_hash !== b.skill.content_hash) warnings.push('skill content_hash differs — the skill itself changed between receipts, so drift mixes skill edits with model drift.');
112
+ if (a.suite.suite_hash !== b.suite.suite_hash) warnings.push('suite_hash differs — the eval suite changed; per-case comparison may be misleading.');
113
+ if (a.skill.name !== b.skill.name) warnings.push(`different skills (${a.skill.name} vs ${b.skill.name}) — comparison is not meaningful.`);
114
+ if ((a.run.judge || {}).samples <= 1 || (b.run.judge || {}).samples <= 1) warnings.push('one or both receipts are single-sample (no bands) — non-overlap can only be trusted when both sides are sampled.');
115
+ if (warnings.length) {
116
+ L.push('> **⚠ Caveats**');
117
+ for (const w of warnings) L.push(`> - ${w}`);
118
+ L.push('');
119
+ }
120
+
121
+ const nWithin = perCase.filter((r) => isWithinNoise(r.verdict)).length;
122
+ const nFloor = perCase.filter((r) => r.verdict === WITHIN_NOISE_FLOOR).length;
123
+ L.push(`## Headline`);
124
+ L.push('');
125
+ L.push(`**${headlineVerdict(perCase)}**`);
126
+ L.push('');
127
+ L.push(`with_skill mean moved ${fmt(headlineDelta)} (${bandStr(aAgg)} → ${bandStr(bAgg)}; band = suite dispersion). Per-case band-overlap verdicts: ${regressions.length} regression(s), ${perCase.filter((r) => r.verdict === 'improvement').length} improvement(s), ${nWithin} within noise${nFloor ? ` (${nFloor} of them band-separated but below the ${EFFECT_FLOOR} effect floor)` : ''}.`);
128
+ L.push('');
129
+
130
+ L.push(`## Per-case with_skill (band overlap → verdict)`);
131
+ L.push('');
132
+ L.push(`| case | ${labelA} (mean ± sd) | ${labelB} (mean ± sd) | Δ | verdict |`);
133
+ L.push(`|---|---|---|---|---|`);
134
+ for (const r of perCase) {
135
+ const flag = r.verdict === 'regression' ? '🔻 regression' : r.verdict === 'improvement' ? '🔼 improvement' : r.verdict === WITHIN_NOISE_FLOOR ? 'within noise (below floor)' : r.verdict === 'within noise' ? 'within noise' : 'n/a';
136
+ L.push(`| \`${r.id}\` | ${r.before ? bandStr(r.before) : 'n/a'} | ${r.after ? bandStr(r.after) : 'n/a'} | ${fmt(r.delta)} | ${flag} |`);
137
+ }
138
+ L.push('');
139
+
140
+ if (regressions.length) {
141
+ L.push(`## Regressions (${regressions.length}) — bands do not overlap`);
142
+ L.push('');
143
+ for (const r of regressions) {
144
+ L.push(`- \`${r.id}\`: ${bandStr(r.before)} → ${bandStr(r.after)} (${fmt(r.delta)})`);
145
+ }
146
+ L.push('');
147
+ }
148
+
149
+ return { markdown: L.join('\n'), perCase, headlineDelta, regressions };
150
+ }
151
+
152
+ module.exports = { buildDriftReport, withSkillBands };
package/lib/init.js ADDED
@@ -0,0 +1,128 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ const fs = require('fs');
5
+ const path = require('path');
6
+
7
+ // `driftproof init <dir>` scaffolding.
8
+ //
9
+ // Produces a complete, runnable skill skeleton the way agentskills.io expects it:
10
+ // <dir>/SKILL.md — the skill instructions (a stub, if none exists)
11
+ // <dir>/evals/evals.json — an eval suite with 3 example cases, each rubric
12
+ // anchored at 0.80 (the scoring convention the
13
+ // example suite and Report #001 use)
14
+ // <dir>/.driftproofrc — per-project run defaults (budget, models, samples)
15
+ //
16
+ // NEVER overwrites an existing file — every write is guarded, and an existing
17
+ // path is reported as "skipped". Safe to re-run.
18
+
19
+ // JSON carries no comments, so guidance rides in `_`-prefixed keys. The runner's
20
+ // suite loader (lib/skill.js normalizeCases) reads only id/prompt/rubric/
21
+ // pass_threshold and ignores everything else, so these keys are inert at run time
22
+ // and exist purely to guide the author editing the file.
23
+ function evalsTemplate(skillName) {
24
+ return {
25
+ skill: skillName,
26
+ version: '0.1.0',
27
+ format: 'agentskills.io/evals',
28
+ _guidance: [
29
+ 'Each case must be grounded in a claim your SKILL.md actually makes.',
30
+ 'Aim for graded difficulty: a good skill should land with_skill ~0.7-0.9, not 1.0 (saturation hides drift).',
31
+ "Anchor every rubric at 0.80 = 'fully correct'; reserve 0.81-1.00 for exemplary work only, and cap clear errors low.",
32
+ 'The baseline is the SAME prompt with no SKILL.md — so a case only measures the skill if the skill is what makes it pass.',
33
+ 'See AUTHORING.md (https://driftproofhq.com/authoring.html) for the full fair-suite guide.',
34
+ ],
35
+ cases: [
36
+ {
37
+ id: 'example-core-claim',
38
+ _comment: 'Replace with a case that exercises the MAIN thing your skill teaches. The prompt should be answerable badly without the skill and well with it.',
39
+ prompt: 'Describe the task here — a realistic request a user would make that your skill is meant to help with.',
40
+ rubric: "State exactly what a good response must do, drawn from your SKILL.md. SCORING ANCHOR (apply strictly): a fully correct, idiomatic response scores 0.80; award 0.81-0.90 only if it is ALSO exemplary; 0.91-1.00 only if flawless and exceptional (rare). Subtract ~0.2 for each concrete error the skill is supposed to prevent.",
41
+ pass_threshold: 0.7,
42
+ },
43
+ {
44
+ id: 'example-common-mistake',
45
+ _comment: 'Target a specific mistake your skill is supposed to prevent. The baseline (no skill) should fall into the trap; the skill should avoid it.',
46
+ prompt: 'Describe a task where the obvious answer is wrong in a way your skill corrects.',
47
+ rubric: "Identify the specific correct behaviour your skill mandates here. SCORING ANCHOR (apply strictly): if the response makes the mistake the skill exists to prevent, cap at 0.3; a correct response scores 0.80; 0.81-1.00 only if exemplary. Subtract ~0.2 per additional error.",
48
+ pass_threshold: 0.7,
49
+ },
50
+ {
51
+ id: 'example-edge-case',
52
+ _comment: 'A harder edge case with real headroom, so the suite is not saturated. Keep it fair: no undocumented expectations, no gotchas the SKILL.md never mentions.',
53
+ prompt: 'Describe an edge case your skill handles that a naive answer would get subtly wrong.',
54
+ rubric: "Name the subtle requirement here, grounded in a claim the SKILL.md makes. SCORING ANCHOR (apply strictly): a correct handling scores 0.80; 0.81-1.00 only if additionally flawless and exceptional; cap a response that misses the edge case at 0.5.",
55
+ pass_threshold: 0.7,
56
+ },
57
+ ],
58
+ };
59
+ }
60
+
61
+ function skillTemplate(skillName) {
62
+ return `---
63
+ name: ${skillName}
64
+ version: 0.1.0
65
+ description: One line describing what this skill teaches an agent to do.
66
+ ---
67
+
68
+ # ${skillName}
69
+
70
+ Replace this stub with your skill's instructions. A skill is the guidance you
71
+ would give an agent so it follows your conventions — the more concrete and
72
+ checkable the claims, the better a suite can measure whether the model still
73
+ honours them.
74
+
75
+ ## Rules
76
+
77
+ 1. State the first rule your skill enforces.
78
+ 2. State the second.
79
+ 3. Keep each rule specific enough that an eval case can check it.
80
+
81
+ ## Notes
82
+
83
+ Every case in \`evals/evals.json\` should be grounded in a rule above. If a case
84
+ tests something this file never says, the suite is unfair — see AUTHORING.md.
85
+ `;
86
+ }
87
+
88
+ function rcTemplate() {
89
+ return {
90
+ _comment: 'Per-project driftproof defaults. CLI flags override these. See `driftproof help`.',
91
+ models: 'claude-haiku-4-5',
92
+ samples: 5,
93
+ max_usd: 2,
94
+ };
95
+ }
96
+
97
+ // Write `content` to `file` only if it does not already exist. Records the path
98
+ // in `created` or `skipped`.
99
+ function writeIfAbsent(file, content, created, skipped) {
100
+ if (fs.existsSync(file)) { skipped.push(file); return; }
101
+ fs.mkdirSync(path.dirname(file), { recursive: true });
102
+ fs.writeFileSync(file, content);
103
+ created.push(file);
104
+ }
105
+
106
+ // Scaffold into `targetDir`. Returns { dir, created:[abs...], skipped:[abs...] }.
107
+ function scaffoldInit(targetDir) {
108
+ const dir = path.resolve(targetDir);
109
+ const skillName = path.basename(dir);
110
+ const created = [];
111
+ const skipped = [];
112
+
113
+ writeIfAbsent(path.join(dir, 'SKILL.md'), skillTemplate(skillName), created, skipped);
114
+ writeIfAbsent(
115
+ path.join(dir, 'evals', 'evals.json'),
116
+ JSON.stringify(evalsTemplate(skillName), null, 2) + '\n',
117
+ created, skipped,
118
+ );
119
+ writeIfAbsent(
120
+ path.join(dir, '.driftproofrc'),
121
+ JSON.stringify(rcTemplate(), null, 2) + '\n',
122
+ created, skipped,
123
+ );
124
+
125
+ return { dir, created, skipped };
126
+ }
127
+
128
+ module.exports = { scaffoldInit, evalsTemplate, skillTemplate, rcTemplate };
package/lib/json.js ADDED
@@ -0,0 +1,141 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ // JSON salvage + retry/timeout helpers.
5
+ //
6
+ // PORTED (then genericized) from a private codebase's aiHelpers.js:
7
+ // - safeParseJSON (trailing-comma + unescaped-newline repair)
8
+ // - extractJsonArray (streaming brace-matcher that salvages a truncated
9
+ // array of objects — LLMs routinely overrun max_tokens)
10
+ // The private origin carried no business logic in these functions; they are pure
11
+ // string/JSON utilities. Scanned clean against the confidentiality deny-list.
12
+
13
+ function sleep(ms) {
14
+ return new Promise((r) => setTimeout(r, ms));
15
+ }
16
+
17
+ // Parse JSON with basic repair for common LLM output quirks. Throws if the text
18
+ // is unsalvageable so callers can fall back rather than silently accept garbage.
19
+ function safeParseJSON(str) {
20
+ try { return JSON.parse(str); } catch (_e) { /* fall through to repairs */ }
21
+ // Strip a ```json ... ``` fence if present.
22
+ const fence = String(str).match(/```(?:json)?\s*([\s\S]*?)```/i);
23
+ let s = fence ? fence[1] : String(str);
24
+ // Repair trailing commas before ] or }.
25
+ let fixed = s.replace(/,\s*([\]}])/g, '$1');
26
+ try { return JSON.parse(fixed); } catch (_e) { /* keep trying */ }
27
+ // Repair unescaped newlines inside string values.
28
+ fixed = fixed.replace(/(?<=": ")([\s\S]*?)(?="[,}])/g, (m) => m.replace(/\n/g, '\\n'));
29
+ try { return JSON.parse(fixed); } catch (_e) { /* give up */ }
30
+ throw new Error('Invalid JSON (unsalvageable)');
31
+ }
32
+
33
+ // Extract as many complete objects as possible from a JSON array in `text`,
34
+ // even when the array is truncated mid-stream. Returns { items, truncated }.
35
+ // A well-formed array parses on the happy path; otherwise a brace-matcher walks
36
+ // the string and keeps every object that closed cleanly.
37
+ function extractJsonArray(text) {
38
+ if (!text || typeof text !== 'string') return { items: [], truncated: false };
39
+ let s = text.trim();
40
+ const fence = s.match(/```(?:json)?\s*([\s\S]*?)```/i);
41
+ if (fence) s = fence[1].trim();
42
+ try {
43
+ const whole = JSON.parse(s);
44
+ if (Array.isArray(whole)) return { items: whole, truncated: false };
45
+ } catch (_e) { /* salvage below */ }
46
+ const start = s.indexOf('[');
47
+ if (start === -1) return { items: [], truncated: false };
48
+ const items = [];
49
+ let depth = 0, inStr = false, esc = false, objStart = -1, closed = false;
50
+ for (let i = start + 1; i < s.length; i++) {
51
+ const c = s[i];
52
+ if (inStr) {
53
+ if (esc) esc = false;
54
+ else if (c === '\\') esc = true;
55
+ else if (c === '"') inStr = false;
56
+ continue;
57
+ }
58
+ if (c === '"') { inStr = true; continue; }
59
+ if (c === '{') { if (depth === 0) objStart = i; depth++; }
60
+ else if (c === '}') {
61
+ depth--;
62
+ if (depth === 0 && objStart !== -1) {
63
+ try { items.push(JSON.parse(s.slice(objStart, i + 1))); } catch (_e) { /* skip broken object */ }
64
+ objStart = -1;
65
+ }
66
+ } else if (c === ']' && depth === 0) { closed = true; break; }
67
+ }
68
+ return { items, truncated: !closed };
69
+ }
70
+
71
+ // Extract a single JSON object (the first balanced {...}) from free text, with
72
+ // salvage. Returns the parsed object or throws.
73
+ function extractJsonObject(text) {
74
+ if (!text || typeof text !== 'string') throw new Error('no text to parse');
75
+ let s = text.trim();
76
+ const fence = s.match(/```(?:json)?\s*([\s\S]*?)```/i);
77
+ if (fence) s = fence[1].trim();
78
+ const start = s.indexOf('{');
79
+ if (start === -1) throw new Error('no JSON object found');
80
+ let depth = 0, inStr = false, esc = false;
81
+ for (let i = start; i < s.length; i++) {
82
+ const c = s[i];
83
+ if (inStr) {
84
+ if (esc) esc = false;
85
+ else if (c === '\\') esc = true;
86
+ else if (c === '"') inStr = false;
87
+ continue;
88
+ }
89
+ if (c === '"') inStr = true;
90
+ else if (c === '{') depth++;
91
+ else if (c === '}') {
92
+ depth--;
93
+ if (depth === 0) return safeParseJSON(s.slice(start, i + 1));
94
+ }
95
+ }
96
+ // Unterminated object — hand the tail to the repairer.
97
+ return safeParseJSON(s.slice(start));
98
+ }
99
+
100
+ // Run an async thunk with a wall-clock timeout. Rejects with a TIMEOUT error if
101
+ // it does not settle in time.
102
+ function withTimeout(promiseFactory, ms, label = 'operation') {
103
+ return new Promise((resolve, reject) => {
104
+ let done = false;
105
+ const t = setTimeout(() => {
106
+ if (done) return;
107
+ done = true;
108
+ const e = new Error(`${label} timed out after ${ms}ms`);
109
+ e.code = 'TIMEOUT';
110
+ reject(e);
111
+ }, ms);
112
+ Promise.resolve()
113
+ .then(promiseFactory)
114
+ .then((v) => { if (!done) { done = true; clearTimeout(t); resolve(v); } })
115
+ .catch((err) => { if (!done) { done = true; clearTimeout(t); reject(err); } });
116
+ });
117
+ }
118
+
119
+ // Retry an async thunk with exponential backoff. Retries only when shouldRetry
120
+ // returns true for the caught error (default: 429 / rate-limit / timeout).
121
+ async function withRetry(fn, { tries = 3, baseDelayMs = 1000, shouldRetry } = {}) {
122
+ const retryable = shouldRetry || ((err) => {
123
+ const msg = String((err && err.message) || err || '');
124
+ return (err && err.status === 429) || /429|rate.?limit|timeout|ETIMEDOUT|ECONNRESET/i.test(msg);
125
+ });
126
+ let lastErr;
127
+ for (let attempt = 0; attempt < tries; attempt++) {
128
+ try {
129
+ return await fn(attempt);
130
+ } catch (err) {
131
+ lastErr = err;
132
+ if (attempt === tries - 1 || !retryable(err)) throw err;
133
+ await sleep(baseDelayMs * Math.pow(2, attempt));
134
+ }
135
+ }
136
+ throw lastErr;
137
+ }
138
+
139
+ module.exports = {
140
+ sleep, safeParseJSON, extractJsonArray, extractJsonObject, withTimeout, withRetry,
141
+ };
package/lib/judge.js ADDED
@@ -0,0 +1,126 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ const { complete, surfaceLabel } = require('./provider');
5
+ const { extractJsonObject } = require('./json');
6
+ const { sha256 } = require('./canonical');
7
+ const { mean, stddev } = require('./stats');
8
+
9
+ // Rubric-based LLM judge.
10
+ //
11
+ // REWRITTEN from scratch. Only the generic SHAPE of a private "gap report"
12
+ // grader was reused: (system role) + (material + rubric) + "return ONLY JSON" +
13
+ // salvage-parse + clamp/validate the score. All domain content (sales/committee/
14
+ // deal) was dropped — none of it applies here. Scanned clean against the deny-list.
15
+ //
16
+ // The judge reads a model's OUTPUT for one eval case and grades it against that
17
+ // case's rubric, returning a normalized score in [0,1] plus a short reason.
18
+
19
+ const JUDGE_SYSTEM =
20
+ 'You are a strict, fair grader. You are given a TASK, a model RESPONSE to that task, '
21
+ + 'and a RUBRIC describing what a good response must do. Grade only against the rubric. '
22
+ + 'Be objective and specific: reward exactly what the rubric asks for and nothing else. '
23
+ + 'Return your grade as JSON only.';
24
+
25
+ // Build the grading prompt. Kept deterministic so the same (task, response,
26
+ // rubric) always yields the same prompt and thus a stable rubric_hash.
27
+ function buildJudgePrompt({ task, response, rubric }) {
28
+ return [
29
+ 'TASK GIVEN TO THE MODEL:',
30
+ '"""',
31
+ String(task || '').trim(),
32
+ '"""',
33
+ '',
34
+ 'MODEL RESPONSE TO GRADE:',
35
+ '"""',
36
+ String(response || '').trim(),
37
+ '"""',
38
+ '',
39
+ 'RUBRIC (grade strictly against this):',
40
+ '"""',
41
+ String(rubric || '').trim(),
42
+ '"""',
43
+ '',
44
+ 'Return ONLY this JSON object, no prose before or after:',
45
+ '{',
46
+ ' "score": <number 0.0 to 1.0, fraction of the rubric satisfied>,',
47
+ ' "pass": <true if the response substantially meets the rubric, else false>,',
48
+ ' "reason": "<one sentence, <=30 words, citing the specific rubric points met or missed>"',
49
+ '}',
50
+ ].join('\n');
51
+ }
52
+
53
+ // The rubric_hash recorded in the receipt binds a grade to the EXACT grading
54
+ // instruction used, so a later reader can tell whether two receipts were graded
55
+ // the same way. It hashes the judge system prompt + the case rubric text.
56
+ function rubricHash(rubric) {
57
+ return sha256(JUDGE_SYSTEM + '\n---\n' + String(rubric || '').trim());
58
+ }
59
+
60
+ function clamp01(n) {
61
+ const x = Number(n);
62
+ if (!Number.isFinite(x)) return 0;
63
+ return Math.max(0, Math.min(1, x));
64
+ }
65
+
66
+ // Judge settings for the CURRENT surface. Determinism where the surface allows:
67
+ // on the api surface we pin temperature 0 for judge calls (and record it); on
68
+ // the cli surface sampling params are surface-controlled and cannot be set, so
69
+ // temperature is null and `sampling` says so. Recorded into every receipt.
70
+ function judgeSettings(samples) {
71
+ const surface = surfaceLabel();
72
+ if (surface === 'api') {
73
+ return { samples, temperature: 0, sampling: 'api-temperature-0', surface };
74
+ }
75
+ return { samples, temperature: null, sampling: 'surface-controlled', surface };
76
+ }
77
+
78
+ // Grade one response once. Returns { score, reason, raw }.
79
+ // `raw` is the judge's verbatim output text (hashed into the receipt for
80
+ // transcript auditability, and optionally retained under --keep-transcripts).
81
+ async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature }) {
82
+ const prompt = buildJudgePrompt({ task, response, rubric });
83
+ const { text } = await complete({ system: JUDGE_SYSTEM, prompt, model, maxTokens: 400, timeoutMs, temperature });
84
+ let parsed;
85
+ try {
86
+ parsed = extractJsonObject(text);
87
+ } catch (_e) {
88
+ // Unsalvageable judge output → conservative 0 (a judge that can't be parsed
89
+ // must never silently "pass"), tagged so the caller can see it happened.
90
+ return { score: 0, reason: 'judge output unparseable', unparsed: true, raw: String(text || '') };
91
+ }
92
+ return { score: clamp01(parsed.score), reason: String(parsed.reason || '').slice(0, 300), raw: String(text || '') };
93
+ }
94
+
95
+ // Grade a response N times and return the sampled distribution:
96
+ // { samples:[scores], mean, stddev, reason, judge_settings, model_id, rubric_hash }
97
+ // `mean` ± `stddev` is the per-case confidence band (raw spread of the N scores)
98
+ // used by the borderline-outcome rule and per-case drift band-overlap logic.
99
+ async function gradeSamples({ task, response, rubric, model, samples = 5, timeoutMs = 120000 }) {
100
+ const settings = judgeSettings(samples);
101
+ const scores = [];
102
+ const reasons = [];
103
+ const rawTexts = [];
104
+ for (let i = 0; i < samples; i++) {
105
+ const r = await gradeOnce({ task, response, rubric, model, timeoutMs, temperature: settings.temperature === null ? undefined : settings.temperature });
106
+ scores.push(r.score);
107
+ reasons.push(r.reason);
108
+ rawTexts.push(r.raw || '');
109
+ }
110
+ return {
111
+ samples: scores,
112
+ mean: mean(scores),
113
+ stddev: stddev(scores),
114
+ reason: reasons[0] || '',
115
+ // v0.3 transcript auditability: verbatim judge outputs + their sha256 hashes,
116
+ // one per sample. `sample_texts` is transient (retained only under
117
+ // --keep-transcripts); `sample_hashes` goes into the receipt.
118
+ sample_texts: rawTexts,
119
+ sample_hashes: rawTexts.map((t) => sha256(t)),
120
+ judge_settings: settings,
121
+ model_id: model,
122
+ rubric_hash: rubricHash(rubric),
123
+ };
124
+ }
125
+
126
+ module.exports = { gradeOnce, gradeSamples, judgeSettings, rubricHash, buildJudgePrompt, JUDGE_SYSTEM };
package/lib/models.js ADDED
@@ -0,0 +1,125 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ const fs = require('fs');
5
+ const path = require('path');
6
+ const { resolveModel } = require('./provider');
7
+
8
+ // The model registry (config/models.json) is the single source of truth for
9
+ // which models Driftproof knows about, their per-MTok prices, tier, and whether
10
+ // they are eligible to serve as the judge. The cost guard reads prices here; the
11
+ // runner marks each receipt registered/unregistered against it; the release
12
+ // trigger appends auto-discovered ids to it.
13
+
14
+ // The registry ships inside the package (config/models.json), so it resolves
15
+ // correctly from a global/npx install — path is relative to this module, not the
16
+ // caller's CWD. A user can point at their own registry with DRIFTPROOF_REGISTRY.
17
+ const REGISTRY_PATH = process.env.DRIFTPROOF_REGISTRY
18
+ ? path.resolve(process.env.DRIFTPROOF_REGISTRY)
19
+ : path.join(__dirname, '..', 'config', 'models.json');
20
+
21
+ // Conservative default price for an UNREGISTERED model, per MTok. A budget guard
22
+ // must never UNDER-estimate, so an unknown id is costed as if it were the most
23
+ // expensive model we know about (Fable-tier). This is deliberately pessimistic:
24
+ // an unregistered id should over-project and, if anything, abort early rather
25
+ // than quietly overspend. Documented in config/models.json.default_price_note.
26
+ const DEFAULT_PRICE = { input: 10.0, output: 50.0 };
27
+
28
+ let _cache = null;
29
+ function loadRegistry(force) {
30
+ if (_cache && !force) return _cache;
31
+ const raw = JSON.parse(fs.readFileSync(REGISTRY_PATH, 'utf8'));
32
+ const models = Array.isArray(raw.models) ? raw.models : [];
33
+ const byId = Object.create(null);
34
+ for (const m of models) byId[m.id] = m;
35
+ _cache = { ...raw, models, byId };
36
+ return _cache;
37
+ }
38
+
39
+ // Infer the model family from an id. Used to find a same-family predecessor when
40
+ // the release trigger discovers a new model.
41
+ function familyOf(id) {
42
+ const m = String(id).match(/claude-(fable|mythos|opus|sonnet|haiku)/i);
43
+ return m ? m[1].toLowerCase() : 'unknown';
44
+ }
45
+
46
+ // Resolve a (possibly aliased) model id against the registry.
47
+ // returns { id: <canonical>, entry: <registry row|null>, registered: bool }
48
+ function resolveRegistry(modelId) {
49
+ const reg = loadRegistry();
50
+ const canonical = resolveModel(modelId);
51
+ const entry = reg.byId[canonical] || reg.byId[modelId] || null;
52
+ return { id: canonical, entry, registered: !!entry };
53
+ }
54
+
55
+ // The value stamped into receipt.run.registry.
56
+ function registryStatus(modelId) {
57
+ return resolveRegistry(modelId).registered ? 'registered' : 'unregistered';
58
+ }
59
+
60
+ // Price per MTok for a model: the registry price when registered, else the
61
+ // conservative default. Returns { input, output, registered }.
62
+ function priceForModel(modelId) {
63
+ const { entry } = resolveRegistry(modelId);
64
+ if (entry) return { input: entry.input_price, output: entry.output_price, registered: true };
65
+ return { input: DEFAULT_PRICE.input, output: DEFAULT_PRICE.output, registered: false };
66
+ }
67
+
68
+ // Whether a model is allowed to be the judge (fixed-judge policy — only the
69
+ // cheap Haiku judge is judge_eligible; see docs/judge-policy.html).
70
+ function isJudgeEligible(modelId) {
71
+ const { entry } = resolveRegistry(modelId);
72
+ return !!(entry && entry.judge_eligible);
73
+ }
74
+
75
+ // The servable same-family predecessor of a model: the registered model in the
76
+ // same family with the most recent `released` date strictly before this one
77
+ // (falling back to any other same-family model). Null when the family has no
78
+ // other member — the trigger then falls back to a cross-family "capability gap"
79
+ // pairing. `modelId` need not itself be in the registry.
80
+ function familyPredecessor(modelId) {
81
+ const reg = loadRegistry();
82
+ const canonical = resolveModel(modelId);
83
+ const fam = familyOf(canonical);
84
+ const self = reg.byId[canonical] || null;
85
+ const selfDate = self && self.released ? self.released : null;
86
+ let pool = reg.models.filter((m) => familyOf(m.id) === fam && m.id !== canonical);
87
+ if (!pool.length) return null;
88
+ // Prefer members released strictly before the new model when we know its date.
89
+ if (selfDate) {
90
+ const before = pool.filter((m) => m.released && m.released < selfDate);
91
+ if (before.length) pool = before;
92
+ }
93
+ const dated = pool.filter((m) => m.released);
94
+ const chooseFrom = dated.length ? dated : pool;
95
+ chooseFrom.sort((a, b) => String(b.released || '').localeCompare(String(a.released || '')));
96
+ return chooseFrom[0].id;
97
+ }
98
+
99
+ // Append a newly-discovered model id to the registry file (release trigger).
100
+ // `firstSeen` is the first-seen date recorded as `released`. Idempotent: a no-op
101
+ // if the id already exists. Returns the added entry (or null if already present).
102
+ function addDiscoveredModel(id, { firstSeen = null, registryPath = REGISTRY_PATH } = {}) {
103
+ const raw = JSON.parse(fs.readFileSync(registryPath, 'utf8'));
104
+ if (raw.models.some((m) => m.id === id)) return null;
105
+ const family = familyOf(id);
106
+ // Best-effort tier from family; unknown families default to frontier pricing.
107
+ const tier = family === 'haiku' ? 'cheap' : family === 'sonnet' ? 'standard' : 'frontier';
108
+ const price = family === 'haiku' ? { input: 1.0, output: 5.0 }
109
+ : family === 'sonnet' ? { input: 3.0, output: 15.0 }
110
+ : { input: 5.0, output: 25.0 };
111
+ const entry = {
112
+ id, family, provider: 'anthropic', released: firstSeen,
113
+ input_price: price.input, output_price: price.output, tier,
114
+ judge_eligible: false, auto_added: true,
115
+ };
116
+ raw.models.push(entry);
117
+ fs.writeFileSync(registryPath, JSON.stringify(raw, null, 2) + '\n');
118
+ _cache = null; // invalidate cache so subsequent lookups see the new entry
119
+ return entry;
120
+ }
121
+
122
+ module.exports = {
123
+ loadRegistry, resolveRegistry, registryStatus, priceForModel, isJudgeEligible,
124
+ familyOf, familyPredecessor, addDiscoveredModel, DEFAULT_PRICE, REGISTRY_PATH,
125
+ };