driftproof 0.8.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +45 -5
- package/bin/driftproof +191 -22
- package/config.js +26 -2
- package/lib/checks.js +28 -6
- package/lib/diff.js +58 -4
- package/lib/importers.js +8 -3
- package/lib/judge.js +74 -15
- package/lib/models.js +52 -11
- package/lib/provider.js +265 -103
- package/lib/receipt.js +51 -11
- package/lib/run.js +231 -26
- package/lib/skill.js +26 -4
- package/lib/stats.js +12 -3
- package/lib/usage.js +56 -4
- package/lib/value.js +4 -2
- package/lib/verdict.js +38 -7
- package/package.json +3 -2
- package/spec/RECEIPT.md +68 -11
- package/spec/receipt.schema.json +299 -54
- package/spec/receipt.v0.5.schema.json +1318 -0
package/lib/diff.js
CHANGED
|
@@ -59,7 +59,10 @@ function withSkillBands(receipt) {
|
|
|
59
59
|
// which is exactly the v0.1 weakness v0.2 fixes).
|
|
60
60
|
function aggWithBand(receipt) {
|
|
61
61
|
const a = receipt.results.aggregates.with_skill;
|
|
62
|
-
|
|
62
|
+
// v0.6 (spec 026 AC-7): a null band is carried as null and rendered as what
|
|
63
|
+
// it is; a legacy receipt with no aggregate band keeps its 0 (the archive's
|
|
64
|
+
// rendering is unmoved, AC-16).
|
|
65
|
+
return { mean: a.mean_score, stddev: a.stddev === null ? null : (a.stddev || 0), cases: a.case_count };
|
|
63
66
|
}
|
|
64
67
|
|
|
65
68
|
function fmt(n) { return n == null ? 'n/a' : (n >= 0 ? '+' : '') + n.toFixed(3); }
|
|
@@ -81,11 +84,41 @@ function pct(n) { return n == null ? 'n/a' : n.toFixed(3); }
|
|
|
81
84
|
// not from one case's draws).
|
|
82
85
|
|
|
83
86
|
function bandStr(x) {
|
|
87
|
+
if (x && x.mean == null) return 'n/a (0 cases)';
|
|
88
|
+
if (x && x.stddev == null) return `${x.mean.toFixed(3)} ± n/a (${x.cases === 1 ? '1 case' : 'no band'})`;
|
|
84
89
|
const label = x && x.source;
|
|
85
90
|
return `${x.mean.toFixed(3)} ± ${x.stddev.toFixed(3)}${label ? ` (${label})` : ''}`;
|
|
86
91
|
}
|
|
87
92
|
function short(h) { return h ? String(h).slice(0, 12) : 'n/a'; }
|
|
88
93
|
|
|
94
|
+
// ── THE JUDGE IS PART OF THE INSTRUMENT (spec 026 AC-12, A4) ─────────────────
|
|
95
|
+
// Two receipts graded by different judges are two measurements with different
|
|
96
|
+
// instruments, and a verdict across them says nothing about the skill.
|
|
97
|
+
// lib/reuse.js's triage already says a changed judge means regrade; the
|
|
98
|
+
// differ did not ask. It asks here: run.judge.model_id (on canonical ids, so
|
|
99
|
+
// an alias and its dated form are one judge), run.judge.prompt_template_hash,
|
|
100
|
+
// run.judge.temperature, and every shared case's judge.rubric_hash. A field
|
|
101
|
+
// that differs names itself in the caveats and suppresses every per-case
|
|
102
|
+
// verdict. A pre-v0.6 receipt carries no template hash: the report says the
|
|
103
|
+
// template is unrecorded on that side and still compares the rest.
|
|
104
|
+
const canonicalJudgeId = (id) => String(id == null ? '' : id).replace(/-\d{8}$/, '');
|
|
105
|
+
function judgeDiffers(a, b, ids, aB, bB) {
|
|
106
|
+
const ja = (a.run && a.run.judge) || {}; const jb = (b.run && b.run.judge) || {};
|
|
107
|
+
const problems = [];
|
|
108
|
+
const idA = ja.model_id || (a.results.cases.find((c) => c.judge && c.judge.model_id) || { judge: {} }).judge.model_id;
|
|
109
|
+
const idB = jb.model_id || (b.results.cases.find((c) => c.judge && c.judge.model_id) || { judge: {} }).judge.model_id;
|
|
110
|
+
if (idA && idB && canonicalJudgeId(idA) !== canonicalJudgeId(idB)) problems.push(`run.judge.model_id differs (${idA} vs ${idB})`);
|
|
111
|
+
const unrecorded = [];
|
|
112
|
+
if (!ja.prompt_template_hash) unrecorded.push('A'); if (!jb.prompt_template_hash) unrecorded.push('B');
|
|
113
|
+
if (ja.prompt_template_hash && jb.prompt_template_hash && ja.prompt_template_hash !== jb.prompt_template_hash) problems.push(`run.judge.prompt_template_hash differs (${short(ja.prompt_template_hash)} vs ${short(jb.prompt_template_hash)}): the grading template is a different judge`);
|
|
114
|
+
if (ja.temperature !== undefined && jb.temperature !== undefined && ja.temperature !== jb.temperature) problems.push(`run.judge.temperature differs (${ja.temperature} vs ${jb.temperature})`);
|
|
115
|
+
const rubricA = Object.fromEntries(a.results.cases.filter((c) => c.mode === 'with_skill' && c.judge).map((c) => [c.id, c.judge.rubric_hash]));
|
|
116
|
+
const rubricB = Object.fromEntries(b.results.cases.filter((c) => c.mode === 'with_skill' && c.judge).map((c) => [c.id, c.judge.rubric_hash]));
|
|
117
|
+
const moved = ids.filter((id) => rubricA[id] && rubricB[id] && rubricA[id] !== rubricB[id]);
|
|
118
|
+
if (moved.length) problems.push(`judge.rubric_hash differs on ${moved.length} shared case(s) (${moved.slice(0, 3).map((x) => `\`${x}\``).join(', ')}${moved.length > 3 ? ', …' : ''})`);
|
|
119
|
+
return { problems, unrecorded };
|
|
120
|
+
}
|
|
121
|
+
|
|
89
122
|
// The headline is a SUMMARY of the per-case band-overlap verdicts — the credibility
|
|
90
123
|
// core lives per case (judge-sample bands), and the headline just aggregates it.
|
|
91
124
|
// It deliberately does NOT run a separate band test on the aggregate mean: the
|
|
@@ -146,9 +179,15 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
146
179
|
// improvement verdicts are never computed from evidence we did not run.
|
|
147
180
|
const levelOf = (r) => r.verification_level || 'TESTED';
|
|
148
181
|
const belowTested = [[labelA, a], [labelB, b]].filter(([, r]) => levelOf(r) !== 'TESTED');
|
|
182
|
+
// Spec 026 AC-4: an INCOMPLETE receipt on either side computes no per-case
|
|
183
|
+
// verdict. RECEIPT.md has said so since v0.3.1; the differ did not ask.
|
|
184
|
+
const incompleteSides = [[labelA, a], [labelB, b]].filter(([, r]) => r.run && r.run.status === 'incomplete');
|
|
185
|
+
// Spec 026 AC-12: a different judge on either side suppresses the verdict.
|
|
186
|
+
const judge = judgeDiffers(a, b, ids, aB, bB);
|
|
187
|
+
const judgeProblem = judge.problems.length ? judge.problems.join('; ') : null;
|
|
149
188
|
// A REFUSAL IS A RESULT, and it suppresses the verdict exactly as an
|
|
150
189
|
// untested input does: no delta is asserted, and the reason travels with it.
|
|
151
|
-
const measured = belowTested.length === 0 && !refusal;
|
|
190
|
+
const measured = belowTested.length === 0 && !refusal && incompleteSides.length === 0 && !judgeProblem;
|
|
152
191
|
|
|
153
192
|
const perCase = ids.map((id) => {
|
|
154
193
|
const before = aB[id] || null;
|
|
@@ -250,6 +289,15 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
250
289
|
if (belowTested.length) {
|
|
251
290
|
warnings.push(`verdicts NOT computed — ${belowTested.map(([l, r]) => `${l} is ${levelOf(r)}${r.run && r.run.source ? ` (${r.run.source})` : ''}`).join('; ')}. Drift verdicts require TESTED receipts on both sides; declared numbers are shown as context only (see /interop.html).`);
|
|
252
291
|
}
|
|
292
|
+
if (incompleteSides.length) {
|
|
293
|
+
warnings.push(`verdicts NOT computed — ${incompleteSides.map(([l, r]) => `${l} is incomplete (run.status incomplete; it excluded ${r.run.failed_case_count || 0} case(s) whose arm could not be measured)`).join('; ')}. A receipt with an unmeasured arm is not evidence for a per-case verdict; its aggregates are shown as context only.`);
|
|
294
|
+
}
|
|
295
|
+
if (judgeProblem) {
|
|
296
|
+
warnings.push(`verdicts NOT computed — the two receipts were graded by different judges: ${judgeProblem}. A verdict across two instruments says nothing about the skill; regrade one side with the other's judge (lib/reuse.js triage: regrade).`);
|
|
297
|
+
}
|
|
298
|
+
if (judge.unrecorded.length) {
|
|
299
|
+
warnings.push(`judge template unrecorded on ${judge.unrecorded.map((x) => (x === 'A' ? labelA : labelB)).join(' and ')} (a pre-v0.6 receipt carries no run.judge.prompt_template_hash); the judge model id and the rubric hashes are compared, the template is not.`);
|
|
300
|
+
}
|
|
253
301
|
// Cross-provider / cross-surface disclosure (Phase 6). A comparison across
|
|
254
302
|
// providers is a skill-DURABILITY comparison across substrates, not model drift
|
|
255
303
|
// over time; across surfaces, sampling control differs. Both are flagged so a
|
|
@@ -281,7 +329,11 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
281
329
|
// THE HEADLINE NAMES THE CAUSE THAT ACTUALLY APPLIES. `belowTested.length`
|
|
282
330
|
// was printed unconditionally, so a refused pair got "0 receipt(s) below
|
|
283
331
|
// TESTED" — a statement measurably false of the receipts it was given.
|
|
284
|
-
if (
|
|
332
|
+
if (incompleteSides.length) {
|
|
333
|
+
// An incomplete side is a fact about that receipt alone (spec 026 AC-4)
|
|
334
|
+
// and leads whatever else is wrong with the pair.
|
|
335
|
+
L.push(`**NOT MEASURED — ${incompleteSides.map(([l]) => l).join(' and ')} incomplete: a case's arm could not be measured, so no per-case verdict is computed.**`);
|
|
336
|
+
} else if (refusal) {
|
|
285
337
|
// The reason is a complete sentence and already ends by saying no verdict
|
|
286
338
|
// is asserted; prefixing that again produced "REFUSED — no verdict is
|
|
287
339
|
// asserted. the baseline arm did not reproduce…" — a duplicated clause and
|
|
@@ -289,6 +341,8 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
289
341
|
L.push(`**REFUSED — ${refusal.reason}**`);
|
|
290
342
|
} else if (belowTested.length) {
|
|
291
343
|
L.push(`**NOT MEASURED — ${belowTested.length} receipt(s) below TESTED. Drift verdicts require TESTED receipts on both sides; the declared numbers above are context, not band-verified evidence.**`);
|
|
344
|
+
} else if (judgeProblem) {
|
|
345
|
+
L.push(`**NOT MEASURED — different judges: ${judgeProblem}. No verdict crosses two instruments.**`);
|
|
292
346
|
} else {
|
|
293
347
|
L.push('**NOT MEASURED — no verdict is asserted.**');
|
|
294
348
|
}
|
|
@@ -327,4 +381,4 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
327
381
|
};
|
|
328
382
|
}
|
|
329
383
|
|
|
330
|
-
module.exports = { buildDriftReport, withSkillBands, revisionPairProblem };
|
|
384
|
+
module.exports = { buildDriftReport, withSkillBands, revisionPairProblem, judgeDiffers };
|
package/lib/importers.js
CHANGED
|
@@ -17,7 +17,7 @@
|
|
|
17
17
|
// compatibility contract) are documented in docs/interop.md.
|
|
18
18
|
|
|
19
19
|
const { RECEIPT_SCHEMA_VERSION, RUNNER_VERSION, SUITE_FORMAT } = require('../config');
|
|
20
|
-
const { sealReceipt } = require('./receipt');
|
|
20
|
+
const { sealReceipt, BAND_RULE } = require('./receipt');
|
|
21
21
|
const { mean, stddev, round, combineUncertainty, aggregateBands } = require('./stats');
|
|
22
22
|
const { outcomeFor } = require('./run');
|
|
23
23
|
const { inferProvider } = require('./provider');
|
|
@@ -64,11 +64,16 @@ function importedReceipt({ tool, skillName, skillVersion, suiteFormat, caseCount
|
|
|
64
64
|
date_utc: dateUtc,
|
|
65
65
|
registry: registryStatus(modelId),
|
|
66
66
|
transcripts: 'none',
|
|
67
|
-
|
|
67
|
+
// v0.6: the source tool's grader is the judge that ran; its template is
|
|
68
|
+
// not ours to hash (null, the one place the schema allows it).
|
|
69
|
+
judge: { ...judgeBlock, model_id: (cases.find((c) => c.judge && c.judge.model_id) || { judge: { model_id: 'unknown' } }).judge.model_id, prompt_template_hash: null },
|
|
70
|
+
// v0.6 (spec 026 AC-1): what answered. An external tool's harness did;
|
|
71
|
+
// nothing here attested a model, and nothing was spawned by us.
|
|
72
|
+
answered_by: { kind: 'external', attested: false, reported_model: null, reported_models: null, isolation: 'none' },
|
|
68
73
|
},
|
|
69
74
|
results: {
|
|
70
75
|
cases,
|
|
71
|
-
aggregates: { with_skill: aggregateMode(withSkill), baseline: aggregateMode(baseline) },
|
|
76
|
+
aggregates: { with_skill: aggregateMode(withSkill), baseline: aggregateMode(baseline), band_rule: BAND_RULE },
|
|
72
77
|
},
|
|
73
78
|
comparison,
|
|
74
79
|
verification_level: 'DECLARED',
|
package/lib/judge.js
CHANGED
|
@@ -58,10 +58,31 @@ function rubricHash(rubric) {
|
|
|
58
58
|
return sha256(JUDGE_SYSTEM + '\n---\n' + String(rubric || '').trim());
|
|
59
59
|
}
|
|
60
60
|
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
61
|
+
// v0.6 (spec 026 AC-11): a digest over the grading TEMPLATE with its three
|
|
62
|
+
// slots empty, recorded once per run as run.judge.prompt_template_hash. The
|
|
63
|
+
// rubric_hash above binds a grade to the case's rubric; this binds every grade
|
|
64
|
+
// of the run to the words around it. A different template is a different judge,
|
|
65
|
+
// and `diff` computes no verdict across one.
|
|
66
|
+
function promptTemplateHash() {
|
|
67
|
+
return sha256(JUDGE_SYSTEM + '\n---\n' + buildJudgePrompt({ task: '', response: '', rubric: '' }));
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
// The score a judge reply carries, or null when it carries none (spec 026
|
|
71
|
+
// AC-3, F2). A non-numeric or non-finite value is not a score; a number outside
|
|
72
|
+
// [0, 1] is on a scale the rubric did not ask for and is not clamped into one
|
|
73
|
+
// (85 is not 1.0). This replaced clamp01, whose silent 0 for a non-finite value
|
|
74
|
+
// scored an absent output as the worst possible one.
|
|
75
|
+
// Stop reasons that mean the surface cut the reply at its output cap: the
|
|
76
|
+
// Messages API's and the claude CLI's `max_tokens`, Chat Completions'
|
|
77
|
+
// `length`. A cut reply is a partial answer (spec 026 AC-8, F4).
|
|
78
|
+
const TRUNCATION_STOP_REASONS = new Set(['max_tokens', 'length']);
|
|
79
|
+
function isTruncated(stopReason) { return TRUNCATION_STOP_REASONS.has(String(stopReason || '')); }
|
|
80
|
+
|
|
81
|
+
function scoreOf(parsed) {
|
|
82
|
+
const x = Number(parsed && parsed.score);
|
|
83
|
+
if (!Number.isFinite(x)) return null;
|
|
84
|
+
if (x < 0 || x > 1) return null;
|
|
85
|
+
return x;
|
|
65
86
|
}
|
|
66
87
|
|
|
67
88
|
// Judge settings for the JUDGE model's surface. Determinism where the surface
|
|
@@ -78,25 +99,50 @@ function judgeSettings(samples, judgeModel) {
|
|
|
78
99
|
return { samples, temperature: null, sampling: 'surface-controlled', surface };
|
|
79
100
|
}
|
|
80
101
|
|
|
81
|
-
// Grade one response once. Returns { score, reason, raw }
|
|
102
|
+
// Grade one response once. Returns { score, reason, raw, ... } for a measured
|
|
103
|
+
// sample, or { unmeasured: true, reason, raw, ... } when the reply carries no
|
|
104
|
+
// score: empty, unparseable, no numeric score, or a score outside [0, 1]. A
|
|
105
|
+
// zero asserts a measurement ("the response satisfied none of the rubric");
|
|
106
|
+
// each of these is the absence of one, and no sample enters any statistic
|
|
107
|
+
// (spec 026 AC-3). The 2026-07 rule that an unparseable judge must never
|
|
108
|
+
// silently pass is kept by the stronger rule: it never silently scores at all.
|
|
82
109
|
// `raw` is the judge's verbatim output text (hashed into the receipt for
|
|
83
110
|
// transcript auditability, and optionally retained under --keep-transcripts).
|
|
84
|
-
async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature }) {
|
|
111
|
+
async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature, trusted = false }) {
|
|
85
112
|
const prompt = buildJudgePrompt({ task, response, rubric });
|
|
86
|
-
const
|
|
113
|
+
const out = await complete({ system: JUDGE_SYSTEM, prompt, model, maxTokens: 400, timeoutMs, temperature, trusted });
|
|
114
|
+
const { text, attempts, usage } = out;
|
|
115
|
+
const base = { raw: String(text || ''), attempts: attempts || 1, usage, reply: out };
|
|
116
|
+
// A judge reply cut at its output cap carries no complete grade.
|
|
117
|
+
if (isTruncated(out.stopReason)) {
|
|
118
|
+
return { ...base, unmeasured: true, reason: `judge output truncated at the output cap (stop_reason ${out.stopReason})` };
|
|
119
|
+
}
|
|
120
|
+
if (!String(text || '').trim()) {
|
|
121
|
+
return { ...base, unmeasured: true, reason: 'judge output empty (the surface returned no text)' };
|
|
122
|
+
}
|
|
87
123
|
let parsed;
|
|
88
124
|
try {
|
|
89
125
|
parsed = extractJsonObject(text);
|
|
90
126
|
} catch (_e) {
|
|
91
|
-
|
|
92
|
-
// must never silently "pass"), tagged so the caller can see it happened.
|
|
93
|
-
return { score: 0, reason: 'judge output unparseable', unparsed: true, raw: String(text || ''), attempts: attempts || 1, usage };
|
|
127
|
+
return { ...base, unmeasured: true, reason: 'judge output unparseable' };
|
|
94
128
|
}
|
|
95
|
-
|
|
129
|
+
const score = scoreOf(parsed);
|
|
130
|
+
if (score === null) {
|
|
131
|
+
const x = parsed && parsed.score;
|
|
132
|
+
const reason = x === undefined || x === null ? 'judge output carries no numeric score'
|
|
133
|
+
: !Number.isFinite(Number(x)) ? `judge output carries no numeric score (score ${JSON.stringify(x)})`
|
|
134
|
+
: `judge score ${x} outside [0, 1]`;
|
|
135
|
+
return { ...base, unmeasured: true, reason };
|
|
136
|
+
}
|
|
137
|
+
return { ...base, score, reason: String(parsed.reason || '').slice(0, 300) };
|
|
96
138
|
}
|
|
97
139
|
|
|
98
140
|
// Grade a response N times and return the sampled distribution:
|
|
99
141
|
// { samples:[scores], mean, stddev, reason, judge_settings, model_id, rubric_hash }
|
|
142
|
+
// or, when any sample is unmeasured, { unmeasured: true, reason, ... } with NO
|
|
143
|
+
// samples: a partial sample set must never become a band, and a draw one of
|
|
144
|
+
// whose judge samples carried no score is unmeasured as a whole (spec 026
|
|
145
|
+
// AC-3). The remaining samples are not taken.
|
|
100
146
|
// `mean` ± `stddev` is the per-case confidence band (raw spread of the N scores)
|
|
101
147
|
// used by the borderline-outcome rule and per-case drift band-overlap logic.
|
|
102
148
|
// NO DEFAULT TIMEOUT HERE (spec 017 AC-2). This defaulted to 120000, which
|
|
@@ -104,17 +150,20 @@ async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature
|
|
|
104
150
|
// JUDGE calls timed out on the api policy while running on a CLI surface, which
|
|
105
151
|
// is the second shadowing site and the one #007's prep session had not found.
|
|
106
152
|
// Passing `undefined` through lets lib/provider.js resolve the declared policy.
|
|
107
|
-
|
|
153
|
+
// `trusted` (spec 022) is plumbed to complete() untouched: the judge takes the
|
|
154
|
+
// same spawn path as the generation it grades, so --trusted-skill is whole-run.
|
|
155
|
+
async function gradeSamples({ task, response, rubric, model, samples = 5, timeoutMs, trusted = false }) {
|
|
108
156
|
const settings = judgeSettings(samples, model);
|
|
109
157
|
const scores = [];
|
|
110
158
|
const reasons = [];
|
|
111
159
|
const rawTexts = [];
|
|
112
160
|
const usages = [];
|
|
161
|
+
const replies = []; // v0.6: what answered each judge call, for the runner's attestation
|
|
113
162
|
let attemptsTotal = 0;
|
|
114
163
|
for (let i = 0; i < samples; i++) {
|
|
115
164
|
let r;
|
|
116
165
|
try {
|
|
117
|
-
r = await gradeOnce({ task, response, rubric, model, timeoutMs, temperature: settings.temperature === null ? undefined : settings.temperature });
|
|
166
|
+
r = await gradeOnce({ task, response, rubric, model, timeoutMs, temperature: settings.temperature === null ? undefined : settings.temperature, trusted });
|
|
118
167
|
} catch (e) {
|
|
119
168
|
// A judge sample that persistently failed (e.g. timed out after retries):
|
|
120
169
|
// tag the error so the runner can charge for the spend and mark the whole
|
|
@@ -124,9 +173,18 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
|
|
|
124
173
|
}
|
|
125
174
|
attemptsTotal += r.attempts || 1;
|
|
126
175
|
usages.push(r.usage || null);
|
|
176
|
+
replies.push(r.reply || null);
|
|
177
|
+
rawTexts.push(r.raw || '');
|
|
178
|
+
if (r.unmeasured) {
|
|
179
|
+
return {
|
|
180
|
+
unmeasured: true, reason: r.reason, samples: [], mean: null, stddev: null,
|
|
181
|
+
sample_texts: rawTexts, sample_hashes: rawTexts.map((t) => sha256(t)),
|
|
182
|
+
judge_settings: settings, model_id: model, rubric_hash: rubricHash(rubric),
|
|
183
|
+
attempts: attemptsTotal, usage: sumUsage(usages), replies,
|
|
184
|
+
};
|
|
185
|
+
}
|
|
127
186
|
scores.push(r.score);
|
|
128
187
|
reasons.push(r.reason);
|
|
129
|
-
rawTexts.push(r.raw || '');
|
|
130
188
|
}
|
|
131
189
|
return {
|
|
132
190
|
samples: scores,
|
|
@@ -147,7 +205,8 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
|
|
|
147
205
|
// EXCLUDED from every skill-value figure (lib/value.js): it is a cost we
|
|
148
206
|
// impose to measure, not a cost of running the skill.
|
|
149
207
|
usage: sumUsage(usages),
|
|
208
|
+
replies,
|
|
150
209
|
};
|
|
151
210
|
}
|
|
152
211
|
|
|
153
|
-
module.exports = { gradeOnce, gradeSamples, judgeSettings, rubricHash, buildJudgePrompt, JUDGE_SYSTEM };
|
|
212
|
+
module.exports = { gradeOnce, gradeSamples, judgeSettings, rubricHash, buildJudgePrompt, promptTemplateHash, scoreOf, isTruncated, TRUNCATION_STOP_REASONS, JUDGE_SYSTEM };
|
package/lib/models.js
CHANGED
|
@@ -66,6 +66,26 @@ function resolveRegistry(modelId) {
|
|
|
66
66
|
return { id: canonical, entry, registered: !!entry };
|
|
67
67
|
}
|
|
68
68
|
|
|
69
|
+
// Spec 026 AC-10 (F6): an id that is not a model does not run. After alias
|
|
70
|
+
// resolution the id must have the contract's model-id shape (letters, digits,
|
|
71
|
+
// . _ -) and resolve in the registry; otherwise the run is refused before any
|
|
72
|
+
// call, naming the id and the absolute path of the registry consulted
|
|
73
|
+
// (DRIFTPROOF_REGISTRY when set, else the packaged config/models.json).
|
|
74
|
+
// `registry: "unregistered"` stays a legal receipt value for IMPORTED receipts,
|
|
75
|
+
// whose model field is another tool's word; a run of ours never writes it.
|
|
76
|
+
const MODEL_ID_SHAPE = /^[A-Za-z0-9._-]+$/;
|
|
77
|
+
function assertRegistered(modelId, role = 'model') {
|
|
78
|
+
const given = String(modelId == null ? '' : modelId);
|
|
79
|
+
const { id, registered } = MODEL_ID_SHAPE.test(given) ? resolveRegistry(given) : { id: given, registered: false };
|
|
80
|
+
if (!MODEL_ID_SHAPE.test(given) || !registered) {
|
|
81
|
+
const why = !MODEL_ID_SHAPE.test(given) ? 'is not a model id (letters, digits, . _ - only)' : 'is not in the model registry';
|
|
82
|
+
const e = new Error(`${role} "${given}"${id !== given ? ` (resolved "${id}")` : ''} ${why}: ${REGISTRY_PATH}. An unregistered model does not run; add a registry row (see config/models.json) or point DRIFTPROOF_REGISTRY at a registry that carries it.`);
|
|
83
|
+
e.code = 'UNREGISTERED_MODEL'; e.model = given; e.registry = REGISTRY_PATH;
|
|
84
|
+
throw e;
|
|
85
|
+
}
|
|
86
|
+
return id;
|
|
87
|
+
}
|
|
88
|
+
|
|
69
89
|
// The value stamped into receipt.run.registry.
|
|
70
90
|
function registryStatus(modelId) {
|
|
71
91
|
return resolveRegistry(modelId).registered ? 'registered' : 'unregistered';
|
|
@@ -140,23 +160,43 @@ function familyPredecessor(modelId) {
|
|
|
140
160
|
return chooseFrom[0].id;
|
|
141
161
|
}
|
|
142
162
|
|
|
143
|
-
//
|
|
144
|
-
//
|
|
145
|
-
//
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
163
|
+
// THE PRICE OF A MODEL NOBODY HAS PRICED YET.
|
|
164
|
+
//
|
|
165
|
+
// READ THIS BEFORE TRUSTING A NUMBER IT RETURNS. Nothing here reads a price
|
|
166
|
+
// SOURCE. The family and the tier are inferred from the id, and the rate is
|
|
167
|
+
// looked up in the table below, so the result is a GUESS whose only input is the
|
|
168
|
+
// string. It is extracted into its own function, and named, because the guess
|
|
169
|
+
// was being made silently inside a function whose job looked like registration.
|
|
170
|
+
//
|
|
171
|
+
// KNOWN DEFECT, spec 021 F-W2, deliberately NOT fixed here. The Anthropic branch
|
|
172
|
+
// falls through to the OPUS rate of 5/25 for every family that is not `haiku` or
|
|
173
|
+
// `sonnet`. Fable's published rate is 10/50, exactly twice opus, which is the
|
|
174
|
+
// whole of the "the watcher read exactly half the snapshot on both fields"
|
|
175
|
+
// observation of 2026-09-03: it is not a halving and not a misread, it is a
|
|
176
|
+
// wrong-tier fallback. `DEFAULT_PRICE` above IS the conservative upper bound the
|
|
177
|
+
// comment on this function used to claim, and it is not consulted. Recorded on
|
|
178
|
+
// the spec 021 carry list, tagged 026, with the reason it is left alone.
|
|
179
|
+
//
|
|
180
|
+
// scripts/release-watch.js no longer registers anything, and records what this
|
|
181
|
+
// returns as `price_verified: false` beside a source naming this function.
|
|
182
|
+
function inferredPrice(id) {
|
|
149
183
|
const family = familyOf(id);
|
|
150
184
|
const provider = inferProviderFromId(id);
|
|
151
|
-
// Best-effort tier from family; unknown families default to frontier pricing.
|
|
152
|
-
// A newly-discovered id gets a CONSERVATIVE (upper-bound) price so the budget
|
|
153
|
-
// guard never under-projects an unknown model — exact rates are set by hand
|
|
154
|
-
// when the model is reviewed for a published run.
|
|
155
185
|
const tier = /haiku|luna|mini|nano/.test(family) || /luna|mini|nano/.test(String(id)) ? 'cheap'
|
|
156
186
|
: /sonnet|terra/.test(family) ? 'standard' : 'frontier';
|
|
157
187
|
const price = provider === 'openai'
|
|
158
188
|
? (tier === 'cheap' ? { input: 1.0, output: 6.0 } : tier === 'standard' ? { input: 2.5, output: 15.0 } : { input: 5.0, output: 30.0 })
|
|
159
189
|
: (family === 'haiku' ? { input: 1.0, output: 5.0 } : family === 'sonnet' ? { input: 3.0, output: 15.0 } : { input: 5.0, output: 25.0 });
|
|
190
|
+
return { family, provider, tier, price };
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
// Append a newly-discovered model id to the registry file (release trigger).
|
|
194
|
+
// `firstSeen` is the first-seen date recorded as `released`. Idempotent: a no-op
|
|
195
|
+
// if the id already exists. Returns the added entry (or null if already present).
|
|
196
|
+
function addDiscoveredModel(id, { firstSeen = null, registryPath = REGISTRY_PATH } = {}) {
|
|
197
|
+
const raw = JSON.parse(fs.readFileSync(registryPath, 'utf8'));
|
|
198
|
+
if (raw.models.some((m) => m.id === id)) return null;
|
|
199
|
+
const { family, provider, tier, price } = inferredPrice(id);
|
|
160
200
|
const entry = {
|
|
161
201
|
id, family, provider, released: firstSeen,
|
|
162
202
|
input_price: price.input, output_price: price.output, tier,
|
|
@@ -171,5 +211,6 @@ function addDiscoveredModel(id, { firstSeen = null, registryPath = REGISTRY_PATH
|
|
|
171
211
|
module.exports = {
|
|
172
212
|
loadRegistry, resolveRegistry, registryStatus, priceForModel, isJudgeEligible,
|
|
173
213
|
assertJudgeEligible, providerForModel, providerConfig, inferProviderFromId,
|
|
174
|
-
familyOf, familyPredecessor, addDiscoveredModel, DEFAULT_PRICE, REGISTRY_PATH,
|
|
214
|
+
familyOf, familyPredecessor, addDiscoveredModel, inferredPrice, DEFAULT_PRICE, REGISTRY_PATH,
|
|
215
|
+
assertRegistered, MODEL_ID_SHAPE,
|
|
175
216
|
};
|