driftproof 0.3.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +63 -19
- package/bin/driftproof +66 -4
- package/config/models.json +14 -4
- package/config.js +55 -5
- package/lib/checks.js +50 -0
- package/lib/diff.js +28 -7
- package/lib/export.js +52 -0
- package/lib/importers.js +207 -0
- package/lib/judge.js +35 -13
- package/lib/models.js +58 -8
- package/lib/provider.js +318 -45
- package/lib/receipt.js +29 -4
- package/lib/run.js +108 -27
- package/lib/skill.js +12 -2
- package/lib/skillCost.js +31 -0
- package/lib/stub.js +24 -3
- package/lib/usage.js +168 -0
- package/lib/value.js +502 -0
- package/lib/verdict.js +10 -1
- package/package.json +1 -1
- package/spec/RECEIPT.md +116 -8
- package/spec/receipt.schema.json +809 -59
- package/spec/receipt.v0.3.1.schema.json +642 -0
- package/spec/receipt.v0.3.schema.json +211 -0
package/lib/importers.js
ADDED
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// Receipt interop — importers (Phase 7). Convert results from neighboring eval
|
|
5
|
+
// tools into valid Driftproof receipts with HONEST epistemics:
|
|
6
|
+
//
|
|
7
|
+
// - verification_level DECLARED (never TESTED — we did not run, hash, or
|
|
8
|
+
// judge the generations),
|
|
9
|
+
// - run.surface "external" + run.source "imported/<tool>",
|
|
10
|
+
// - generation/judge-sample hashes OMITTED, content/suite/rubric hashes null —
|
|
11
|
+
// hashes are never fabricated,
|
|
12
|
+
// - no baseline mode in the source tool → empty baseline aggregate + null
|
|
13
|
+
// comparison, never a fabricated 0-baseline.
|
|
14
|
+
//
|
|
15
|
+
// The per-tool field mappings (and the assumed-shape disclosure — neither tool
|
|
16
|
+
// publishes a frozen results schema, so the checked-in fixtures ARE the
|
|
17
|
+
// compatibility contract) are documented in docs/interop.md.
|
|
18
|
+
|
|
19
|
+
const { RECEIPT_SCHEMA_VERSION, RUNNER_VERSION, SUITE_FORMAT } = require('../config');
|
|
20
|
+
const { sealReceipt } = require('./receipt');
|
|
21
|
+
const { mean, stddev, round, combineUncertainty, aggregateBands } = require('./stats');
|
|
22
|
+
const { outcomeFor } = require('./run');
|
|
23
|
+
const { inferProvider } = require('./provider');
|
|
24
|
+
const { registryStatus } = require('./models');
|
|
25
|
+
|
|
26
|
+
const IMPORT_TOOLS = ['agent-skills-eval', 'skillgrade'];
|
|
27
|
+
|
|
28
|
+
// Aggregate a list of imported cases the same way lib/receipt.aggregate does
|
|
29
|
+
// (suite dispersion band). An empty mode aggregates to the conventional empty
|
|
30
|
+
// aggregate (case_count 0) — its absence is signalled there, and in the null
|
|
31
|
+
// comparison, not by fabricated numbers.
|
|
32
|
+
function aggregateMode(cases) {
|
|
33
|
+
const band = aggregateBands(cases.map((c) => ({ mean: c.mean, stddev: c.stddev || 0, n: c.samples.length })));
|
|
34
|
+
return {
|
|
35
|
+
case_count: cases.length,
|
|
36
|
+
pass_count: cases.filter((c) => c.outcome === 'pass').length,
|
|
37
|
+
borderline_count: cases.filter((c) => c.outcome === 'borderline').length,
|
|
38
|
+
mean_score: band.mean,
|
|
39
|
+
stddev: band.stddev,
|
|
40
|
+
};
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
// Shared receipt shell for both importers. `judgeBlock` describes the SOURCE
|
|
44
|
+
// tool's grading (samples = what it actually did), never our sampled judge.
|
|
45
|
+
function importedReceipt({ tool, skillName, skillVersion, suiteFormat, caseCount, modelId, dateUtc, judgeBlock, cases, comparison }) {
|
|
46
|
+
const withSkill = cases.filter((c) => c.mode === 'with_skill');
|
|
47
|
+
const baseline = cases.filter((c) => c.mode === 'baseline');
|
|
48
|
+
const receipt = {
|
|
49
|
+
schema_version: RECEIPT_SCHEMA_VERSION,
|
|
50
|
+
skill: {
|
|
51
|
+
name: skillName,
|
|
52
|
+
version: skillVersion || 'unknown',
|
|
53
|
+
// Never seen the skill bytes — null, never fabricated.
|
|
54
|
+
content_hash: null,
|
|
55
|
+
},
|
|
56
|
+
suite: { format: suiteFormat, suite_hash: null, case_count: caseCount },
|
|
57
|
+
run: {
|
|
58
|
+
model_id: modelId,
|
|
59
|
+
model_release_date: null,
|
|
60
|
+
provider: inferProvider(modelId),
|
|
61
|
+
surface: 'external',
|
|
62
|
+
source: `imported/${tool}`,
|
|
63
|
+
runner_version: RUNNER_VERSION,
|
|
64
|
+
date_utc: dateUtc,
|
|
65
|
+
registry: registryStatus(modelId),
|
|
66
|
+
transcripts: 'none',
|
|
67
|
+
judge: judgeBlock,
|
|
68
|
+
},
|
|
69
|
+
results: {
|
|
70
|
+
cases,
|
|
71
|
+
aggregates: { with_skill: aggregateMode(withSkill), baseline: aggregateMode(baseline) },
|
|
72
|
+
},
|
|
73
|
+
comparison,
|
|
74
|
+
verification_level: 'DECLARED',
|
|
75
|
+
receipt_hash: '',
|
|
76
|
+
};
|
|
77
|
+
return sealReceipt(receipt);
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
// ── agent-skills-eval (darkrishabh/agent-skills-eval) ────────────────────────
|
|
81
|
+
// Same suite lineage as Driftproof (agentskills.io evals.json). Runs each eval
|
|
82
|
+
// with_skill and without_skill; an LLM judge grades BINARY per assertion. The
|
|
83
|
+
// converter reads the rolled-up benchmark artifact:
|
|
84
|
+
// { skill_name, version?, target, judge, timestamp?, evals: [
|
|
85
|
+
// { id, with_skill: { pass, assertions?: [{assertion, pass, reasoning?}] },
|
|
86
|
+
// without_skill: { ... } } ] }
|
|
87
|
+
// Per (eval, mode): mean = fraction of assertions passed (else 1/0 from the
|
|
88
|
+
// eval-level pass); samples = [mean] — the ONE grade their judge produced,
|
|
89
|
+
// never resampled; stddev 0; outcome from their pass verdict.
|
|
90
|
+
function modeScore(grading) {
|
|
91
|
+
const asserts = Array.isArray(grading.assertions) ? grading.assertions : [];
|
|
92
|
+
if (asserts.length) return round(asserts.filter((a) => a.pass === true).length / asserts.length);
|
|
93
|
+
return grading.pass === true ? 1 : 0;
|
|
94
|
+
}
|
|
95
|
+
function importAgentSkillsEval(data, { importedAt } = {}) {
|
|
96
|
+
if (!data || !Array.isArray(data.evals)) throw new Error('agent-skills-eval import: expected { skill_name, target, judge, evals: [...] } (see docs/interop.md)');
|
|
97
|
+
const judgeModel = data.judge || 'unknown';
|
|
98
|
+
const cases = [];
|
|
99
|
+
for (const ev of data.evals) {
|
|
100
|
+
for (const [theirMode, ourMode] of [['with_skill', 'with_skill'], ['without_skill', 'baseline']]) {
|
|
101
|
+
const grading = ev[theirMode];
|
|
102
|
+
if (!grading) continue;
|
|
103
|
+
const m = modeScore(grading);
|
|
104
|
+
const failing = (grading.assertions || []).find((a) => a.pass === false);
|
|
105
|
+
const c = {
|
|
106
|
+
id: String(ev.id),
|
|
107
|
+
mode: ourMode,
|
|
108
|
+
outcome: grading.pass === true ? 'pass' : 'fail',
|
|
109
|
+
score: m, mean: m, stddev: 0,
|
|
110
|
+
samples: [m],
|
|
111
|
+
threshold: null,
|
|
112
|
+
judge: { model_id: judgeModel, rubric_hash: null },
|
|
113
|
+
};
|
|
114
|
+
if (failing && failing.reasoning) c.reason = String(failing.reasoning).slice(0, 300);
|
|
115
|
+
cases.push(c);
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
const withMeans = cases.filter((c) => c.mode === 'with_skill').map((c) => c.mean);
|
|
119
|
+
const baseMeans = cases.filter((c) => c.mode === 'baseline').map((c) => c.mean);
|
|
120
|
+
const hasBaseline = baseMeans.length > 0;
|
|
121
|
+
const wAgg = aggregateBands(cases.filter((c) => c.mode === 'with_skill').map((c) => ({ mean: c.mean, stddev: 0, n: 1 })));
|
|
122
|
+
const bAgg = aggregateBands(cases.filter((c) => c.mode === 'baseline').map((c) => ({ mean: c.mean, stddev: 0, n: 1 })));
|
|
123
|
+
const comparison = {
|
|
124
|
+
with_skill_score: round(mean(withMeans)),
|
|
125
|
+
baseline_score: hasBaseline ? round(mean(baseMeans)) : null,
|
|
126
|
+
delta: hasBaseline ? round(mean(withMeans) - mean(baseMeans)) : null,
|
|
127
|
+
delta_uncertainty: hasBaseline ? combineUncertainty(wAgg.stddev, bAgg.stddev) : null,
|
|
128
|
+
};
|
|
129
|
+
return importedReceipt({
|
|
130
|
+
tool: 'agent-skills-eval',
|
|
131
|
+
skillName: data.skill_name || 'unknown',
|
|
132
|
+
skillVersion: data.version,
|
|
133
|
+
suiteFormat: SUITE_FORMAT, // agentskills.io/evals — same suite lineage
|
|
134
|
+
caseCount: data.evals.length,
|
|
135
|
+
modelId: data.target || 'unknown',
|
|
136
|
+
dateUtc: data.timestamp || importedAt || new Date().toISOString(),
|
|
137
|
+
judgeBlock: { samples: 1, temperature: null, sampling: 'external', surface: 'external' },
|
|
138
|
+
cases,
|
|
139
|
+
comparison,
|
|
140
|
+
});
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
// ── skillgrade (mgechev/skillgrade) ──────────────────────────────────────────
|
|
144
|
+
// "Unit tests for your agent skills": N trials per task, each trial's reward =
|
|
145
|
+
// weighted grader scores (0..1), compared against a threshold. NO baseline mode
|
|
146
|
+
// — so the receipt carries an empty baseline aggregate and a null comparison.
|
|
147
|
+
// Trials are real repeated runs, so per-trial rewards map onto samples[]
|
|
148
|
+
// honestly (a genuine cross-trial band). Expected shape:
|
|
149
|
+
// { skill, agent, grader_model?, threshold?, timestamp?, tasks: [
|
|
150
|
+
// { name, threshold?, trials: [ { reward }, ... ] } ] }
|
|
151
|
+
function importSkillgrade(data, { importedAt } = {}) {
|
|
152
|
+
if (!data || !Array.isArray(data.tasks)) throw new Error('skillgrade import: expected { skill, agent, tasks: [...] } (see docs/interop.md)');
|
|
153
|
+
const graderModel = data.grader_model || 'unknown';
|
|
154
|
+
const defaultThreshold = typeof data.threshold === 'number' ? data.threshold : 0.8;
|
|
155
|
+
const cases = [];
|
|
156
|
+
let maxTrials = 1;
|
|
157
|
+
for (const task of data.tasks) {
|
|
158
|
+
const rewards = (task.trials || []).map((t) => (typeof t === 'number' ? t : t.reward)).filter((r) => typeof r === 'number');
|
|
159
|
+
if (!rewards.length) continue;
|
|
160
|
+
maxTrials = Math.max(maxTrials, rewards.length);
|
|
161
|
+
const m = round(mean(rewards));
|
|
162
|
+
const sd = round(stddev(rewards));
|
|
163
|
+
const threshold = typeof task.threshold === 'number' ? task.threshold : defaultThreshold;
|
|
164
|
+
cases.push({
|
|
165
|
+
id: String(task.name),
|
|
166
|
+
mode: 'with_skill',
|
|
167
|
+
// Same outcome rule as a Driftproof run (borderline when the threshold
|
|
168
|
+
// sits inside mean ± stddev) — a deterministic READING of their numbers.
|
|
169
|
+
outcome: outcomeFor(m, sd, threshold),
|
|
170
|
+
score: m, mean: m, stddev: sd,
|
|
171
|
+
samples: rewards.map((r) => round(r)),
|
|
172
|
+
threshold,
|
|
173
|
+
judge: { model_id: graderModel, rubric_hash: null },
|
|
174
|
+
});
|
|
175
|
+
}
|
|
176
|
+
return importedReceipt({
|
|
177
|
+
tool: 'skillgrade',
|
|
178
|
+
skillName: data.skill || 'unknown',
|
|
179
|
+
skillVersion: data.version,
|
|
180
|
+
suiteFormat: 'skillgrade/eval.yaml',
|
|
181
|
+
caseCount: data.tasks.length,
|
|
182
|
+
// skillgrade names an AGENT CLI (claude/gemini/codex), not a model id —
|
|
183
|
+
// imported verbatim (or the results' model field when present).
|
|
184
|
+
modelId: data.model || data.agent || 'unknown',
|
|
185
|
+
dateUtc: data.timestamp || importedAt || new Date().toISOString(),
|
|
186
|
+
judgeBlock: { samples: maxTrials, temperature: null, sampling: 'external', surface: 'external' },
|
|
187
|
+
cases,
|
|
188
|
+
// No baseline mode exists in skillgrade — nulls, never a fabricated 0.
|
|
189
|
+
comparison: {
|
|
190
|
+
with_skill_score: round(mean(cases.map((c) => c.mean))),
|
|
191
|
+
baseline_score: null,
|
|
192
|
+
delta: null,
|
|
193
|
+
delta_uncertainty: null,
|
|
194
|
+
},
|
|
195
|
+
});
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
// Dispatch. `from` must name a supported tool.
|
|
199
|
+
function importResults(data, { from, importedAt } = {}) {
|
|
200
|
+
switch (from) {
|
|
201
|
+
case 'agent-skills-eval': return importAgentSkillsEval(data, { importedAt });
|
|
202
|
+
case 'skillgrade': return importSkillgrade(data, { importedAt });
|
|
203
|
+
default: throw new Error(`unknown import source "${from}" — supported: ${IMPORT_TOOLS.join(', ')}`);
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
module.exports = { importResults, importAgentSkillsEval, importSkillgrade, IMPORT_TOOLS };
|
package/lib/judge.js
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
// SPDX-License-Identifier: Apache-2.0
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
|
-
const { complete,
|
|
4
|
+
const { complete, surfaceForModel } = require('./provider');
|
|
5
5
|
const { extractJsonObject } = require('./json');
|
|
6
6
|
const { sha256 } = require('./canonical');
|
|
7
7
|
const { mean, stddev } = require('./stats');
|
|
8
|
+
const { sumUsage } = require('./usage');
|
|
8
9
|
|
|
9
10
|
// Rubric-based LLM judge.
|
|
10
11
|
//
|
|
@@ -63,13 +64,15 @@ function clamp01(n) {
|
|
|
63
64
|
return Math.max(0, Math.min(1, x));
|
|
64
65
|
}
|
|
65
66
|
|
|
66
|
-
// Judge settings for the
|
|
67
|
-
// on
|
|
68
|
-
//
|
|
69
|
-
// temperature is null and `sampling`
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
67
|
+
// Judge settings for the JUDGE model's surface. Determinism where the surface
|
|
68
|
+
// allows: on an api surface (Anthropic `api` or `openai-api`) we pin temperature 0
|
|
69
|
+
// for judge calls (and record it); on a cli/subscription surface sampling params
|
|
70
|
+
// are surface-controlled and cannot be set, so temperature is null and `sampling`
|
|
71
|
+
// says so. `judgeModel` defaults to the fixed Haiku judge, whose surface is the
|
|
72
|
+
// Anthropic axis. Recorded into every receipt.
|
|
73
|
+
function judgeSettings(samples, judgeModel) {
|
|
74
|
+
const surface = surfaceForModel(judgeModel || 'claude-haiku-4-5');
|
|
75
|
+
if (surface === 'api' || surface === 'openai-api') {
|
|
73
76
|
return { samples, temperature: 0, sampling: 'api-temperature-0', surface };
|
|
74
77
|
}
|
|
75
78
|
return { samples, temperature: null, sampling: 'surface-controlled', surface };
|
|
@@ -80,16 +83,16 @@ function judgeSettings(samples) {
|
|
|
80
83
|
// transcript auditability, and optionally retained under --keep-transcripts).
|
|
81
84
|
async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature }) {
|
|
82
85
|
const prompt = buildJudgePrompt({ task, response, rubric });
|
|
83
|
-
const { text } = await complete({ system: JUDGE_SYSTEM, prompt, model, maxTokens: 400, timeoutMs, temperature });
|
|
86
|
+
const { text, attempts, usage } = await complete({ system: JUDGE_SYSTEM, prompt, model, maxTokens: 400, timeoutMs, temperature });
|
|
84
87
|
let parsed;
|
|
85
88
|
try {
|
|
86
89
|
parsed = extractJsonObject(text);
|
|
87
90
|
} catch (_e) {
|
|
88
91
|
// Unsalvageable judge output → conservative 0 (a judge that can't be parsed
|
|
89
92
|
// must never silently "pass"), tagged so the caller can see it happened.
|
|
90
|
-
return { score: 0, reason: 'judge output unparseable', unparsed: true, raw: String(text || '') };
|
|
93
|
+
return { score: 0, reason: 'judge output unparseable', unparsed: true, raw: String(text || ''), attempts: attempts || 1, usage };
|
|
91
94
|
}
|
|
92
|
-
return { score: clamp01(parsed.score), reason: String(parsed.reason || '').slice(0, 300), raw: String(text || '') };
|
|
95
|
+
return { score: clamp01(parsed.score), reason: String(parsed.reason || '').slice(0, 300), raw: String(text || ''), attempts: attempts || 1, usage };
|
|
93
96
|
}
|
|
94
97
|
|
|
95
98
|
// Grade a response N times and return the sampled distribution:
|
|
@@ -97,12 +100,25 @@ async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature
|
|
|
97
100
|
// `mean` ± `stddev` is the per-case confidence band (raw spread of the N scores)
|
|
98
101
|
// used by the borderline-outcome rule and per-case drift band-overlap logic.
|
|
99
102
|
async function gradeSamples({ task, response, rubric, model, samples = 5, timeoutMs = 120000 }) {
|
|
100
|
-
const settings = judgeSettings(samples);
|
|
103
|
+
const settings = judgeSettings(samples, model);
|
|
101
104
|
const scores = [];
|
|
102
105
|
const reasons = [];
|
|
103
106
|
const rawTexts = [];
|
|
107
|
+
const usages = [];
|
|
108
|
+
let attemptsTotal = 0;
|
|
104
109
|
for (let i = 0; i < samples; i++) {
|
|
105
|
-
|
|
110
|
+
let r;
|
|
111
|
+
try {
|
|
112
|
+
r = await gradeOnce({ task, response, rubric, model, timeoutMs, temperature: settings.temperature === null ? undefined : settings.temperature });
|
|
113
|
+
} catch (e) {
|
|
114
|
+
// A judge sample that persistently failed (e.g. timed out after retries):
|
|
115
|
+
// tag the error so the runner can charge for the spend and mark the whole
|
|
116
|
+
// case failed_timeout (a partial sample set must never become a band).
|
|
117
|
+
if (e && typeof e === 'object') { e.phase = 'judge'; e.judgeAttempts = attemptsTotal + (e.attempts || 1); }
|
|
118
|
+
throw e;
|
|
119
|
+
}
|
|
120
|
+
attemptsTotal += r.attempts || 1;
|
|
121
|
+
usages.push(r.usage || null);
|
|
106
122
|
scores.push(r.score);
|
|
107
123
|
reasons.push(r.reason);
|
|
108
124
|
rawTexts.push(r.raw || '');
|
|
@@ -120,6 +136,12 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
|
|
|
120
136
|
judge_settings: settings,
|
|
121
137
|
model_id: model,
|
|
122
138
|
rubric_hash: rubricHash(rubric),
|
|
139
|
+
attempts: attemptsTotal,
|
|
140
|
+
// v0.4: the measurement overhead of grading this one case — the SUM over all
|
|
141
|
+
// N judge calls. Recorded in the receipt as the case's `judge_usage` and
|
|
142
|
+
// EXCLUDED from every skill-value figure (lib/value.js): it is a cost we
|
|
143
|
+
// impose to measure, not a cost of running the skill.
|
|
144
|
+
usage: sumUsage(usages),
|
|
123
145
|
};
|
|
124
146
|
}
|
|
125
147
|
|
package/lib/models.js
CHANGED
|
@@ -37,10 +37,24 @@ function loadRegistry(force) {
|
|
|
37
37
|
}
|
|
38
38
|
|
|
39
39
|
// Infer the model family from an id. Used to find a same-family predecessor when
|
|
40
|
-
// the release trigger discovers a new model.
|
|
40
|
+
// the release trigger discovers a new model. Covers both providers:
|
|
41
|
+
// anthropic — claude-<family>-… → the family word (opus/sonnet/haiku/…)
|
|
42
|
+
// openai — gpt-<major.minor>-… → gpt-<major.minor> (so sol/terra/luna variants group)
|
|
41
43
|
function familyOf(id) {
|
|
42
|
-
const
|
|
43
|
-
|
|
44
|
+
const s = String(id);
|
|
45
|
+
const c = s.match(/claude-(fable|mythos|opus|sonnet|haiku)/i);
|
|
46
|
+
if (c) return c[1].toLowerCase();
|
|
47
|
+
const g = s.match(/^(gpt-\d+(?:\.\d+)?)/i);
|
|
48
|
+
if (g) return g[1].toLowerCase();
|
|
49
|
+
const o = s.match(/^(o[1-9]+)/i);
|
|
50
|
+
if (o) return o[1].toLowerCase();
|
|
51
|
+
return 'unknown';
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
// Infer the provider from an id when the registry does not carry it (prefix-based,
|
|
55
|
+
// mirrors lib/provider.inferProvider — kept here too so models.js has no cycle).
|
|
56
|
+
function inferProviderFromId(id) {
|
|
57
|
+
return /^(gpt-|o[1-9]|chatgpt|codex|text-|davinci|omni)/i.test(String(id)) ? 'openai' : 'anthropic';
|
|
44
58
|
}
|
|
45
59
|
|
|
46
60
|
// Resolve a (possibly aliased) model id against the registry.
|
|
@@ -72,6 +86,36 @@ function isJudgeEligible(modelId) {
|
|
|
72
86
|
return !!(entry && entry.judge_eligible);
|
|
73
87
|
}
|
|
74
88
|
|
|
89
|
+
// Enforce the fixed-judge policy at a published-run boundary: throw unless the
|
|
90
|
+
// chosen judge is judge_eligible in the registry. This is what makes "attempting
|
|
91
|
+
// an OpenAI judge errors" true — no OpenAI model is judge_eligible, so a cross-
|
|
92
|
+
// provider report can only ever grade with the pinned Haiku judge. Callers on the
|
|
93
|
+
// published/report path (prepare-report, prepare-report-002) invoke this; the dev
|
|
94
|
+
// `run` command's self-judge is intentionally not gated by it.
|
|
95
|
+
function assertJudgeEligible(modelId) {
|
|
96
|
+
if (!isJudgeEligible(modelId)) {
|
|
97
|
+
const { id } = resolveRegistry(modelId);
|
|
98
|
+
const e = new Error(`judge model "${id}" is not judge_eligible: the judge is fixed to a judge_eligible model (claude-haiku-*); no OpenAI model may serve as the judge (see docs/judge-policy.html).`);
|
|
99
|
+
e.code = 'JUDGE_INELIGIBLE';
|
|
100
|
+
throw e;
|
|
101
|
+
}
|
|
102
|
+
return true;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
// The provider for a model: the registry `provider` when registered, else inferred
|
|
106
|
+
// from the id prefix. Stamped into receipts as run.provider.
|
|
107
|
+
function providerForModel(modelId) {
|
|
108
|
+
const { entry, id } = resolveRegistry(modelId);
|
|
109
|
+
return (entry && entry.provider) || inferProviderFromId(id);
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
// Per-provider config block from the registry (config/models.json `providers`):
|
|
113
|
+
// { base_url, api_key_env, surfaces }. Null when the registry omits the provider.
|
|
114
|
+
function providerConfig(provider) {
|
|
115
|
+
const reg = loadRegistry();
|
|
116
|
+
return (reg.providers && reg.providers[provider]) || null;
|
|
117
|
+
}
|
|
118
|
+
|
|
75
119
|
// The servable same-family predecessor of a model: the registered model in the
|
|
76
120
|
// same family with the most recent `released` date strictly before this one
|
|
77
121
|
// (falling back to any other same-family model). Null when the family has no
|
|
@@ -103,13 +147,18 @@ function addDiscoveredModel(id, { firstSeen = null, registryPath = REGISTRY_PATH
|
|
|
103
147
|
const raw = JSON.parse(fs.readFileSync(registryPath, 'utf8'));
|
|
104
148
|
if (raw.models.some((m) => m.id === id)) return null;
|
|
105
149
|
const family = familyOf(id);
|
|
150
|
+
const provider = inferProviderFromId(id);
|
|
106
151
|
// Best-effort tier from family; unknown families default to frontier pricing.
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
152
|
+
// A newly-discovered id gets a CONSERVATIVE (upper-bound) price so the budget
|
|
153
|
+
// guard never under-projects an unknown model — exact rates are set by hand
|
|
154
|
+
// when the model is reviewed for a published run.
|
|
155
|
+
const tier = /haiku|luna|mini|nano/.test(family) || /luna|mini|nano/.test(String(id)) ? 'cheap'
|
|
156
|
+
: /sonnet|terra/.test(family) ? 'standard' : 'frontier';
|
|
157
|
+
const price = provider === 'openai'
|
|
158
|
+
? (tier === 'cheap' ? { input: 1.0, output: 6.0 } : tier === 'standard' ? { input: 2.5, output: 15.0 } : { input: 5.0, output: 30.0 })
|
|
159
|
+
: (family === 'haiku' ? { input: 1.0, output: 5.0 } : family === 'sonnet' ? { input: 3.0, output: 15.0 } : { input: 5.0, output: 25.0 });
|
|
111
160
|
const entry = {
|
|
112
|
-
id, family, provider
|
|
161
|
+
id, family, provider, released: firstSeen,
|
|
113
162
|
input_price: price.input, output_price: price.output, tier,
|
|
114
163
|
judge_eligible: false, auto_added: true,
|
|
115
164
|
};
|
|
@@ -121,5 +170,6 @@ function addDiscoveredModel(id, { firstSeen = null, registryPath = REGISTRY_PATH
|
|
|
121
170
|
|
|
122
171
|
module.exports = {
|
|
123
172
|
loadRegistry, resolveRegistry, registryStatus, priceForModel, isJudgeEligible,
|
|
173
|
+
assertJudgeEligible, providerForModel, providerConfig, inferProviderFromId,
|
|
124
174
|
familyOf, familyPredecessor, addDiscoveredModel, DEFAULT_PRICE, REGISTRY_PATH,
|
|
125
175
|
};
|