driftproof 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/diff.js CHANGED
@@ -59,7 +59,10 @@ function withSkillBands(receipt) {
59
59
  // which is exactly the v0.1 weakness v0.2 fixes).
60
60
  function aggWithBand(receipt) {
61
61
  const a = receipt.results.aggregates.with_skill;
62
- return { mean: a.mean_score, stddev: a.stddev || 0 };
62
+ // v0.6 (spec 026 AC-7): a null band is carried as null and rendered as what
63
+ // it is; a legacy receipt with no aggregate band keeps its 0 (the archive's
64
+ // rendering is unmoved, AC-16).
65
+ return { mean: a.mean_score, stddev: a.stddev === null ? null : (a.stddev || 0), cases: a.case_count };
63
66
  }
64
67
 
65
68
  function fmt(n) { return n == null ? 'n/a' : (n >= 0 ? '+' : '') + n.toFixed(3); }
@@ -81,11 +84,41 @@ function pct(n) { return n == null ? 'n/a' : n.toFixed(3); }
81
84
  // not from one case's draws).
82
85
 
83
86
  function bandStr(x) {
87
+ if (x && x.mean == null) return 'n/a (0 cases)';
88
+ if (x && x.stddev == null) return `${x.mean.toFixed(3)} ± n/a (${x.cases === 1 ? '1 case' : 'no band'})`;
84
89
  const label = x && x.source;
85
90
  return `${x.mean.toFixed(3)} ± ${x.stddev.toFixed(3)}${label ? ` (${label})` : ''}`;
86
91
  }
87
92
  function short(h) { return h ? String(h).slice(0, 12) : 'n/a'; }
88
93
 
94
+ // ── THE JUDGE IS PART OF THE INSTRUMENT (spec 026 AC-12, A4) ─────────────────
95
+ // Two receipts graded by different judges are two measurements with different
96
+ // instruments, and a verdict across them says nothing about the skill.
97
+ // lib/reuse.js's triage already says a changed judge means regrade; the
98
+ // differ did not ask. It asks here: run.judge.model_id (on canonical ids, so
99
+ // an alias and its dated form are one judge), run.judge.prompt_template_hash,
100
+ // run.judge.temperature, and every shared case's judge.rubric_hash. A field
101
+ // that differs names itself in the caveats and suppresses every per-case
102
+ // verdict. A pre-v0.6 receipt carries no template hash: the report says the
103
+ // template is unrecorded on that side and still compares the rest.
104
+ const canonicalJudgeId = (id) => String(id == null ? '' : id).replace(/-\d{8}$/, '');
105
+ function judgeDiffers(a, b, ids, aB, bB) {
106
+ const ja = (a.run && a.run.judge) || {}; const jb = (b.run && b.run.judge) || {};
107
+ const problems = [];
108
+ const idA = ja.model_id || (a.results.cases.find((c) => c.judge && c.judge.model_id) || { judge: {} }).judge.model_id;
109
+ const idB = jb.model_id || (b.results.cases.find((c) => c.judge && c.judge.model_id) || { judge: {} }).judge.model_id;
110
+ if (idA && idB && canonicalJudgeId(idA) !== canonicalJudgeId(idB)) problems.push(`run.judge.model_id differs (${idA} vs ${idB})`);
111
+ const unrecorded = [];
112
+ if (!ja.prompt_template_hash) unrecorded.push('A'); if (!jb.prompt_template_hash) unrecorded.push('B');
113
+ if (ja.prompt_template_hash && jb.prompt_template_hash && ja.prompt_template_hash !== jb.prompt_template_hash) problems.push(`run.judge.prompt_template_hash differs (${short(ja.prompt_template_hash)} vs ${short(jb.prompt_template_hash)}): the grading template is a different judge`);
114
+ if (ja.temperature !== undefined && jb.temperature !== undefined && ja.temperature !== jb.temperature) problems.push(`run.judge.temperature differs (${ja.temperature} vs ${jb.temperature})`);
115
+ const rubricA = Object.fromEntries(a.results.cases.filter((c) => c.mode === 'with_skill' && c.judge).map((c) => [c.id, c.judge.rubric_hash]));
116
+ const rubricB = Object.fromEntries(b.results.cases.filter((c) => c.mode === 'with_skill' && c.judge).map((c) => [c.id, c.judge.rubric_hash]));
117
+ const moved = ids.filter((id) => rubricA[id] && rubricB[id] && rubricA[id] !== rubricB[id]);
118
+ if (moved.length) problems.push(`judge.rubric_hash differs on ${moved.length} shared case(s) (${moved.slice(0, 3).map((x) => `\`${x}\``).join(', ')}${moved.length > 3 ? ', …' : ''})`);
119
+ return { problems, unrecorded };
120
+ }
121
+
89
122
  // The headline is a SUMMARY of the per-case band-overlap verdicts — the credibility
90
123
  // core lives per case (judge-sample bands), and the headline just aggregates it.
91
124
  // It deliberately does NOT run a separate band test on the aggregate mean: the
@@ -146,9 +179,15 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
146
179
  // improvement verdicts are never computed from evidence we did not run.
147
180
  const levelOf = (r) => r.verification_level || 'TESTED';
148
181
  const belowTested = [[labelA, a], [labelB, b]].filter(([, r]) => levelOf(r) !== 'TESTED');
182
+ // Spec 026 AC-4: an INCOMPLETE receipt on either side computes no per-case
183
+ // verdict. RECEIPT.md has said so since v0.3.1; the differ did not ask.
184
+ const incompleteSides = [[labelA, a], [labelB, b]].filter(([, r]) => r.run && r.run.status === 'incomplete');
185
+ // Spec 026 AC-12: a different judge on either side suppresses the verdict.
186
+ const judge = judgeDiffers(a, b, ids, aB, bB);
187
+ const judgeProblem = judge.problems.length ? judge.problems.join('; ') : null;
149
188
  // A REFUSAL IS A RESULT, and it suppresses the verdict exactly as an
150
189
  // untested input does: no delta is asserted, and the reason travels with it.
151
- const measured = belowTested.length === 0 && !refusal;
190
+ const measured = belowTested.length === 0 && !refusal && incompleteSides.length === 0 && !judgeProblem;
152
191
 
153
192
  const perCase = ids.map((id) => {
154
193
  const before = aB[id] || null;
@@ -250,6 +289,15 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
250
289
  if (belowTested.length) {
251
290
  warnings.push(`verdicts NOT computed — ${belowTested.map(([l, r]) => `${l} is ${levelOf(r)}${r.run && r.run.source ? ` (${r.run.source})` : ''}`).join('; ')}. Drift verdicts require TESTED receipts on both sides; declared numbers are shown as context only (see /interop.html).`);
252
291
  }
292
+ if (incompleteSides.length) {
293
+ warnings.push(`verdicts NOT computed — ${incompleteSides.map(([l, r]) => `${l} is incomplete (run.status incomplete; it excluded ${r.run.failed_case_count || 0} case(s) whose arm could not be measured)`).join('; ')}. A receipt with an unmeasured arm is not evidence for a per-case verdict; its aggregates are shown as context only.`);
294
+ }
295
+ if (judgeProblem) {
296
+ warnings.push(`verdicts NOT computed — the two receipts were graded by different judges: ${judgeProblem}. A verdict across two instruments says nothing about the skill; regrade one side with the other's judge (lib/reuse.js triage: regrade).`);
297
+ }
298
+ if (judge.unrecorded.length) {
299
+ warnings.push(`judge template unrecorded on ${judge.unrecorded.map((x) => (x === 'A' ? labelA : labelB)).join(' and ')} (a pre-v0.6 receipt carries no run.judge.prompt_template_hash); the judge model id and the rubric hashes are compared, the template is not.`);
300
+ }
253
301
  // Cross-provider / cross-surface disclosure (Phase 6). A comparison across
254
302
  // providers is a skill-DURABILITY comparison across substrates, not model drift
255
303
  // over time; across surfaces, sampling control differs. Both are flagged so a
@@ -281,7 +329,11 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
281
329
  // THE HEADLINE NAMES THE CAUSE THAT ACTUALLY APPLIES. `belowTested.length`
282
330
  // was printed unconditionally, so a refused pair got "0 receipt(s) below
283
331
  // TESTED" — a statement measurably false of the receipts it was given.
284
- if (refusal) {
332
+ if (incompleteSides.length) {
333
+ // An incomplete side is a fact about that receipt alone (spec 026 AC-4)
334
+ // and leads whatever else is wrong with the pair.
335
+ L.push(`**NOT MEASURED — ${incompleteSides.map(([l]) => l).join(' and ')} incomplete: a case's arm could not be measured, so no per-case verdict is computed.**`);
336
+ } else if (refusal) {
285
337
  // The reason is a complete sentence and already ends by saying no verdict
286
338
  // is asserted; prefixing that again produced "REFUSED — no verdict is
287
339
  // asserted. the baseline arm did not reproduce…" — a duplicated clause and
@@ -289,6 +341,8 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
289
341
  L.push(`**REFUSED — ${refusal.reason}**`);
290
342
  } else if (belowTested.length) {
291
343
  L.push(`**NOT MEASURED — ${belowTested.length} receipt(s) below TESTED. Drift verdicts require TESTED receipts on both sides; the declared numbers above are context, not band-verified evidence.**`);
344
+ } else if (judgeProblem) {
345
+ L.push(`**NOT MEASURED — different judges: ${judgeProblem}. No verdict crosses two instruments.**`);
292
346
  } else {
293
347
  L.push('**NOT MEASURED — no verdict is asserted.**');
294
348
  }
@@ -327,4 +381,4 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
327
381
  };
328
382
  }
329
383
 
330
- module.exports = { buildDriftReport, withSkillBands, revisionPairProblem };
384
+ module.exports = { buildDriftReport, withSkillBands, revisionPairProblem, judgeDiffers };
package/lib/importers.js CHANGED
@@ -17,7 +17,7 @@
17
17
  // compatibility contract) are documented in docs/interop.md.
18
18
 
19
19
  const { RECEIPT_SCHEMA_VERSION, RUNNER_VERSION, SUITE_FORMAT } = require('../config');
20
- const { sealReceipt } = require('./receipt');
20
+ const { sealReceipt, BAND_RULE } = require('./receipt');
21
21
  const { mean, stddev, round, combineUncertainty, aggregateBands } = require('./stats');
22
22
  const { outcomeFor } = require('./run');
23
23
  const { inferProvider } = require('./provider');
@@ -64,11 +64,16 @@ function importedReceipt({ tool, skillName, skillVersion, suiteFormat, caseCount
64
64
  date_utc: dateUtc,
65
65
  registry: registryStatus(modelId),
66
66
  transcripts: 'none',
67
- judge: judgeBlock,
67
+ // v0.6: the source tool's grader is the judge that ran; its template is
68
+ // not ours to hash (null, the one place the schema allows it).
69
+ judge: { ...judgeBlock, model_id: (cases.find((c) => c.judge && c.judge.model_id) || { judge: { model_id: 'unknown' } }).judge.model_id, prompt_template_hash: null },
70
+ // v0.6 (spec 026 AC-1): what answered. An external tool's harness did;
71
+ // nothing here attested a model, and nothing was spawned by us.
72
+ answered_by: { kind: 'external', attested: false, reported_model: null, reported_models: null, isolation: 'none' },
68
73
  },
69
74
  results: {
70
75
  cases,
71
- aggregates: { with_skill: aggregateMode(withSkill), baseline: aggregateMode(baseline) },
76
+ aggregates: { with_skill: aggregateMode(withSkill), baseline: aggregateMode(baseline), band_rule: BAND_RULE },
72
77
  },
73
78
  comparison,
74
79
  verification_level: 'DECLARED',
package/lib/judge.js CHANGED
@@ -58,10 +58,31 @@ function rubricHash(rubric) {
58
58
  return sha256(JUDGE_SYSTEM + '\n---\n' + String(rubric || '').trim());
59
59
  }
60
60
 
61
- function clamp01(n) {
62
- const x = Number(n);
63
- if (!Number.isFinite(x)) return 0;
64
- return Math.max(0, Math.min(1, x));
61
+ // v0.6 (spec 026 AC-11): a digest over the grading TEMPLATE with its three
62
+ // slots empty, recorded once per run as run.judge.prompt_template_hash. The
63
+ // rubric_hash above binds a grade to the case's rubric; this binds every grade
64
+ // of the run to the words around it. A different template is a different judge,
65
+ // and `diff` computes no verdict across one.
66
+ function promptTemplateHash() {
67
+ return sha256(JUDGE_SYSTEM + '\n---\n' + buildJudgePrompt({ task: '', response: '', rubric: '' }));
68
+ }
69
+
70
+ // The score a judge reply carries, or null when it carries none (spec 026
71
+ // AC-3, F2). A non-numeric or non-finite value is not a score; a number outside
72
+ // [0, 1] is on a scale the rubric did not ask for and is not clamped into one
73
+ // (85 is not 1.0). This replaced clamp01, whose silent 0 for a non-finite value
74
+ // scored an absent output as the worst possible one.
75
+ // Stop reasons that mean the surface cut the reply at its output cap: the
76
+ // Messages API's and the claude CLI's `max_tokens`, Chat Completions'
77
+ // `length`. A cut reply is a partial answer (spec 026 AC-8, F4).
78
+ const TRUNCATION_STOP_REASONS = new Set(['max_tokens', 'length']);
79
+ function isTruncated(stopReason) { return TRUNCATION_STOP_REASONS.has(String(stopReason || '')); }
80
+
81
+ function scoreOf(parsed) {
82
+ const x = Number(parsed && parsed.score);
83
+ if (!Number.isFinite(x)) return null;
84
+ if (x < 0 || x > 1) return null;
85
+ return x;
65
86
  }
66
87
 
67
88
  // Judge settings for the JUDGE model's surface. Determinism where the surface
@@ -78,25 +99,50 @@ function judgeSettings(samples, judgeModel) {
78
99
  return { samples, temperature: null, sampling: 'surface-controlled', surface };
79
100
  }
80
101
 
81
- // Grade one response once. Returns { score, reason, raw }.
102
+ // Grade one response once. Returns { score, reason, raw, ... } for a measured
103
+ // sample, or { unmeasured: true, reason, raw, ... } when the reply carries no
104
+ // score: empty, unparseable, no numeric score, or a score outside [0, 1]. A
105
+ // zero asserts a measurement ("the response satisfied none of the rubric");
106
+ // each of these is the absence of one, and no sample enters any statistic
107
+ // (spec 026 AC-3). The 2026-07 rule that an unparseable judge must never
108
+ // silently pass is kept by the stronger rule: it never silently scores at all.
82
109
  // `raw` is the judge's verbatim output text (hashed into the receipt for
83
110
  // transcript auditability, and optionally retained under --keep-transcripts).
84
- async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature }) {
111
+ async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature, trusted = false }) {
85
112
  const prompt = buildJudgePrompt({ task, response, rubric });
86
- const { text, attempts, usage } = await complete({ system: JUDGE_SYSTEM, prompt, model, maxTokens: 400, timeoutMs, temperature });
113
+ const out = await complete({ system: JUDGE_SYSTEM, prompt, model, maxTokens: 400, timeoutMs, temperature, trusted });
114
+ const { text, attempts, usage } = out;
115
+ const base = { raw: String(text || ''), attempts: attempts || 1, usage, reply: out };
116
+ // A judge reply cut at its output cap carries no complete grade.
117
+ if (isTruncated(out.stopReason)) {
118
+ return { ...base, unmeasured: true, reason: `judge output truncated at the output cap (stop_reason ${out.stopReason})` };
119
+ }
120
+ if (!String(text || '').trim()) {
121
+ return { ...base, unmeasured: true, reason: 'judge output empty (the surface returned no text)' };
122
+ }
87
123
  let parsed;
88
124
  try {
89
125
  parsed = extractJsonObject(text);
90
126
  } catch (_e) {
91
- // Unsalvageable judge output → conservative 0 (a judge that can't be parsed
92
- // must never silently "pass"), tagged so the caller can see it happened.
93
- return { score: 0, reason: 'judge output unparseable', unparsed: true, raw: String(text || ''), attempts: attempts || 1, usage };
127
+ return { ...base, unmeasured: true, reason: 'judge output unparseable' };
94
128
  }
95
- return { score: clamp01(parsed.score), reason: String(parsed.reason || '').slice(0, 300), raw: String(text || ''), attempts: attempts || 1, usage };
129
+ const score = scoreOf(parsed);
130
+ if (score === null) {
131
+ const x = parsed && parsed.score;
132
+ const reason = x === undefined || x === null ? 'judge output carries no numeric score'
133
+ : !Number.isFinite(Number(x)) ? `judge output carries no numeric score (score ${JSON.stringify(x)})`
134
+ : `judge score ${x} outside [0, 1]`;
135
+ return { ...base, unmeasured: true, reason };
136
+ }
137
+ return { ...base, score, reason: String(parsed.reason || '').slice(0, 300) };
96
138
  }
97
139
 
98
140
  // Grade a response N times and return the sampled distribution:
99
141
  // { samples:[scores], mean, stddev, reason, judge_settings, model_id, rubric_hash }
142
+ // or, when any sample is unmeasured, { unmeasured: true, reason, ... } with NO
143
+ // samples: a partial sample set must never become a band, and a draw one of
144
+ // whose judge samples carried no score is unmeasured as a whole (spec 026
145
+ // AC-3). The remaining samples are not taken.
100
146
  // `mean` ± `stddev` is the per-case confidence band (raw spread of the N scores)
101
147
  // used by the borderline-outcome rule and per-case drift band-overlap logic.
102
148
  // NO DEFAULT TIMEOUT HERE (spec 017 AC-2). This defaulted to 120000, which
@@ -104,17 +150,20 @@ async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature
104
150
  // JUDGE calls timed out on the api policy while running on a CLI surface, which
105
151
  // is the second shadowing site and the one #007's prep session had not found.
106
152
  // Passing `undefined` through lets lib/provider.js resolve the declared policy.
107
- async function gradeSamples({ task, response, rubric, model, samples = 5, timeoutMs }) {
153
+ // `trusted` (spec 022) is plumbed to complete() untouched: the judge takes the
154
+ // same spawn path as the generation it grades, so --trusted-skill is whole-run.
155
+ async function gradeSamples({ task, response, rubric, model, samples = 5, timeoutMs, trusted = false }) {
108
156
  const settings = judgeSettings(samples, model);
109
157
  const scores = [];
110
158
  const reasons = [];
111
159
  const rawTexts = [];
112
160
  const usages = [];
161
+ const replies = []; // v0.6: what answered each judge call, for the runner's attestation
113
162
  let attemptsTotal = 0;
114
163
  for (let i = 0; i < samples; i++) {
115
164
  let r;
116
165
  try {
117
- r = await gradeOnce({ task, response, rubric, model, timeoutMs, temperature: settings.temperature === null ? undefined : settings.temperature });
166
+ r = await gradeOnce({ task, response, rubric, model, timeoutMs, temperature: settings.temperature === null ? undefined : settings.temperature, trusted });
118
167
  } catch (e) {
119
168
  // A judge sample that persistently failed (e.g. timed out after retries):
120
169
  // tag the error so the runner can charge for the spend and mark the whole
@@ -124,9 +173,18 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
124
173
  }
125
174
  attemptsTotal += r.attempts || 1;
126
175
  usages.push(r.usage || null);
176
+ replies.push(r.reply || null);
177
+ rawTexts.push(r.raw || '');
178
+ if (r.unmeasured) {
179
+ return {
180
+ unmeasured: true, reason: r.reason, samples: [], mean: null, stddev: null,
181
+ sample_texts: rawTexts, sample_hashes: rawTexts.map((t) => sha256(t)),
182
+ judge_settings: settings, model_id: model, rubric_hash: rubricHash(rubric),
183
+ attempts: attemptsTotal, usage: sumUsage(usages), replies,
184
+ };
185
+ }
127
186
  scores.push(r.score);
128
187
  reasons.push(r.reason);
129
- rawTexts.push(r.raw || '');
130
188
  }
131
189
  return {
132
190
  samples: scores,
@@ -147,7 +205,8 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
147
205
  // EXCLUDED from every skill-value figure (lib/value.js): it is a cost we
148
206
  // impose to measure, not a cost of running the skill.
149
207
  usage: sumUsage(usages),
208
+ replies,
150
209
  };
151
210
  }
152
211
 
153
- module.exports = { gradeOnce, gradeSamples, judgeSettings, rubricHash, buildJudgePrompt, JUDGE_SYSTEM };
212
+ module.exports = { gradeOnce, gradeSamples, judgeSettings, rubricHash, buildJudgePrompt, promptTemplateHash, scoreOf, isTruncated, TRUNCATION_STOP_REASONS, JUDGE_SYSTEM };
package/lib/models.js CHANGED
@@ -66,6 +66,26 @@ function resolveRegistry(modelId) {
66
66
  return { id: canonical, entry, registered: !!entry };
67
67
  }
68
68
 
69
+ // Spec 026 AC-10 (F6): an id that is not a model does not run. After alias
70
+ // resolution the id must have the contract's model-id shape (letters, digits,
71
+ // . _ -) and resolve in the registry; otherwise the run is refused before any
72
+ // call, naming the id and the absolute path of the registry consulted
73
+ // (DRIFTPROOF_REGISTRY when set, else the packaged config/models.json).
74
+ // `registry: "unregistered"` stays a legal receipt value for IMPORTED receipts,
75
+ // whose model field is another tool's word; a run of ours never writes it.
76
+ const MODEL_ID_SHAPE = /^[A-Za-z0-9._-]+$/;
77
+ function assertRegistered(modelId, role = 'model') {
78
+ const given = String(modelId == null ? '' : modelId);
79
+ const { id, registered } = MODEL_ID_SHAPE.test(given) ? resolveRegistry(given) : { id: given, registered: false };
80
+ if (!MODEL_ID_SHAPE.test(given) || !registered) {
81
+ const why = !MODEL_ID_SHAPE.test(given) ? 'is not a model id (letters, digits, . _ - only)' : 'is not in the model registry';
82
+ const e = new Error(`${role} "${given}"${id !== given ? ` (resolved "${id}")` : ''} ${why}: ${REGISTRY_PATH}. An unregistered model does not run; add a registry row (see config/models.json) or point DRIFTPROOF_REGISTRY at a registry that carries it.`);
83
+ e.code = 'UNREGISTERED_MODEL'; e.model = given; e.registry = REGISTRY_PATH;
84
+ throw e;
85
+ }
86
+ return id;
87
+ }
88
+
69
89
  // The value stamped into receipt.run.registry.
70
90
  function registryStatus(modelId) {
71
91
  return resolveRegistry(modelId).registered ? 'registered' : 'unregistered';
@@ -140,23 +160,43 @@ function familyPredecessor(modelId) {
140
160
  return chooseFrom[0].id;
141
161
  }
142
162
 
143
- // Append a newly-discovered model id to the registry file (release trigger).
144
- // `firstSeen` is the first-seen date recorded as `released`. Idempotent: a no-op
145
- // if the id already exists. Returns the added entry (or null if already present).
146
- function addDiscoveredModel(id, { firstSeen = null, registryPath = REGISTRY_PATH } = {}) {
147
- const raw = JSON.parse(fs.readFileSync(registryPath, 'utf8'));
148
- if (raw.models.some((m) => m.id === id)) return null;
163
+ // THE PRICE OF A MODEL NOBODY HAS PRICED YET.
164
+ //
165
+ // READ THIS BEFORE TRUSTING A NUMBER IT RETURNS. Nothing here reads a price
166
+ // SOURCE. The family and the tier are inferred from the id, and the rate is
167
+ // looked up in the table below, so the result is a GUESS whose only input is the
168
+ // string. It is extracted into its own function, and named, because the guess
169
+ // was being made silently inside a function whose job looked like registration.
170
+ //
171
+ // KNOWN DEFECT, spec 021 F-W2, deliberately NOT fixed here. The Anthropic branch
172
+ // falls through to the OPUS rate of 5/25 for every family that is not `haiku` or
173
+ // `sonnet`. Fable's published rate is 10/50, exactly twice opus, which is the
174
+ // whole of the "the watcher read exactly half the snapshot on both fields"
175
+ // observation of 2026-09-03: it is not a halving and not a misread, it is a
176
+ // wrong-tier fallback. `DEFAULT_PRICE` above IS the conservative upper bound the
177
+ // comment on this function used to claim, and it is not consulted. Recorded on
178
+ // the spec 021 carry list, tagged 026, with the reason it is left alone.
179
+ //
180
+ // scripts/release-watch.js no longer registers anything, and records what this
181
+ // returns as `price_verified: false` beside a source naming this function.
182
+ function inferredPrice(id) {
149
183
  const family = familyOf(id);
150
184
  const provider = inferProviderFromId(id);
151
- // Best-effort tier from family; unknown families default to frontier pricing.
152
- // A newly-discovered id gets a CONSERVATIVE (upper-bound) price so the budget
153
- // guard never under-projects an unknown model — exact rates are set by hand
154
- // when the model is reviewed for a published run.
155
185
  const tier = /haiku|luna|mini|nano/.test(family) || /luna|mini|nano/.test(String(id)) ? 'cheap'
156
186
  : /sonnet|terra/.test(family) ? 'standard' : 'frontier';
157
187
  const price = provider === 'openai'
158
188
  ? (tier === 'cheap' ? { input: 1.0, output: 6.0 } : tier === 'standard' ? { input: 2.5, output: 15.0 } : { input: 5.0, output: 30.0 })
159
189
  : (family === 'haiku' ? { input: 1.0, output: 5.0 } : family === 'sonnet' ? { input: 3.0, output: 15.0 } : { input: 5.0, output: 25.0 });
190
+ return { family, provider, tier, price };
191
+ }
192
+
193
+ // Append a newly-discovered model id to the registry file (release trigger).
194
+ // `firstSeen` is the first-seen date recorded as `released`. Idempotent: a no-op
195
+ // if the id already exists. Returns the added entry (or null if already present).
196
+ function addDiscoveredModel(id, { firstSeen = null, registryPath = REGISTRY_PATH } = {}) {
197
+ const raw = JSON.parse(fs.readFileSync(registryPath, 'utf8'));
198
+ if (raw.models.some((m) => m.id === id)) return null;
199
+ const { family, provider, tier, price } = inferredPrice(id);
160
200
  const entry = {
161
201
  id, family, provider, released: firstSeen,
162
202
  input_price: price.input, output_price: price.output, tier,
@@ -171,5 +211,6 @@ function addDiscoveredModel(id, { firstSeen = null, registryPath = REGISTRY_PATH
171
211
  module.exports = {
172
212
  loadRegistry, resolveRegistry, registryStatus, priceForModel, isJudgeEligible,
173
213
  assertJudgeEligible, providerForModel, providerConfig, inferProviderFromId,
174
- familyOf, familyPredecessor, addDiscoveredModel, DEFAULT_PRICE, REGISTRY_PATH,
214
+ familyOf, familyPredecessor, addDiscoveredModel, inferredPrice, DEFAULT_PRICE, REGISTRY_PATH,
215
+ assertRegistered, MODEL_ID_SHAPE,
175
216
  };