driftproof 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,207 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ // Receipt interop — importers (Phase 7). Convert results from neighboring eval
5
+ // tools into valid Driftproof receipts with HONEST epistemics:
6
+ //
7
+ // - verification_level DECLARED (never TESTED — we did not run, hash, or
8
+ // judge the generations),
9
+ // - run.surface "external" + run.source "imported/<tool>",
10
+ // - generation/judge-sample hashes OMITTED, content/suite/rubric hashes null —
11
+ // hashes are never fabricated,
12
+ // - no baseline mode in the source tool → empty baseline aggregate + null
13
+ // comparison, never a fabricated 0-baseline.
14
+ //
15
+ // The per-tool field mappings (and the assumed-shape disclosure — neither tool
16
+ // publishes a frozen results schema, so the checked-in fixtures ARE the
17
+ // compatibility contract) are documented in docs/interop.md.
18
+
19
+ const { RECEIPT_SCHEMA_VERSION, RUNNER_VERSION, SUITE_FORMAT } = require('../config');
20
+ const { sealReceipt } = require('./receipt');
21
+ const { mean, stddev, round, combineUncertainty, aggregateBands } = require('./stats');
22
+ const { outcomeFor } = require('./run');
23
+ const { inferProvider } = require('./provider');
24
+ const { registryStatus } = require('./models');
25
+
26
+ const IMPORT_TOOLS = ['agent-skills-eval', 'skillgrade'];
27
+
28
+ // Aggregate a list of imported cases the same way lib/receipt.aggregate does
29
+ // (suite dispersion band). An empty mode aggregates to the conventional empty
30
+ // aggregate (case_count 0) — its absence is signalled there, and in the null
31
+ // comparison, not by fabricated numbers.
32
+ function aggregateMode(cases) {
33
+ const band = aggregateBands(cases.map((c) => ({ mean: c.mean, stddev: c.stddev || 0, n: c.samples.length })));
34
+ return {
35
+ case_count: cases.length,
36
+ pass_count: cases.filter((c) => c.outcome === 'pass').length,
37
+ borderline_count: cases.filter((c) => c.outcome === 'borderline').length,
38
+ mean_score: band.mean,
39
+ stddev: band.stddev,
40
+ };
41
+ }
42
+
43
+ // Shared receipt shell for both importers. `judgeBlock` describes the SOURCE
44
+ // tool's grading (samples = what it actually did), never our sampled judge.
45
+ function importedReceipt({ tool, skillName, skillVersion, suiteFormat, caseCount, modelId, dateUtc, judgeBlock, cases, comparison }) {
46
+ const withSkill = cases.filter((c) => c.mode === 'with_skill');
47
+ const baseline = cases.filter((c) => c.mode === 'baseline');
48
+ const receipt = {
49
+ schema_version: RECEIPT_SCHEMA_VERSION,
50
+ skill: {
51
+ name: skillName,
52
+ version: skillVersion || 'unknown',
53
+ // Never seen the skill bytes — null, never fabricated.
54
+ content_hash: null,
55
+ },
56
+ suite: { format: suiteFormat, suite_hash: null, case_count: caseCount },
57
+ run: {
58
+ model_id: modelId,
59
+ model_release_date: null,
60
+ provider: inferProvider(modelId),
61
+ surface: 'external',
62
+ source: `imported/${tool}`,
63
+ runner_version: RUNNER_VERSION,
64
+ date_utc: dateUtc,
65
+ registry: registryStatus(modelId),
66
+ transcripts: 'none',
67
+ judge: judgeBlock,
68
+ },
69
+ results: {
70
+ cases,
71
+ aggregates: { with_skill: aggregateMode(withSkill), baseline: aggregateMode(baseline) },
72
+ },
73
+ comparison,
74
+ verification_level: 'DECLARED',
75
+ receipt_hash: '',
76
+ };
77
+ return sealReceipt(receipt);
78
+ }
79
+
80
+ // ── agent-skills-eval (darkrishabh/agent-skills-eval) ────────────────────────
81
+ // Same suite lineage as Driftproof (agentskills.io evals.json). Runs each eval
82
+ // with_skill and without_skill; an LLM judge grades BINARY per assertion. The
83
+ // converter reads the rolled-up benchmark artifact:
84
+ // { skill_name, version?, target, judge, timestamp?, evals: [
85
+ // { id, with_skill: { pass, assertions?: [{assertion, pass, reasoning?}] },
86
+ // without_skill: { ... } } ] }
87
+ // Per (eval, mode): mean = fraction of assertions passed (else 1/0 from the
88
+ // eval-level pass); samples = [mean] — the ONE grade their judge produced,
89
+ // never resampled; stddev 0; outcome from their pass verdict.
90
+ function modeScore(grading) {
91
+ const asserts = Array.isArray(grading.assertions) ? grading.assertions : [];
92
+ if (asserts.length) return round(asserts.filter((a) => a.pass === true).length / asserts.length);
93
+ return grading.pass === true ? 1 : 0;
94
+ }
95
+ function importAgentSkillsEval(data, { importedAt } = {}) {
96
+ if (!data || !Array.isArray(data.evals)) throw new Error('agent-skills-eval import: expected { skill_name, target, judge, evals: [...] } (see docs/interop.md)');
97
+ const judgeModel = data.judge || 'unknown';
98
+ const cases = [];
99
+ for (const ev of data.evals) {
100
+ for (const [theirMode, ourMode] of [['with_skill', 'with_skill'], ['without_skill', 'baseline']]) {
101
+ const grading = ev[theirMode];
102
+ if (!grading) continue;
103
+ const m = modeScore(grading);
104
+ const failing = (grading.assertions || []).find((a) => a.pass === false);
105
+ const c = {
106
+ id: String(ev.id),
107
+ mode: ourMode,
108
+ outcome: grading.pass === true ? 'pass' : 'fail',
109
+ score: m, mean: m, stddev: 0,
110
+ samples: [m],
111
+ threshold: null,
112
+ judge: { model_id: judgeModel, rubric_hash: null },
113
+ };
114
+ if (failing && failing.reasoning) c.reason = String(failing.reasoning).slice(0, 300);
115
+ cases.push(c);
116
+ }
117
+ }
118
+ const withMeans = cases.filter((c) => c.mode === 'with_skill').map((c) => c.mean);
119
+ const baseMeans = cases.filter((c) => c.mode === 'baseline').map((c) => c.mean);
120
+ const hasBaseline = baseMeans.length > 0;
121
+ const wAgg = aggregateBands(cases.filter((c) => c.mode === 'with_skill').map((c) => ({ mean: c.mean, stddev: 0, n: 1 })));
122
+ const bAgg = aggregateBands(cases.filter((c) => c.mode === 'baseline').map((c) => ({ mean: c.mean, stddev: 0, n: 1 })));
123
+ const comparison = {
124
+ with_skill_score: round(mean(withMeans)),
125
+ baseline_score: hasBaseline ? round(mean(baseMeans)) : null,
126
+ delta: hasBaseline ? round(mean(withMeans) - mean(baseMeans)) : null,
127
+ delta_uncertainty: hasBaseline ? combineUncertainty(wAgg.stddev, bAgg.stddev) : null,
128
+ };
129
+ return importedReceipt({
130
+ tool: 'agent-skills-eval',
131
+ skillName: data.skill_name || 'unknown',
132
+ skillVersion: data.version,
133
+ suiteFormat: SUITE_FORMAT, // agentskills.io/evals — same suite lineage
134
+ caseCount: data.evals.length,
135
+ modelId: data.target || 'unknown',
136
+ dateUtc: data.timestamp || importedAt || new Date().toISOString(),
137
+ judgeBlock: { samples: 1, temperature: null, sampling: 'external', surface: 'external' },
138
+ cases,
139
+ comparison,
140
+ });
141
+ }
142
+
143
+ // ── skillgrade (mgechev/skillgrade) ──────────────────────────────────────────
144
+ // "Unit tests for your agent skills": N trials per task, each trial's reward =
145
+ // weighted grader scores (0..1), compared against a threshold. NO baseline mode
146
+ // — so the receipt carries an empty baseline aggregate and a null comparison.
147
+ // Trials are real repeated runs, so per-trial rewards map onto samples[]
148
+ // honestly (a genuine cross-trial band). Expected shape:
149
+ // { skill, agent, grader_model?, threshold?, timestamp?, tasks: [
150
+ // { name, threshold?, trials: [ { reward }, ... ] } ] }
151
+ function importSkillgrade(data, { importedAt } = {}) {
152
+ if (!data || !Array.isArray(data.tasks)) throw new Error('skillgrade import: expected { skill, agent, tasks: [...] } (see docs/interop.md)');
153
+ const graderModel = data.grader_model || 'unknown';
154
+ const defaultThreshold = typeof data.threshold === 'number' ? data.threshold : 0.8;
155
+ const cases = [];
156
+ let maxTrials = 1;
157
+ for (const task of data.tasks) {
158
+ const rewards = (task.trials || []).map((t) => (typeof t === 'number' ? t : t.reward)).filter((r) => typeof r === 'number');
159
+ if (!rewards.length) continue;
160
+ maxTrials = Math.max(maxTrials, rewards.length);
161
+ const m = round(mean(rewards));
162
+ const sd = round(stddev(rewards));
163
+ const threshold = typeof task.threshold === 'number' ? task.threshold : defaultThreshold;
164
+ cases.push({
165
+ id: String(task.name),
166
+ mode: 'with_skill',
167
+ // Same outcome rule as a Driftproof run (borderline when the threshold
168
+ // sits inside mean ± stddev) — a deterministic READING of their numbers.
169
+ outcome: outcomeFor(m, sd, threshold),
170
+ score: m, mean: m, stddev: sd,
171
+ samples: rewards.map((r) => round(r)),
172
+ threshold,
173
+ judge: { model_id: graderModel, rubric_hash: null },
174
+ });
175
+ }
176
+ return importedReceipt({
177
+ tool: 'skillgrade',
178
+ skillName: data.skill || 'unknown',
179
+ skillVersion: data.version,
180
+ suiteFormat: 'skillgrade/eval.yaml',
181
+ caseCount: data.tasks.length,
182
+ // skillgrade names an AGENT CLI (claude/gemini/codex), not a model id —
183
+ // imported verbatim (or the results' model field when present).
184
+ modelId: data.model || data.agent || 'unknown',
185
+ dateUtc: data.timestamp || importedAt || new Date().toISOString(),
186
+ judgeBlock: { samples: maxTrials, temperature: null, sampling: 'external', surface: 'external' },
187
+ cases,
188
+ // No baseline mode exists in skillgrade — nulls, never a fabricated 0.
189
+ comparison: {
190
+ with_skill_score: round(mean(cases.map((c) => c.mean))),
191
+ baseline_score: null,
192
+ delta: null,
193
+ delta_uncertainty: null,
194
+ },
195
+ });
196
+ }
197
+
198
+ // Dispatch. `from` must name a supported tool.
199
+ function importResults(data, { from, importedAt } = {}) {
200
+ switch (from) {
201
+ case 'agent-skills-eval': return importAgentSkillsEval(data, { importedAt });
202
+ case 'skillgrade': return importSkillgrade(data, { importedAt });
203
+ default: throw new Error(`unknown import source "${from}" — supported: ${IMPORT_TOOLS.join(', ')}`);
204
+ }
205
+ }
206
+
207
+ module.exports = { importResults, importAgentSkillsEval, importSkillgrade, IMPORT_TOOLS };
package/lib/judge.js CHANGED
@@ -1,10 +1,11 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
- const { complete, surfaceLabel } = require('./provider');
4
+ const { complete, surfaceForModel } = require('./provider');
5
5
  const { extractJsonObject } = require('./json');
6
6
  const { sha256 } = require('./canonical');
7
7
  const { mean, stddev } = require('./stats');
8
+ const { sumUsage } = require('./usage');
8
9
 
9
10
  // Rubric-based LLM judge.
10
11
  //
@@ -63,13 +64,15 @@ function clamp01(n) {
63
64
  return Math.max(0, Math.min(1, x));
64
65
  }
65
66
 
66
- // Judge settings for the CURRENT surface. Determinism where the surface allows:
67
- // on the api surface we pin temperature 0 for judge calls (and record it); on
68
- // the cli surface sampling params are surface-controlled and cannot be set, so
69
- // temperature is null and `sampling` says so. Recorded into every receipt.
70
- function judgeSettings(samples) {
71
- const surface = surfaceLabel();
72
- if (surface === 'api') {
67
+ // Judge settings for the JUDGE model's surface. Determinism where the surface
68
+ // allows: on an api surface (Anthropic `api` or `openai-api`) we pin temperature 0
69
+ // for judge calls (and record it); on a cli/subscription surface sampling params
70
+ // are surface-controlled and cannot be set, so temperature is null and `sampling`
71
+ // says so. `judgeModel` defaults to the fixed Haiku judge, whose surface is the
72
+ // Anthropic axis. Recorded into every receipt.
73
+ function judgeSettings(samples, judgeModel) {
74
+ const surface = surfaceForModel(judgeModel || 'claude-haiku-4-5');
75
+ if (surface === 'api' || surface === 'openai-api') {
73
76
  return { samples, temperature: 0, sampling: 'api-temperature-0', surface };
74
77
  }
75
78
  return { samples, temperature: null, sampling: 'surface-controlled', surface };
@@ -80,16 +83,16 @@ function judgeSettings(samples) {
80
83
  // transcript auditability, and optionally retained under --keep-transcripts).
81
84
  async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature }) {
82
85
  const prompt = buildJudgePrompt({ task, response, rubric });
83
- const { text } = await complete({ system: JUDGE_SYSTEM, prompt, model, maxTokens: 400, timeoutMs, temperature });
86
+ const { text, attempts, usage } = await complete({ system: JUDGE_SYSTEM, prompt, model, maxTokens: 400, timeoutMs, temperature });
84
87
  let parsed;
85
88
  try {
86
89
  parsed = extractJsonObject(text);
87
90
  } catch (_e) {
88
91
  // Unsalvageable judge output → conservative 0 (a judge that can't be parsed
89
92
  // must never silently "pass"), tagged so the caller can see it happened.
90
- return { score: 0, reason: 'judge output unparseable', unparsed: true, raw: String(text || '') };
93
+ return { score: 0, reason: 'judge output unparseable', unparsed: true, raw: String(text || ''), attempts: attempts || 1, usage };
91
94
  }
92
- return { score: clamp01(parsed.score), reason: String(parsed.reason || '').slice(0, 300), raw: String(text || '') };
95
+ return { score: clamp01(parsed.score), reason: String(parsed.reason || '').slice(0, 300), raw: String(text || ''), attempts: attempts || 1, usage };
93
96
  }
94
97
 
95
98
  // Grade a response N times and return the sampled distribution:
@@ -97,12 +100,25 @@ async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature
97
100
  // `mean` ± `stddev` is the per-case confidence band (raw spread of the N scores)
98
101
  // used by the borderline-outcome rule and per-case drift band-overlap logic.
99
102
  async function gradeSamples({ task, response, rubric, model, samples = 5, timeoutMs = 120000 }) {
100
- const settings = judgeSettings(samples);
103
+ const settings = judgeSettings(samples, model);
101
104
  const scores = [];
102
105
  const reasons = [];
103
106
  const rawTexts = [];
107
+ const usages = [];
108
+ let attemptsTotal = 0;
104
109
  for (let i = 0; i < samples; i++) {
105
- const r = await gradeOnce({ task, response, rubric, model, timeoutMs, temperature: settings.temperature === null ? undefined : settings.temperature });
110
+ let r;
111
+ try {
112
+ r = await gradeOnce({ task, response, rubric, model, timeoutMs, temperature: settings.temperature === null ? undefined : settings.temperature });
113
+ } catch (e) {
114
+ // A judge sample that persistently failed (e.g. timed out after retries):
115
+ // tag the error so the runner can charge for the spend and mark the whole
116
+ // case failed_timeout (a partial sample set must never become a band).
117
+ if (e && typeof e === 'object') { e.phase = 'judge'; e.judgeAttempts = attemptsTotal + (e.attempts || 1); }
118
+ throw e;
119
+ }
120
+ attemptsTotal += r.attempts || 1;
121
+ usages.push(r.usage || null);
106
122
  scores.push(r.score);
107
123
  reasons.push(r.reason);
108
124
  rawTexts.push(r.raw || '');
@@ -120,6 +136,12 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
120
136
  judge_settings: settings,
121
137
  model_id: model,
122
138
  rubric_hash: rubricHash(rubric),
139
+ attempts: attemptsTotal,
140
+ // v0.4: the measurement overhead of grading this one case — the SUM over all
141
+ // N judge calls. Recorded in the receipt as the case's `judge_usage` and
142
+ // EXCLUDED from every skill-value figure (lib/value.js): it is a cost we
143
+ // impose to measure, not a cost of running the skill.
144
+ usage: sumUsage(usages),
123
145
  };
124
146
  }
125
147
 
package/lib/models.js CHANGED
@@ -37,10 +37,24 @@ function loadRegistry(force) {
37
37
  }
38
38
 
39
39
  // Infer the model family from an id. Used to find a same-family predecessor when
40
- // the release trigger discovers a new model.
40
+ // the release trigger discovers a new model. Covers both providers:
41
+ // anthropic — claude-<family>-… → the family word (opus/sonnet/haiku/…)
42
+ // openai — gpt-<major.minor>-… → gpt-<major.minor> (so sol/terra/luna variants group)
41
43
  function familyOf(id) {
42
- const m = String(id).match(/claude-(fable|mythos|opus|sonnet|haiku)/i);
43
- return m ? m[1].toLowerCase() : 'unknown';
44
+ const s = String(id);
45
+ const c = s.match(/claude-(fable|mythos|opus|sonnet|haiku)/i);
46
+ if (c) return c[1].toLowerCase();
47
+ const g = s.match(/^(gpt-\d+(?:\.\d+)?)/i);
48
+ if (g) return g[1].toLowerCase();
49
+ const o = s.match(/^(o[1-9]+)/i);
50
+ if (o) return o[1].toLowerCase();
51
+ return 'unknown';
52
+ }
53
+
54
+ // Infer the provider from an id when the registry does not carry it (prefix-based,
55
+ // mirrors lib/provider.inferProvider — kept here too so models.js has no cycle).
56
+ function inferProviderFromId(id) {
57
+ return /^(gpt-|o[1-9]|chatgpt|codex|text-|davinci|omni)/i.test(String(id)) ? 'openai' : 'anthropic';
44
58
  }
45
59
 
46
60
  // Resolve a (possibly aliased) model id against the registry.
@@ -72,6 +86,36 @@ function isJudgeEligible(modelId) {
72
86
  return !!(entry && entry.judge_eligible);
73
87
  }
74
88
 
89
+ // Enforce the fixed-judge policy at a published-run boundary: throw unless the
90
+ // chosen judge is judge_eligible in the registry. This is what makes "attempting
91
+ // an OpenAI judge errors" true — no OpenAI model is judge_eligible, so a cross-
92
+ // provider report can only ever grade with the pinned Haiku judge. Callers on the
93
+ // published/report path (prepare-report, prepare-report-002) invoke this; the dev
94
+ // `run` command's self-judge is intentionally not gated by it.
95
+ function assertJudgeEligible(modelId) {
96
+ if (!isJudgeEligible(modelId)) {
97
+ const { id } = resolveRegistry(modelId);
98
+ const e = new Error(`judge model "${id}" is not judge_eligible: the judge is fixed to a judge_eligible model (claude-haiku-*); no OpenAI model may serve as the judge (see docs/judge-policy.html).`);
99
+ e.code = 'JUDGE_INELIGIBLE';
100
+ throw e;
101
+ }
102
+ return true;
103
+ }
104
+
105
+ // The provider for a model: the registry `provider` when registered, else inferred
106
+ // from the id prefix. Stamped into receipts as run.provider.
107
+ function providerForModel(modelId) {
108
+ const { entry, id } = resolveRegistry(modelId);
109
+ return (entry && entry.provider) || inferProviderFromId(id);
110
+ }
111
+
112
+ // Per-provider config block from the registry (config/models.json `providers`):
113
+ // { base_url, api_key_env, surfaces }. Null when the registry omits the provider.
114
+ function providerConfig(provider) {
115
+ const reg = loadRegistry();
116
+ return (reg.providers && reg.providers[provider]) || null;
117
+ }
118
+
75
119
  // The servable same-family predecessor of a model: the registered model in the
76
120
  // same family with the most recent `released` date strictly before this one
77
121
  // (falling back to any other same-family model). Null when the family has no
@@ -103,13 +147,18 @@ function addDiscoveredModel(id, { firstSeen = null, registryPath = REGISTRY_PATH
103
147
  const raw = JSON.parse(fs.readFileSync(registryPath, 'utf8'));
104
148
  if (raw.models.some((m) => m.id === id)) return null;
105
149
  const family = familyOf(id);
150
+ const provider = inferProviderFromId(id);
106
151
  // Best-effort tier from family; unknown families default to frontier pricing.
107
- const tier = family === 'haiku' ? 'cheap' : family === 'sonnet' ? 'standard' : 'frontier';
108
- const price = family === 'haiku' ? { input: 1.0, output: 5.0 }
109
- : family === 'sonnet' ? { input: 3.0, output: 15.0 }
110
- : { input: 5.0, output: 25.0 };
152
+ // A newly-discovered id gets a CONSERVATIVE (upper-bound) price so the budget
153
+ // guard never under-projects an unknown model — exact rates are set by hand
154
+ // when the model is reviewed for a published run.
155
+ const tier = /haiku|luna|mini|nano/.test(family) || /luna|mini|nano/.test(String(id)) ? 'cheap'
156
+ : /sonnet|terra/.test(family) ? 'standard' : 'frontier';
157
+ const price = provider === 'openai'
158
+ ? (tier === 'cheap' ? { input: 1.0, output: 6.0 } : tier === 'standard' ? { input: 2.5, output: 15.0 } : { input: 5.0, output: 30.0 })
159
+ : (family === 'haiku' ? { input: 1.0, output: 5.0 } : family === 'sonnet' ? { input: 3.0, output: 15.0 } : { input: 5.0, output: 25.0 });
111
160
  const entry = {
112
- id, family, provider: 'anthropic', released: firstSeen,
161
+ id, family, provider, released: firstSeen,
113
162
  input_price: price.input, output_price: price.output, tier,
114
163
  judge_eligible: false, auto_added: true,
115
164
  };
@@ -121,5 +170,6 @@ function addDiscoveredModel(id, { firstSeen = null, registryPath = REGISTRY_PATH
121
170
 
122
171
  module.exports = {
123
172
  loadRegistry, resolveRegistry, registryStatus, priceForModel, isJudgeEligible,
173
+ assertJudgeEligible, providerForModel, providerConfig, inferProviderFromId,
124
174
  familyOf, familyPredecessor, addDiscoveredModel, DEFAULT_PRICE, REGISTRY_PATH,
125
175
  };