openmerit 0.1.2 → 0.1.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +100 -18
- package/dist/cli.js +26 -153
- package/dist/daemon.js +56 -14
- package/dist/diagnostics.js +227 -0
- package/dist/llm.js +6 -2
- package/dist/pi-config.js +46 -0
- package/dist/pi-trials.js +30 -10
- package/dist/policy.js +71 -1
- package/dist/routes.js +15 -0
- package/dist/standalone.js +224 -0
- package/dist/store.js +42 -10
- package/dist/strategist.js +1 -1
- package/dist/trials.js +13 -4
- package/examples/task.example.json +1 -0
- package/extension/openmerit.ts +182 -29
- package/instructions/OPENMERIT.md +9 -0
- package/instructions/openmerit.policy.json +6 -2
- package/package.json +1 -1
- package/rules.md +4 -0
|
@@ -0,0 +1,224 @@
|
|
|
1
|
+
/** Provider-neutral standalone task trials executed through Pi routes. */
|
|
2
|
+
import { readFileSync } from "node:fs";
|
|
3
|
+
import { openRouterBenchmarkCandidates, relevantBenchmarks } from "./benchmarks.js";
|
|
4
|
+
import { fetchCatalog, saveSnapshot } from "./catalog.js";
|
|
5
|
+
import { paretoFrontier, pickBest, pickFallback } from "./frontier.js";
|
|
6
|
+
import { JUDGE_PREFS } from "./judge.js";
|
|
7
|
+
import { loadKey, PiCliChatClient } from "./llm.js";
|
|
8
|
+
import { availablePiRoutes, PiHarnessAdapter, recordedActiveTask, runPiTrial, scorePiRun } from "./pi-trials.js";
|
|
9
|
+
import { loadPolicy, providerAllowed } from "./policy.js";
|
|
10
|
+
import { buildRecommendation } from "./recommend.js";
|
|
11
|
+
import { applyRouteOverrides, enrichRouteEntry, routeCatalog, routeKey, routeLabel } from "./routes.js";
|
|
12
|
+
import { appendJsonl, paths, readJson, taskKey } from "./store.js";
|
|
13
|
+
import { pickNext, STRAT_PREFS } from "./strategist.js";
|
|
14
|
+
import { budgetOk, recordMeritSpend, recordTrialSpend, trialBudgetOk } from "./trials.js";
|
|
15
|
+
function optionalOpenRouterKey() {
|
|
16
|
+
try {
|
|
17
|
+
return loadKey();
|
|
18
|
+
}
|
|
19
|
+
catch {
|
|
20
|
+
return null;
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
function configuredRoutes(policy) {
|
|
24
|
+
const state = readJson(paths.harnessState(), {});
|
|
25
|
+
let routes = state.routes?.length ? state.routes : availablePiRoutes(undefined, policy.pi.provider_extensions);
|
|
26
|
+
if (state.currentRoute && !routes.some((route) => routeKey(route) === routeKey(state.currentRoute)))
|
|
27
|
+
routes = [state.currentRoute, ...routes];
|
|
28
|
+
routes = applyRouteOverrides(routes, policy.route_overrides);
|
|
29
|
+
const unique = new Map(routes.map((route) => [routeKey(route), route]));
|
|
30
|
+
const currentRoute = state.currentRoute
|
|
31
|
+
? applyRouteOverrides([state.currentRoute], policy.route_overrides)[0] : null;
|
|
32
|
+
return { routes: [...unique.values()], currentRoute };
|
|
33
|
+
}
|
|
34
|
+
function selectorKey(selector) {
|
|
35
|
+
if (!selector)
|
|
36
|
+
return null;
|
|
37
|
+
if (typeof selector === "string")
|
|
38
|
+
return selector;
|
|
39
|
+
const modelId = selector.modelId ?? selector.model_id;
|
|
40
|
+
return selector.provider && modelId ? `${selector.provider}:${modelId}` : null;
|
|
41
|
+
}
|
|
42
|
+
function matchingEntries(cat, selector) {
|
|
43
|
+
const exact = cat.get(selector);
|
|
44
|
+
if (exact)
|
|
45
|
+
return [exact];
|
|
46
|
+
return [...cat.values()].filter((entry) => entry.id === selector || entry.route?.meritId === selector ||
|
|
47
|
+
(entry.route && `${entry.route.provider}/${entry.route.modelId}` === selector));
|
|
48
|
+
}
|
|
49
|
+
function resolveInitial(cat, cfg, currentRoute) {
|
|
50
|
+
const exactKey = selectorKey(cfg.initial_route);
|
|
51
|
+
if (cfg.initial_route && !exactKey)
|
|
52
|
+
throw new Error("initial_route must include provider and modelId");
|
|
53
|
+
if (exactKey) {
|
|
54
|
+
const exact = cat.get(exactKey);
|
|
55
|
+
if (!exact)
|
|
56
|
+
throw new Error(`initial route ${exactKey} is not eligible in Pi`);
|
|
57
|
+
if (cfg.initial_model !== exact.id && cfg.initial_model !== exactKey)
|
|
58
|
+
throw new Error(`initial_route ${exactKey} resolves to ${exact.id}, not initial_model ${cfg.initial_model}`);
|
|
59
|
+
return exact;
|
|
60
|
+
}
|
|
61
|
+
const matches = matchingEntries(cat, cfg.initial_model);
|
|
62
|
+
if (matches.length === 0)
|
|
63
|
+
throw new Error(`initial_model ${cfg.initial_model} is not eligible in Pi`);
|
|
64
|
+
if (matches.length === 1)
|
|
65
|
+
return matches[0];
|
|
66
|
+
if (currentRoute) {
|
|
67
|
+
const current = matches.find((entry) => entry.route && routeKey(entry.route) === routeKey(currentRoute));
|
|
68
|
+
if (current)
|
|
69
|
+
return current;
|
|
70
|
+
}
|
|
71
|
+
throw new Error(`initial_model ${cfg.initial_model} has multiple Pi routes; set initial_route to one of: ` +
|
|
72
|
+
matches.map((entry) => routeKey(entry.route)).join(", "));
|
|
73
|
+
}
|
|
74
|
+
function resolveMeritRoute(cat, prefs, label) {
|
|
75
|
+
for (const pref of prefs) {
|
|
76
|
+
if (!pref)
|
|
77
|
+
continue;
|
|
78
|
+
const matches = matchingEntries(cat, pref);
|
|
79
|
+
if (matches.length)
|
|
80
|
+
return matches.sort((a, b) => (a.priceKnown === false ? 1 : 0) - (b.priceKnown === false ? 1 : 0) || a.price - b.price)[0];
|
|
81
|
+
}
|
|
82
|
+
for (const pref of prefs) {
|
|
83
|
+
if (!pref)
|
|
84
|
+
continue;
|
|
85
|
+
const fuzzy = [...cat.values()].find((entry) => entry.id.includes(pref));
|
|
86
|
+
if (fuzzy)
|
|
87
|
+
return fuzzy;
|
|
88
|
+
}
|
|
89
|
+
const fallback = [...cat.values()].sort((a, b) => (a.priceKnown === false ? 1 : 0) - (b.priceKnown === false ? 1 : 0) || a.price - b.price)[0];
|
|
90
|
+
if (fallback)
|
|
91
|
+
return fallback;
|
|
92
|
+
throw new Error(`could not resolve ${label} route from Pi's eligible models`);
|
|
93
|
+
}
|
|
94
|
+
async function buildCatalog(routes, key) {
|
|
95
|
+
let cat = routeCatalog(routes);
|
|
96
|
+
if (!key)
|
|
97
|
+
return cat;
|
|
98
|
+
try {
|
|
99
|
+
const live = await fetchCatalog(key);
|
|
100
|
+
saveSnapshot(live);
|
|
101
|
+
cat = new Map([...cat].map(([id, entry]) => [id,
|
|
102
|
+
entry.route?.provider === "openrouter" ? enrichRouteEntry(entry, live.get(entry.id)) : entry]));
|
|
103
|
+
}
|
|
104
|
+
catch (error) {
|
|
105
|
+
console.log(`OpenRouter enrichment unavailable: ${error.message}`);
|
|
106
|
+
}
|
|
107
|
+
return cat;
|
|
108
|
+
}
|
|
109
|
+
export async function runStandaloneTrial(taskFile, rounds) {
|
|
110
|
+
const cfg = JSON.parse(readFileSync(taskFile, "utf8"));
|
|
111
|
+
if (!cfg.task?.trim() || !cfg.eval?.trim() || !cfg.initial_model?.trim())
|
|
112
|
+
throw new Error("task JSON requires task, eval, and initial_model");
|
|
113
|
+
if (!Number.isInteger(rounds) || rounds < 1)
|
|
114
|
+
throw new Error("--rounds must be a positive integer");
|
|
115
|
+
const policy = loadPolicy(paths.policy());
|
|
116
|
+
const key = optionalOpenRouterKey();
|
|
117
|
+
const { routes, currentRoute } = configuredRoutes(policy);
|
|
118
|
+
let cat = await buildCatalog(routes, key);
|
|
119
|
+
for (const [id, entry] of [...cat]) {
|
|
120
|
+
if (!entry.route || !providerAllowed(policy, entry.id, entry.route.provider) ||
|
|
121
|
+
(entry.priceKnown !== false && entry.price > (cfg.max_usd_per_m ?? policy.max_usd_per_m)))
|
|
122
|
+
cat.delete(id);
|
|
123
|
+
}
|
|
124
|
+
if (cat.size === 0)
|
|
125
|
+
throw new Error("Pi exposes no model routes allowed by the current policy");
|
|
126
|
+
const initial = resolveInitial(cat, cfg, currentRoute);
|
|
127
|
+
const judge = resolveMeritRoute(cat, [cfg.judge_model, policy.judge_model, ...JUDGE_PREFS, initial.id], "judge");
|
|
128
|
+
const strategist = resolveMeritRoute(cat, [cfg.strategist_model, policy.strategist_model, ...STRAT_PREFS, initial.id], "strategist");
|
|
129
|
+
const client = new PiCliChatClient(undefined, recordMeritSpend, policy.pi.provider_extensions);
|
|
130
|
+
const harness = new PiHarnessAdapter(policy.pi.provider_extensions);
|
|
131
|
+
const { category, benchmarks } = relevantBenchmarks(cfg.task);
|
|
132
|
+
let benchmarkCandidates = [];
|
|
133
|
+
if (key) {
|
|
134
|
+
try {
|
|
135
|
+
benchmarkCandidates = await openRouterBenchmarkCandidates(key, category, new Map([...cat.values()].map((entry) => [entry.id, entry])));
|
|
136
|
+
}
|
|
137
|
+
catch (error) {
|
|
138
|
+
console.log(`OpenRouter benchmark shortlist unavailable: ${error.message}`);
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
console.log(`routes: ${cat.size} eligible in Pi${key ? " (OpenRouter enrichment enabled)" : ""}`);
|
|
142
|
+
console.log(`initial: ${routeLabel(initial.route)} | judge: ${routeLabel(judge.route)} | ` +
|
|
143
|
+
`strategist: ${routeLabel(strategist.route)} | benchmarks: ${benchmarks.join(", ")}`);
|
|
144
|
+
const tKey = taskKey(cfg.task);
|
|
145
|
+
const tried = new Set();
|
|
146
|
+
const results = [];
|
|
147
|
+
const failedProviders = new Set();
|
|
148
|
+
for (let i = 0; i < rounds; i++) {
|
|
149
|
+
const daily = budgetOk(policy);
|
|
150
|
+
if (!daily.ok) {
|
|
151
|
+
console.log(`budget: ${daily.reason}; stopping`);
|
|
152
|
+
break;
|
|
153
|
+
}
|
|
154
|
+
let entry;
|
|
155
|
+
let why;
|
|
156
|
+
if (i === 0) {
|
|
157
|
+
entry = initial;
|
|
158
|
+
why = "initial route";
|
|
159
|
+
}
|
|
160
|
+
else {
|
|
161
|
+
const pick = await pickNext(key ?? "", strategist.route, cfg.task, cfg.eval, benchmarks, results, cat, tried, cfg.max_usd_per_m ?? policy.max_usd_per_m, failedProviders, benchmarkCandidates, client);
|
|
162
|
+
if (!pick)
|
|
163
|
+
break;
|
|
164
|
+
entry = pick.route ? cat.get(routeKey(pick.route)) : cat.get(pick.model);
|
|
165
|
+
why = pick.why;
|
|
166
|
+
}
|
|
167
|
+
const route = entry.route;
|
|
168
|
+
const keyForRoute = routeKey(route);
|
|
169
|
+
if (tried.has(keyForRoute))
|
|
170
|
+
break;
|
|
171
|
+
if (i > 0) {
|
|
172
|
+
const admission = trialBudgetOk(policy, entry, cfg.task.length);
|
|
173
|
+
if (!admission.ok) {
|
|
174
|
+
tried.add(keyForRoute);
|
|
175
|
+
console.log(`[round ${i + 1}/${rounds}] ${routeLabel(route)} skipped: ${admission.reason}`);
|
|
176
|
+
continue;
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
console.log(`[round ${i + 1}/${rounds}] ${routeLabel(route)} (${why}) ...`);
|
|
180
|
+
const observed = i === 0 ? recordedActiveTask(cfg.task, route) : null;
|
|
181
|
+
if (i === 0)
|
|
182
|
+
console.log(observed
|
|
183
|
+
? " using the exact provider route from Pi's active session trace"
|
|
184
|
+
: " active trace unavailable; running the initial route in a fresh Pi session");
|
|
185
|
+
const outcome = observed
|
|
186
|
+
? await scorePiRun(key ?? "", judge.route, cfg.task, cfg.eval, entry.id, entry, observed, "trace", [], client)
|
|
187
|
+
: await runPiTrial(key ?? "", judge.route, cfg.task, cfg.eval, entry.id, entry, cfg.cwd ?? process.cwd(), [], [], harness, client);
|
|
188
|
+
tried.add(keyForRoute);
|
|
189
|
+
results.push(outcome.point);
|
|
190
|
+
// Reusing the active Pi answer is observation, not a new candidate call.
|
|
191
|
+
if (!observed)
|
|
192
|
+
recordTrialSpend(outcome.costUsd);
|
|
193
|
+
appendJsonl(paths.trials(), { ...outcome.point, taskKey: tKey });
|
|
194
|
+
if (outcome.sessionId)
|
|
195
|
+
console.log(` pi trace session: ${outcome.sessionId}`);
|
|
196
|
+
if (outcome.error) {
|
|
197
|
+
failedProviders.add(route.provider);
|
|
198
|
+
console.log(` FAILED: ${outcome.error.slice(0, 160)}`);
|
|
199
|
+
}
|
|
200
|
+
else {
|
|
201
|
+
const price = outcome.point.priceKnown === false ? "price unknown" : `$${outcome.point.price}/M`;
|
|
202
|
+
console.log(` score=${outcome.point.score.toFixed(2)} ${price} ` +
|
|
203
|
+
`run=$${outcome.costUsd.toFixed(4)} ${(outcome.point.why ?? "").slice(0, 100)}`);
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
const scored = results.filter((point) => point.score > 0);
|
|
207
|
+
if (scored.length === 0)
|
|
208
|
+
throw new Error("no trials completed");
|
|
209
|
+
const frontier = paretoFrontier(scored);
|
|
210
|
+
const best = pickBest(scored);
|
|
211
|
+
const fallback = pickFallback(scored, best, policy.fallback.min_score);
|
|
212
|
+
console.log("\n=== pareto frontier (quality up, price and latency down) ===");
|
|
213
|
+
for (const point of frontier)
|
|
214
|
+
console.log(` ${(point.route ? routeLabel(point.route) : point.model).padEnd(52)} ` +
|
|
215
|
+
`score=${point.score.toFixed(2)} ${point.priceKnown === false ? "price unknown" : `$${point.price.toFixed(2)}/M`}`);
|
|
216
|
+
console.log(`\nBEST FIT : ${best.route ? routeLabel(best.route) : best.model}`);
|
|
217
|
+
console.log(fallback ? `FALLBACK : ${fallback.route ? routeLabel(fallback.route) : fallback.model}` : "FALLBACK : n/a");
|
|
218
|
+
const recommendation = buildRecommendation(tKey, cfg.task.slice(0, 120), initial.id, results, policy, null, initial.route);
|
|
219
|
+
if (recommendation) {
|
|
220
|
+
appendJsonl(paths.recommendations(), recommendation);
|
|
221
|
+
console.log(`\nrecommendation -> ${paths.recommendations()} (auto=${recommendation.policy.autoApply})`);
|
|
222
|
+
}
|
|
223
|
+
return { points: results, recommendation };
|
|
224
|
+
}
|
package/dist/store.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
/** State directory layout + JSONL persistence. All state lives under OPENMERIT_HOME (default ~/.openmerit). */
|
|
2
2
|
import { createHash } from "node:crypto";
|
|
3
|
-
import { appendFileSync, existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
|
|
3
|
+
import { appendFileSync, existsSync, mkdirSync, readFileSync, renameSync, rmSync, writeFileSync, } from "node:fs";
|
|
4
4
|
import { homedir } from "node:os";
|
|
5
5
|
import { join } from "node:path";
|
|
6
6
|
export function stateDir() {
|
|
@@ -38,20 +38,52 @@ export function appendJsonl(file, obj) {
|
|
|
38
38
|
mkdirSync(join(file, ".."), { recursive: true });
|
|
39
39
|
appendFileSync(file, JSON.stringify(obj) + "\n");
|
|
40
40
|
}
|
|
41
|
+
/**
|
|
42
|
+
* Read append-only state without letting one interrupted or malformed line
|
|
43
|
+
* hide the remaining history. Diagnostics can surface invalid line numbers.
|
|
44
|
+
*/
|
|
45
|
+
export function readJsonlReport(file) {
|
|
46
|
+
if (!existsSync(file))
|
|
47
|
+
return { records: [], invalidLines: [] };
|
|
48
|
+
const records = [];
|
|
49
|
+
const invalidLines = [];
|
|
50
|
+
for (const [index, line] of readFileSync(file, "utf8").split("\n").entries()) {
|
|
51
|
+
if (!line.trim())
|
|
52
|
+
continue;
|
|
53
|
+
try {
|
|
54
|
+
records.push(JSON.parse(line));
|
|
55
|
+
}
|
|
56
|
+
catch {
|
|
57
|
+
invalidLines.push(index + 1);
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
return { records, invalidLines };
|
|
61
|
+
}
|
|
41
62
|
export function readJsonl(file) {
|
|
63
|
+
return readJsonlReport(file).records;
|
|
64
|
+
}
|
|
65
|
+
export function readJsonReport(file, fallback) {
|
|
42
66
|
if (!existsSync(file))
|
|
43
|
-
return
|
|
44
|
-
|
|
45
|
-
.
|
|
46
|
-
|
|
47
|
-
|
|
67
|
+
return { value: fallback, exists: false, valid: true };
|
|
68
|
+
try {
|
|
69
|
+
return { value: JSON.parse(readFileSync(file, "utf8")), exists: true, valid: true };
|
|
70
|
+
}
|
|
71
|
+
catch (error) {
|
|
72
|
+
return { value: fallback, exists: true, valid: false, error: error.message };
|
|
73
|
+
}
|
|
48
74
|
}
|
|
49
75
|
export function readJson(file, fallback) {
|
|
50
|
-
|
|
51
|
-
return fallback;
|
|
52
|
-
return JSON.parse(readFileSync(file, "utf8"));
|
|
76
|
+
return readJsonReport(file, fallback).value;
|
|
53
77
|
}
|
|
54
78
|
export function writeJson(file, obj) {
|
|
55
79
|
mkdirSync(join(file, ".."), { recursive: true });
|
|
56
|
-
|
|
80
|
+
const temp = `${file}.tmp-${process.pid}-${Date.now()}`;
|
|
81
|
+
try {
|
|
82
|
+
writeFileSync(temp, JSON.stringify(obj, null, 2) + "\n");
|
|
83
|
+
renameSync(temp, file);
|
|
84
|
+
}
|
|
85
|
+
finally {
|
|
86
|
+
if (existsSync(temp))
|
|
87
|
+
rmSync(temp, { force: true });
|
|
88
|
+
}
|
|
57
89
|
}
|
package/dist/strategist.js
CHANGED
|
@@ -21,7 +21,7 @@ ROUTES (route id | logical model | blended $/1M tokens | ctx):
|
|
|
21
21
|
/** Pick the next model to trial, or null when the catalog is exhausted. */
|
|
22
22
|
export async function pickNext(key, stratModel, task, rubric, benchmarks, results, cat, tried, maxPrice, failedVendors, extraCandidates, client = directChatClient(key)) {
|
|
23
23
|
const entryKey = (c) => c.route ? routeKey(c.route) : c.id;
|
|
24
|
-
const ok = (c) =>
|
|
24
|
+
const ok = (c) => c.priceKnown !== false && c.price <= maxPrice &&
|
|
25
25
|
!tried.has(entryKey(c)) &&
|
|
26
26
|
c.ctx >= 4096 &&
|
|
27
27
|
!failedVendors.has(c.route?.provider ?? c.id.split("/")[0]);
|
package/dist/trials.js
CHANGED
|
@@ -1,12 +1,15 @@
|
|
|
1
1
|
/** Shadow trials: run candidate models against observed/declared tasks on the background track. */
|
|
2
2
|
import { directChatClient } from "./llm.js";
|
|
3
3
|
import { judge, parseObj } from "./judge.js";
|
|
4
|
-
import { paths,
|
|
4
|
+
import { paths, readJsonReport, writeJson } from "./store.js";
|
|
5
5
|
function today() {
|
|
6
6
|
return new Date().toISOString().slice(0, 10);
|
|
7
7
|
}
|
|
8
8
|
export function budgetOk(policy) {
|
|
9
|
-
const
|
|
9
|
+
const report = readJsonReport(paths.ledger(), { date: today(), trials: 0, usd: 0 });
|
|
10
|
+
if (!report.valid)
|
|
11
|
+
return { ok: false, reason: "trial ledger is malformed; run `openmerit doctor`" };
|
|
12
|
+
const ledger = report.value;
|
|
10
13
|
if (ledger.date !== today())
|
|
11
14
|
return { ok: true }; // new day resets
|
|
12
15
|
if (ledger.trials >= policy.budgets.max_trials_per_day)
|
|
@@ -31,7 +34,10 @@ export function trialBudgetOk(policy, entry, inputChars) {
|
|
|
31
34
|
return { ok: true, estimatedUsd };
|
|
32
35
|
}
|
|
33
36
|
export function recordTrialSpend(usd) {
|
|
34
|
-
|
|
37
|
+
const report = readJsonReport(paths.ledger(), { date: today(), trials: 0, usd: 0 });
|
|
38
|
+
if (!report.valid)
|
|
39
|
+
throw new Error("trial ledger is malformed; refusing to reset spend");
|
|
40
|
+
let ledger = report.value;
|
|
35
41
|
if (ledger.date !== today())
|
|
36
42
|
ledger = { date: today(), trials: 0, usd: 0 };
|
|
37
43
|
ledger.trials += 1;
|
|
@@ -42,7 +48,10 @@ export function recordTrialSpend(usd) {
|
|
|
42
48
|
export function recordMeritSpend(usd) {
|
|
43
49
|
if (!(usd > 0))
|
|
44
50
|
return;
|
|
45
|
-
|
|
51
|
+
const report = readJsonReport(paths.ledger(), { date: today(), trials: 0, usd: 0 });
|
|
52
|
+
if (!report.valid)
|
|
53
|
+
throw new Error("trial ledger is malformed; refusing to reset spend");
|
|
54
|
+
let ledger = report.value;
|
|
46
55
|
if (ledger.date !== today())
|
|
47
56
|
ledger = { date: today(), trials: 0, usd: 0 };
|
|
48
57
|
ledger.usd = Math.round((ledger.usd + usd) * 1e6) / 1e6;
|
|
@@ -2,5 +2,6 @@
|
|
|
2
2
|
"task": "Write a Python function `def word_ladder(begin, end, words)` that returns the SHORTEST transformation sequence from `begin` to `end`, where consecutive words differ by exactly one character and every intermediate word must be in `words`. Return [] if no path exists. If begin == end return [begin]. All words are the same length, lowercase a-z. Output only runnable code, no explanation.",
|
|
3
3
|
"eval": "Score 1.0 requires ALL of: (a) algorithm guaranteed to find a shortest path (BFS or equivalent, NOT DFS/greedy); (b) returned path starts with begin and ends with end with all intermediates in words; (c) returns [] when impossible; (d) begin==end returns [begin]; (e) begin need not be in words; (f) syntactically valid, runnable Python, no external imports, function named word_ladder; (g) output contains only code. Deduct ~0.2 per missing criterion; non-shortest-path algorithms score at most 0.5.",
|
|
4
4
|
"initial_model": "openai/gpt-4o-mini",
|
|
5
|
+
"initial_route": "openai:gpt-4o-mini",
|
|
5
6
|
"max_usd_per_m": 20
|
|
6
7
|
}
|