ruvnet-brain 4.5.3 → 4.5.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/bin/install.mjs +148 -31
- package/config/model-router/catalog.template.json +126 -52
- package/config/model-router/policy.default.mjs +94 -75
- package/config/model-router/qualification-contract.json +124 -0
- package/config/model-router/routing-eval-cases.json +275 -0
- package/config/model-router/routing-policy.template.json +76 -0
- package/config/model-router/weekly-analyst-instruction.md +60 -0
- package/data/model-catalog.json +44 -49
- package/package.json +3 -1
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/hooks/codex-hooks.json +40 -3
- package/plugin/hooks/hook-contracts.json +218 -19
- package/plugin/hooks/hooks.json +51 -2
- package/plugin/scripts/agentdb-recall.mjs +101 -30
- package/plugin/scripts/codex-hook-adapter.mjs +18 -9
- package/plugin/scripts/continuity-hook-policy.mjs +9 -0
- package/plugin/scripts/continuity-journal.mjs +33 -33
- package/plugin/scripts/ground-ruvnet.sh +5 -5
- package/plugin/scripts/hook-shim.mjs +4 -30
- package/plugin/scripts/project-capture-queue.mjs +333 -0
- package/plugin/scripts/project-progression-contract.mjs +1 -1
- package/plugin/scripts/project-progression-hook.mjs +3 -3
- package/plugin/scripts/project-progression-producer.mjs +53 -28
- package/plugin/scripts/project-progression-session-start.mjs +44 -4
- package/plugin/scripts/project-progression-store.mjs +14 -0
- package/plugin/scripts/project-transition-hook.mjs +204 -0
- package/plugin/scripts/session-snapshot-hook.mjs +44 -272
- package/plugin/scripts/session-start-budget.mjs +2 -2
- package/plugin/scripts/turn-outcome-capture.mjs +125 -47
- package/plugin/scripts/turn-transport-journal.mjs +106 -0
- package/scripts/codex-hook-trust-reconcile.mjs +247 -0
- package/scripts/codex-routed.sh +3 -36
- package/scripts/goldie-weekly.sh +8 -64
- package/scripts/metaharness-router.mjs +7 -1
- package/scripts/model-analyst-sandbox.mjs +54 -0
- package/scripts/model-currency-evidence.mjs +139 -0
- package/scripts/model-currency.mjs +230 -0
- package/scripts/model-native-catalog.mjs +111 -0
- package/scripts/model-native-qualification.mjs +251 -0
- package/scripts/model-router-agent-hook.mjs +136 -0
- package/scripts/model-router-dispatch.mjs +161 -0
- package/scripts/model-router-engine.mjs +155 -104
- package/scripts/model-routing-eval.mjs +108 -0
- package/scripts/model-routing-gateway.mjs +420 -0
- package/scripts/model-routing-launchers.mjs +174 -0
- package/scripts/model-routing-policy-promotion.mjs +203 -0
- package/scripts/model-weekly-analyst.mjs +299 -0
- package/scripts/model-weekly-assessment.mjs +91 -0
- package/scripts/model-weekly-cycle.mjs +183 -0
- package/scripts/model-weekly-qualification.mjs +362 -0
- package/scripts/native-subscription-usage.mjs +57 -0
- package/scripts/release-qualification-contract.mjs +54 -0
- package/scripts/security-guidance-codex-compat.mjs +142 -0
- package/scripts/user-model-prompt-hook.mjs +69 -0
package/scripts/codex-routed.sh
CHANGED
|
@@ -1,38 +1,5 @@
|
|
|
1
1
|
#!/usr/bin/env bash
|
|
2
|
-
#
|
|
3
|
-
|
|
4
|
-
# per-prompt model-selection path available to it. Part of the MetaHarness router (see
|
|
5
|
-
# scripts/model-router-engine.mjs, the harness-neutral prompt -> {model, reason} decision
|
|
6
|
-
# engine). This script only CONSULTS the engine and LAUNCHES codex — all selection logic
|
|
7
|
-
# (features, policy, pricing) lives in the engine, not here.
|
|
8
|
-
#
|
|
9
|
-
# Contract: never block Stuart's launch. If the engine errors, times out, or returns a
|
|
10
|
-
# null model, fall straight through to plain `codex` with no --model flag — a missing
|
|
11
|
-
# routing decision must never be worse than no routing at all.
|
|
12
|
-
set -uo pipefail
|
|
2
|
+
# One routed native noninteractive worker. No silent fallback; model and effort are enforced together.
|
|
3
|
+
set -euo pipefail
|
|
13
4
|
DIR="$(cd "$(dirname "$0")" && pwd)"
|
|
14
|
-
|
|
15
|
-
DECISION=$(node "$DIR/model-router-engine.mjs" --harness codex --prompt "$*" --json 2>/dev/null)
|
|
16
|
-
|
|
17
|
-
# Extract .model + a short .reason without a jq dependency (Stuart directive). python3 ships
|
|
18
|
-
# with macOS by default and its json module handles arbitrary reason text (quotes, unicode)
|
|
19
|
-
# more safely than hand-rolled JS string escaping in a node -e one-liner would. Single call,
|
|
20
|
-
# tab-delimited output, so we only shell out once per launch.
|
|
21
|
-
IFS=$'\t' read -r M REASON <<< "$(printf '%s' "$DECISION" | python3 -c '
|
|
22
|
-
import json, sys
|
|
23
|
-
try:
|
|
24
|
-
d = json.load(sys.stdin)
|
|
25
|
-
m = d.get("model") or ""
|
|
26
|
-
r = (d.get("reason") or "").replace("\n", " ").replace("\t", " ")[:80]
|
|
27
|
-
print(f"{m}\t{r}")
|
|
28
|
-
except Exception:
|
|
29
|
-
print("\t")
|
|
30
|
-
' 2>/dev/null)"
|
|
31
|
-
|
|
32
|
-
if [ -z "${M:-}" ] || [ "$M" = "None" ]; then
|
|
33
|
-
# Engine failed / no policy resolved a model — never block, just launch codex unrouted.
|
|
34
|
-
exec codex "$@"
|
|
35
|
-
fi
|
|
36
|
-
|
|
37
|
-
printf '\x1b[2m🧭 codex-routed → %s (%s)\x1b[0m\n' "$M" "$REASON"
|
|
38
|
-
exec codex --model "$M" "$@"
|
|
5
|
+
exec node "$DIR/model-router-dispatch.mjs" --harness codex -- "$@"
|
package/scripts/goldie-weekly.sh
CHANGED
|
@@ -1,67 +1,11 @@
|
|
|
1
1
|
#!/bin/bash
|
|
2
|
-
#
|
|
3
|
-
#
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
# drift flags + radar + the dated brief. Pure data, no LLM, no key needed.
|
|
10
|
-
# 2. JUDGMENT (headless Claude, subscription-covered): answers the brief's three standing
|
|
11
|
-
# questions (bucket count, best-model-per-bucket per public evals, radar adoption) with real
|
|
12
|
-
# web research, APPENDED to the same brief as a PROPOSAL — never auto-applied to policy.mjs.
|
|
13
|
-
# Skippable (GOLDIE_SKIP_JUDGMENT=1) and its failure never hides layer 1's result.
|
|
14
|
-
set -u
|
|
15
|
-
cd /Users/stuartkerr/Code/ruvnet-brain || exit 1
|
|
16
|
-
mkdir -p logs
|
|
17
|
-
LOG=logs/goldie.log
|
|
18
|
-
TODAY=$(date +%F)
|
|
19
|
-
BRIEF="$HOME/.claude/model-router/goldie/$TODAY.md"
|
|
20
|
-
|
|
21
|
-
echo "===== goldie-weekly — $(date -u +%FT%TZ) =====" >> "$LOG"
|
|
22
|
-
|
|
23
|
-
# ── Layer 1: deterministic refresh (must succeed for the run to count) ──
|
|
24
|
-
if ! /usr/local/bin/node scripts/goldie-research.mjs >> "$LOG" 2>&1; then
|
|
25
|
-
sh scripts/notify.sh "🔴 Goldie FAILED — model catalog is going stale" \
|
|
26
|
-
"goldie-research.mjs could not produce this week's brief (OpenRouter fetch or catalog write failed). The router is now running on last week's picture. See logs/goldie.log." \
|
|
27
|
-
urgent "rotating_light" || true
|
|
2
|
+
# Existing Goldie launchd label/calendar stays unchanged. Run installed, guarded native cycle;
|
|
3
|
+
# policy promotion is a separate evidence-qualified boundary. No legacy model/API proposer.
|
|
4
|
+
set -eu
|
|
5
|
+
CYCLE="$HOME/.claude/model-router/bin/model-weekly-cycle.mjs"
|
|
6
|
+
NODE_BINARY="${RUVNET_NODE_BINARY:-/opt/homebrew/opt/node@24/bin/node}"
|
|
7
|
+
if [ ! -f "$CYCLE" ] || [ ! -x "$NODE_BINARY" ]; then
|
|
8
|
+
echo 'Weekly model cycle is not installed; prior policy retained.' >&2
|
|
28
9
|
exit 1
|
|
29
10
|
fi
|
|
30
|
-
|
|
31
|
-
# ── Layer 2: judgment (best-effort, never blocks; appends to the brief) ──
|
|
32
|
-
JUDGMENT="skipped"
|
|
33
|
-
if [ "${GOLDIE_SKIP_JUDGMENT:-0}" != "1" ] && command -v claude >/dev/null 2>&1; then
|
|
34
|
-
PROMPT="You are Goldie, the weekly model-landscape researcher for a prompt->model router.
|
|
35
|
-
Read $BRIEF (this week's deterministic data) and ~/.claude/model-router/catalog.json (the candidate
|
|
36
|
-
catalog) and ~/.claude/model-router/policy.default.mjs (the current placeholder policy). Then use web
|
|
37
|
-
search on the current public evaluations (Artificial Analysis, LMArena, SWE-bench and similar) to
|
|
38
|
-
answer the brief's three standing questions with sources and dates:
|
|
39
|
-
(1) how many BUCKETS should prompts be classified into and what are they;
|
|
40
|
-
(2) the best model per bucket right now on capability-per-cost-per-speed, split into: covered by a
|
|
41
|
-
Claude Max subscription (claude-code harness), covered by a ChatGPT/Codex subscription (codex
|
|
42
|
-
harness), and cheapest-capable OpenRouter API model;
|
|
43
|
-
(3) whether any radar model in the brief deserves wiring up, and what that requires.
|
|
44
|
-
Rules: cite sources with dates for every claim; distinguish MEASURED numbers from vendor claims;
|
|
45
|
-
recommendations are PROPOSALS for catalog.json/policy.mjs — do not edit any file. End with a
|
|
46
|
-
'## Proposed policy changes' section in plain, reviewable prose.
|
|
47
|
-
Write your full answer to stdout as markdown."
|
|
48
|
-
# env -u ANTHROPIC_API_KEY: a stray/stale API key in the environment makes headless claude bill
|
|
49
|
-
# (or fail on) the API instead of riding the Claude Max login — the exact "spend where the
|
|
50
|
-
# subscription is free" mistake this system exists to kill. Found live 2026-07-12: an invalid
|
|
51
|
-
# inherited key failed the whole judgment layer with "Invalid API key". Subscription, always.
|
|
52
|
-
if OUT=$(timeout 900 env -u ANTHROPIC_API_KEY claude -p "$PROMPT" --model sonnet --allowed-tools "WebSearch,WebFetch,Read" 2>>"$LOG"); then
|
|
53
|
-
{ echo ""; echo "---"; echo ""; echo "# Judgment layer (headless Claude, $(date -u +%FT%TZ))"; echo ""; echo "$OUT"; } >> "$BRIEF"
|
|
54
|
-
JUDGMENT="ok"
|
|
55
|
-
else
|
|
56
|
-
JUDGMENT="FAILED (see logs/goldie.log)"
|
|
57
|
-
{ echo ""; echo "---"; echo ""; echo "# Judgment layer: FAILED this week ($(date -u +%FT%TZ)) — deterministic data above still fresh."; } >> "$BRIEF"
|
|
58
|
-
fi
|
|
59
|
-
fi
|
|
60
|
-
|
|
61
|
-
# ── Exactly one summary push per run — success included (silence is never a signal) ──
|
|
62
|
-
HEADLINES=$(grep -E "PRICE DRIFT|NOT FOUND" "$BRIEF" | head -3)
|
|
63
|
-
sh scripts/notify.sh "🧭 Goldie ran — model catalog refreshed" \
|
|
64
|
-
"Weekly brief: $BRIEF. Judgment layer: $JUDGMENT.${HEADLINES:+ ATTENTION: $HEADLINES}" \
|
|
65
|
-
default "compass" || true
|
|
66
|
-
echo "===== goldie-weekly done (judgment: $JUDGMENT) =====" >> "$LOG"
|
|
67
|
-
exit 0
|
|
11
|
+
exec "$NODE_BINARY" "$CYCLE" --router-dir "$HOME/.claude/model-router"
|
|
@@ -142,7 +142,13 @@ export async function route(prompt, candidates, profile, { qualityBar = 0.7, k =
|
|
|
142
142
|
}
|
|
143
143
|
|
|
144
144
|
const prices = effectivePrices(candidates, profile);
|
|
145
|
-
const
|
|
145
|
+
const allowed = new Set(candidates.map((m) => m.id));
|
|
146
|
+
const constrainedRows = rows.map((row) => ({ ...row,
|
|
147
|
+
scores: Object.fromEntries(Object.entries(row.scores).filter(([id]) => allowed.has(id))),
|
|
148
|
+
})).filter((row) => Object.keys(row.scores).length);
|
|
149
|
+
if (constrainedRows.length < MIN_LABELS) return { routedBy: 'COLD-START',
|
|
150
|
+
reason: 'Insufficient labels for policy-eligible models', labels: constrainedRows.length };
|
|
151
|
+
const router = mod.Router.fromExamples(constrainedRows, prices, { k, qualityBar });
|
|
146
152
|
const pick = router.route(await embed(prompt));
|
|
147
153
|
|
|
148
154
|
return {
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
// DISTINCT-FROM: native-subscription-usage.mjs — isolated child tool-denial trust, never an inference turn.
|
|
2
|
+
import fs from 'node:fs';
|
|
3
|
+
import os from 'node:os';
|
|
4
|
+
import path from 'node:path';
|
|
5
|
+
import { spawn } from 'node:child_process';
|
|
6
|
+
import { fileURLToPath } from 'node:url';
|
|
7
|
+
export const TOOL_DENIAL = { hookSpecificOutput: { hookEventName: 'PreToolUse', permissionDecision: 'deny', permissionDecisionReason: 'Weekly analyst tools are prohibited; analyse supplied evidence only.' } };
|
|
8
|
+
const quote = (value) => `'${value.replaceAll("'", "'\\''")}'`;
|
|
9
|
+
export function createAnalystHome(runDir) {
|
|
10
|
+
const requestedHome = path.join(runDir, 'native-home'); fs.mkdirSync(requestedHome, { mode: 0o700 }); const home = fs.realpathSync(requestedHome);
|
|
11
|
+
const source = path.join(os.homedir(), '.codex');
|
|
12
|
+
const auth = fs.realpathSync(path.join(source, 'auth.json'));
|
|
13
|
+
if (!fs.statSync(auth).isFile()) throw new Error('Canonical native OAuth file unavailable');
|
|
14
|
+
fs.symlinkSync(auth, path.join(home, 'auth.json'));
|
|
15
|
+
fs.copyFileSync(path.join(source, 'models_cache.json'), path.join(home, 'models_cache.json')); fs.chmodSync(path.join(home, 'models_cache.json'), 0o600);
|
|
16
|
+
fs.writeFileSync(path.join(home, 'config.toml'), '', { mode: 0o600 });
|
|
17
|
+
const command = `${quote(process.execPath)} ${quote(fileURLToPath(import.meta.url))} --deny`;
|
|
18
|
+
fs.writeFileSync(path.join(home, 'hooks.json'), JSON.stringify({ hooks: { PreToolUse: [{ matcher: '.*', hooks: [{ type: 'command', command, timeout: 5 }] }] } }), { mode: 0o600 });
|
|
19
|
+
return { home, command };
|
|
20
|
+
}
|
|
21
|
+
export function sameNativeConfigPath(actual, expected, platform = process.platform) {
|
|
22
|
+
if (typeof actual !== 'string' || !actual) return false;
|
|
23
|
+
const paths = platform === 'win32' ? path.win32 : path.posix;
|
|
24
|
+
const normalize = value => { const resolved = paths.resolve(value); return platform === 'win32' ? resolved.toLowerCase() : resolved; };
|
|
25
|
+
return normalize(actual) === normalize(expected);
|
|
26
|
+
}
|
|
27
|
+
export async function trustAnalystDenial({ home, command, env = process.env, spawnHost = spawn, timeoutMs = 8000 }) {
|
|
28
|
+
const child = spawnHost('codex', ['app-server', '--strict-config', '-c', 'features.plugins=false', '-c', 'service_tier="default"', '--listen', 'stdio://'], { cwd: home, env: { ...env, CODEX_HOME: home }, stdio: ['pipe', 'pipe', 'pipe'], shell: false });
|
|
29
|
+
let id = 0, buffer = ''; const pending = new Map(); let failure;
|
|
30
|
+
const fail = () => { failure = new Error('Private native hook metadata transport failed'); for (const p of pending.values()) p.reject(failure); pending.clear(); };
|
|
31
|
+
child.on('error', fail); child.on('exit', fail); child.stderr.on('data', () => {}); child.stdin.on('error', fail);
|
|
32
|
+
child.stdout.on('data', (chunk) => { buffer += chunk; if (buffer.length > 1048576) return fail(); let split;
|
|
33
|
+
while ((split = buffer.indexOf('\n')) >= 0) { const line = buffer.slice(0, split); buffer = buffer.slice(split + 1); let r; try { r = JSON.parse(line); } catch { continue; }
|
|
34
|
+
const p = pending.get(r.id); if (!p) continue; pending.delete(r.id); r.error ? p.reject(new Error('Private native hook metadata request rejected')) : p.resolve(r.result);
|
|
35
|
+
}
|
|
36
|
+
});
|
|
37
|
+
const request = (method, params) => new Promise((resolve, reject) => { if (failure) return reject(failure); const n = ++id; pending.set(n, { resolve, reject }); child.stdin.write(JSON.stringify({ id: n, method, params }) + '\n'); });
|
|
38
|
+
const timer = setTimeout(() => { fail(); child.kill(); }, timeoutMs);
|
|
39
|
+
const oneHook = (result) => { const entries = result?.data ?? []; const hooks = entries.flatMap((e) => e.hooks ?? []);
|
|
40
|
+
if (entries.some((e) => e.errors?.length) || hooks.length !== 1 || hooks[0].eventName !== 'preToolUse' || hooks[0].matcher !== '.*' || hooks[0].command !== command || !hooks[0].enabled || !/^sha256:[a-f0-9]{64}$/.test(hooks[0].currentHash)) throw new Error('Private deny hook is not the sole enabled hook'); return hooks[0]; };
|
|
41
|
+
try {
|
|
42
|
+
await request('initialize', { clientInfo: { name: 'weekly_analyst_sandbox', version: '1' }, capabilities: { experimentalApi: true } }); child.stdin.write('{"method":"initialized"}\n');
|
|
43
|
+
const first = oneHook(await request('hooks/list', { cwds: [home] }));
|
|
44
|
+
const config = await request('config/read', { cwd: home, includeLayers: true });
|
|
45
|
+
const layer = config.layers?.find((l) => sameNativeConfigPath(l.name?.file, path.join(home, 'config.toml')));
|
|
46
|
+
if (!layer?.version) throw new Error('Private native config version missing');
|
|
47
|
+
await request('config/batchWrite', { filePath: path.join(home, 'config.toml'), expectedVersion: layer.version, reloadUserConfig: true,
|
|
48
|
+
edits: [{ keyPath: `hooks.state.${JSON.stringify(first.key)}.trusted_hash`, value: first.currentHash, mergeStrategy: 'replace' }] });
|
|
49
|
+
const final = oneHook(await request('hooks/list', { cwds: [home] }));
|
|
50
|
+
if (final.trustStatus !== 'trusted' || final.currentHash !== first.currentHash) throw new Error('Private deny hook trust not accepted');
|
|
51
|
+
return { trusted: true, hookKey: final.key, currentHash: final.currentHash, inferenceStarted: false };
|
|
52
|
+
} finally { clearTimeout(timer); child.stdin.end(); child.kill(); }
|
|
53
|
+
}
|
|
54
|
+
if (process.argv.includes('--deny')) console.log(JSON.stringify(TOOL_DENIAL));
|
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
// DISTINCT-FROM: scripts/refresh-model-catalog.mjs — independent effort-specific evaluation evidence, never router authority.
|
|
2
|
+
import { createHash } from 'node:crypto';
|
|
3
|
+
|
|
4
|
+
export const WEEK_MS = 7 * 24 * 60 * 60 * 1000;
|
|
5
|
+
export const digest = (bytes) => createHash('sha256').update(bytes).digest('hex');
|
|
6
|
+
const finite = (value) => typeof value === 'number' && Number.isFinite(value) && value >= 0;
|
|
7
|
+
|
|
8
|
+
/** Captured AA Next Flight contract. No eval/script execution; a changed schema fails closed. */
|
|
9
|
+
export function parseArtificialAnalysis(html, { url, checkedAt, identityBindings = {} } = {}) {
|
|
10
|
+
const version = html.match(/Artificial Analysis Intelligence Index v(\d+\.\d+(?:\.\d+)?)/)?.[1];
|
|
11
|
+
if (!version) throw new Error('AA Intelligence Index version is absent');
|
|
12
|
+
const rows = new Map();
|
|
13
|
+
const walk = (value) => {
|
|
14
|
+
if (Array.isArray(value)) { for (const item of value) walk(item); return; }
|
|
15
|
+
if (!value || typeof value !== 'object') return;
|
|
16
|
+
if (Array.isArray(value.intelligenceIndexEvaluations) && typeof value.id === 'string') {
|
|
17
|
+
rows.set(value.id, value); return;
|
|
18
|
+
}
|
|
19
|
+
for (const child of Object.values(value)) walk(child);
|
|
20
|
+
};
|
|
21
|
+
for (const match of html.matchAll(/self\.__next_f\.push\((.*?)\)<\/script>/gs)) {
|
|
22
|
+
try {
|
|
23
|
+
const envelope = JSON.parse(match[1]);
|
|
24
|
+
if (typeof envelope[1] !== 'string' || !envelope[1].includes('intelligenceIndexEvaluations')) continue;
|
|
25
|
+
const line = envelope[1];
|
|
26
|
+
walk(JSON.parse(line.slice(line.indexOf(':') + 1)));
|
|
27
|
+
} catch { /* unrelated or chunked flight frame; required rows checked below */ }
|
|
28
|
+
}
|
|
29
|
+
const source = { url, checkedAt, sha256: digest(html), parser: 'aa-next-flight-v1' };
|
|
30
|
+
const records = [];
|
|
31
|
+
for (const row of rows.values()) {
|
|
32
|
+
const effortLabel = typeof row.suffix === 'string' ? row.suffix : row.name?.match(/\((Low|Medium|High|Xhigh|Max|Minimal|None)(?:,|\))/i)?.[1];
|
|
33
|
+
const effort = effortLabel?.split(',')[0]?.trim().toLowerCase();
|
|
34
|
+
if (!['low', 'medium', 'high', 'xhigh', 'max', 'minimal', 'none'].includes(effort)) continue;
|
|
35
|
+
if (!row.release?.slug || !finite(row.intelligenceIndex) || row.intelligenceIndexIsEstimated !== false) continue;
|
|
36
|
+
const benchmarks = row.intelligenceIndexEvaluations.filter((b) => typeof b.slug === 'string'
|
|
37
|
+
&& Number.isFinite(b.score) && finite(b.costPerTask) && finite(b.timePerTask))
|
|
38
|
+
.map((b) => ({ suite: b.slug, score: b.score, costUsd: b.costPerTask, timeSeconds: b.timePerTask }));
|
|
39
|
+
const binding = identityBindings[row.release.slug];
|
|
40
|
+
if (binding && (typeof binding.model !== 'string' || !binding.evidence || !binding.checkedAt)) {
|
|
41
|
+
throw new Error(`Invalid native identity binding: ${row.release.slug}`);
|
|
42
|
+
}
|
|
43
|
+
records.push({
|
|
44
|
+
sourceModelId: row.id, sourceReleaseSlug: row.release.slug, sourceName: row.name,
|
|
45
|
+
model: binding?.model ?? null, identityEvidence: binding ?? null, effort,
|
|
46
|
+
benchmark: { suite: 'artificial-analysis-intelligence-index', version },
|
|
47
|
+
quality: { intelligenceIndex: row.intelligenceIndex, estimated: false }, benchmarks,
|
|
48
|
+
costPerTaskUsd: finite(row.intelligenceIndexCostPerTask?.cost?.total) ? row.intelligenceIndexCostPerTask.cost.total : null,
|
|
49
|
+
timePerTaskSeconds: finite(row.intelligenceIndexTimePerTask) ? row.intelligenceIndexTimePerTask : null,
|
|
50
|
+
speedTokensPerSecond: finite(row.outputSpeedVariance?.median ?? row.medianOutputSpeed) ? (row.outputSpeedVariance?.median ?? row.medianOutputSpeed) : null,
|
|
51
|
+
inputUsdPerMillion: finite(row.price1mInputTokens) ? row.price1mInputTokens : null,
|
|
52
|
+
outputUsdPerMillion: finite(row.price1mOutputTokens) ? row.price1mOutputTokens : null,
|
|
53
|
+
source,
|
|
54
|
+
});
|
|
55
|
+
}
|
|
56
|
+
if (records.length === 0 || !records.some((r) => r.benchmarks.length > 0 && r.costPerTaskUsd !== null && r.timePerTaskSeconds !== null)) {
|
|
57
|
+
throw new Error('AA effort benchmark matrix absent or incomplete');
|
|
58
|
+
}
|
|
59
|
+
return { source, records };
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export function parseInventory(body, source) {
|
|
63
|
+
const data = JSON.parse(body).data;
|
|
64
|
+
if (!Array.isArray(data) || data.length < 50) throw new Error('OpenRouter inventory is suspiciously thin');
|
|
65
|
+
const models = data.map((row) => {
|
|
66
|
+
const input = Number(row.pricing?.prompt); const output = Number(row.pricing?.completion);
|
|
67
|
+
if (typeof row.id !== 'string' || !row.id || !Number.isFinite(input) || !Number.isFinite(output)) throw new Error('Invalid OpenRouter model/pricing');
|
|
68
|
+
return { id: row.id, pricing: { inputUsdPerMillion: finite(input) ? input * 1e6 : null, outputUsdPerMillion: finite(output) ? output * 1e6 : null },
|
|
69
|
+
supportedParameters: Array.isArray(row.supported_parameters) ? row.supported_parameters.filter((p) => typeof p === 'string') : [] };
|
|
70
|
+
});
|
|
71
|
+
return { checkedAt: source.checkedAt, source, models, discoveryOnly: true };
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/** No benchmark row can certify native access, supported effort, or default-selection eligibility. */
|
|
75
|
+
export function currencyStatus(record, now = Date.now()) {
|
|
76
|
+
const fresh = (section) => {
|
|
77
|
+
const checked = Date.parse(section?.checkedAt);
|
|
78
|
+
return Number.isFinite(checked) && checked <= now && now - checked < WEEK_MS;
|
|
79
|
+
};
|
|
80
|
+
const inventoryFresh = record?.schemaVersion === 1 && fresh(record.inventory)
|
|
81
|
+
&& record.inventory.discoveryOnly === true && record.inventory.models?.length >= 50;
|
|
82
|
+
const evaluationsFresh = record?.schemaVersion === 1 && fresh(record.evaluations)
|
|
83
|
+
&& record.evaluations.sources?.length > 0 && record.evaluations.records?.length > 0
|
|
84
|
+
&& record.evaluations.sources.every((source) => fresh(source) && /^[a-f0-9]{64}$/.test(source.sha256 ?? ''));
|
|
85
|
+
const failed = !!record?.lastAttempt && record.lastAttempt.status !== 'complete';
|
|
86
|
+
return { status: inventoryFresh && evaluationsFresh && !failed ? 'current' : 'stale', inventoryFresh: !!inventoryFresh,
|
|
87
|
+
evaluationsFresh: !!evaluationsFresh, selectionQualified: false, maxAgeMs: WEEK_MS,
|
|
88
|
+
errors: record?.lastAttempt?.errors ?? [], assessment: record?.assessment ?? null, reason: failed ? 'last refresh incomplete' : inventoryFresh && evaluationsFresh ? 'fresh independent evidence' : 'weekly evidence required' };
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/** Coding-agent workloads are a separate suite and harness/configuration, never model-index aliases. */
|
|
92
|
+
export function parseCodingAgentEvidence(html, methodology, { url, checkedAt, identityBindings = {} } = {}) {
|
|
93
|
+
const version = methodology.match(/Coding Agent Index v(\d+\.\d+(?:\.\d+)?)/)?.[1];
|
|
94
|
+
if (!version) throw new Error('Coding Agent Index methodology version absent');
|
|
95
|
+
const rows = new Map();
|
|
96
|
+
const walk = (value) => {
|
|
97
|
+
if (Array.isArray(value)) { for (const child of value) walk(child); return; }
|
|
98
|
+
if (!value || typeof value !== 'object') return;
|
|
99
|
+
if (typeof value.hostModelSlug === 'string' && typeof value.id === 'string' && value.indexScore !== undefined) { rows.set(value.id, value); return; }
|
|
100
|
+
for (const child of Object.values(value)) walk(child);
|
|
101
|
+
};
|
|
102
|
+
for (const match of html.matchAll(/self\.__next_f\.push\((.*?)\)<\/script>/gs)) {
|
|
103
|
+
try {
|
|
104
|
+
const envelope = JSON.parse(match[1]);
|
|
105
|
+
if (typeof envelope[1] !== 'string') continue;
|
|
106
|
+
for (const line of envelope[1].split('\n')) {
|
|
107
|
+
if (!line.includes('hostModelSlug')) continue;
|
|
108
|
+
try { walk(JSON.parse(line.slice(line.indexOf(':') + 1))); } catch { /* unrelated frame */ }
|
|
109
|
+
}
|
|
110
|
+
} catch { /* required contract checked below */ }
|
|
111
|
+
}
|
|
112
|
+
const source = { url, checkedAt, sha256: digest(html), parser: 'aa-coding-agents-flight-v1' };
|
|
113
|
+
const records = [];
|
|
114
|
+
for (const row of rows.values()) {
|
|
115
|
+
if (!finite(row.indexScore) || row.indexScore > 1 || !finite(row.mean?.costUsd) || !finite(row.mean?.agentWallTimeSec)
|
|
116
|
+
|| !row.display?.model || !Array.isArray(row.evals) || row.evals.length !== row.indexComponentCount || !row.versions
|
|
117
|
+
|| row.evals.some((e) => !e.datasetIndexName || !finite(e.mean?.reward))) {
|
|
118
|
+
throw new Error(`Incomplete coding-agent record: ${row.id}`);
|
|
119
|
+
}
|
|
120
|
+
const nativeHost = row.agentName === 'Codex' && row.provider === 'openai' ? 'codex'
|
|
121
|
+
: row.agentName === 'Claude Code' && row.provider === 'anthropic' ? 'claude-code' : null;
|
|
122
|
+
const binding = nativeHost ? identityBindings[row.modelRelease?.slug] : null;
|
|
123
|
+
if (binding && (!binding.model || !binding.evidence || !binding.checkedAt)) throw new Error('Invalid coding-agent native identity binding');
|
|
124
|
+
const effort = nativeHost ? row.display.model.match(/\((low|medium|high|xhigh|max|minimal|none)\)/i)?.[1]?.toLowerCase() ?? null : null;
|
|
125
|
+
records.push({ sourceId: row.id, harness: row.agentName, provider: row.provider, nativeHost,
|
|
126
|
+
sourceModelSlug: row.hostModelSlug, sourceReleaseSlug: row.modelRelease?.slug ?? null,
|
|
127
|
+
displayModel: row.display.model, configurationLabel: row.displayLabel, effort,
|
|
128
|
+
model: binding?.model ?? null, identityEvidence: binding ?? null,
|
|
129
|
+
fallback: /fallback/i.test(row.displayLabel ?? row.display.model),
|
|
130
|
+
benchmark: { suite: 'artificial-analysis-coding-agent-index', version }, codingAgentIndexFraction: row.indexScore,
|
|
131
|
+
apiBenchmarkCostPerTaskUsd: row.mean.costUsd, timePerTaskSeconds: row.mean.agentWallTimeSec,
|
|
132
|
+
versions: row.versions, components: row.evals.map((e) => ({ suite: e.datasetIndexName,
|
|
133
|
+
dataset: e.refDatasetName, score: e.mean.reward, mean: e.mean })), source,
|
|
134
|
+
selectionQualified: false });
|
|
135
|
+
}
|
|
136
|
+
if (!records.length) throw new Error('Coding-agent structured records absent');
|
|
137
|
+
return { source, records, coverage: { scope: 'public serialized agent/configuration records; native entitlements unverified',
|
|
138
|
+
recordCount: records.length, renderedChartSelectionVerified: false }, selectionQualified: false };
|
|
139
|
+
}
|
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// DISTINCT-FROM: scripts/model-router-catalog.mjs — per-user evidence currency, never candidate/default mutation.
|
|
3
|
+
import fs from 'node:fs';
|
|
4
|
+
import path from 'node:path';
|
|
5
|
+
import os from 'node:os';
|
|
6
|
+
import { spawn } from 'node:child_process';
|
|
7
|
+
import { randomUUID } from 'node:crypto';
|
|
8
|
+
import { buildWeeklyAssessment } from './model-weekly-assessment.mjs';
|
|
9
|
+
import { fileURLToPath } from 'node:url';
|
|
10
|
+
import { WEEK_MS, digest, parseInventory, parseArtificialAnalysis, parseCodingAgentEvidence, currencyStatus } from './model-currency-evidence.mjs';
|
|
11
|
+
|
|
12
|
+
export const DEFAULT_ROUTER_DIR = path.join(os.homedir(), '.claude', 'model-router');
|
|
13
|
+
export const AA_URLS = [
|
|
14
|
+
'https://artificialanalysis.ai/models/comparisons',
|
|
15
|
+
'https://artificialanalysis.ai/models/comparisons/claude-fable-5-1-medium-vs-claude-opus-5-medium',
|
|
16
|
+
'https://artificialanalysis.ai/models/releases/comparisons/gpt-6-1-sol-vs-claude-sonnet-5-5',
|
|
17
|
+
'https://artificialanalysis.ai/models/releases/comparisons/gpt-6-luna-vs-gpt-6-astra',
|
|
18
|
+
'https://artificialanalysis.ai/models/releases/comparisons/claude-opus-5-5-vs-gpt-6-astra',
|
|
19
|
+
];
|
|
20
|
+
export const OFFICIAL_SOURCES = [
|
|
21
|
+
{ provider: 'openai', url: 'https://developers.openai.com/api/docs/models' },
|
|
22
|
+
{ provider: 'anthropic', url: 'https://platform.claude.com/docs/en/models/overview' },
|
|
23
|
+
];
|
|
24
|
+
export const AGENT_SOURCE_URLS = [
|
|
25
|
+
'https://artificialanalysis.ai/agents/coding-agents',
|
|
26
|
+
'https://artificialanalysis.ai/methodology/coding-agents-benchmarking',
|
|
27
|
+
];
|
|
28
|
+
export const CODING_BENCHMARK_SOURCES = [
|
|
29
|
+
{ suite: 'vulcanbench', url: 'https://vulcanbench.com/' },
|
|
30
|
+
{ suite: 'vulcanbench-swe-v4', url: 'https://vulcanbench.com/benchmarks/swe-v4-gpt61-sol-v318.html' },
|
|
31
|
+
{ suite: 'terminal-bench', url: 'https://www.tbench.ai/' },
|
|
32
|
+
{ suite: 'swe-bench', url: 'https://www.swebench.com/' },
|
|
33
|
+
];
|
|
34
|
+
const INVENTORY_URL = 'https://openrouter.ai/api/v1/models';
|
|
35
|
+
const LOCK_MS = 10 * 60 * 1000;
|
|
36
|
+
const RETRY_MS = 60 * 60 * 1000;
|
|
37
|
+
const SELF = fileURLToPath(import.meta.url);
|
|
38
|
+
function readRecord(routerDir) {
|
|
39
|
+
try {
|
|
40
|
+
const target = path.join(routerDir, 'currency.json');
|
|
41
|
+
if (fs.statSync(target).size > 8 * 1024 * 1024) return null;
|
|
42
|
+
return JSON.parse(fs.readFileSync(target, 'utf8'));
|
|
43
|
+
} catch { return null; }
|
|
44
|
+
}
|
|
45
|
+
export function readCurrencyStatus({ routerDir = DEFAULT_ROUTER_DIR, now = Date.now() } = {}) {
|
|
46
|
+
return currencyStatus(readRecord(routerDir), now);
|
|
47
|
+
}
|
|
48
|
+
function atomicWrite(target, bytes, beforeCommit = () => {}) {
|
|
49
|
+
fs.mkdirSync(path.dirname(target), { recursive: true, mode: 0o700 });
|
|
50
|
+
const tmp = `${target}.tmp-${process.pid}-${Date.now()}`;
|
|
51
|
+
try {
|
|
52
|
+
const fd = fs.openSync(tmp, 'wx', 0o600);
|
|
53
|
+
try { fs.writeFileSync(fd, bytes); fs.fsyncSync(fd); } finally { fs.closeSync(fd); }
|
|
54
|
+
beforeCommit();
|
|
55
|
+
fs.renameSync(tmp, target);
|
|
56
|
+
} finally { try { fs.unlinkSync(tmp); } catch { /* renamed */ } }
|
|
57
|
+
}
|
|
58
|
+
// This guard is held only for synchronous filesystem transactions, never during fetches.
|
|
59
|
+
// Never reap it: a crash inside this tiny transaction must fail closed rather than overlap writers.
|
|
60
|
+
function transaction(routerDir, operation) {
|
|
61
|
+
fs.mkdirSync(routerDir, { recursive: true, mode: 0o700 });
|
|
62
|
+
const guard = path.join(routerDir, 'currency-mutation.lock');
|
|
63
|
+
try { fs.mkdirSync(guard, { mode: 0o700 }); } catch (error) { if (error.code === 'EEXIST') return null; throw error; }
|
|
64
|
+
try { return operation(); } finally { fs.rmdirSync(guard); }
|
|
65
|
+
}
|
|
66
|
+
function ownerPath(routerDir) { return path.join(routerDir, 'currency-refresh-owner.json'); }
|
|
67
|
+
function readOwner(routerDir) {
|
|
68
|
+
try { return JSON.parse(fs.readFileSync(ownerPath(routerDir), 'utf8')); }
|
|
69
|
+
catch (error) { if (error.code === 'ENOENT') return null; throw error; }
|
|
70
|
+
}
|
|
71
|
+
function claim(routerDir, now) {
|
|
72
|
+
return transaction(routerDir, () => {
|
|
73
|
+
const owner = readOwner(routerDir);
|
|
74
|
+
if (owner && (!Number.isFinite(owner.claimedAt) || now - owner.claimedAt <= LOCK_MS)) return null;
|
|
75
|
+
const next = { token: randomUUID(), claimedAt: now };
|
|
76
|
+
atomicWrite(ownerPath(routerDir), JSON.stringify(next));
|
|
77
|
+
return next.token;
|
|
78
|
+
});
|
|
79
|
+
}
|
|
80
|
+
function release(routerDir, token) {
|
|
81
|
+
return transaction(routerDir, () => {
|
|
82
|
+
if (readOwner(routerDir)?.token !== token) return false;
|
|
83
|
+
fs.unlinkSync(ownerPath(routerDir)); return true;
|
|
84
|
+
});
|
|
85
|
+
}
|
|
86
|
+
function fencedWrite(routerDir, token, target, bytes) {
|
|
87
|
+
const written = transaction(routerDir, () => {
|
|
88
|
+
if (readOwner(routerDir)?.token !== token) throw new Error('refresh superseded: ownership token changed');
|
|
89
|
+
atomicWrite(target, bytes, () => {
|
|
90
|
+
if (readOwner(routerDir)?.token !== token) throw new Error('refresh superseded before commit');
|
|
91
|
+
}); return true;
|
|
92
|
+
});
|
|
93
|
+
if (!written) throw new Error('refresh transaction busy; no write committed');
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
/** Prompt path: local bounded read and one detached worker; never await a network request. */
|
|
97
|
+
export function maybeLaunchCurrencyRefresh({ routerDir = DEFAULT_ROUTER_DIR, now = Date.now(), launch = spawn } = {}) {
|
|
98
|
+
const record = readRecord(routerDir); const status = currencyStatus(record, now);
|
|
99
|
+
if (status.status === 'current') return { ...status, launched: false };
|
|
100
|
+
const attempted = Date.parse(record?.lastAttempt?.checkedAt);
|
|
101
|
+
if (Number.isFinite(attempted) && attempted <= now && now - attempted < RETRY_MS) return { ...status, launched: false, deferred: 'retry cooldown' };
|
|
102
|
+
const lock = claim(routerDir, now);
|
|
103
|
+
if (!lock) return { ...status, launched: false, deferred: fs.existsSync(path.join(routerDir, 'currency-mutation.lock'))
|
|
104
|
+
? 'refresh transaction blocked; inspect mutation guard before recovery' : 'refresh already running' };
|
|
105
|
+
try {
|
|
106
|
+
const child = launch(process.execPath, [SELF, '--refresh', '--claim-token', lock, '--router-dir', routerDir], { detached: true, stdio: 'ignore' });
|
|
107
|
+
child.once?.('error', () => release(routerDir, lock));
|
|
108
|
+
child.unref();
|
|
109
|
+
return { ...status, launched: true };
|
|
110
|
+
} catch (error) { release(routerDir, lock); return { ...status, launched: false, errors: [error.message] }; }
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
export async function refreshModelCurrency({ routerDir = DEFAULT_ROUTER_DIR, now = Date.now(), fetchImpl = fetch,
|
|
114
|
+
identityBindings, aaUrls = AA_URLS, officialSources = OFFICIAL_SOURCES, agentSourceUrls = AGENT_SOURCE_URLS, codingBenchmarkSources = CODING_BENCHMARK_SOURCES, claimToken = null } = {}) {
|
|
115
|
+
const lock = claimToken ?? claim(routerDir, now);
|
|
116
|
+
if (!lock) return { action: 'busy', status: 'stale', reason: 'refresh ownership or transaction guard unavailable' };
|
|
117
|
+
try {
|
|
118
|
+
const prior = readRecord(routerDir) ?? { schemaVersion: 1 };
|
|
119
|
+
const checkedAt = new Date(now).toISOString(); const errors = [];
|
|
120
|
+
let bindings = identityBindings;
|
|
121
|
+
if (!bindings) {
|
|
122
|
+
try { bindings = JSON.parse(fs.readFileSync(path.join(routerDir, 'identity-bindings.json'), 'utf8')); }
|
|
123
|
+
catch { bindings = {}; }
|
|
124
|
+
}
|
|
125
|
+
const collect = async (url) => {
|
|
126
|
+
const response = await fetchImpl(url, { signal: AbortSignal.timeout(20_000) });
|
|
127
|
+
if (!response.ok) throw new Error(`HTTP ${response.status}`);
|
|
128
|
+
const bytes = await response.text();
|
|
129
|
+
if (bytes.length > 6 * 1024 * 1024) throw new Error('source exceeds 6 MiB limit');
|
|
130
|
+
const source = { url, checkedAt, sha256: digest(bytes) };
|
|
131
|
+
fencedWrite(routerDir, lock, path.join(routerDir, 'evidence', `${source.sha256}.${url === INVENTORY_URL ? 'json' : 'html'}`), bytes);
|
|
132
|
+
return { source, bytes };
|
|
133
|
+
};
|
|
134
|
+
const results = await Promise.allSettled([INVENTORY_URL, ...aaUrls, ...officialSources.map((s) => s.url), ...agentSourceUrls, ...codingBenchmarkSources.map((s) => s.url)].map(collect));
|
|
135
|
+
let inventory = prior.inventory; let evaluations = prior.evaluations;
|
|
136
|
+
if (results[0].status === 'fulfilled') {
|
|
137
|
+
try { inventory = parseInventory(results[0].value.bytes, results[0].value.source); }
|
|
138
|
+
catch (error) { errors.push(`inventory: ${error.message}`); }
|
|
139
|
+
} else errors.push(`inventory: ${results[0].reason.message}`);
|
|
140
|
+
const parsed = [];
|
|
141
|
+
for (let i = 1; i <= aaUrls.length; i++) {
|
|
142
|
+
const result = results[i];
|
|
143
|
+
try {
|
|
144
|
+
if (result.status !== 'fulfilled') throw result.reason;
|
|
145
|
+
parsed.push(parseArtificialAnalysis(result.value.bytes, { ...result.value.source, identityBindings: bindings }));
|
|
146
|
+
} catch (error) { errors.push(`evaluations ${aaUrls[i - 1]}: ${error.message}`); }
|
|
147
|
+
}
|
|
148
|
+
// All requested pages must parse before certifying a new matrix; partial pages are evidence only.
|
|
149
|
+
if (parsed.length === aaUrls.length && parsed.length > 0) {
|
|
150
|
+
const records = new Map();
|
|
151
|
+
for (const page of parsed) for (const record of page.records) records.set(record.sourceModelId, record);
|
|
152
|
+
evaluations = { checkedAt, sources: parsed.map((p) => p.source), records: [...records.values()],
|
|
153
|
+
selectionQualified: false, limitation: 'Independent benchmark evidence; native access and supported effort require separate verification. Arena is not collected.' };
|
|
154
|
+
}
|
|
155
|
+
let official = prior.officialSources;
|
|
156
|
+
const publicDocs = [];
|
|
157
|
+
for (let i = 0; i < officialSources.length; i++) {
|
|
158
|
+
const result = results[1 + aaUrls.length + i];
|
|
159
|
+
if (result.status === 'fulfilled') publicDocs.push({ ...result.value.source, provider: officialSources[i].provider,
|
|
160
|
+
scope: 'official public/API documentation; not native subscription access', semanticallyQualified: false });
|
|
161
|
+
else errors.push(`official ${officialSources[i].url}: ${result.reason.message}`);
|
|
162
|
+
}
|
|
163
|
+
if (publicDocs.length === officialSources.length) official = { checkedAt, sources: publicDocs };
|
|
164
|
+
let agents = prior.agentSources;
|
|
165
|
+
if (agentSourceUrls.length) {
|
|
166
|
+
const agentResults = results.slice(1 + aaUrls.length + officialSources.length, 1 + aaUrls.length + officialSources.length + agentSourceUrls.length);
|
|
167
|
+
try {
|
|
168
|
+
if (agentResults.length !== 2) throw new Error('coding-agent source and methodology pair required');
|
|
169
|
+
for (const result of agentResults) if (result.status !== 'fulfilled') throw result.reason;
|
|
170
|
+
const parsedAgents = parseCodingAgentEvidence(agentResults[0].value.bytes, agentResults[1].value.bytes,
|
|
171
|
+
{ ...agentResults[0].value.source, identityBindings: bindings });
|
|
172
|
+
agents = { checkedAt, sources: agentResults.map((r) => r.value.source), ...parsedAgents };
|
|
173
|
+
} catch (error) { errors.push(`coding agents: ${error.message}`); }
|
|
174
|
+
}
|
|
175
|
+
const additional = [];
|
|
176
|
+
for (let i = 0; i < codingBenchmarkSources.length; i++) {
|
|
177
|
+
const result = results[1 + aaUrls.length + officialSources.length + agentSourceUrls.length + i];
|
|
178
|
+
if (result.status === 'fulfilled' && result.value.bytes.trim().length >= 100) {
|
|
179
|
+
additional.push({ ...result.value.source, suite: codingBenchmarkSources[i].suite,
|
|
180
|
+
scope: 'raw source archive; independent suite, score parsing and semantic qualification not implemented' });
|
|
181
|
+
} else errors.push(`coding benchmark ${codingBenchmarkSources[i].url}: ${result.status === 'rejected' ? result.reason.message : 'empty or truncated source'}`);
|
|
182
|
+
}
|
|
183
|
+
if (additional.length === codingBenchmarkSources.length && agents && agents !== prior.agentSources) agents = { ...agents, additionalSources: additional };
|
|
184
|
+
else if (additional.length !== codingBenchmarkSources.length) agents = prior.agentSources;
|
|
185
|
+
const instructionPath = path.join(routerDir, 'weekly-analyst-instruction.md');
|
|
186
|
+
let instruction; let instructionSource = 'packaged-fallback';
|
|
187
|
+
try {
|
|
188
|
+
const fd = fs.openSync(instructionPath, 'r');
|
|
189
|
+
try {
|
|
190
|
+
const buffer = Buffer.alloc(128 * 1024 + 1);
|
|
191
|
+
const size = fs.readSync(fd, buffer, 0, buffer.length, 0);
|
|
192
|
+
if (size > 128 * 1024) throw new Error('effective weekly instruction exceeds 128 KiB');
|
|
193
|
+
instruction = buffer.subarray(0, size).toString('utf8');
|
|
194
|
+
if (!instruction.trim()) throw new Error('effective weekly instruction is empty');
|
|
195
|
+
instructionSource = 'effective-per-user-file';
|
|
196
|
+
} finally { fs.closeSync(fd); }
|
|
197
|
+
} catch (error) {
|
|
198
|
+
if (error.code !== 'ENOENT') { errors.push(`instruction: ${error.message}`); instruction = undefined; instructionSource = 'fallback-after-read-error'; }
|
|
199
|
+
}
|
|
200
|
+
const next = { schemaVersion: 1, maxAgeMs: WEEK_MS, inventory, evaluations, officialSources: official, agentSources: agents,
|
|
201
|
+
lastAttempt: { checkedAt, status: errors.length ? (inventory === prior.inventory && evaluations === prior.evaluations ? 'failed' : 'partial') : 'complete', errors } };
|
|
202
|
+
let priorPolicyBytes = null; let policy = null;
|
|
203
|
+
try { priorPolicyBytes = fs.readFileSync(path.join(routerDir, 'routing-policy.json'), 'utf8'); policy = JSON.parse(priorPolicyBytes); }
|
|
204
|
+
catch { /* no policy: report missing allocation, never synthesize one */ }
|
|
205
|
+
const assessment = buildWeeklyAssessment({ currency: next, policy, priorPolicyBytes, now, previousAssessment: prior.assessment, instruction, instructionSource });
|
|
206
|
+
const assessmentDir = path.join(routerDir, 'assessments', `${checkedAt.replaceAll(':', '-')}-${lock}`);
|
|
207
|
+
for (const [name, bytes] of [['report.json', JSON.stringify(assessment.report, null, 2)],
|
|
208
|
+
['proposal.json', JSON.stringify(assessment.proposal, null, 2)], ['report.md', assessment.markdown],
|
|
209
|
+
['instruction.md', assessment.instruction], ['prior-policy.json', priorPolicyBytes ?? 'null']]) {
|
|
210
|
+
fencedWrite(routerDir, lock, path.join(assessmentDir, name), bytes);
|
|
211
|
+
}
|
|
212
|
+
if (!fs.existsSync(instructionPath)) fencedWrite(routerDir, lock, instructionPath, assessment.instruction);
|
|
213
|
+
next.assessment = { instructionPath, instructionSha256: assessment.report.instructionSha256, instructionSource, checkedAt, signature: assessment.report.signature, reportPath: path.join(assessmentDir, 'report.json'),
|
|
214
|
+
proposalPath: path.join(assessmentDir, 'proposal.json'), status: 'complete-unqualified', analystExecuted: false,
|
|
215
|
+
notification: assessment.report.notification };
|
|
216
|
+
fencedWrite(routerDir, lock, path.join(routerDir, 'currency.json'), `${JSON.stringify(next, null, 2)}\n`);
|
|
217
|
+
return { action: 'refreshed', ...currencyStatus(next, now), lastAttempt: next.lastAttempt };
|
|
218
|
+
} finally { release(routerDir, lock); }
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === SELF) {
|
|
222
|
+
const dirIndex = process.argv.indexOf('--router-dir');
|
|
223
|
+
const routerDir = dirIndex >= 0 ? process.argv[dirIndex + 1] : DEFAULT_ROUTER_DIR;
|
|
224
|
+
if (!routerDir || !path.isAbsolute(routerDir)) throw new Error('--router-dir must be absolute');
|
|
225
|
+
if (process.argv.includes('--refresh')) {
|
|
226
|
+
refreshModelCurrency({ routerDir, claimToken: process.argv.includes('--claim-token') ? process.argv[process.argv.indexOf('--claim-token') + 1] : null }).then((result) => {
|
|
227
|
+
console.log(JSON.stringify(result)); if (result.status === 'stale') process.exitCode = 1;
|
|
228
|
+
}).catch((error) => { console.error(error.message); process.exitCode = 1; });
|
|
229
|
+
} else console.log(JSON.stringify(process.argv.includes('--catch-up') ? maybeLaunchCurrencyRefresh({ routerDir }) : readCurrencyStatus({ routerDir })));
|
|
230
|
+
}
|