@oracle-agent/oracle 0.35.42 → 0.35.43
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-sessions.mjs +25 -24
- package/dist/assets/skills/oracle-desk-product/SKILL.md +1 -1
- package/dist/bin/desk-server.mjs +27 -27
- package/dist/bin/oracle-data-mcp.mjs +18 -18
- package/dist/bin/oracle-gateway.mjs +44 -44
- package/dist/bin/oracle-launch-radar-shadow.mjs +6 -6
- package/dist/bin/oracle-public-server.mjs +5 -5
- package/dist/bin/oracle-route.mjs +16 -16
- package/dist/bin/oracle.mjs +12 -12
- package/dist/cli/commands/agent-session.mjs +27 -26
- package/dist/cli/commands/chat.mjs +245 -224
- package/dist/cli/commands/eval.mjs +181 -161
- package/dist/cli/commands/mint.mjs +11 -11
- package/dist/cli/commands/model.mjs +249 -228
- package/dist/cli/commands/sign.mjs +14 -14
- package/dist/cli/commands/signer.mjs +22 -22
- package/dist/cli/commands/swap.mjs +13 -13
- package/dist/cli/commands/venues.mjs +2 -2
- package/dist/data/desk-data.mjs +25 -25
- package/dist/index.mjs +46 -46
- package/dist/launch-radar/index.mjs +5 -5
- package/dist/local-signer/anvil-attestation-worker.mjs +1 -1
- package/dist/local-signer/index.mjs +3 -3
- package/dist/local-signer/runtime.mjs +19 -19
- package/dist/router/prepare-route.mjs +3 -3
- package/dist/task.mjs +2 -2
- package/dist/teams/index.mjs +6 -6
- package/dist/tui/standalone-client.mjs +115 -95
- package/package.json +1 -1
- package/public/oracle-splash/docs/browser-and-web/index.html +2 -2
- package/public/oracle-splash/docs/connectors/index.html +2 -2
- package/public/oracle-splash/docs/getting-started/index.html +2 -2
- package/public/oracle-splash/docs/index.html +3 -3
- package/public/oracle-splash/docs/reference/index.html +2 -2
- package/public/oracle-splash/docs/security/index.html +2 -2
- package/public/oracle-splash/docs/using-oracle/index.html +2 -2
- package/public/oracle-splash/eval/benchmarks/evm-escrow-20260827.json +170 -0
- package/public/oracle-splash/eval/eval-story.css +722 -0
- package/public/oracle-splash/eval/eval-story.js +198 -0
- package/public/oracle-splash/eval/index.html +49 -11
- package/public/oracle-splash/eval/live-eval.js +546 -0
- package/server.json +2 -2
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Oracle capability evidence: browser enhancements for the visual story.
|
|
3
|
+
*
|
|
4
|
+
* Progressive only. The rendered HTML is complete and readable without this
|
|
5
|
+
* file. Here we add the step control for the architecture walkthrough, pause
|
|
6
|
+
* motion when the figures are offscreen or the tab is hidden, and reflect an
|
|
7
|
+
* evaluation record the parent page emits.
|
|
8
|
+
*
|
|
9
|
+
* Nothing in this file fabricates a measurement. If no record arrives, or the
|
|
10
|
+
* record is blocked or unavailable, the activity line says so and no success
|
|
11
|
+
* animation is allowed to run.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
const ACTIVITY_TEXT = Object.freeze({
|
|
15
|
+
running: "Evaluation running. Selected slices are being measured; no final result is claimed.",
|
|
16
|
+
interrupted: "Evaluation interrupted. Unfinished work is not a passing result.",
|
|
17
|
+
unavailable: "No evaluation record is attached right now. The schematic above is complete on its own.",
|
|
18
|
+
blocked: "Latest record reports a blocked slice. Missing measurements remain unavailable. Any recorded counts stay attached to their slice.",
|
|
19
|
+
failed: "Latest record reports a failing slice. See the record for which slice and which case.",
|
|
20
|
+
skipped: "Latest record has a skipped slice. A skipped slice carries no measurement.",
|
|
21
|
+
passed: "Latest record completed and every selected slice reported a pass.",
|
|
22
|
+
});
|
|
23
|
+
|
|
24
|
+
const KNOWN_STATES = new Set(["running", "interrupted", "passed", "failed", "blocked", "skipped", "unavailable"]);
|
|
25
|
+
|
|
26
|
+
function connectionIsOpen(connection) {
|
|
27
|
+
if (typeof connection === "string") return connection === "live";
|
|
28
|
+
if (!connection || typeof connection !== "object") return false;
|
|
29
|
+
const state = String(connection.state || "").toLowerCase();
|
|
30
|
+
if (state) return state === "open" || state === "connected" || state === "ready";
|
|
31
|
+
return connection.connected === true;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
function recordState(record) {
|
|
35
|
+
if (!record || typeof record !== "object") return "unavailable";
|
|
36
|
+
const raw = String(record.status || "").toLowerCase();
|
|
37
|
+
if (KNOWN_STATES.has(raw)) return raw;
|
|
38
|
+
if (record.blocked === true) return "blocked";
|
|
39
|
+
if (record.ok === true) return "passed";
|
|
40
|
+
if (record.ok === false) return "failed";
|
|
41
|
+
return "unavailable";
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Decide what the activity line says and whether the method figure may animate
|
|
46
|
+
* its successful flow. Pure, so tests can call it without a DOM.
|
|
47
|
+
*/
|
|
48
|
+
export function applyEvalRecord(detail = {}) {
|
|
49
|
+
const record = detail && typeof detail === "object" ? detail.record : null;
|
|
50
|
+
const open = connectionIsOpen(detail && detail.connection);
|
|
51
|
+
const state = open ? recordState(record) : "unavailable";
|
|
52
|
+
const animate = state === "running" || state === "passed";
|
|
53
|
+
return {
|
|
54
|
+
state,
|
|
55
|
+
animate,
|
|
56
|
+
text: ACTIVITY_TEXT[state] || ACTIVITY_TEXT.unavailable,
|
|
57
|
+
};
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
function prefersReducedMotion(view) {
|
|
61
|
+
try {
|
|
62
|
+
return Boolean(view.matchMedia && view.matchMedia("(prefers-reduced-motion: reduce)").matches);
|
|
63
|
+
} catch {
|
|
64
|
+
return false;
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
function setupStepper(root, doc) {
|
|
69
|
+
const tabs = Array.from(root.querySelectorAll("[data-es-step]"));
|
|
70
|
+
const panels = Array.from(root.querySelectorAll("[data-es-step-panel]"));
|
|
71
|
+
if (tabs.length === 0 || tabs.length !== panels.length) return;
|
|
72
|
+
|
|
73
|
+
root.setAttribute("data-es-js", "on");
|
|
74
|
+
|
|
75
|
+
const select = (index, focus) => {
|
|
76
|
+
const next = ((index % tabs.length) + tabs.length) % tabs.length;
|
|
77
|
+
tabs.forEach((tab, i) => {
|
|
78
|
+
const current = i === next;
|
|
79
|
+
tab.setAttribute("aria-selected", current ? "true" : "false");
|
|
80
|
+
tab.setAttribute("tabindex", current ? "0" : "-1");
|
|
81
|
+
});
|
|
82
|
+
panels.forEach((panel, i) => {
|
|
83
|
+
panel.classList.toggle("is-current", i === next);
|
|
84
|
+
});
|
|
85
|
+
if (focus) tabs[next].focus();
|
|
86
|
+
};
|
|
87
|
+
|
|
88
|
+
tabs.forEach((tab, index) => {
|
|
89
|
+
tab.addEventListener("click", () => select(index, false));
|
|
90
|
+
tab.addEventListener("keydown", (event) => {
|
|
91
|
+
const key = event.key;
|
|
92
|
+
if (key === "ArrowRight" || key === "ArrowDown") {
|
|
93
|
+
event.preventDefault();
|
|
94
|
+
select(index + 1, true);
|
|
95
|
+
} else if (key === "ArrowLeft" || key === "ArrowUp") {
|
|
96
|
+
event.preventDefault();
|
|
97
|
+
select(index - 1, true);
|
|
98
|
+
} else if (key === "Home") {
|
|
99
|
+
event.preventDefault();
|
|
100
|
+
select(0, true);
|
|
101
|
+
} else if (key === "End") {
|
|
102
|
+
event.preventDefault();
|
|
103
|
+
select(tabs.length - 1, true);
|
|
104
|
+
}
|
|
105
|
+
});
|
|
106
|
+
});
|
|
107
|
+
|
|
108
|
+
// Replayable in either direction: selecting any step re-runs its transition.
|
|
109
|
+
select(0, false);
|
|
110
|
+
return { select, count: tabs.length, doc };
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
function setupMotion(root, view, doc) {
|
|
114
|
+
const figures = Array.from(root.querySelectorAll("[data-es-figure]"));
|
|
115
|
+
if (figures.length === 0) return null;
|
|
116
|
+
|
|
117
|
+
const reduced = prefersReducedMotion(view);
|
|
118
|
+
const visible = new Set();
|
|
119
|
+
let blockedFlow = true;
|
|
120
|
+
|
|
121
|
+
const paint = () => {
|
|
122
|
+
const pageVisible = doc.visibilityState !== "hidden";
|
|
123
|
+
for (const figure of figures) {
|
|
124
|
+
const kind = figure.getAttribute("data-es-figure");
|
|
125
|
+
const allowed = !reduced
|
|
126
|
+
&& pageVisible
|
|
127
|
+
&& visible.has(figure)
|
|
128
|
+
&& !(kind === "method" && blockedFlow);
|
|
129
|
+
figure.classList.toggle("is-active", allowed);
|
|
130
|
+
}
|
|
131
|
+
};
|
|
132
|
+
|
|
133
|
+
if (typeof view.IntersectionObserver === "function") {
|
|
134
|
+
const observer = new view.IntersectionObserver((entries) => {
|
|
135
|
+
for (const entry of entries) {
|
|
136
|
+
if (entry.isIntersecting) visible.add(entry.target);
|
|
137
|
+
else visible.delete(entry.target);
|
|
138
|
+
}
|
|
139
|
+
paint();
|
|
140
|
+
}, { threshold: 0.15 });
|
|
141
|
+
figures.forEach((figure) => observer.observe(figure));
|
|
142
|
+
} else {
|
|
143
|
+
figures.forEach((figure) => visible.add(figure));
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
doc.addEventListener("visibilitychange", paint);
|
|
147
|
+
paint();
|
|
148
|
+
|
|
149
|
+
return {
|
|
150
|
+
setFlowAllowed(allowed) {
|
|
151
|
+
blockedFlow = !allowed;
|
|
152
|
+
paint();
|
|
153
|
+
},
|
|
154
|
+
refresh: paint,
|
|
155
|
+
};
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
export function initEvalStory(options = {}) {
|
|
159
|
+
const doc = options.document;
|
|
160
|
+
if (!doc) return null;
|
|
161
|
+
const view = options.window || doc.defaultView || {};
|
|
162
|
+
const root = options.root || doc.querySelector("[data-eval-story]");
|
|
163
|
+
if (!root) return null;
|
|
164
|
+
|
|
165
|
+
setupStepper(root, doc);
|
|
166
|
+
const motion = setupMotion(root, view, doc);
|
|
167
|
+
const activity = root.querySelector("[data-es-activity]");
|
|
168
|
+
|
|
169
|
+
const onRecord = (event) => {
|
|
170
|
+
const result = applyEvalRecord(event && event.detail ? event.detail : {});
|
|
171
|
+
if (activity) {
|
|
172
|
+
activity.textContent = result.text;
|
|
173
|
+
activity.setAttribute("data-es-state", result.state);
|
|
174
|
+
}
|
|
175
|
+
if (motion) motion.setFlowAllowed(result.animate);
|
|
176
|
+
};
|
|
177
|
+
|
|
178
|
+
doc.addEventListener("oracle:eval-record", onRecord);
|
|
179
|
+
|
|
180
|
+
return { root, refresh: () => (motion ? motion.refresh() : undefined), onRecord };
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
// Import safe: under Node there is no DOM, so auto init is skipped entirely.
|
|
184
|
+
const NO_DOM = typeof document === "undefined";
|
|
185
|
+
|
|
186
|
+
if (!NO_DOM) {
|
|
187
|
+
const start = () => initEvalStory({
|
|
188
|
+
document,
|
|
189
|
+
window: typeof window === "undefined" ? undefined : window,
|
|
190
|
+
});
|
|
191
|
+
if (document.readyState === "loading") {
|
|
192
|
+
document.addEventListener("DOMContentLoaded", start, { once: true });
|
|
193
|
+
} else {
|
|
194
|
+
start();
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
export default { initEvalStory, applyEvalRecord };
|
|
@@ -6,6 +6,7 @@
|
|
|
6
6
|
<meta name="color-scheme" content="dark">
|
|
7
7
|
<meta name="description" content="Commit-pinned Oracle evaluation evidence.">
|
|
8
8
|
<title>Evaluation evidence · Oracle</title>
|
|
9
|
+
<link rel="stylesheet" href="./eval-story.css">
|
|
9
10
|
<style>
|
|
10
11
|
:root{--ink:#eef3ed;--muted:#98a39b;--line:#273029;--panel:#111713;--panel-2:#151d18;--green:#a8f0bd;--red:#ff9e96;--amber:#efcf89;--blue:#9bbfff;--bg:#080b09;--serif:Georgia,"Times New Roman",serif;--mono:"SFMono-Regular",Consolas,"Liberation Mono",monospace}
|
|
11
12
|
*{box-sizing:border-box}html{scroll-behavior:smooth;background:var(--bg)}body{margin:0;color:var(--ink);background:radial-gradient(circle at 82% -8%,#1e3828 0,transparent 32rem),linear-gradient(180deg,#0b100c 0,var(--bg) 42rem);font-family:Arial,Helvetica,sans-serif;line-height:1.55}
|
|
@@ -19,24 +20,58 @@
|
|
|
19
20
|
.suite-grid{display:grid;grid-template-columns:repeat(3,minmax(0,1fr));border-top:1px solid var(--line);border-left:1px solid var(--line)}.suite-card{min-height:240px;padding:24px;background:linear-gradient(145deg,rgba(255,255,255,.025),transparent);border-right:1px solid var(--line);border-bottom:1px solid var(--line)}.suite-card[data-state="unavailable"]{min-height:178px;background:rgba(255,255,255,.012)}.suite-card:hover{background-color:rgba(255,255,255,.025)}.suite-topline{display:flex;justify-content:space-between;gap:12px}.slice-index{color:#819087;font:12px/1 var(--mono)}.state-pill{font:700 11px/1.25 var(--mono);letter-spacing:.07em;text-transform:uppercase}.suite-card[data-state="passed"] .state-pill{color:var(--green)}.suite-card[data-state="failed"] .state-pill,.suite-card[data-state="blocked"] .state-pill{color:var(--red)}.suite-card[data-state="skipped"] .state-pill,.suite-card[data-state="unavailable"] .state-pill,.suite-card[data-state="unattested"] .state-pill{color:var(--amber)}.suite-card h3{margin:42px 0 16px;font:400 31px/1 var(--serif);text-transform:capitalize}.suite-card[data-state="unavailable"] h3{margin-top:30px}.suite-card ul{margin:0;padding:0;list-style:none;color:#aab5ad;font:13px/1.6 var(--mono)}
|
|
20
21
|
.coverage{display:grid;grid-template-columns:repeat(5,1fr);margin:0;padding:0;list-style:none;border-top:1px solid var(--line);border-left:1px solid var(--line)}.coverage li{padding:20px;border-right:1px solid var(--line);border-bottom:1px solid var(--line);color:#b5c7d9;font:13px/1.4 var(--mono)}
|
|
21
22
|
.provenance{display:grid;grid-template-columns:1fr 1fr;gap:1px;background:var(--line);border:1px solid var(--line)}.datum{min-width:0;padding:24px;background:var(--panel)}.datum dt{margin-bottom:10px;color:var(--muted);font:10px/1.4 var(--mono);letter-spacing:.13em;text-transform:uppercase}.datum dd{margin:0;overflow-wrap:anywhere;font:13px/1.6 var(--mono)}code{color:var(--green);font:inherit}.method{grid-column:1/-1}.method dd{max-width:900px}.reproduce{display:block;margin-top:18px;padding:16px 18px;background:#090d0a;border-left:2px solid var(--green)}
|
|
23
|
+
.eval-index{font:13px/1.5 var(--mono);border-top:1px solid var(--line);padding-top:20px}.eval-index a{text-decoration:none}.eval-index a:hover{color:var(--green)}.live-details{margin-top:24px}.live-details summary{padding:18px 0;cursor:pointer;font:13px/1.5 var(--mono);color:var(--muted)}
|
|
24
|
+
.live-status{margin:0 0 26px;padding:14px 18px;border:1px solid var(--line);border-left:3px solid var(--blue);background:var(--panel);font:12px/1.5 var(--mono)}.live-status[data-phase="waiting"]{border-left-color:var(--muted)}.live-status[data-phase="cached"]{border-left-color:var(--amber);color:#e4d8bb}.live-status[data-phase="disconnected"]{border-left-color:var(--red);color:#f4c6c1}
|
|
25
|
+
.live-controls{display:flex;flex-wrap:wrap;align-items:center;gap:14px;margin:0 0 26px}.live-controls label{color:var(--muted);font:10px/1.4 var(--mono);letter-spacing:.13em;text-transform:uppercase}.live-controls select{min-width:min(420px,100%);padding:10px 12px;color:var(--ink);background:var(--panel-2);border:1px solid var(--line);font:12px/1.4 var(--mono)}.live-controls select:focus-visible{outline:2px solid var(--green);outline-offset:3px}
|
|
26
|
+
.live-note{margin:10px 0 0;color:var(--amber);font:11px/1.5 var(--mono)}.live-provenance{margin:0 0 34px}
|
|
27
|
+
.live-grid{display:grid;grid-template-columns:repeat(3,minmax(0,1fr));border-top:1px solid var(--line);border-left:1px solid var(--line)}.live-card{min-height:132px;padding:20px;border-right:1px solid var(--line);border-bottom:1px solid var(--line);background:linear-gradient(145deg,rgba(255,255,255,.02),transparent)}.live-card-head{display:flex;justify-content:space-between;gap:12px;align-items:baseline}.live-card h3{margin:0;font:400 24px/1 var(--serif);text-transform:capitalize}.live-card p{margin:12px 0 0;color:#aab5ad;font:12px/1.6 var(--mono);overflow-wrap:anywhere}.live-pill{font:700 10px/1.3 var(--mono);letter-spacing:.07em;text-transform:uppercase}.live-card[data-tone="amber"] .live-pill{color:var(--amber)}.live-card[data-tone="red"] .live-pill{color:var(--red)}.live-card[data-tone="blue"] .live-pill{color:var(--blue)}.live-card[data-tone="muted"] .live-pill{color:var(--muted)}
|
|
28
|
+
.pinned{margin:0;border-top:1px solid var(--line)}.pinned summary{padding:34px 0;cursor:pointer;color:var(--muted);font:11px/1.5 var(--mono);letter-spacing:.14em;text-transform:uppercase}.pinned summary:focus-visible{outline:2px solid var(--ink);outline-offset:4px}.pinned summary strong{display:block;margin-top:8px;color:var(--ink);font:400 clamp(28px,4vw,44px)/1.05 var(--serif);letter-spacing:-.03em;text-transform:none}.pinned section:first-of-type{border-top:0}
|
|
22
29
|
footer{padding:38px 0 54px;display:flex;justify-content:space-between;gap:24px;color:#738078;font:11px/1.6 var(--mono);text-transform:uppercase;letter-spacing:.1em}
|
|
23
|
-
@media (max-width:820px){.hero{min-height:auto;grid-template-columns:1fr;padding-top:80px}.warning,.section-head{grid-template-columns:1fr}.suite-grid{grid-template-columns:1fr 1fr}.coverage{grid-template-columns:1fr 1fr}.provenance{grid-template-columns:1fr}.method{grid-column:auto}}
|
|
24
|
-
@media (max-width:540px){.wrap{width:min(100% - 24px,1180px)}nav{padding:18px 0}.nav-meta{display:none}h1{font-size:clamp(42px,14vw,60px);line-height:.94}.hero{padding-top:58px;padding-bottom:38px;gap:28px}.warning{margin-bottom:46px;padding:18px}.suite-grid,.coverage{grid-template-columns:1fr}.suite-card{min-height:190px}.suite-card[data-state="unavailable"]{min-height:0;padding:16px 20px}.suite-card[data-state="unavailable"] h3{margin:18px 0 7px;font-size:25px}.suite-card[data-state="unavailable"] ul{font-size:12px}.section-head{margin-bottom:28px}section{padding:52px 0}.provenance{display:block}.datum+.datum{border-top:1px solid var(--line)}footer{flex-direction:column;font-size:12px}}
|
|
30
|
+
@media (max-width:820px){.hero{min-height:auto;grid-template-columns:1fr;padding-top:80px}.warning,.section-head{grid-template-columns:1fr}.suite-grid{grid-template-columns:1fr 1fr}.live-grid{grid-template-columns:1fr 1fr}.coverage{grid-template-columns:1fr 1fr}.provenance{grid-template-columns:1fr}.method{grid-column:auto}}
|
|
31
|
+
@media (max-width:540px){.wrap{width:min(100% - 24px,1180px)}nav{padding:18px 0}.nav-meta{display:none}h1{font-size:clamp(42px,14vw,60px);line-height:.94}.hero{padding-top:58px;padding-bottom:38px;gap:28px}.warning{margin-bottom:46px;padding:18px}.suite-grid,.coverage,.live-grid{grid-template-columns:1fr}.suite-card{min-height:190px}.suite-card[data-state="unavailable"]{min-height:0;padding:16px 20px}.suite-card[data-state="unavailable"] h3{margin:18px 0 7px;font-size:25px}.suite-card[data-state="unavailable"] ul{font-size:12px}.section-head{margin-bottom:28px}section{padding:52px 0}.provenance{display:block}.datum+.datum{border-top:1px solid var(--line)}.live-controls select{min-width:100%}footer{flex-direction:column;font-size:12px}}
|
|
25
32
|
@media (prefers-reduced-motion: reduce){html{scroll-behavior:auto}*,*:before,*:after{animation-duration:.01ms!important;animation-iteration-count:1!important;transition-duration:.01ms!important}}
|
|
26
33
|
</style>
|
|
27
34
|
</head>
|
|
28
35
|
<body>
|
|
29
|
-
<a class="skip" href="#
|
|
36
|
+
<a class="skip" href="#live-eval">Skip to live evaluation</a>
|
|
30
37
|
<div class="wrap">
|
|
31
|
-
<nav aria-label="Evaluation navigation"><a class="brand" href="../">Oracle / Eval</a><span class="nav-meta">
|
|
38
|
+
<nav aria-label="Evaluation navigation"><a class="brand" href="../">Oracle / Eval</a><span class="nav-meta">live run updates · immutable records</span></nav>
|
|
32
39
|
<header class="hero">
|
|
33
|
-
<div><p class="eyebrow">Public evaluation record</p><h1>Evaluation <em>evidence</em></h1><p class="dek">
|
|
34
|
-
<aside class="
|
|
40
|
+
<div><p class="eyebrow">Public evaluation record</p><h1>Evaluation <em>evidence</em></h1><p class="dek">Live run updates as each evaluation progresses, plus immutable records of every run that came before. Missing evidence remains visible as missing.</p></div>
|
|
41
|
+
<aside class="eval-index" aria-label="Explore the evidence"><span class="kicker">Inspect the harness</span><p><a href="#live-eval">01 / Live runs</a></p><p><a href="#eval-story-method">02 / Evaluation method</a></p><p><a href="#eval-story-rlm">03 / Bounded RLM research</a></p><p><a href="#eval-story-ladder">04 / Crypto capability layers</a></p><p><a href="#eval-story-benchmark">05 / Protocol case study</a></p></aside>
|
|
35
42
|
</header>
|
|
36
|
-
<aside class="warning" aria-labelledby="historical-title"><h2 id="historical-title">Historical snapshot</h2><p><strong>This is evidence for the report commit only.</strong> The exact report commit <code>6a9feb3a90b596b22744777143d6d909bc4d31df</code> does not match the current tree. Do not read this page as a claim about untested changes.</p></aside>
|
|
37
43
|
</div>
|
|
38
44
|
<main id="evidence">
|
|
39
|
-
<section aria-labelledby="
|
|
45
|
+
<section id="live-eval" aria-labelledby="live-title"><div class="wrap"><div class="section-head"><span class="kicker">01 / Live evaluation</span><div><h2 id="live-title">Live evaluation</h2><p>Follow each selected slice from queued to its recorded outcome. Open the provenance panel for the exact source, timestamps, and hash-checked JSON. Producer-recorded results are not independent attestation.</p></div></div>
|
|
46
|
+
<p class="live-status" id="live-feed-status" data-phase="waiting" role="status" aria-live="polite">Waiting for the first published run.</p>
|
|
47
|
+
<div class="live-controls"><label for="live-run-select">Run history</label><select id="live-run-select" aria-describedby="live-feed-status"><option value="latest" selected>Follow latest</option></select></div>
|
|
48
|
+
<div class="live-grid" id="live-slices" aria-live="polite"></div>
|
|
49
|
+
<details class="live-details"><summary>Run provenance and downloadable evidence</summary>
|
|
50
|
+
<dl class="provenance live-provenance">
|
|
51
|
+
<div class="datum"><dt>Recorded status</dt><dd id="live-run-status">not published</dd></div>
|
|
52
|
+
<div class="datum"><dt>Feed liveness</dt><dd id="live-liveness">not published</dd></div>
|
|
53
|
+
<div class="datum"><dt>Target</dt><dd id="live-target">not published</dd></div>
|
|
54
|
+
<div class="datum"><dt>Exact commit</dt><dd id="live-commit">not published</dd></div>
|
|
55
|
+
<div class="datum"><dt>Working tree</dt><dd id="live-dirty">not published</dd></div>
|
|
56
|
+
<div class="datum"><dt>Package version</dt><dd id="live-version">not published</dd><p class="live-note">Source package version, not a confirmed installed release. Scope is development only, never an npm equivalence claim.</p></div>
|
|
57
|
+
<div class="datum"><dt>Started at</dt><dd id="live-started">not published</dd></div>
|
|
58
|
+
<div class="datum"><dt>Updated at</dt><dd id="live-updated">not published</dd></div>
|
|
59
|
+
<div class="datum method"><dt>Downloadable record</dt><dd><a id="live-record" hidden>View the record for this run</a><span id="live-record-none">No validated record is available for this run.</span></dd></div>
|
|
60
|
+
</dl></details>
|
|
61
|
+
<p class="live-note">Recorded results, not an independent audit. Unselected slices remain not run. Models, fixtures, and live chain reads are identified separately below.</p>
|
|
62
|
+
</div></section>
|
|
63
|
+
<div class="wrap"><div class="eval-story" data-eval-story>
|
|
64
|
+
<section class="es-section" id="eval-story-method" aria-labelledby="eval-story-method-title"><header class="es-head"><p class="es-kicker">Method</p><h2 id="eval-story-method-title">How the evaluation runs</h2><p class="es-lede">Every published number comes out of the same path. The run pins its own source, isolates the runner, records each slice separately, and seals the result with a hash you can recompute.</p></header><figure class="es-fig" data-es-figure="method"><svg class="es-svg" viewBox="0 0 960 196" role="img" preserveAspectRatio="xMidYMid meet" aria-label="Schematic of the evaluation pipeline: record source identity, isolated runner, measured slices, immutable JSON revision, hash check, public evidence."><line class="es-axis" x1="30" y1="142" x2="930" y2="142" /><text class="es-axis-label" x="30" y="162">start</text><text class="es-axis-label" x="930" y="162" text-anchor="end">published record</text><line class="es-rail" x1="120" y1="92" x2="244.8" y2="92" /><line class="es-rail-flow" data-es-flow="0" x1="120" y1="92" x2="244.8" y2="92" /><line class="es-rail" x1="268.8" y1="92" x2="393.6" y2="92" /><line class="es-rail-flow" data-es-flow="1" x1="268.8" y1="92" x2="393.6" y2="92" /><line class="es-rail" x1="417.6" y1="92" x2="542.4000000000001" y2="92" /><line class="es-rail-flow" data-es-flow="2" x1="417.6" y1="92" x2="542.4000000000001" y2="92" /><line class="es-rail" x1="566.4000000000001" y1="92" x2="691.2" y2="92" /><line class="es-rail-flow" data-es-flow="3" x1="566.4000000000001" y1="92" x2="691.2" y2="92" /><line class="es-rail" x1="715.2" y1="92" x2="840" y2="92" /><line class="es-rail-flow" data-es-flow="4" x1="715.2" y1="92" x2="840" y2="92" /><rect class="es-node" data-es-node="0" x="101" y="85" width="14" height="14" /><rect class="es-node" data-es-node="1" x="249.8" y="85" width="14" height="14" /><rect class="es-node" data-es-node="2" x="398.6" y="85" width="14" height="14" /><rect class="es-node" data-es-node="3" x="547.4000000000001" y="85" width="14" height="14" /><rect class="es-node" data-es-node="4" x="696.2" y="85" width="14" height="14" /><rect class="es-node" data-es-node="5" x="845" y="85" width="14" height="14" /><text class="es-node-label" x="108" y="68" text-anchor="middle">Pin source</text><text class="es-node-index" x="108" y="120" text-anchor="middle">01</text><text class="es-node-label" x="256.8" y="68" text-anchor="middle">Isolated runner</text><text class="es-node-index" x="256.8" y="120" text-anchor="middle">02</text><text class="es-node-label" x="405.6" y="68" text-anchor="middle">Measured slices</text><text class="es-node-index" x="405.6" y="120" text-anchor="middle">03</text><text class="es-node-label" x="554.4000000000001" y="68" text-anchor="middle">Immutable revision</text><text class="es-node-index" x="554.4000000000001" y="120" text-anchor="middle">04</text><text class="es-node-label" x="703.2" y="68" text-anchor="middle">Hash check</text><text class="es-node-index" x="703.2" y="120" text-anchor="middle">05</text><text class="es-node-label" x="852" y="68" text-anchor="middle">Public evidence</text><text class="es-node-index" x="852" y="120" text-anchor="middle">06</text></svg><figcaption><p class="es-fig-label"><span>FIG. 1</span><span>Production method for one evaluation record</span><span class="es-scroll-hint">scroll the figure sideways</span></p><p class="es-caveat"><strong>Method schematic.</strong> The stages are drawn to explain the pipeline shape. They are not actual packet counts, timings, or case totals from any run.</p></figcaption></figure><div class="es-split"><ol class="es-steps"><li><p class="es-step-title"><span class="es-step-num">01</span>Pin source</p><p class="es-step-note">The source commit, working-tree state, and source package version identify the run.</p></li><li><p class="es-step-title"><span class="es-step-num">02</span>Isolated runner</p><p class="es-step-note">Child evaluations get a trimmed environment, not the ambient one.</p></li><li><p class="es-step-title"><span class="es-step-num">03</span>Measured slices</p><p class="es-step-note">Each slice reports its own cases, passes, failures, and blocks.</p></li><li><p class="es-step-title"><span class="es-step-num">04</span>Immutable revision</p><p class="es-step-note">Each update is written as a new JSON revision. Earlier revisions are never overwritten.</p></li><li><p class="es-step-title"><span class="es-step-num">05</span>Hash check</p><p class="es-step-note">SHA-256 covers the exact published bytes so mismatches are detectable.</p></li><li><p class="es-step-title"><span class="es-step-num">06</span>Public evidence</p><p class="es-step-note">The same file the site reads is the file you can download.</p></li></ol><aside class="es-aside"><h3>Statuses stay honest</h3><p>A slice reports one of these. A missing or blocked slice keeps its own status and never reads zero, because zero would look like a measured failure instead of an absence.</p><ul class="es-status-list"><li><span class="es-status" data-es-status="passed">passed</span></li><li><span class="es-status" data-es-status="failed">failed</span></li><li><span class="es-status" data-es-status="blocked">blocked</span></li><li><span class="es-status" data-es-status="skipped">skipped</span></li><li><span class="es-status" data-es-status="unavailable">unavailable</span></li></ul><p class="es-aside-note" data-es-activity>No evaluation record is attached to this page yet. The schematic above is complete on its own.</p><p class="es-source">Source: src/cli/commands/eval.mjs</p></aside></div></section>
|
|
65
|
+
<section class="es-section" id="eval-story-rlm" aria-labelledby="eval-story-rlm-title"><header class="es-head"><p class="es-kicker">Architecture</p><h2 id="eval-story-rlm-title">One parent. Bounded research.</h2><p class="es-lede">Oracle can hand a research question to a child that reads, then returns a result. The parent keeps the tools that matter. The point is delegation you can reason about, not an agent tree that grows on its own.</p></header><figure class="es-fig" data-es-figure="rlm"><svg class="es-svg" viewBox="0 0 960 292" role="img" preserveAspectRatio="xMidYMid meet" aria-label="Architecture walkthrough: a parent session binds variables, sends a goal and context to one depth-1 child with read only tools, receives a structured result, while signing, file writes, and nested spawning stay refused for the child."><g class="es-group" data-es-part="parent"><rect class="es-node es-node-lg" x="40" y="40" width="18" height="18" /><text class="es-node-label" x="70" y="46">Parent session</text><text class="es-node-sub" x="70" y="66">coordinates tools and approvals</text></g><g class="es-group" data-es-part="context"><rect class="es-panel-box" x="40" y="96" width="262" height="88" /><text class="es-node-sub" x="58" y="122">bound variables</text><rect class="es-chip" x="58" y="136" width="68" height="26" /><text class="es-chip-label" x="92" y="153" text-anchor="middle">pair</text><rect class="es-chip" x="138" y="136" width="68" height="26" /><text class="es-chip-label" x="172" y="153" text-anchor="middle">route</text><rect class="es-chip" x="218" y="136" width="68" height="26" /><text class="es-chip-label" x="252" y="153" text-anchor="middle">limits</text></g><line class="es-rail" x1="310" y1="112" x2="454" y2="112" /><line class="es-rail-flow" data-es-flow="out" x1="310" y1="112" x2="454" y2="112" /><path class="es-arrow" d="M454 106 L466 112 L454 118 Z" /><text class="es-axis-label" x="310" y="102">goal, context, vars</text><line class="es-rail" x1="454" y1="170" x2="322" y2="170" /><line class="es-rail-flow" data-es-flow="back" x1="454" y1="170" x2="322" y2="170" /><path class="es-arrow" d="M322 164 L310 170 L322 176 Z" /><text class="es-axis-label" x="310" y="190">structured result</text><g class="es-group" data-es-part="child"><rect class="es-panel-box" x="470" y="66" width="256" height="130" /><rect class="es-node" x="490" y="90" width="14" height="14" /><text class="es-node-label" x="516" y="102">Child, depth 1</text><text class="es-node-sub" x="490" y="136">read only tools</text><text class="es-node-sub" x="490" y="160">4 steps default, 8 at most</text></g><g class="es-group" data-es-part="denied"><line class="es-deny-line" x1="762" y1="34" x2="762" y2="222" /><text class="es-deny-title" x="782" y="46">refused for children</text><rect class="es-deny-box" x="782" y="65" width="172" height="30" /><text class="es-deny-label" x="794" y="85">sign a transaction</text><rect class="es-deny-box" x="782" y="107" width="172" height="30" /><text class="es-deny-label" x="794" y="127">write a file</text><rect class="es-deny-box" x="782" y="149" width="172" height="30" /><text class="es-deny-label" x="794" y="169">spawn another child</text></g><line class="es-axis" x1="40" y1="238" x2="920" y2="238" /><text class="es-axis-label" x="40" y="258">depth ceiling 1</text><text class="es-axis-label" x="920" y="258" text-anchor="end">the host decides, not the prompt</text></svg><figcaption><p class="es-fig-label"><span>FIG. 2</span><span>Parent, one bounded child, and the refused edge</span><span class="es-scroll-hint">scroll the figure sideways</span></p><p class="es-caveat"><strong>Architecture walkthrough, not a live execution.</strong> No tool output is replayed here and nothing on this page is a recording of a session.</p></figcaption></figure><div class="es-stepper"><div class="es-tabs" role="tablist" aria-label="Recursive language model walkthrough steps"><button class="es-tab" type="button" role="tab" id="es-rlm-bind-tab" aria-controls="es-rlm-bind-panel" aria-selected="true" tabindex="0" data-es-step="0"><span class="es-tab-num">01</span><span class="es-tab-label">Bind context</span></button><button class="es-tab" type="button" role="tab" id="es-rlm-run-tab" aria-controls="es-rlm-run-panel" aria-selected="false" tabindex="-1" data-es-step="1"><span class="es-tab-num">02</span><span class="es-tab-label">Run child</span></button><button class="es-tab" type="button" role="tab" id="es-rlm-return-tab" aria-controls="es-rlm-return-panel" aria-selected="false" tabindex="-1" data-es-step="2"><span class="es-tab-num">03</span><span class="es-tab-label">Return evidence</span></button><button class="es-tab" type="button" role="tab" id="es-rlm-refuse-tab" aria-controls="es-rlm-refuse-panel" aria-selected="false" tabindex="-1" data-es-step="3"><span class="es-tab-num">04</span><span class="es-tab-label">Refuse unsafe tools</span></button></div><div class="es-panels" data-es-panels><div class="es-panel" role="tabpanel" id="es-rlm-bind-panel" aria-labelledby="es-rlm-bind-tab" data-es-step-panel="0" tabindex="0"><p class="es-panel-kicker">Parent holds the working set</p><p class="es-panel-body">The parent binds named variables and keeps them across turns. Values are capped at 16 KB each, so context stays a small readable object rather than a growing transcript.</p><p class="es-source">Source: rlm.bind / rlm.get / rlm.vars in src/tui/rlm.mjs</p></div><div class="es-panel" role="tabpanel" id="es-rlm-run-panel" aria-labelledby="es-rlm-run-tab" data-es-step-panel="1" tabindex="0"><p class="es-panel-kicker">One bounded research branch</p><p class="es-panel-body">rlm.run hands the child a packet: goal, context, the bound variables, and an allow list of read only tools. The child runs at depth 1 with a step budget of 4 by default and 8 at most.</p><p class="es-source">Source: runRlmTool packet build in src/tui/rlm.mjs</p></div><div class="es-panel" role="tabpanel" id="es-rlm-return-panel" aria-labelledby="es-rlm-return-tab" data-es-step-panel="2" tabindex="0"><p class="es-panel-kicker">Structured result, then control returns</p><p class="es-panel-body">Only wait=true is supported, so the child answer comes back into the same parent turn as a stored result with a handle. Background runs are refused rather than queued.</p><p class="es-source">Source: wait=false refusal in src/tui/rlm.mjs</p></div><div class="es-panel" role="tabpanel" id="es-rlm-refuse-panel" aria-labelledby="es-rlm-refuse-tab" data-es-step-panel="3" tabindex="0"><p class="es-panel-kicker">The boundary is code, not a prompt</p><p class="es-panel-body">A child cannot call rlm.run again, cannot sign, and cannot write files. Those refusals are evaluated by the host before the tool is dispatched.</p><p class="es-source">Source: evaluateRlmChildTool in src/tui/rlm.mjs</p></div></div></div><div class="es-facts"><div class="es-fact"><p class="es-fact-value">Depth ceiling 1</p><p class="es-fact-note">A child cannot open another child. The ceiling is a constant, so raising the environment variable cannot exceed it.</p><p class="es-source">Source: src/tui/rlm.mjs</p></div><div class="es-fact"><p class="es-fact-value">12 deterministic checks</p><p class="es-fact-note">Bind, run, result, refusals, and the depth ceiling are graded with a stub child runner. No model call, no network, and no signing take place, so these checks prove mechanics rather than research quality.</p><p class="es-source">Source: src/eval/rlm-eval.mjs</p></div></div></section>
|
|
66
|
+
<section class="es-section" id="eval-story-ladder" aria-labelledby="eval-story-ladder-title"><header class="es-head"><p class="es-kicker">Coverage</p><h2 id="eval-story-ladder-title">Crypto capability, measured in layers</h2><p class="es-lede">Different layers carry different weight. We keep them apart on purpose: code that was exercised, corpora that exist, and work that is still ahead. There is no single number that blends them.</p></header><figure class="es-fig es-fig-plain" data-es-figure="ladder"><ol class="es-ladder"><li class="es-layer" data-es-state="measured"><div class="es-layer-head"><p class="es-layer-title">Implementation checks</p><p class="es-layer-count">12 checks</p></div><div class="es-matrix" aria-hidden="true"><span class="es-cell" data-es-cell="on"></span></div><p class="es-layer-state">measured in the release run</p><p class="es-layer-detail">Bind, run, result, depth ceiling, and refusal mechanics run with a stub child runner. Deterministic and repeatable.</p><p class="es-source">Source: src/eval/rlm-eval.mjs</p></li><li class="es-layer" data-es-state="measured"><div class="es-layer-head"><p class="es-layer-title">Read only preparation</p><p class="es-layer-count">3 cases</p></div><div class="es-matrix" aria-hidden="true"><span class="es-cell" data-es-cell="on"></span></div><p class="es-layer-state">measured in the release run</p><p class="es-layer-detail">Ethereum, Base, and Arbitrum swap intents are prepared and simulated from an unfunded address. The expected result is execution reverted: STF. Nothing is signed and nothing is broadcast.</p><p class="es-source">Source: scripts/prepare-fidelity-eval.mjs</p></li><li class="es-layer" data-es-state="corpus"><div class="es-layer-head"><p class="es-layer-title">Policy decisions</p><p class="es-layer-count">45 cases</p></div><div class="es-matrix" aria-hidden="true"><span class="es-cell" data-es-cell="on"></span></div><p class="es-layer-state">corpus size, not a fresh pass</p><p class="es-layer-detail">An evidence bound classifier that grades the next safe action and the highest lifecycle state already shown. This is a corpus size, not a result from today.</p><p class="es-source">Source: environments/oracle_harness_eval/oracle_harness_eval_v2.py</p></li><li class="es-layer" data-es-state="corpus"><div class="es-layer-head"><p class="es-layer-title">Mock tool behavior</p><p class="es-layer-count">36 cases</p></div><div class="es-matrix" aria-hidden="true"><span class="es-cell" data-es-cell="on"></span></div><p class="es-layer-state">corpus size, not a fresh pass</p><p class="es-layer-detail">Twelve reads, twelve simulations, and twelve execute to prepare traps against inert mock tools whose sign and broadcast paths always deny. Also a corpus size, not a result from today.</p><p class="es-source">Source: environments/oracle_harness_eval/oracle_tool_safety.py</p></li><li class="es-layer" data-es-state="planned"><div class="es-layer-head"><p class="es-layer-title">Model trajectories</p><p class="es-layer-count">not yet published</p></div><div class="es-matrix" aria-hidden="true"><span class="es-cell" data-es-cell="off"></span></div><p class="es-layer-state">planned, in development</p><p class="es-layer-detail">Full agent runs on crypto tasks, graded end to end. This layer is in development and we have not published it as a comprehensive benchmark.</p><p class="es-source">Source: in development</p></li></ol><figcaption><p class="es-fig-label"><span>FIG. 3</span><span>Evidence layers, kept separate</span></p><p class="es-caveat">Case counts for the policy and tool suites are a <strong>corpus size</strong>, meaning how many cases exist, not a pass count from a run today. The tool suite uses <strong>inert mock tools</strong> whose sign and broadcast paths always deny.</p></figcaption></figure><div class="es-next"><h3>Next: agentic crypto tasks</h3><p>We are building our own task set for crypto agents. It is incomplete and in development, and we are not claiming a result from it. What it will grade:</p><ul class="es-measures"><li><p class="es-measure-title">Research and data honesty</p><p class="es-measure-note">Does the answer match what the tools actually returned, including gaps.</p></li><li><p class="es-measure-title">Exact unsigned preparation</p><p class="es-measure-note">Are chain, route, amounts, and calldata right in the artifact it builds.</p></li><li><p class="es-measure-title">Policy and confirmation boundaries</p><p class="es-measure-note">Does it stop where a person has to confirm, every time.</p></li><li><p class="es-measure-title">Receipt truth</p><p class="es-measure-note">Does it report what settled, without describing work it did not do.</p></li><li><p class="es-measure-title">RLM task completion</p><p class="es-measure-note">Does the bounded child branch return usable evidence to the parent.</p></li><li><p class="es-measure-title">Protocol build acceptance</p><p class="es-measure-note">Does generated contract work pass tests it never saw.</p></li></ul><p class="es-next-note">The primary measures are task completion, correctness of the artifact, forbidden tool calls, evidence fidelity, and refusals that were not warranted. Performance is evaluated by correct outcomes and verifiable evidence, not by a race to produce an answer.</p></div></section>
|
|
67
|
+
<section class="es-section" id="eval-story-benchmark" aria-labelledby="eval-story-benchmark-title"><header class="es-head"><p class="es-kicker">Case study</p><h2 id="eval-story-benchmark-title">Protocol building: historical case study</h2><p class="es-lede">Historical custom EVM milestone escrow task. Local compile and tests only, no deployment. Existing generated artifacts revalidated offline, not fresh model generations. Oracle means MCP plus protocol-builder skill, not a native-harness head-to-head.</p></header><figure class="es-fig" data-es-figure="benchmark"><table class="es-table"><caption>Hidden test acceptance by arm. The scale starts at zero and every value is taken from the record.</caption><thead><tr><th scope="col">Arm</th><th scope="col">Hidden tests</th><th scope="col">Own tests</th><th scope="col">Targeted source review</th></tr></thead><tbody><tr data-es-arm="codex-clean" data-es-oracle="no"><th scope="row"><span class="es-arm-label">Codex</span><span class="es-arm-meta">Codex CLI 0.150.1, GPT-5.6-Sol</span></th><td class="es-bar-cell" data-label="Hidden tests"><span class="es-bar-track"><span class="es-bar-fill" data-es-bar style="--es-fill:100%"></span></span><span class="es-bar-value">28 / 28</span></td><td class="es-own" data-label="Own tests">8 / 8</td><td class="es-audit" data-label="Historical source review">One literal checks-effects-interactions ordering miss in fund()</td></tr><tr data-es-arm="codex-oracle-clean" data-es-oracle="yes"><th scope="row"><span class="es-arm-label">Codex + Oracle</span><span class="es-arm-meta">Codex CLI 0.150.1, GPT-5.6-Sol</span></th><td class="es-bar-cell" data-label="Hidden tests"><span class="es-bar-track"><span class="es-bar-fill" data-es-bar style="--es-fill:100%"></span></span><span class="es-bar-value">28 / 28</span></td><td class="es-own" data-label="Own tests">9 / 9</td><td class="es-audit" data-label="Historical source review">No targeted specification gaps found</td></tr><tr data-es-arm="claude-clean" data-es-oracle="no"><th scope="row"><span class="es-arm-label">Claude</span><span class="es-arm-meta">Claude Code 2.1.247, Opus 5</span></th><td class="es-bar-cell" data-label="Hidden tests"><span class="es-bar-track"><span class="es-bar-fill" data-es-bar style="--es-fill:100%"></span></span><span class="es-bar-value">28 / 28</span></td><td class="es-own" data-label="Own tests">82 / 82</td><td class="es-audit" data-label="Historical source review">No targeted specification gaps found</td></tr><tr data-es-arm="claude-oracle-clean" data-es-oracle="yes"><th scope="row"><span class="es-arm-label">Claude + Oracle</span><span class="es-arm-meta">Claude Code 2.1.247, Opus 5</span></th><td class="es-bar-cell" data-label="Hidden tests"><span class="es-bar-track"><span class="es-bar-fill" data-es-bar style="--es-fill:100%"></span></span><span class="es-bar-value">28 / 28</span></td><td class="es-own" data-label="Own tests">91 / 91</td><td class="es-audit" data-label="Historical source review">No targeted specification gaps found</td></tr></tbody></table><figcaption><p class="es-fig-label"><span>FIG. 4</span><span>Hidden test acceptance, zero baseline</span></p><p class="es-caveat"><strong>Historical run, measured 2026-08-27.</strong> One custom EVM task. One run per arm. Local compile and tests only, no deploy, and not a formal security audit. This compares Claude and Codex with and without Oracle as an MCP server and skill. It is not an Oracle model against another model.</p><p class="es-caveat"><strong>Self generated tests measure assurance effort, not code quality.</strong> An arm can write a small suite it passes easily, so the own tests column is kept separate from the hidden tests it never saw.</p></figcaption></figure><p class="es-evidence"><a href="./benchmarks/evm-escrow-20260827.json">Download the comparison evidence</a></p></section>
|
|
68
|
+
</div></div>
|
|
69
|
+
<details class="pinned" id="pinned-release"><summary class="wrap"><span>Release evidence, separate from development runs</span><strong>Pinned release snapshot</strong></summary>
|
|
70
|
+
<div class="wrap">
|
|
71
|
+
<aside class="verdict" data-state="attention" aria-label="Snapshot verdict"><span class="verdict-label">Verdict</span><p class="verdict-value">PINNED RELEASE EVIDENCE · NOT CURRENT TREE</p><span class="nav-meta">8 reported suites · 9 required slices</span></aside>
|
|
72
|
+
<aside class="warning" aria-labelledby="historical-title"><h2 id="historical-title">Historical snapshot</h2><p><strong>This is evidence for the report commit only.</strong> The exact report commit <code>6a9feb3a90b596b22744777143d6d909bc4d31df</code> does not match the current tree. Do not read this page as a claim about untested changes.</p></aside>
|
|
73
|
+
</div>
|
|
74
|
+
<section aria-labelledby="suite-title"><div class="wrap"><div class="section-head"><span class="kicker">02 / Evidence map</span><div><h2 id="suite-title">Expected eval slices</h2><p>Every expected slice has a place. Absent, blocked, skipped, and unavailable results are neutral or adverse states—never green and never counted as zero.</p></div></div><div class="suite-grid">
|
|
40
75
|
<article class="suite-card" data-slice="guards" data-state="unattested" aria-labelledby="suite-guards">
|
|
41
76
|
<div class="suite-topline"><span class="slice-index">01</span><span class="state-pill">recorded pass · unattested</span></div>
|
|
42
77
|
<h3 id="suite-guards">guards</h3>
|
|
@@ -88,8 +123,8 @@
|
|
|
88
123
|
<ul><li>No result was recorded for this slice.</li></ul>
|
|
89
124
|
</article>
|
|
90
125
|
</div></div></section>
|
|
91
|
-
<section aria-labelledby="coverage-title"><div class="wrap"><div class="section-head"><span class="kicker">
|
|
92
|
-
<section aria-labelledby="provenance-title"><div class="wrap"><div class="section-head"><span class="kicker">
|
|
126
|
+
<section aria-labelledby="coverage-title"><div class="wrap"><div class="section-head"><span class="kicker">03 / Attack surface</span><div><h2 id="coverage-title">Adversarial coverage</h2><p>Classes explicitly named by the committed report. No inferred coverage and no synthetic score.</p></div></div><ul class="coverage"><li>exact-confirmation</li><li>artifact-mutation</li><li>replay-idempotency</li><li>receipt-finality</li><li>autonomous-mint-gates</li><li>exact-nft-contract</li><li>hyperliquid-position-drift</li></ul></div></section>
|
|
127
|
+
<section aria-labelledby="provenance-title"><div class="wrap"><div class="section-head"><span class="kicker">04 / Provenance</span><div><h2 id="provenance-title">Pin the claim</h2><p>The timestamp, package, commit, and reproduction command define the boundary of this evidence.</p></div></div><dl class="provenance">
|
|
93
128
|
<div class="datum"><dt>Exact report commit</dt><dd><a href="https://github.com/demi-hl/oracle/commit/6a9feb3a90b596b22744777143d6d909bc4d31df">6a9feb3a90b596b22744777143d6d909bc4d31df</a></dd></div>
|
|
94
129
|
<div class="datum"><dt>Generated at</dt><dd><time datetime="2026-09-05T01:10:54.826Z">2026-09-05T01:10:54.826Z</time></dd></div>
|
|
95
130
|
<div class="datum"><dt>Package version</dt><dd><a href="https://www.npmjs.com/package/@oracle-agent/oracle/v/0.35.41">@oracle-agent/oracle · 0.35.41</a></dd></div>
|
|
@@ -98,7 +133,10 @@
|
|
|
98
133
|
<div class="datum method"><dt>Methodology and reproduction</dt><dd>Exact-commit evidence generated by the deterministic oracle eval all aggregate. Fidelity uses read-only chain calls and changes no chain state.<code class="reproduce">npx @oracle-agent/oracle@0.35.41 eval all --mode playwright --json</code></dd></div>
|
|
99
134
|
<div class="datum method"><dt>Download records</dt><dd><a href="./releases/oracle-0.35.41.json">Full 0.35.41 release report</a> · <a href="./archive/oracle-0.35.40.json">Archived 0.35.40 report</a> · <a href="./archive/oracle-0.35.41-blocked-attempt.json">Prior blocked attempt</a>. The public records declare their local-path redaction and retain original stdout hashes. The first fresh run failed closed during Chromium process cleanup; the separately recorded retry passed.</dd></div>
|
|
100
135
|
</dl></div></section>
|
|
136
|
+
</details>
|
|
101
137
|
</main>
|
|
102
|
-
<div class="wrap"><footer><span>Oracle evaluation evidence</span><span>
|
|
138
|
+
<div class="wrap"><footer><span>Oracle evaluation evidence</span><span>Live run updates · immutable records</span></footer></div>
|
|
139
|
+
<script type="module" src="./eval-story.js"></script>
|
|
140
|
+
<script type="module" src="./live-eval.js"></script>
|
|
103
141
|
</body>
|
|
104
142
|
</html>
|