opencode-agent-skill 9.0.0 → 11.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +116 -0
- package/README.md +742 -675
- package/bin/ocskill.mjs +354 -5
- package/docs/V11-PERCEPTION-ADAPTIVE-EXECUTION.md +75 -0
- package/docs/V11-PERCEPTION-ADAPTIVE.md +220 -0
- package/evals/router-triggers.json +82 -0
- package/evals/routing.json +76 -0
- package/evals/v11/tasks.json +122 -0
- package/global-config/AGENTS.md +78 -163
- package/global-config/agents/merge-arbiter.md +12 -0
- package/global-config/agents/visual-verifier.md +12 -0
- package/global-config/plugins/ues-router/index.js +683 -59
- package/global-config/plugins/ues-router/router.js +62 -3
- package/global-config/plugins/ues-router/runtime-guard.js +265 -0
- package/global-config/skills/browser-qa/SKILL.md +14 -0
- package/global-config/skills/browser-qa/references/workflow.md +11 -0
- package/global-config/skills/browser-security/SKILL.md +12 -0
- package/global-config/skills/component-visual-testing/SKILL.md +10 -0
- package/global-config/skills/design-source/SKILL.md +10 -0
- package/global-config/skills/design-source/references/workflow.md +12 -0
- package/global-config/skills/dynamic-workflow/SKILL.md +18 -0
- package/global-config/skills/dynamic-workflow/references/workflow.md +19 -0
- package/global-config/skills/responsive-verification/SKILL.md +10 -0
- package/global-config/skills/skill-authoring/SKILL.md +12 -0
- package/global-config/skills/skill-evaluation/SKILL.md +17 -0
- package/global-config/skills/visual-fidelity/SKILL.md +14 -0
- package/global-config/skills/visual-fidelity/references/workflow.md +14 -0
- package/lib/benchmark-confidence.mjs +49 -11
- package/lib/browser-adapter.mjs +82 -0
- package/lib/browser-runtime.mjs +193 -0
- package/lib/capability-registry.mjs +109 -0
- package/lib/context-engine-v11.mjs +146 -0
- package/lib/context-manifest.mjs +16 -3
- package/lib/control-center.mjs +12 -2
- package/lib/dynamic-workflow.mjs +179 -0
- package/lib/eval-ablation.mjs +146 -0
- package/lib/eval-report.mjs +83 -0
- package/lib/eval-telemetry.mjs +64 -0
- package/lib/evidence-budget.mjs +84 -0
- package/lib/evidence-store.mjs +178 -0
- package/lib/hermes-bridge.mjs +45 -1
- package/lib/model-config.mjs +9 -1
- package/lib/model-policy.mjs +58 -1
- package/lib/orchestrator-policy.mjs +100 -7
- package/lib/png-diff.mjs +229 -0
- package/lib/prompt-cache.mjs +60 -0
- package/lib/skill-quality.mjs +72 -0
- package/lib/task-engine.mjs +223 -12
- package/lib/ui-inspector.mjs +152 -0
- package/lib/v11-metrics.mjs +64 -0
- package/lib/visual-spec.mjs +159 -0
- package/package.json +11 -5
- package/scripts/eval-ablation.mjs +47 -0
- package/scripts/eval-matrix.mjs +13 -2
- package/scripts/validate-v11-suite.mjs +58 -0
- package/scripts/validate.mjs +16 -4
package/lib/control-center.mjs
CHANGED
|
@@ -3,6 +3,7 @@ import { mkdir, readFile, readdir, writeFile } from "node:fs/promises"
|
|
|
3
3
|
import path from "node:path"
|
|
4
4
|
import { readLearningState } from "./learning-engine.mjs"
|
|
5
5
|
import { readRuntimeEvents } from "./runtime-events.mjs"
|
|
6
|
+
import { evidenceStoreStatus } from "./evidence-store.mjs"
|
|
6
7
|
|
|
7
8
|
function escapeHtml(value) {
|
|
8
9
|
return String(value ?? "")
|
|
@@ -64,18 +65,23 @@ async function collectEvalSummary(root) {
|
|
|
64
65
|
|
|
65
66
|
export async function collectControlCenterData(root = process.cwd()) {
|
|
66
67
|
root = path.resolve(root)
|
|
67
|
-
const [work, learning, evals] = await Promise.all([
|
|
68
|
+
const [work, learning, evals, evidenceStore] = await Promise.all([
|
|
68
69
|
collectWork(root),
|
|
69
70
|
readLearningState(root),
|
|
70
71
|
collectEvalSummary(root),
|
|
72
|
+
evidenceStoreStatus(root),
|
|
71
73
|
])
|
|
72
74
|
return {
|
|
73
|
-
schemaVersion:
|
|
75
|
+
schemaVersion: 2,
|
|
74
76
|
generatedAt: new Date().toISOString(),
|
|
75
77
|
root,
|
|
76
78
|
work,
|
|
77
79
|
learning,
|
|
78
80
|
evals,
|
|
81
|
+
v11: {
|
|
82
|
+
evidenceStore,
|
|
83
|
+
runtime: "perception-adaptive-execution",
|
|
84
|
+
},
|
|
79
85
|
}
|
|
80
86
|
}
|
|
81
87
|
|
|
@@ -103,6 +109,7 @@ details{margin-top:10px;border-top:1px solid #243049;padding-top:8px}summary{cur
|
|
|
103
109
|
<div class="top"><div><h1>UES Control Center</h1><div class="muted" id="root"></div></div><div class="muted" id="generated"></div></div>
|
|
104
110
|
<div class="section"><h2>Long-horizon work</h2><div class="grid" id="work"></div></div>
|
|
105
111
|
<div class="section"><h2>Learning loop</h2><div class="grid" id="learning"></div></div>
|
|
112
|
+
<div class="section"><h2>V11 runtime efficiency</h2><div class="grid" id="v11"></div></div>
|
|
106
113
|
<div class="section"><h2>Recent runtime events</h2><div class="card"><table><thead><tr><th>Work</th><th>Event</th><th>Task</th><th>Time</th></tr></thead><tbody id="events"></tbody></table></div></div>
|
|
107
114
|
<div class="section"><h2>Recent evaluations</h2><div class="card"><table><thead><tr><th>Suite</th><th>Model</th><th>Baseline</th><th>UES</th></tr></thead><tbody id="evals"></tbody></table></div></div>
|
|
108
115
|
</div>
|
|
@@ -126,6 +133,9 @@ function render(next){
|
|
|
126
133
|
if(!data.work.length) work.innerHTML='<div class="card muted">No .ues-work items found.</div>';
|
|
127
134
|
const learning=document.querySelector("#learning");
|
|
128
135
|
learning.innerHTML='<div class="card"><h3>Accepted lessons</h3><strong>'+data.learning.accepted.length+'</strong></div><div class="card"><h3>Proposals</h3><strong>'+data.learning.proposals.length+'</strong></div>';
|
|
136
|
+
const v11=document.querySelector("#v11");
|
|
137
|
+
const store=data.v11?.evidenceStore||{};
|
|
138
|
+
v11.innerHTML='<div class="card"><h3>Evidence store</h3><strong>'+esc(store.entries||0)+'</strong><div class="muted">'+esc(store.bytes||0)+' bytes externalized</div></div><div class="card"><h3>Runtime</h3><strong>V11</strong><div class="muted">'+esc(data.v11?.runtime||'adaptive')+'</div></div>';
|
|
129
139
|
const eventBody=document.querySelector("#events");
|
|
130
140
|
eventBody.innerHTML="";
|
|
131
141
|
for(const item of data.work){
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
function taskFiles(task = {}) {
|
|
2
|
+
if (Array.isArray(task.files)) return task.files
|
|
3
|
+
const files = task.files && typeof task.files === "object" ? task.files : {}
|
|
4
|
+
return [...new Set(["create","modify","test","delete"].flatMap((key) => Array.isArray(files[key]) ? files[key] : []))]
|
|
5
|
+
}
|
|
6
|
+
|
|
7
|
+
function writes(task = {}) {
|
|
8
|
+
const files = task.files && typeof task.files === "object" && !Array.isArray(task.files) ? task.files : null
|
|
9
|
+
if (!files) return Array.isArray(task.files) && task.files.length > 0
|
|
10
|
+
return ["create","modify","delete"].some((key) => Array.isArray(files[key]) && files[key].length > 0)
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
function clampInt(value, fallback, min, max) {
|
|
14
|
+
const parsed = Number(value)
|
|
15
|
+
if (!Number.isFinite(parsed)) return fallback
|
|
16
|
+
return Math.max(min, Math.min(max, Math.round(parsed)))
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
export function classifyWorkflowTask(task = {}) {
|
|
20
|
+
const text = [task.title, task.summary, ...(task.acceptance || [])].filter(Boolean).join(" ").toLowerCase()
|
|
21
|
+
const visual = /(visual|screenshot|pixel|figma|layout|giao diện|ảnh mẫu|image reference)/.test(text)
|
|
22
|
+
const deterministic = task.deterministic === true || /(run test|typecheck|lint|format|generate manifest|build index|verify command|compile|unit test)/.test(text)
|
|
23
|
+
const kind = deterministic ? "deterministic" : visual ? "vision" : "llm"
|
|
24
|
+
const fileCount = taskFiles(task).length
|
|
25
|
+
const acceptanceCount = Array.isArray(task.acceptance) ? task.acceptance.length : 0
|
|
26
|
+
const verificationCount = Array.isArray(task.verification) ? task.verification.length : 0
|
|
27
|
+
const estimatedCost = kind === "deterministic"
|
|
28
|
+
? 1
|
|
29
|
+
: Math.max(
|
|
30
|
+
2,
|
|
31
|
+
Math.min(
|
|
32
|
+
12,
|
|
33
|
+
2 + fileCount + Math.min(3, acceptanceCount) + Math.min(2, verificationCount) + (task.risk === "high" ? 3 : 0),
|
|
34
|
+
),
|
|
35
|
+
)
|
|
36
|
+
return {
|
|
37
|
+
kind,
|
|
38
|
+
estimatedCost,
|
|
39
|
+
writes: writes(task),
|
|
40
|
+
files: taskFiles(task),
|
|
41
|
+
risk: task.risk || "medium",
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
function conflict(a, b) {
|
|
46
|
+
if (!a.writes && !b.writes) return false
|
|
47
|
+
const aa = new Set(a.files)
|
|
48
|
+
return b.files.some((file) => aa.has(file))
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
function executionMode(classification, options) {
|
|
52
|
+
if (classification.kind === "deterministic") return "deterministic"
|
|
53
|
+
if (classification.kind === "vision") {
|
|
54
|
+
return classification.estimatedCost >= options.minVisionAgentCost ? "agent" : "inline"
|
|
55
|
+
}
|
|
56
|
+
return classification.estimatedCost >= options.minAgentCost ? "agent" : "inline"
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
function kindCapacity(selected, classification, options) {
|
|
60
|
+
if (classification.kind === "vision") {
|
|
61
|
+
return selected.filter((item) => item.classification.kind === "vision" && item.execution === "agent").length < options.maxVisionConcurrent
|
|
62
|
+
}
|
|
63
|
+
if (classification.kind === "llm") {
|
|
64
|
+
return selected.filter((item) => item.classification.kind === "llm" && item.execution === "agent").length < options.maxLLMConcurrent
|
|
65
|
+
}
|
|
66
|
+
return true
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
function readyTasks(remaining, byID, complete) {
|
|
70
|
+
return [...remaining]
|
|
71
|
+
.map((id) => byID.get(id))
|
|
72
|
+
.filter((task) => (task.dependsOn || task.dependencies || []).every((dep) => complete.has(dep)))
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
export function planDynamicWorkflow(tasks = [], inputOptions = {}) {
|
|
76
|
+
const options = {
|
|
77
|
+
maxConcurrent: clampInt(inputOptions.maxConcurrent, 4, 1, 16),
|
|
78
|
+
maxLLMConcurrent: clampInt(inputOptions.maxLLMConcurrent, inputOptions.maxConcurrent || 4, 1, 16),
|
|
79
|
+
maxVisionConcurrent: clampInt(inputOptions.maxVisionConcurrent, 2, 1, 8),
|
|
80
|
+
maxWaveCost: clampInt(inputOptions.maxWaveCost, 24, 1, 128),
|
|
81
|
+
minAgentCost: clampInt(inputOptions.minAgentCost, 5, 2, 12),
|
|
82
|
+
minVisionAgentCost: clampInt(inputOptions.minVisionAgentCost, 4, 2, 12),
|
|
83
|
+
deterministicFirst: inputOptions.deterministicFirst !== false,
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
const byID = new Map(tasks.map((task) => [task.id, task]))
|
|
87
|
+
if (byID.size !== tasks.length || tasks.some((task) => !task?.id)) {
|
|
88
|
+
throw new Error("Workflow tasks require unique non-empty ids")
|
|
89
|
+
}
|
|
90
|
+
for (const task of tasks) {
|
|
91
|
+
for (const dep of task.dependsOn || task.dependencies || []) {
|
|
92
|
+
if (!byID.has(dep)) throw new Error("Workflow contains a missing dependency: " + dep)
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
const remaining = new Set(tasks.map((task) => task.id))
|
|
97
|
+
const complete = new Set()
|
|
98
|
+
const waves = []
|
|
99
|
+
let waveIndex = 0
|
|
100
|
+
|
|
101
|
+
while (remaining.size) {
|
|
102
|
+
const ready = readyTasks(remaining, byID, complete)
|
|
103
|
+
if (!ready.length) throw new Error("Workflow contains a dependency cycle or missing dependency")
|
|
104
|
+
|
|
105
|
+
const deterministicReady = ready.filter((task) => classifyWorkflowTask(task).kind === "deterministic")
|
|
106
|
+
const candidates = options.deterministicFirst && deterministicReady.length
|
|
107
|
+
? deterministicReady
|
|
108
|
+
: ready
|
|
109
|
+
|
|
110
|
+
const classified = candidates
|
|
111
|
+
.map((task) => {
|
|
112
|
+
const classification = classifyWorkflowTask(task)
|
|
113
|
+
return { task, classification, execution: executionMode(classification, options) }
|
|
114
|
+
})
|
|
115
|
+
.sort((a, b) => {
|
|
116
|
+
const order = { deterministic: 0, inline: 1, agent: 2 }
|
|
117
|
+
return order[a.execution] - order[b.execution] ||
|
|
118
|
+
a.classification.estimatedCost - b.classification.estimatedCost ||
|
|
119
|
+
a.task.id.localeCompare(b.task.id)
|
|
120
|
+
})
|
|
121
|
+
|
|
122
|
+
const selected = []
|
|
123
|
+
let waveCost = 0
|
|
124
|
+
for (const item of classified) {
|
|
125
|
+
if (selected.length >= options.maxConcurrent) break
|
|
126
|
+
if (selected.some((existing) => conflict(item.classification, existing.classification))) continue
|
|
127
|
+
if (!kindCapacity(selected, item.classification, options)) continue
|
|
128
|
+
if (selected.length && waveCost + item.classification.estimatedCost > options.maxWaveCost) continue
|
|
129
|
+
selected.push(item)
|
|
130
|
+
waveCost += item.classification.estimatedCost
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
if (!selected.length) selected.push(classified[0])
|
|
134
|
+
|
|
135
|
+
const tasksInWave = selected.map(({ task, classification, execution }) => ({
|
|
136
|
+
id: task.id,
|
|
137
|
+
kind: classification.kind,
|
|
138
|
+
execution,
|
|
139
|
+
estimatedCost: classification.estimatedCost,
|
|
140
|
+
writes: classification.writes,
|
|
141
|
+
files: classification.files,
|
|
142
|
+
spawnAgent: execution === "agent",
|
|
143
|
+
rationale:
|
|
144
|
+
execution === "deterministic"
|
|
145
|
+
? "deterministic tool/script work should not consume an agent slot"
|
|
146
|
+
: execution === "inline"
|
|
147
|
+
? "coordination cost exceeds expected benefit for this bounded unit"
|
|
148
|
+
: classification.kind === "vision"
|
|
149
|
+
? "visual judgment requires a bounded vision worker"
|
|
150
|
+
: "independent task size justifies isolated agent execution",
|
|
151
|
+
}))
|
|
152
|
+
|
|
153
|
+
waves.push({
|
|
154
|
+
index: waveIndex++,
|
|
155
|
+
tasks: tasksInWave,
|
|
156
|
+
totalEstimatedCost: tasksInWave.reduce((sum, item) => sum + item.estimatedCost, 0),
|
|
157
|
+
agentSlots: tasksInWave.filter((item) => item.spawnAgent).length,
|
|
158
|
+
visionAgentSlots: tasksInWave.filter((item) => item.spawnAgent && item.kind === "vision").length,
|
|
159
|
+
})
|
|
160
|
+
|
|
161
|
+
for (const { task } of selected) {
|
|
162
|
+
remaining.delete(task.id)
|
|
163
|
+
complete.add(task.id)
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
const flat = waves.flatMap((wave) => wave.tasks)
|
|
168
|
+
return {
|
|
169
|
+
schemaVersion: 2,
|
|
170
|
+
options,
|
|
171
|
+
waves,
|
|
172
|
+
taskCount: tasks.length,
|
|
173
|
+
agentTaskCount: flat.filter((task) => task.spawnAgent).length,
|
|
174
|
+
inlineTaskCount: flat.filter((task) => task.execution === "inline").length,
|
|
175
|
+
deterministicTaskCount: flat.filter((task) => task.execution === "deterministic").length,
|
|
176
|
+
visionAgentTaskCount: flat.filter((task) => task.spawnAgent && task.kind === "vision").length,
|
|
177
|
+
estimatedCoordinationSaved: flat.filter((task) => task.execution !== "agent").reduce((sum, task) => sum + task.estimatedCost, 0),
|
|
178
|
+
}
|
|
179
|
+
}
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
function finite(value) {
|
|
2
|
+
if (value === null || value === undefined || value === "") return null
|
|
3
|
+
const number = Number(value)
|
|
4
|
+
return Number.isFinite(number) ? number : null
|
|
5
|
+
}
|
|
6
|
+
|
|
7
|
+
function modeSummary(value) {
|
|
8
|
+
return value?.summary?.modes?.ues || value?.modes?.ues || null
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
function ratio(candidate, reference) {
|
|
12
|
+
return candidate != null && reference != null && reference > 0
|
|
13
|
+
? candidate / reference
|
|
14
|
+
: null
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
function delta(candidate, reference) {
|
|
18
|
+
return candidate != null && reference != null ? candidate - reference : null
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
export function compareEvalSummaries(reference, candidate, options = {}) {
|
|
22
|
+
const ref = modeSummary(reference)
|
|
23
|
+
const next = modeSummary(candidate)
|
|
24
|
+
if (!ref || !next) throw new Error("reference and candidate must contain a UES mode summary")
|
|
25
|
+
|
|
26
|
+
const passRateTolerance = Math.max(0, Number(options.passRateTolerance || 0))
|
|
27
|
+
const minInitialInputReduction = Math.max(0, Number(options.minInitialInputReduction ?? 0.10))
|
|
28
|
+
const maxTotalTokenRatio = Math.max(0.01, Number(options.maxTotalTokenRatio || 1.05))
|
|
29
|
+
const maxDurationRatio = Math.max(0.01, Number(options.maxDurationRatio || 1.10))
|
|
30
|
+
const minCacheableRatio = options.minCacheableRatio == null ? null : Math.max(0, Math.min(1, Number(options.minCacheableRatio)))
|
|
31
|
+
const minEvidenceReuseRatio = options.minEvidenceReuseRatio == null ? null : Math.max(0, Math.min(1, Number(options.minEvidenceReuseRatio)))
|
|
32
|
+
const maxRepeatedStableRatio = options.maxRepeatedStableRatio == null ? null : Math.max(0, Number(options.maxRepeatedStableRatio))
|
|
33
|
+
|
|
34
|
+
const referencePassRate = finite(ref.passRate)
|
|
35
|
+
const candidatePassRate = finite(next.passRate)
|
|
36
|
+
const referenceInitialInput = finite(ref.avgInitialInputTokens)
|
|
37
|
+
const candidateInitialInput = finite(next.avgInitialInputTokens)
|
|
38
|
+
const referenceTokens = finite(ref.avgTokens)
|
|
39
|
+
const candidateTokens = finite(next.avgTokens)
|
|
40
|
+
const referenceDuration = finite(ref.avgDurationMs)
|
|
41
|
+
const candidateDuration = finite(next.avgDurationMs)
|
|
42
|
+
const referenceCacheableRatio = finite(ref.avgCacheableRatio)
|
|
43
|
+
const candidateCacheableRatio = finite(next.avgCacheableRatio)
|
|
44
|
+
const referenceEvidenceReuseRatio = finite(ref.avgEvidenceReuseRatio)
|
|
45
|
+
const candidateEvidenceReuseRatio = finite(next.avgEvidenceReuseRatio)
|
|
46
|
+
const candidateRepeatedStableChars = finite(next.avgRepeatedStableChars)
|
|
47
|
+
const referenceRepeatedStableRatio = finite(ref.avgRepeatedStableRatio)
|
|
48
|
+
const candidateRepeatedStableRatio = finite(next.avgRepeatedStableRatio)
|
|
49
|
+
|
|
50
|
+
const initialInputRatio = ratio(candidateInitialInput, referenceInitialInput)
|
|
51
|
+
const tokenRatio = ratio(candidateTokens, referenceTokens)
|
|
52
|
+
const durationRatio = ratio(candidateDuration, referenceDuration)
|
|
53
|
+
|
|
54
|
+
const checks = {
|
|
55
|
+
passRatePreserved:
|
|
56
|
+
referencePassRate != null &&
|
|
57
|
+
candidatePassRate != null &&
|
|
58
|
+
candidatePassRate >= referencePassRate - passRateTolerance,
|
|
59
|
+
initialInputReduced:
|
|
60
|
+
initialInputRatio == null
|
|
61
|
+
? null
|
|
62
|
+
: initialInputRatio <= 1 - minInitialInputReduction,
|
|
63
|
+
totalTokensBounded:
|
|
64
|
+
tokenRatio == null ? null : tokenRatio <= maxTotalTokenRatio,
|
|
65
|
+
durationBounded:
|
|
66
|
+
durationRatio == null ? null : durationRatio <= maxDurationRatio,
|
|
67
|
+
cacheableRatioTarget:
|
|
68
|
+
minCacheableRatio == null ? null :
|
|
69
|
+
candidateCacheableRatio == null ? false :
|
|
70
|
+
candidateCacheableRatio >= minCacheableRatio,
|
|
71
|
+
evidenceReuseTarget:
|
|
72
|
+
minEvidenceReuseRatio == null ? null :
|
|
73
|
+
candidateEvidenceReuseRatio == null ? false :
|
|
74
|
+
candidateEvidenceReuseRatio >= minEvidenceReuseRatio,
|
|
75
|
+
repeatedStableTarget:
|
|
76
|
+
maxRepeatedStableRatio == null ? null :
|
|
77
|
+
candidateRepeatedStableRatio == null ? false :
|
|
78
|
+
candidateRepeatedStableRatio <= maxRepeatedStableRatio,
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
const efficiencyEvidence = [
|
|
82
|
+
checks.initialInputReduced,
|
|
83
|
+
checks.totalTokensBounded,
|
|
84
|
+
checks.durationBounded,
|
|
85
|
+
].filter((value) => value !== null)
|
|
86
|
+
|
|
87
|
+
return {
|
|
88
|
+
schemaVersion: 1,
|
|
89
|
+
kind: "ues-eval-ablation",
|
|
90
|
+
thresholds: {
|
|
91
|
+
passRateTolerance,
|
|
92
|
+
minInitialInputReduction,
|
|
93
|
+
maxTotalTokenRatio,
|
|
94
|
+
maxDurationRatio,
|
|
95
|
+
minCacheableRatio,
|
|
96
|
+
minEvidenceReuseRatio,
|
|
97
|
+
maxRepeatedStableRatio,
|
|
98
|
+
},
|
|
99
|
+
reference: {
|
|
100
|
+
passRate: referencePassRate,
|
|
101
|
+
avgInitialInputTokens: referenceInitialInput,
|
|
102
|
+
avgTokens: referenceTokens,
|
|
103
|
+
avgDurationMs: referenceDuration,
|
|
104
|
+
avgCacheableRatio: referenceCacheableRatio,
|
|
105
|
+
avgEvidenceReuseRatio: referenceEvidenceReuseRatio,
|
|
106
|
+
avgRepeatedStableRatio: referenceRepeatedStableRatio,
|
|
107
|
+
},
|
|
108
|
+
candidate: {
|
|
109
|
+
passRate: candidatePassRate,
|
|
110
|
+
avgInitialInputTokens: candidateInitialInput,
|
|
111
|
+
avgTokens: candidateTokens,
|
|
112
|
+
avgDurationMs: candidateDuration,
|
|
113
|
+
avgCacheableRatio: candidateCacheableRatio,
|
|
114
|
+
avgEvidenceReuseRatio: candidateEvidenceReuseRatio,
|
|
115
|
+
avgRepeatedStableChars: candidateRepeatedStableChars,
|
|
116
|
+
avgRepeatedStableRatio: candidateRepeatedStableRatio,
|
|
117
|
+
},
|
|
118
|
+
delta: {
|
|
119
|
+
passRate: delta(candidatePassRate, referencePassRate),
|
|
120
|
+
avgInitialInputTokens: delta(candidateInitialInput, referenceInitialInput),
|
|
121
|
+
avgTokens: delta(candidateTokens, referenceTokens),
|
|
122
|
+
avgDurationMs: delta(candidateDuration, referenceDuration),
|
|
123
|
+
avgCacheableRatio: delta(candidateCacheableRatio, referenceCacheableRatio),
|
|
124
|
+
avgEvidenceReuseRatio: delta(candidateEvidenceReuseRatio, referenceEvidenceReuseRatio),
|
|
125
|
+
avgRepeatedStableRatio: delta(candidateRepeatedStableRatio, referenceRepeatedStableRatio),
|
|
126
|
+
},
|
|
127
|
+
ratios: {
|
|
128
|
+
initialInput: initialInputRatio,
|
|
129
|
+
totalTokens: tokenRatio,
|
|
130
|
+
duration: durationRatio,
|
|
131
|
+
},
|
|
132
|
+
checks,
|
|
133
|
+
telemetrySufficient: checks.initialInputReduced !== null,
|
|
134
|
+
optionalTargetsSatisfied:
|
|
135
|
+
(checks.cacheableRatioTarget == null || checks.cacheableRatioTarget === true) &&
|
|
136
|
+
(checks.evidenceReuseTarget == null || checks.evidenceReuseTarget === true) &&
|
|
137
|
+
(checks.repeatedStableTarget == null || checks.repeatedStableTarget === true),
|
|
138
|
+
gateEligible:
|
|
139
|
+
checks.passRatePreserved === true &&
|
|
140
|
+
checks.initialInputReduced === true &&
|
|
141
|
+
efficiencyEvidence.every((value) => value === true) &&
|
|
142
|
+
(checks.cacheableRatioTarget == null || checks.cacheableRatioTarget === true) &&
|
|
143
|
+
(checks.evidenceReuseTarget == null || checks.evidenceReuseTarget === true) &&
|
|
144
|
+
(checks.repeatedStableTarget == null || checks.repeatedStableTarget === true),
|
|
145
|
+
}
|
|
146
|
+
}
|
package/lib/eval-report.mjs
CHANGED
|
@@ -10,8 +10,24 @@ export function summarizeEvalResults(results) {
|
|
|
10
10
|
toolSamples: 0,
|
|
11
11
|
tokens: 0,
|
|
12
12
|
tokenSamples: 0,
|
|
13
|
+
initialInputTokens: 0,
|
|
14
|
+
initialInputSamples: 0,
|
|
13
15
|
cost: 0,
|
|
14
16
|
costSamples: 0,
|
|
17
|
+
cacheableRatio: 0,
|
|
18
|
+
cacheableRatioSamples: 0,
|
|
19
|
+
repeatedStableChars: 0,
|
|
20
|
+
repeatedStableSamples: 0,
|
|
21
|
+
repeatedStableRatio: 0,
|
|
22
|
+
repeatedStableRatioSamples: 0,
|
|
23
|
+
evidenceReuseRatio: 0,
|
|
24
|
+
evidenceReuseSamples: 0,
|
|
25
|
+
visualRepairAttempts: 0,
|
|
26
|
+
visualRepairSamples: 0,
|
|
27
|
+
contextExpansions: 0,
|
|
28
|
+
contextExpansionSamples: 0,
|
|
29
|
+
modelEscalations: 0,
|
|
30
|
+
modelEscalationSamples: 0,
|
|
15
31
|
}
|
|
16
32
|
bucket.total += 1
|
|
17
33
|
if (item.passed) bucket.passed += 1
|
|
@@ -25,10 +41,45 @@ export function summarizeEvalResults(results) {
|
|
|
25
41
|
bucket.tokens += Number(item.telemetry?.tokens?.total) || 0
|
|
26
42
|
bucket.tokenSamples += 1
|
|
27
43
|
}
|
|
44
|
+
const initialInput = Number(item.telemetry?.firstUsage?.input)
|
|
45
|
+
if (Number.isFinite(initialInput)) {
|
|
46
|
+
bucket.initialInputTokens += initialInput
|
|
47
|
+
bucket.initialInputSamples += 1
|
|
48
|
+
}
|
|
28
49
|
if ((item.telemetry?.costSamples || 0) > 0) {
|
|
29
50
|
bucket.cost += Number(item.telemetry?.cost) || 0
|
|
30
51
|
bucket.costSamples += 1
|
|
31
52
|
}
|
|
53
|
+
|
|
54
|
+
const v11 = item.telemetry?.v11 || {}
|
|
55
|
+
if (Number.isFinite(Number(v11.avgCacheableRatio))) {
|
|
56
|
+
bucket.cacheableRatio += Number(v11.avgCacheableRatio)
|
|
57
|
+
bucket.cacheableRatioSamples += 1
|
|
58
|
+
}
|
|
59
|
+
if (Number.isFinite(Number(v11.repeatedStableChars))) {
|
|
60
|
+
bucket.repeatedStableChars += Number(v11.repeatedStableChars)
|
|
61
|
+
bucket.repeatedStableSamples += 1
|
|
62
|
+
}
|
|
63
|
+
if (Number.isFinite(Number(v11.repeatedStableRatio))) {
|
|
64
|
+
bucket.repeatedStableRatio += Number(v11.repeatedStableRatio)
|
|
65
|
+
bucket.repeatedStableRatioSamples += 1
|
|
66
|
+
}
|
|
67
|
+
if (Number.isFinite(Number(v11.evidenceReuseRatio))) {
|
|
68
|
+
bucket.evidenceReuseRatio += Number(v11.evidenceReuseRatio)
|
|
69
|
+
bucket.evidenceReuseSamples += 1
|
|
70
|
+
}
|
|
71
|
+
if (Number.isFinite(Number(v11.visualRepairAttempts))) {
|
|
72
|
+
bucket.visualRepairAttempts += Number(v11.visualRepairAttempts)
|
|
73
|
+
bucket.visualRepairSamples += 1
|
|
74
|
+
}
|
|
75
|
+
if (Number.isFinite(Number(v11.contextExpansions))) {
|
|
76
|
+
bucket.contextExpansions += Number(v11.contextExpansions)
|
|
77
|
+
bucket.contextExpansionSamples += 1
|
|
78
|
+
}
|
|
79
|
+
if (Number.isFinite(Number(v11.modelEscalations))) {
|
|
80
|
+
bucket.modelEscalations += Number(v11.modelEscalations)
|
|
81
|
+
bucket.modelEscalationSamples += 1
|
|
82
|
+
}
|
|
32
83
|
}
|
|
33
84
|
|
|
34
85
|
for (const bucket of Object.values(modes)) {
|
|
@@ -36,19 +87,51 @@ export function summarizeEvalResults(results) {
|
|
|
36
87
|
bucket.avgDurationMs = bucket.total ? bucket.durationMs / bucket.total : 0
|
|
37
88
|
bucket.avgToolCalls = bucket.toolSamples ? bucket.toolCalls / bucket.toolSamples : null
|
|
38
89
|
bucket.avgTokens = bucket.tokenSamples ? bucket.tokens / bucket.tokenSamples : null
|
|
90
|
+
bucket.avgInitialInputTokens = bucket.initialInputSamples ? bucket.initialInputTokens / bucket.initialInputSamples : null
|
|
39
91
|
bucket.avgCost = bucket.costSamples ? bucket.cost / bucket.costSamples : null
|
|
92
|
+
bucket.avgCacheableRatio = bucket.cacheableRatioSamples ? bucket.cacheableRatio / bucket.cacheableRatioSamples : null
|
|
93
|
+
bucket.avgRepeatedStableChars = bucket.repeatedStableSamples ? bucket.repeatedStableChars / bucket.repeatedStableSamples : null
|
|
94
|
+
bucket.avgRepeatedStableRatio = bucket.repeatedStableRatioSamples ? bucket.repeatedStableRatio / bucket.repeatedStableRatioSamples : null
|
|
95
|
+
bucket.avgEvidenceReuseRatio = bucket.evidenceReuseSamples ? bucket.evidenceReuseRatio / bucket.evidenceReuseSamples : null
|
|
96
|
+
bucket.avgVisualRepairAttempts = bucket.visualRepairSamples ? bucket.visualRepairAttempts / bucket.visualRepairSamples : null
|
|
97
|
+
bucket.avgContextExpansions = bucket.contextExpansionSamples ? bucket.contextExpansions / bucket.contextExpansionSamples : null
|
|
98
|
+
bucket.avgModelEscalations = bucket.modelEscalationSamples ? bucket.modelEscalations / bucket.modelEscalationSamples : null
|
|
40
99
|
bucket.telemetryCoverage = {
|
|
41
100
|
tools: bucket.total ? bucket.toolSamples / bucket.total : 0,
|
|
42
101
|
tokens: bucket.total ? bucket.tokenSamples / bucket.total : 0,
|
|
102
|
+
initialInputTokens: bucket.total ? bucket.initialInputSamples / bucket.total : 0,
|
|
43
103
|
cost: bucket.total ? bucket.costSamples / bucket.total : 0,
|
|
104
|
+
cacheableRatio: bucket.total ? bucket.cacheableRatioSamples / bucket.total : 0,
|
|
105
|
+
repeatedStableChars: bucket.total ? bucket.repeatedStableSamples / bucket.total : 0,
|
|
106
|
+
repeatedStableRatio: bucket.total ? bucket.repeatedStableRatioSamples / bucket.total : 0,
|
|
107
|
+
evidenceReuseRatio: bucket.total ? bucket.evidenceReuseSamples / bucket.total : 0,
|
|
108
|
+
visualRepairAttempts: bucket.total ? bucket.visualRepairSamples / bucket.total : 0,
|
|
109
|
+
contextExpansions: bucket.total ? bucket.contextExpansionSamples / bucket.total : 0,
|
|
110
|
+
modelEscalations: bucket.total ? bucket.modelEscalationSamples / bucket.total : 0,
|
|
44
111
|
}
|
|
45
112
|
delete bucket.durationMs
|
|
46
113
|
delete bucket.toolCalls
|
|
47
114
|
delete bucket.toolSamples
|
|
48
115
|
delete bucket.tokens
|
|
49
116
|
delete bucket.tokenSamples
|
|
117
|
+
delete bucket.initialInputTokens
|
|
118
|
+
delete bucket.initialInputSamples
|
|
50
119
|
delete bucket.cost
|
|
51
120
|
delete bucket.costSamples
|
|
121
|
+
delete bucket.cacheableRatio
|
|
122
|
+
delete bucket.cacheableRatioSamples
|
|
123
|
+
delete bucket.repeatedStableChars
|
|
124
|
+
delete bucket.repeatedStableSamples
|
|
125
|
+
delete bucket.repeatedStableRatio
|
|
126
|
+
delete bucket.repeatedStableRatioSamples
|
|
127
|
+
delete bucket.evidenceReuseRatio
|
|
128
|
+
delete bucket.evidenceReuseSamples
|
|
129
|
+
delete bucket.visualRepairAttempts
|
|
130
|
+
delete bucket.visualRepairSamples
|
|
131
|
+
delete bucket.contextExpansions
|
|
132
|
+
delete bucket.contextExpansionSamples
|
|
133
|
+
delete bucket.modelEscalations
|
|
134
|
+
delete bucket.modelEscalationSamples
|
|
52
135
|
}
|
|
53
136
|
|
|
54
137
|
const byTaskMap = new Map()
|
package/lib/eval-telemetry.mjs
CHANGED
|
@@ -92,6 +92,19 @@ export function parseOpenCodeTelemetry(stdout) {
|
|
|
92
92
|
let cost = 0
|
|
93
93
|
let usageSamples = 0
|
|
94
94
|
let costSamples = 0
|
|
95
|
+
let firstUsage = null
|
|
96
|
+
const v11 = {
|
|
97
|
+
promptCacheSamples: 0,
|
|
98
|
+
cacheableRatioSum: 0,
|
|
99
|
+
stableChars: 0,
|
|
100
|
+
dynamicChars: 0,
|
|
101
|
+
repeatedStableChars: 0,
|
|
102
|
+
evidenceRefs: new Set(),
|
|
103
|
+
evidenceRefOccurrences: 0,
|
|
104
|
+
visualRepairAttempts: 0,
|
|
105
|
+
contextExpansions: 0,
|
|
106
|
+
modelEscalations: 0,
|
|
107
|
+
}
|
|
95
108
|
|
|
96
109
|
parsed.forEach((event, eventIndex) => {
|
|
97
110
|
let eventUsage = null
|
|
@@ -119,6 +132,39 @@ export function parseOpenCodeTelemetry(stdout) {
|
|
|
119
132
|
}
|
|
120
133
|
|
|
121
134
|
if (!eventUsage) eventUsage = usageFromObject(object)
|
|
135
|
+
|
|
136
|
+
const promptCache = object.promptCache && typeof object.promptCache === "object" ? object.promptCache : null
|
|
137
|
+
if (promptCache) {
|
|
138
|
+
const ratio = Number(promptCache.cacheableRatio)
|
|
139
|
+
if (Number.isFinite(ratio)) {
|
|
140
|
+
v11.promptCacheSamples += 1
|
|
141
|
+
v11.cacheableRatioSum += ratio
|
|
142
|
+
}
|
|
143
|
+
const stableChars = Number(promptCache.stableChars)
|
|
144
|
+
if (Number.isFinite(stableChars) && stableChars >= 0) v11.stableChars += stableChars
|
|
145
|
+
const dynamicChars = Number(promptCache.dynamicChars)
|
|
146
|
+
if (Number.isFinite(dynamicChars) && dynamicChars >= 0) v11.dynamicChars += dynamicChars
|
|
147
|
+
const repeated = Number(promptCache.repeatedStableChars)
|
|
148
|
+
if (Number.isFinite(repeated) && repeated >= 0) v11.repeatedStableChars += repeated
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
const scanRefs = (value) => {
|
|
152
|
+
if (typeof value === "string" && /^evidence:sha256:[a-f0-9]{64}$/i.test(value)) {
|
|
153
|
+
v11.evidenceRefOccurrences += 1
|
|
154
|
+
v11.evidenceRefs.add(value.toLowerCase())
|
|
155
|
+
} else if (Array.isArray(value)) {
|
|
156
|
+
for (const item of value) scanRefs(item)
|
|
157
|
+
} else if (value && typeof value === "object") {
|
|
158
|
+
for (const child of Object.values(value)) scanRefs(child)
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
scanRefs(object.evidencePointers)
|
|
162
|
+
scanRefs(object.evidence)
|
|
163
|
+
|
|
164
|
+
const eventName = String(object.type || object.event || object.kind || "").toLowerCase()
|
|
165
|
+
if (eventName.includes("visual") && eventName.includes("repair")) v11.visualRepairAttempts += 1
|
|
166
|
+
if (eventName.includes("context") && (eventName.includes("expand") || eventName.includes("recovery"))) v11.contextExpansions += 1
|
|
167
|
+
if (eventName.includes("model") && eventName.includes("escalat")) v11.modelEscalations += 1
|
|
122
168
|
if (eventCost === null) {
|
|
123
169
|
for (const key of ["cost", "totalCost", "total_cost"]) {
|
|
124
170
|
if (typeof object[key] === "number" && Number.isFinite(object[key])) {
|
|
@@ -130,6 +176,7 @@ export function parseOpenCodeTelemetry(stdout) {
|
|
|
130
176
|
})
|
|
131
177
|
|
|
132
178
|
if (eventUsage) {
|
|
179
|
+
if (!firstUsage && eventUsage.input > 0) firstUsage = { ...eventUsage }
|
|
133
180
|
usageSamples += 1
|
|
134
181
|
tokens.input += eventUsage.input
|
|
135
182
|
tokens.output += eventUsage.output
|
|
@@ -151,8 +198,25 @@ export function parseOpenCodeTelemetry(stdout) {
|
|
|
151
198
|
skillsLoaded: [...skills].sort(),
|
|
152
199
|
subagents: [...subagents].sort(),
|
|
153
200
|
tokens,
|
|
201
|
+
firstUsage,
|
|
154
202
|
usageSamples,
|
|
155
203
|
cost,
|
|
156
204
|
costSamples,
|
|
205
|
+
v11: {
|
|
206
|
+
promptCacheSamples: v11.promptCacheSamples,
|
|
207
|
+
avgCacheableRatio: v11.promptCacheSamples ? v11.cacheableRatioSum / v11.promptCacheSamples : null,
|
|
208
|
+
avgStableChars: v11.promptCacheSamples ? v11.stableChars / v11.promptCacheSamples : null,
|
|
209
|
+
avgDynamicChars: v11.promptCacheSamples ? v11.dynamicChars / v11.promptCacheSamples : null,
|
|
210
|
+
repeatedStableChars: v11.promptCacheSamples ? v11.repeatedStableChars : null,
|
|
211
|
+
repeatedStableRatio: v11.stableChars > 0 ? v11.repeatedStableChars / v11.stableChars : null,
|
|
212
|
+
evidenceRefOccurrences: v11.evidenceRefOccurrences || null,
|
|
213
|
+
uniqueEvidenceRefs: v11.evidenceRefs.size || null,
|
|
214
|
+
evidenceReuseRatio: v11.evidenceRefOccurrences
|
|
215
|
+
? 1 - (v11.evidenceRefs.size / v11.evidenceRefOccurrences)
|
|
216
|
+
: null,
|
|
217
|
+
visualRepairAttempts: v11.visualRepairAttempts || null,
|
|
218
|
+
contextExpansions: v11.contextExpansions || null,
|
|
219
|
+
modelEscalations: v11.modelEscalations || null,
|
|
220
|
+
},
|
|
157
221
|
}
|
|
158
222
|
}
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
const DEFAULT_WEIGHTS = {
|
|
2
|
+
instructions: 0.10,
|
|
3
|
+
task: 0.08,
|
|
4
|
+
declared: 0.34,
|
|
5
|
+
tests: 0.15,
|
|
6
|
+
references: 0.20,
|
|
7
|
+
history: 0.05,
|
|
8
|
+
tools: 0.08,
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
function clamp(value, min, max) {
|
|
12
|
+
return Math.min(max, Math.max(min, value))
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
function normalizeWeights(weights) {
|
|
16
|
+
const total = Object.values(weights).reduce((sum, value) => sum + Math.max(0, Number(value) || 0), 0) || 1
|
|
17
|
+
return Object.fromEntries(Object.entries(weights).map(([key, value]) => [key, Math.max(0, Number(value) || 0) / total]))
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
export function planEvidenceBudget(taskPolicy = {}, task = {}, signals = {}) {
|
|
21
|
+
const base = clamp(
|
|
22
|
+
Number(taskPolicy.contextBudget ?? taskPolicy.profile?.contextBudget ?? signals.contextBudget ?? 20_000),
|
|
23
|
+
4_000,
|
|
24
|
+
48_000,
|
|
25
|
+
)
|
|
26
|
+
const text = [task?.title, task?.summary, ...(task?.acceptance || [])].filter(Boolean).join(" ").toLowerCase()
|
|
27
|
+
const visual = signals.visual === true || signals.vision === true || /(screenshot|figma|visual|pixel|layout|giao diện|hình ảnh|ảnh mẫu)/i.test(text)
|
|
28
|
+
const browser = signals.browser === true || /(browser|playwright|e2e|click|navigation|trình duyệt)/i.test(text)
|
|
29
|
+
const debugging = signals.debugging === true || /(fix|bug|error|regression|debug|lỗi)/i.test(text)
|
|
30
|
+
const highRisk = taskPolicy.risk === "high"
|
|
31
|
+
|
|
32
|
+
const weights = { ...DEFAULT_WEIGHTS }
|
|
33
|
+
if (debugging) {
|
|
34
|
+
weights.tests += 0.07
|
|
35
|
+
weights.references -= 0.04
|
|
36
|
+
weights.history += 0.02
|
|
37
|
+
weights.declared -= 0.05
|
|
38
|
+
}
|
|
39
|
+
if (highRisk) {
|
|
40
|
+
weights.tests += 0.06
|
|
41
|
+
weights.references += 0.04
|
|
42
|
+
weights.tools -= 0.03
|
|
43
|
+
weights.declared -= 0.05
|
|
44
|
+
weights.task -= 0.02
|
|
45
|
+
}
|
|
46
|
+
if (visual || browser) {
|
|
47
|
+
weights.tools += 0.08
|
|
48
|
+
weights.references -= 0.04
|
|
49
|
+
weights.declared -= 0.04
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
const normalized = normalizeWeights(weights)
|
|
53
|
+
const buckets = Object.fromEntries(
|
|
54
|
+
Object.entries(normalized).map(([key, weight]) => [key, Math.max(256, Math.round(base * weight))]),
|
|
55
|
+
)
|
|
56
|
+
const allocated = Object.values(buckets).reduce((sum, value) => sum + value, 0)
|
|
57
|
+
const drift = base - allocated
|
|
58
|
+
buckets.declared = Math.max(256, buckets.declared + drift)
|
|
59
|
+
|
|
60
|
+
return {
|
|
61
|
+
schemaVersion: 1,
|
|
62
|
+
total: base,
|
|
63
|
+
unit: "characters",
|
|
64
|
+
buckets,
|
|
65
|
+
signals: { visual, browser, debugging, highRisk },
|
|
66
|
+
expansion: {
|
|
67
|
+
initial: base,
|
|
68
|
+
diagnose: Math.min(48_000, Math.max(base, Math.round(base * 1.35))),
|
|
69
|
+
deepRecovery: Math.min(48_000, Math.max(20_000, Math.round(base * 1.75))),
|
|
70
|
+
},
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
export function bucketLimit(plan, role, remaining = Infinity) {
|
|
75
|
+
const limit = Number(plan?.buckets?.[role] || 0)
|
|
76
|
+
return Math.max(0, Math.min(limit, Number.isFinite(remaining) ? remaining : limit))
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
export function evidenceValueScore({ relevance = 0, freshness = 0, confidence = 0, chars = 1 } = {}) {
|
|
80
|
+
const signal = Math.max(0, Number(relevance)) * 0.55 +
|
|
81
|
+
Math.max(0, Number(freshness)) * 0.20 +
|
|
82
|
+
Math.max(0, Number(confidence)) * 0.25
|
|
83
|
+
return signal / Math.max(1, Number(chars))
|
|
84
|
+
}
|