opencode-agent-skill 10.0.0 → 11.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +67 -0
- package/README.md +49 -3
- package/bin/ocskill.mjs +330 -5
- package/docs/V11-PERCEPTION-ADAPTIVE-EXECUTION.md +75 -0
- package/docs/V11-PERCEPTION-ADAPTIVE.md +220 -0
- package/evals/router-triggers.json +82 -0
- package/evals/routing.json +76 -0
- package/evals/v11/tasks.json +122 -0
- package/global-config/agents/merge-arbiter.md +12 -0
- package/global-config/agents/visual-verifier.md +12 -0
- package/global-config/plugins/ues-router/index.js +272 -2
- package/global-config/plugins/ues-router/router.js +27 -3
- package/global-config/skills/browser-qa/SKILL.md +14 -0
- package/global-config/skills/browser-qa/references/workflow.md +11 -0
- package/global-config/skills/browser-security/SKILL.md +12 -0
- package/global-config/skills/component-visual-testing/SKILL.md +10 -0
- package/global-config/skills/design-source/SKILL.md +10 -0
- package/global-config/skills/design-source/references/workflow.md +12 -0
- package/global-config/skills/dynamic-workflow/SKILL.md +18 -0
- package/global-config/skills/dynamic-workflow/references/workflow.md +19 -0
- package/global-config/skills/responsive-verification/SKILL.md +10 -0
- package/global-config/skills/skill-authoring/SKILL.md +12 -0
- package/global-config/skills/skill-evaluation/SKILL.md +17 -0
- package/global-config/skills/visual-fidelity/SKILL.md +14 -0
- package/global-config/skills/visual-fidelity/references/workflow.md +14 -0
- package/lib/browser-adapter.mjs +82 -0
- package/lib/browser-runtime.mjs +193 -0
- package/lib/capability-registry.mjs +109 -0
- package/lib/context-engine-v11.mjs +146 -0
- package/lib/context-manifest.mjs +16 -3
- package/lib/control-center.mjs +12 -2
- package/lib/dynamic-workflow.mjs +179 -0
- package/lib/eval-ablation.mjs +43 -1
- package/lib/eval-report.mjs +72 -0
- package/lib/eval-telemetry.mjs +61 -0
- package/lib/evidence-budget.mjs +84 -0
- package/lib/evidence-store.mjs +178 -0
- package/lib/hermes-bridge.mjs +45 -1
- package/lib/model-config.mjs +9 -1
- package/lib/model-policy.mjs +52 -1
- package/lib/orchestrator-policy.mjs +1 -1
- package/lib/png-diff.mjs +229 -0
- package/lib/prompt-cache.mjs +60 -0
- package/lib/skill-quality.mjs +72 -0
- package/lib/task-engine.mjs +78 -4
- package/lib/ui-inspector.mjs +152 -0
- package/lib/v11-metrics.mjs +64 -0
- package/lib/visual-spec.mjs +159 -0
- package/package.json +10 -5
- package/scripts/eval-ablation.mjs +4 -1
- package/scripts/validate-v11-suite.mjs +58 -0
- package/scripts/validate.mjs +16 -4
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
function taskFiles(task = {}) {
|
|
2
|
+
if (Array.isArray(task.files)) return task.files
|
|
3
|
+
const files = task.files && typeof task.files === "object" ? task.files : {}
|
|
4
|
+
return [...new Set(["create","modify","test","delete"].flatMap((key) => Array.isArray(files[key]) ? files[key] : []))]
|
|
5
|
+
}
|
|
6
|
+
|
|
7
|
+
function writes(task = {}) {
|
|
8
|
+
const files = task.files && typeof task.files === "object" && !Array.isArray(task.files) ? task.files : null
|
|
9
|
+
if (!files) return Array.isArray(task.files) && task.files.length > 0
|
|
10
|
+
return ["create","modify","delete"].some((key) => Array.isArray(files[key]) && files[key].length > 0)
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
function clampInt(value, fallback, min, max) {
|
|
14
|
+
const parsed = Number(value)
|
|
15
|
+
if (!Number.isFinite(parsed)) return fallback
|
|
16
|
+
return Math.max(min, Math.min(max, Math.round(parsed)))
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
export function classifyWorkflowTask(task = {}) {
|
|
20
|
+
const text = [task.title, task.summary, ...(task.acceptance || [])].filter(Boolean).join(" ").toLowerCase()
|
|
21
|
+
const visual = /(visual|screenshot|pixel|figma|layout|giao diện|ảnh mẫu|image reference)/.test(text)
|
|
22
|
+
const deterministic = task.deterministic === true || /(run test|typecheck|lint|format|generate manifest|build index|verify command|compile|unit test)/.test(text)
|
|
23
|
+
const kind = deterministic ? "deterministic" : visual ? "vision" : "llm"
|
|
24
|
+
const fileCount = taskFiles(task).length
|
|
25
|
+
const acceptanceCount = Array.isArray(task.acceptance) ? task.acceptance.length : 0
|
|
26
|
+
const verificationCount = Array.isArray(task.verification) ? task.verification.length : 0
|
|
27
|
+
const estimatedCost = kind === "deterministic"
|
|
28
|
+
? 1
|
|
29
|
+
: Math.max(
|
|
30
|
+
2,
|
|
31
|
+
Math.min(
|
|
32
|
+
12,
|
|
33
|
+
2 + fileCount + Math.min(3, acceptanceCount) + Math.min(2, verificationCount) + (task.risk === "high" ? 3 : 0),
|
|
34
|
+
),
|
|
35
|
+
)
|
|
36
|
+
return {
|
|
37
|
+
kind,
|
|
38
|
+
estimatedCost,
|
|
39
|
+
writes: writes(task),
|
|
40
|
+
files: taskFiles(task),
|
|
41
|
+
risk: task.risk || "medium",
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
function conflict(a, b) {
|
|
46
|
+
if (!a.writes && !b.writes) return false
|
|
47
|
+
const aa = new Set(a.files)
|
|
48
|
+
return b.files.some((file) => aa.has(file))
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
function executionMode(classification, options) {
|
|
52
|
+
if (classification.kind === "deterministic") return "deterministic"
|
|
53
|
+
if (classification.kind === "vision") {
|
|
54
|
+
return classification.estimatedCost >= options.minVisionAgentCost ? "agent" : "inline"
|
|
55
|
+
}
|
|
56
|
+
return classification.estimatedCost >= options.minAgentCost ? "agent" : "inline"
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
function kindCapacity(selected, classification, options) {
|
|
60
|
+
if (classification.kind === "vision") {
|
|
61
|
+
return selected.filter((item) => item.classification.kind === "vision" && item.execution === "agent").length < options.maxVisionConcurrent
|
|
62
|
+
}
|
|
63
|
+
if (classification.kind === "llm") {
|
|
64
|
+
return selected.filter((item) => item.classification.kind === "llm" && item.execution === "agent").length < options.maxLLMConcurrent
|
|
65
|
+
}
|
|
66
|
+
return true
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
function readyTasks(remaining, byID, complete) {
|
|
70
|
+
return [...remaining]
|
|
71
|
+
.map((id) => byID.get(id))
|
|
72
|
+
.filter((task) => (task.dependsOn || task.dependencies || []).every((dep) => complete.has(dep)))
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
export function planDynamicWorkflow(tasks = [], inputOptions = {}) {
|
|
76
|
+
const options = {
|
|
77
|
+
maxConcurrent: clampInt(inputOptions.maxConcurrent, 4, 1, 16),
|
|
78
|
+
maxLLMConcurrent: clampInt(inputOptions.maxLLMConcurrent, inputOptions.maxConcurrent || 4, 1, 16),
|
|
79
|
+
maxVisionConcurrent: clampInt(inputOptions.maxVisionConcurrent, 2, 1, 8),
|
|
80
|
+
maxWaveCost: clampInt(inputOptions.maxWaveCost, 24, 1, 128),
|
|
81
|
+
minAgentCost: clampInt(inputOptions.minAgentCost, 5, 2, 12),
|
|
82
|
+
minVisionAgentCost: clampInt(inputOptions.minVisionAgentCost, 4, 2, 12),
|
|
83
|
+
deterministicFirst: inputOptions.deterministicFirst !== false,
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
const byID = new Map(tasks.map((task) => [task.id, task]))
|
|
87
|
+
if (byID.size !== tasks.length || tasks.some((task) => !task?.id)) {
|
|
88
|
+
throw new Error("Workflow tasks require unique non-empty ids")
|
|
89
|
+
}
|
|
90
|
+
for (const task of tasks) {
|
|
91
|
+
for (const dep of task.dependsOn || task.dependencies || []) {
|
|
92
|
+
if (!byID.has(dep)) throw new Error("Workflow contains a missing dependency: " + dep)
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
const remaining = new Set(tasks.map((task) => task.id))
|
|
97
|
+
const complete = new Set()
|
|
98
|
+
const waves = []
|
|
99
|
+
let waveIndex = 0
|
|
100
|
+
|
|
101
|
+
while (remaining.size) {
|
|
102
|
+
const ready = readyTasks(remaining, byID, complete)
|
|
103
|
+
if (!ready.length) throw new Error("Workflow contains a dependency cycle or missing dependency")
|
|
104
|
+
|
|
105
|
+
const deterministicReady = ready.filter((task) => classifyWorkflowTask(task).kind === "deterministic")
|
|
106
|
+
const candidates = options.deterministicFirst && deterministicReady.length
|
|
107
|
+
? deterministicReady
|
|
108
|
+
: ready
|
|
109
|
+
|
|
110
|
+
const classified = candidates
|
|
111
|
+
.map((task) => {
|
|
112
|
+
const classification = classifyWorkflowTask(task)
|
|
113
|
+
return { task, classification, execution: executionMode(classification, options) }
|
|
114
|
+
})
|
|
115
|
+
.sort((a, b) => {
|
|
116
|
+
const order = { deterministic: 0, inline: 1, agent: 2 }
|
|
117
|
+
return order[a.execution] - order[b.execution] ||
|
|
118
|
+
a.classification.estimatedCost - b.classification.estimatedCost ||
|
|
119
|
+
a.task.id.localeCompare(b.task.id)
|
|
120
|
+
})
|
|
121
|
+
|
|
122
|
+
const selected = []
|
|
123
|
+
let waveCost = 0
|
|
124
|
+
for (const item of classified) {
|
|
125
|
+
if (selected.length >= options.maxConcurrent) break
|
|
126
|
+
if (selected.some((existing) => conflict(item.classification, existing.classification))) continue
|
|
127
|
+
if (!kindCapacity(selected, item.classification, options)) continue
|
|
128
|
+
if (selected.length && waveCost + item.classification.estimatedCost > options.maxWaveCost) continue
|
|
129
|
+
selected.push(item)
|
|
130
|
+
waveCost += item.classification.estimatedCost
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
if (!selected.length) selected.push(classified[0])
|
|
134
|
+
|
|
135
|
+
const tasksInWave = selected.map(({ task, classification, execution }) => ({
|
|
136
|
+
id: task.id,
|
|
137
|
+
kind: classification.kind,
|
|
138
|
+
execution,
|
|
139
|
+
estimatedCost: classification.estimatedCost,
|
|
140
|
+
writes: classification.writes,
|
|
141
|
+
files: classification.files,
|
|
142
|
+
spawnAgent: execution === "agent",
|
|
143
|
+
rationale:
|
|
144
|
+
execution === "deterministic"
|
|
145
|
+
? "deterministic tool/script work should not consume an agent slot"
|
|
146
|
+
: execution === "inline"
|
|
147
|
+
? "coordination cost exceeds expected benefit for this bounded unit"
|
|
148
|
+
: classification.kind === "vision"
|
|
149
|
+
? "visual judgment requires a bounded vision worker"
|
|
150
|
+
: "independent task size justifies isolated agent execution",
|
|
151
|
+
}))
|
|
152
|
+
|
|
153
|
+
waves.push({
|
|
154
|
+
index: waveIndex++,
|
|
155
|
+
tasks: tasksInWave,
|
|
156
|
+
totalEstimatedCost: tasksInWave.reduce((sum, item) => sum + item.estimatedCost, 0),
|
|
157
|
+
agentSlots: tasksInWave.filter((item) => item.spawnAgent).length,
|
|
158
|
+
visionAgentSlots: tasksInWave.filter((item) => item.spawnAgent && item.kind === "vision").length,
|
|
159
|
+
})
|
|
160
|
+
|
|
161
|
+
for (const { task } of selected) {
|
|
162
|
+
remaining.delete(task.id)
|
|
163
|
+
complete.add(task.id)
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
const flat = waves.flatMap((wave) => wave.tasks)
|
|
168
|
+
return {
|
|
169
|
+
schemaVersion: 2,
|
|
170
|
+
options,
|
|
171
|
+
waves,
|
|
172
|
+
taskCount: tasks.length,
|
|
173
|
+
agentTaskCount: flat.filter((task) => task.spawnAgent).length,
|
|
174
|
+
inlineTaskCount: flat.filter((task) => task.execution === "inline").length,
|
|
175
|
+
deterministicTaskCount: flat.filter((task) => task.execution === "deterministic").length,
|
|
176
|
+
visionAgentTaskCount: flat.filter((task) => task.spawnAgent && task.kind === "vision").length,
|
|
177
|
+
estimatedCoordinationSaved: flat.filter((task) => task.execution !== "agent").reduce((sum, task) => sum + task.estimatedCost, 0),
|
|
178
|
+
}
|
|
179
|
+
}
|
package/lib/eval-ablation.mjs
CHANGED
|
@@ -27,6 +27,9 @@ export function compareEvalSummaries(reference, candidate, options = {}) {
|
|
|
27
27
|
const minInitialInputReduction = Math.max(0, Number(options.minInitialInputReduction ?? 0.10))
|
|
28
28
|
const maxTotalTokenRatio = Math.max(0.01, Number(options.maxTotalTokenRatio || 1.05))
|
|
29
29
|
const maxDurationRatio = Math.max(0.01, Number(options.maxDurationRatio || 1.10))
|
|
30
|
+
const minCacheableRatio = options.minCacheableRatio == null ? null : Math.max(0, Math.min(1, Number(options.minCacheableRatio)))
|
|
31
|
+
const minEvidenceReuseRatio = options.minEvidenceReuseRatio == null ? null : Math.max(0, Math.min(1, Number(options.minEvidenceReuseRatio)))
|
|
32
|
+
const maxRepeatedStableRatio = options.maxRepeatedStableRatio == null ? null : Math.max(0, Number(options.maxRepeatedStableRatio))
|
|
30
33
|
|
|
31
34
|
const referencePassRate = finite(ref.passRate)
|
|
32
35
|
const candidatePassRate = finite(next.passRate)
|
|
@@ -36,6 +39,13 @@ export function compareEvalSummaries(reference, candidate, options = {}) {
|
|
|
36
39
|
const candidateTokens = finite(next.avgTokens)
|
|
37
40
|
const referenceDuration = finite(ref.avgDurationMs)
|
|
38
41
|
const candidateDuration = finite(next.avgDurationMs)
|
|
42
|
+
const referenceCacheableRatio = finite(ref.avgCacheableRatio)
|
|
43
|
+
const candidateCacheableRatio = finite(next.avgCacheableRatio)
|
|
44
|
+
const referenceEvidenceReuseRatio = finite(ref.avgEvidenceReuseRatio)
|
|
45
|
+
const candidateEvidenceReuseRatio = finite(next.avgEvidenceReuseRatio)
|
|
46
|
+
const candidateRepeatedStableChars = finite(next.avgRepeatedStableChars)
|
|
47
|
+
const referenceRepeatedStableRatio = finite(ref.avgRepeatedStableRatio)
|
|
48
|
+
const candidateRepeatedStableRatio = finite(next.avgRepeatedStableRatio)
|
|
39
49
|
|
|
40
50
|
const initialInputRatio = ratio(candidateInitialInput, referenceInitialInput)
|
|
41
51
|
const tokenRatio = ratio(candidateTokens, referenceTokens)
|
|
@@ -54,6 +64,18 @@ export function compareEvalSummaries(reference, candidate, options = {}) {
|
|
|
54
64
|
tokenRatio == null ? null : tokenRatio <= maxTotalTokenRatio,
|
|
55
65
|
durationBounded:
|
|
56
66
|
durationRatio == null ? null : durationRatio <= maxDurationRatio,
|
|
67
|
+
cacheableRatioTarget:
|
|
68
|
+
minCacheableRatio == null ? null :
|
|
69
|
+
candidateCacheableRatio == null ? false :
|
|
70
|
+
candidateCacheableRatio >= minCacheableRatio,
|
|
71
|
+
evidenceReuseTarget:
|
|
72
|
+
minEvidenceReuseRatio == null ? null :
|
|
73
|
+
candidateEvidenceReuseRatio == null ? false :
|
|
74
|
+
candidateEvidenceReuseRatio >= minEvidenceReuseRatio,
|
|
75
|
+
repeatedStableTarget:
|
|
76
|
+
maxRepeatedStableRatio == null ? null :
|
|
77
|
+
candidateRepeatedStableRatio == null ? false :
|
|
78
|
+
candidateRepeatedStableRatio <= maxRepeatedStableRatio,
|
|
57
79
|
}
|
|
58
80
|
|
|
59
81
|
const efficiencyEvidence = [
|
|
@@ -70,24 +92,37 @@ export function compareEvalSummaries(reference, candidate, options = {}) {
|
|
|
70
92
|
minInitialInputReduction,
|
|
71
93
|
maxTotalTokenRatio,
|
|
72
94
|
maxDurationRatio,
|
|
95
|
+
minCacheableRatio,
|
|
96
|
+
minEvidenceReuseRatio,
|
|
97
|
+
maxRepeatedStableRatio,
|
|
73
98
|
},
|
|
74
99
|
reference: {
|
|
75
100
|
passRate: referencePassRate,
|
|
76
101
|
avgInitialInputTokens: referenceInitialInput,
|
|
77
102
|
avgTokens: referenceTokens,
|
|
78
103
|
avgDurationMs: referenceDuration,
|
|
104
|
+
avgCacheableRatio: referenceCacheableRatio,
|
|
105
|
+
avgEvidenceReuseRatio: referenceEvidenceReuseRatio,
|
|
106
|
+
avgRepeatedStableRatio: referenceRepeatedStableRatio,
|
|
79
107
|
},
|
|
80
108
|
candidate: {
|
|
81
109
|
passRate: candidatePassRate,
|
|
82
110
|
avgInitialInputTokens: candidateInitialInput,
|
|
83
111
|
avgTokens: candidateTokens,
|
|
84
112
|
avgDurationMs: candidateDuration,
|
|
113
|
+
avgCacheableRatio: candidateCacheableRatio,
|
|
114
|
+
avgEvidenceReuseRatio: candidateEvidenceReuseRatio,
|
|
115
|
+
avgRepeatedStableChars: candidateRepeatedStableChars,
|
|
116
|
+
avgRepeatedStableRatio: candidateRepeatedStableRatio,
|
|
85
117
|
},
|
|
86
118
|
delta: {
|
|
87
119
|
passRate: delta(candidatePassRate, referencePassRate),
|
|
88
120
|
avgInitialInputTokens: delta(candidateInitialInput, referenceInitialInput),
|
|
89
121
|
avgTokens: delta(candidateTokens, referenceTokens),
|
|
90
122
|
avgDurationMs: delta(candidateDuration, referenceDuration),
|
|
123
|
+
avgCacheableRatio: delta(candidateCacheableRatio, referenceCacheableRatio),
|
|
124
|
+
avgEvidenceReuseRatio: delta(candidateEvidenceReuseRatio, referenceEvidenceReuseRatio),
|
|
125
|
+
avgRepeatedStableRatio: delta(candidateRepeatedStableRatio, referenceRepeatedStableRatio),
|
|
91
126
|
},
|
|
92
127
|
ratios: {
|
|
93
128
|
initialInput: initialInputRatio,
|
|
@@ -96,9 +131,16 @@ export function compareEvalSummaries(reference, candidate, options = {}) {
|
|
|
96
131
|
},
|
|
97
132
|
checks,
|
|
98
133
|
telemetrySufficient: checks.initialInputReduced !== null,
|
|
134
|
+
optionalTargetsSatisfied:
|
|
135
|
+
(checks.cacheableRatioTarget == null || checks.cacheableRatioTarget === true) &&
|
|
136
|
+
(checks.evidenceReuseTarget == null || checks.evidenceReuseTarget === true) &&
|
|
137
|
+
(checks.repeatedStableTarget == null || checks.repeatedStableTarget === true),
|
|
99
138
|
gateEligible:
|
|
100
139
|
checks.passRatePreserved === true &&
|
|
101
140
|
checks.initialInputReduced === true &&
|
|
102
|
-
efficiencyEvidence.every((value) => value === true)
|
|
141
|
+
efficiencyEvidence.every((value) => value === true) &&
|
|
142
|
+
(checks.cacheableRatioTarget == null || checks.cacheableRatioTarget === true) &&
|
|
143
|
+
(checks.evidenceReuseTarget == null || checks.evidenceReuseTarget === true) &&
|
|
144
|
+
(checks.repeatedStableTarget == null || checks.repeatedStableTarget === true),
|
|
103
145
|
}
|
|
104
146
|
}
|
package/lib/eval-report.mjs
CHANGED
|
@@ -14,6 +14,20 @@ export function summarizeEvalResults(results) {
|
|
|
14
14
|
initialInputSamples: 0,
|
|
15
15
|
cost: 0,
|
|
16
16
|
costSamples: 0,
|
|
17
|
+
cacheableRatio: 0,
|
|
18
|
+
cacheableRatioSamples: 0,
|
|
19
|
+
repeatedStableChars: 0,
|
|
20
|
+
repeatedStableSamples: 0,
|
|
21
|
+
repeatedStableRatio: 0,
|
|
22
|
+
repeatedStableRatioSamples: 0,
|
|
23
|
+
evidenceReuseRatio: 0,
|
|
24
|
+
evidenceReuseSamples: 0,
|
|
25
|
+
visualRepairAttempts: 0,
|
|
26
|
+
visualRepairSamples: 0,
|
|
27
|
+
contextExpansions: 0,
|
|
28
|
+
contextExpansionSamples: 0,
|
|
29
|
+
modelEscalations: 0,
|
|
30
|
+
modelEscalationSamples: 0,
|
|
17
31
|
}
|
|
18
32
|
bucket.total += 1
|
|
19
33
|
if (item.passed) bucket.passed += 1
|
|
@@ -36,6 +50,36 @@ export function summarizeEvalResults(results) {
|
|
|
36
50
|
bucket.cost += Number(item.telemetry?.cost) || 0
|
|
37
51
|
bucket.costSamples += 1
|
|
38
52
|
}
|
|
53
|
+
|
|
54
|
+
const v11 = item.telemetry?.v11 || {}
|
|
55
|
+
if (Number.isFinite(Number(v11.avgCacheableRatio))) {
|
|
56
|
+
bucket.cacheableRatio += Number(v11.avgCacheableRatio)
|
|
57
|
+
bucket.cacheableRatioSamples += 1
|
|
58
|
+
}
|
|
59
|
+
if (Number.isFinite(Number(v11.repeatedStableChars))) {
|
|
60
|
+
bucket.repeatedStableChars += Number(v11.repeatedStableChars)
|
|
61
|
+
bucket.repeatedStableSamples += 1
|
|
62
|
+
}
|
|
63
|
+
if (Number.isFinite(Number(v11.repeatedStableRatio))) {
|
|
64
|
+
bucket.repeatedStableRatio += Number(v11.repeatedStableRatio)
|
|
65
|
+
bucket.repeatedStableRatioSamples += 1
|
|
66
|
+
}
|
|
67
|
+
if (Number.isFinite(Number(v11.evidenceReuseRatio))) {
|
|
68
|
+
bucket.evidenceReuseRatio += Number(v11.evidenceReuseRatio)
|
|
69
|
+
bucket.evidenceReuseSamples += 1
|
|
70
|
+
}
|
|
71
|
+
if (Number.isFinite(Number(v11.visualRepairAttempts))) {
|
|
72
|
+
bucket.visualRepairAttempts += Number(v11.visualRepairAttempts)
|
|
73
|
+
bucket.visualRepairSamples += 1
|
|
74
|
+
}
|
|
75
|
+
if (Number.isFinite(Number(v11.contextExpansions))) {
|
|
76
|
+
bucket.contextExpansions += Number(v11.contextExpansions)
|
|
77
|
+
bucket.contextExpansionSamples += 1
|
|
78
|
+
}
|
|
79
|
+
if (Number.isFinite(Number(v11.modelEscalations))) {
|
|
80
|
+
bucket.modelEscalations += Number(v11.modelEscalations)
|
|
81
|
+
bucket.modelEscalationSamples += 1
|
|
82
|
+
}
|
|
39
83
|
}
|
|
40
84
|
|
|
41
85
|
for (const bucket of Object.values(modes)) {
|
|
@@ -45,11 +89,25 @@ export function summarizeEvalResults(results) {
|
|
|
45
89
|
bucket.avgTokens = bucket.tokenSamples ? bucket.tokens / bucket.tokenSamples : null
|
|
46
90
|
bucket.avgInitialInputTokens = bucket.initialInputSamples ? bucket.initialInputTokens / bucket.initialInputSamples : null
|
|
47
91
|
bucket.avgCost = bucket.costSamples ? bucket.cost / bucket.costSamples : null
|
|
92
|
+
bucket.avgCacheableRatio = bucket.cacheableRatioSamples ? bucket.cacheableRatio / bucket.cacheableRatioSamples : null
|
|
93
|
+
bucket.avgRepeatedStableChars = bucket.repeatedStableSamples ? bucket.repeatedStableChars / bucket.repeatedStableSamples : null
|
|
94
|
+
bucket.avgRepeatedStableRatio = bucket.repeatedStableRatioSamples ? bucket.repeatedStableRatio / bucket.repeatedStableRatioSamples : null
|
|
95
|
+
bucket.avgEvidenceReuseRatio = bucket.evidenceReuseSamples ? bucket.evidenceReuseRatio / bucket.evidenceReuseSamples : null
|
|
96
|
+
bucket.avgVisualRepairAttempts = bucket.visualRepairSamples ? bucket.visualRepairAttempts / bucket.visualRepairSamples : null
|
|
97
|
+
bucket.avgContextExpansions = bucket.contextExpansionSamples ? bucket.contextExpansions / bucket.contextExpansionSamples : null
|
|
98
|
+
bucket.avgModelEscalations = bucket.modelEscalationSamples ? bucket.modelEscalations / bucket.modelEscalationSamples : null
|
|
48
99
|
bucket.telemetryCoverage = {
|
|
49
100
|
tools: bucket.total ? bucket.toolSamples / bucket.total : 0,
|
|
50
101
|
tokens: bucket.total ? bucket.tokenSamples / bucket.total : 0,
|
|
51
102
|
initialInputTokens: bucket.total ? bucket.initialInputSamples / bucket.total : 0,
|
|
52
103
|
cost: bucket.total ? bucket.costSamples / bucket.total : 0,
|
|
104
|
+
cacheableRatio: bucket.total ? bucket.cacheableRatioSamples / bucket.total : 0,
|
|
105
|
+
repeatedStableChars: bucket.total ? bucket.repeatedStableSamples / bucket.total : 0,
|
|
106
|
+
repeatedStableRatio: bucket.total ? bucket.repeatedStableRatioSamples / bucket.total : 0,
|
|
107
|
+
evidenceReuseRatio: bucket.total ? bucket.evidenceReuseSamples / bucket.total : 0,
|
|
108
|
+
visualRepairAttempts: bucket.total ? bucket.visualRepairSamples / bucket.total : 0,
|
|
109
|
+
contextExpansions: bucket.total ? bucket.contextExpansionSamples / bucket.total : 0,
|
|
110
|
+
modelEscalations: bucket.total ? bucket.modelEscalationSamples / bucket.total : 0,
|
|
53
111
|
}
|
|
54
112
|
delete bucket.durationMs
|
|
55
113
|
delete bucket.toolCalls
|
|
@@ -60,6 +118,20 @@ export function summarizeEvalResults(results) {
|
|
|
60
118
|
delete bucket.initialInputSamples
|
|
61
119
|
delete bucket.cost
|
|
62
120
|
delete bucket.costSamples
|
|
121
|
+
delete bucket.cacheableRatio
|
|
122
|
+
delete bucket.cacheableRatioSamples
|
|
123
|
+
delete bucket.repeatedStableChars
|
|
124
|
+
delete bucket.repeatedStableSamples
|
|
125
|
+
delete bucket.repeatedStableRatio
|
|
126
|
+
delete bucket.repeatedStableRatioSamples
|
|
127
|
+
delete bucket.evidenceReuseRatio
|
|
128
|
+
delete bucket.evidenceReuseSamples
|
|
129
|
+
delete bucket.visualRepairAttempts
|
|
130
|
+
delete bucket.visualRepairSamples
|
|
131
|
+
delete bucket.contextExpansions
|
|
132
|
+
delete bucket.contextExpansionSamples
|
|
133
|
+
delete bucket.modelEscalations
|
|
134
|
+
delete bucket.modelEscalationSamples
|
|
63
135
|
}
|
|
64
136
|
|
|
65
137
|
const byTaskMap = new Map()
|
package/lib/eval-telemetry.mjs
CHANGED
|
@@ -93,6 +93,18 @@ export function parseOpenCodeTelemetry(stdout) {
|
|
|
93
93
|
let usageSamples = 0
|
|
94
94
|
let costSamples = 0
|
|
95
95
|
let firstUsage = null
|
|
96
|
+
const v11 = {
|
|
97
|
+
promptCacheSamples: 0,
|
|
98
|
+
cacheableRatioSum: 0,
|
|
99
|
+
stableChars: 0,
|
|
100
|
+
dynamicChars: 0,
|
|
101
|
+
repeatedStableChars: 0,
|
|
102
|
+
evidenceRefs: new Set(),
|
|
103
|
+
evidenceRefOccurrences: 0,
|
|
104
|
+
visualRepairAttempts: 0,
|
|
105
|
+
contextExpansions: 0,
|
|
106
|
+
modelEscalations: 0,
|
|
107
|
+
}
|
|
96
108
|
|
|
97
109
|
parsed.forEach((event, eventIndex) => {
|
|
98
110
|
let eventUsage = null
|
|
@@ -120,6 +132,39 @@ export function parseOpenCodeTelemetry(stdout) {
|
|
|
120
132
|
}
|
|
121
133
|
|
|
122
134
|
if (!eventUsage) eventUsage = usageFromObject(object)
|
|
135
|
+
|
|
136
|
+
const promptCache = object.promptCache && typeof object.promptCache === "object" ? object.promptCache : null
|
|
137
|
+
if (promptCache) {
|
|
138
|
+
const ratio = Number(promptCache.cacheableRatio)
|
|
139
|
+
if (Number.isFinite(ratio)) {
|
|
140
|
+
v11.promptCacheSamples += 1
|
|
141
|
+
v11.cacheableRatioSum += ratio
|
|
142
|
+
}
|
|
143
|
+
const stableChars = Number(promptCache.stableChars)
|
|
144
|
+
if (Number.isFinite(stableChars) && stableChars >= 0) v11.stableChars += stableChars
|
|
145
|
+
const dynamicChars = Number(promptCache.dynamicChars)
|
|
146
|
+
if (Number.isFinite(dynamicChars) && dynamicChars >= 0) v11.dynamicChars += dynamicChars
|
|
147
|
+
const repeated = Number(promptCache.repeatedStableChars)
|
|
148
|
+
if (Number.isFinite(repeated) && repeated >= 0) v11.repeatedStableChars += repeated
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
const scanRefs = (value) => {
|
|
152
|
+
if (typeof value === "string" && /^evidence:sha256:[a-f0-9]{64}$/i.test(value)) {
|
|
153
|
+
v11.evidenceRefOccurrences += 1
|
|
154
|
+
v11.evidenceRefs.add(value.toLowerCase())
|
|
155
|
+
} else if (Array.isArray(value)) {
|
|
156
|
+
for (const item of value) scanRefs(item)
|
|
157
|
+
} else if (value && typeof value === "object") {
|
|
158
|
+
for (const child of Object.values(value)) scanRefs(child)
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
scanRefs(object.evidencePointers)
|
|
162
|
+
scanRefs(object.evidence)
|
|
163
|
+
|
|
164
|
+
const eventName = String(object.type || object.event || object.kind || "").toLowerCase()
|
|
165
|
+
if (eventName.includes("visual") && eventName.includes("repair")) v11.visualRepairAttempts += 1
|
|
166
|
+
if (eventName.includes("context") && (eventName.includes("expand") || eventName.includes("recovery"))) v11.contextExpansions += 1
|
|
167
|
+
if (eventName.includes("model") && eventName.includes("escalat")) v11.modelEscalations += 1
|
|
123
168
|
if (eventCost === null) {
|
|
124
169
|
for (const key of ["cost", "totalCost", "total_cost"]) {
|
|
125
170
|
if (typeof object[key] === "number" && Number.isFinite(object[key])) {
|
|
@@ -157,5 +202,21 @@ export function parseOpenCodeTelemetry(stdout) {
|
|
|
157
202
|
usageSamples,
|
|
158
203
|
cost,
|
|
159
204
|
costSamples,
|
|
205
|
+
v11: {
|
|
206
|
+
promptCacheSamples: v11.promptCacheSamples,
|
|
207
|
+
avgCacheableRatio: v11.promptCacheSamples ? v11.cacheableRatioSum / v11.promptCacheSamples : null,
|
|
208
|
+
avgStableChars: v11.promptCacheSamples ? v11.stableChars / v11.promptCacheSamples : null,
|
|
209
|
+
avgDynamicChars: v11.promptCacheSamples ? v11.dynamicChars / v11.promptCacheSamples : null,
|
|
210
|
+
repeatedStableChars: v11.promptCacheSamples ? v11.repeatedStableChars : null,
|
|
211
|
+
repeatedStableRatio: v11.stableChars > 0 ? v11.repeatedStableChars / v11.stableChars : null,
|
|
212
|
+
evidenceRefOccurrences: v11.evidenceRefOccurrences || null,
|
|
213
|
+
uniqueEvidenceRefs: v11.evidenceRefs.size || null,
|
|
214
|
+
evidenceReuseRatio: v11.evidenceRefOccurrences
|
|
215
|
+
? 1 - (v11.evidenceRefs.size / v11.evidenceRefOccurrences)
|
|
216
|
+
: null,
|
|
217
|
+
visualRepairAttempts: v11.visualRepairAttempts || null,
|
|
218
|
+
contextExpansions: v11.contextExpansions || null,
|
|
219
|
+
modelEscalations: v11.modelEscalations || null,
|
|
220
|
+
},
|
|
160
221
|
}
|
|
161
222
|
}
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
const DEFAULT_WEIGHTS = {
|
|
2
|
+
instructions: 0.10,
|
|
3
|
+
task: 0.08,
|
|
4
|
+
declared: 0.34,
|
|
5
|
+
tests: 0.15,
|
|
6
|
+
references: 0.20,
|
|
7
|
+
history: 0.05,
|
|
8
|
+
tools: 0.08,
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
function clamp(value, min, max) {
|
|
12
|
+
return Math.min(max, Math.max(min, value))
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
function normalizeWeights(weights) {
|
|
16
|
+
const total = Object.values(weights).reduce((sum, value) => sum + Math.max(0, Number(value) || 0), 0) || 1
|
|
17
|
+
return Object.fromEntries(Object.entries(weights).map(([key, value]) => [key, Math.max(0, Number(value) || 0) / total]))
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
export function planEvidenceBudget(taskPolicy = {}, task = {}, signals = {}) {
|
|
21
|
+
const base = clamp(
|
|
22
|
+
Number(taskPolicy.contextBudget ?? taskPolicy.profile?.contextBudget ?? signals.contextBudget ?? 20_000),
|
|
23
|
+
4_000,
|
|
24
|
+
48_000,
|
|
25
|
+
)
|
|
26
|
+
const text = [task?.title, task?.summary, ...(task?.acceptance || [])].filter(Boolean).join(" ").toLowerCase()
|
|
27
|
+
const visual = signals.visual === true || signals.vision === true || /(screenshot|figma|visual|pixel|layout|giao diện|hình ảnh|ảnh mẫu)/i.test(text)
|
|
28
|
+
const browser = signals.browser === true || /(browser|playwright|e2e|click|navigation|trình duyệt)/i.test(text)
|
|
29
|
+
const debugging = signals.debugging === true || /(fix|bug|error|regression|debug|lỗi)/i.test(text)
|
|
30
|
+
const highRisk = taskPolicy.risk === "high"
|
|
31
|
+
|
|
32
|
+
const weights = { ...DEFAULT_WEIGHTS }
|
|
33
|
+
if (debugging) {
|
|
34
|
+
weights.tests += 0.07
|
|
35
|
+
weights.references -= 0.04
|
|
36
|
+
weights.history += 0.02
|
|
37
|
+
weights.declared -= 0.05
|
|
38
|
+
}
|
|
39
|
+
if (highRisk) {
|
|
40
|
+
weights.tests += 0.06
|
|
41
|
+
weights.references += 0.04
|
|
42
|
+
weights.tools -= 0.03
|
|
43
|
+
weights.declared -= 0.05
|
|
44
|
+
weights.task -= 0.02
|
|
45
|
+
}
|
|
46
|
+
if (visual || browser) {
|
|
47
|
+
weights.tools += 0.08
|
|
48
|
+
weights.references -= 0.04
|
|
49
|
+
weights.declared -= 0.04
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
const normalized = normalizeWeights(weights)
|
|
53
|
+
const buckets = Object.fromEntries(
|
|
54
|
+
Object.entries(normalized).map(([key, weight]) => [key, Math.max(256, Math.round(base * weight))]),
|
|
55
|
+
)
|
|
56
|
+
const allocated = Object.values(buckets).reduce((sum, value) => sum + value, 0)
|
|
57
|
+
const drift = base - allocated
|
|
58
|
+
buckets.declared = Math.max(256, buckets.declared + drift)
|
|
59
|
+
|
|
60
|
+
return {
|
|
61
|
+
schemaVersion: 1,
|
|
62
|
+
total: base,
|
|
63
|
+
unit: "characters",
|
|
64
|
+
buckets,
|
|
65
|
+
signals: { visual, browser, debugging, highRisk },
|
|
66
|
+
expansion: {
|
|
67
|
+
initial: base,
|
|
68
|
+
diagnose: Math.min(48_000, Math.max(base, Math.round(base * 1.35))),
|
|
69
|
+
deepRecovery: Math.min(48_000, Math.max(20_000, Math.round(base * 1.75))),
|
|
70
|
+
},
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
export function bucketLimit(plan, role, remaining = Infinity) {
|
|
75
|
+
const limit = Number(plan?.buckets?.[role] || 0)
|
|
76
|
+
return Math.max(0, Math.min(limit, Number.isFinite(remaining) ? remaining : limit))
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
export function evidenceValueScore({ relevance = 0, freshness = 0, confidence = 0, chars = 1 } = {}) {
|
|
80
|
+
const signal = Math.max(0, Number(relevance)) * 0.55 +
|
|
81
|
+
Math.max(0, Number(freshness)) * 0.20 +
|
|
82
|
+
Math.max(0, Number(confidence)) * 0.25
|
|
83
|
+
return signal / Math.max(1, Number(chars))
|
|
84
|
+
}
|