opencode-agent-skill 9.0.0 → 11.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/CHANGELOG.md +116 -0
  2. package/README.md +742 -675
  3. package/bin/ocskill.mjs +354 -5
  4. package/docs/V11-PERCEPTION-ADAPTIVE-EXECUTION.md +75 -0
  5. package/docs/V11-PERCEPTION-ADAPTIVE.md +220 -0
  6. package/evals/router-triggers.json +82 -0
  7. package/evals/routing.json +76 -0
  8. package/evals/v11/tasks.json +122 -0
  9. package/global-config/AGENTS.md +78 -163
  10. package/global-config/agents/merge-arbiter.md +12 -0
  11. package/global-config/agents/visual-verifier.md +12 -0
  12. package/global-config/plugins/ues-router/index.js +683 -59
  13. package/global-config/plugins/ues-router/router.js +62 -3
  14. package/global-config/plugins/ues-router/runtime-guard.js +265 -0
  15. package/global-config/skills/browser-qa/SKILL.md +14 -0
  16. package/global-config/skills/browser-qa/references/workflow.md +11 -0
  17. package/global-config/skills/browser-security/SKILL.md +12 -0
  18. package/global-config/skills/component-visual-testing/SKILL.md +10 -0
  19. package/global-config/skills/design-source/SKILL.md +10 -0
  20. package/global-config/skills/design-source/references/workflow.md +12 -0
  21. package/global-config/skills/dynamic-workflow/SKILL.md +18 -0
  22. package/global-config/skills/dynamic-workflow/references/workflow.md +19 -0
  23. package/global-config/skills/responsive-verification/SKILL.md +10 -0
  24. package/global-config/skills/skill-authoring/SKILL.md +12 -0
  25. package/global-config/skills/skill-evaluation/SKILL.md +17 -0
  26. package/global-config/skills/visual-fidelity/SKILL.md +14 -0
  27. package/global-config/skills/visual-fidelity/references/workflow.md +14 -0
  28. package/lib/benchmark-confidence.mjs +49 -11
  29. package/lib/browser-adapter.mjs +82 -0
  30. package/lib/browser-runtime.mjs +193 -0
  31. package/lib/capability-registry.mjs +109 -0
  32. package/lib/context-engine-v11.mjs +146 -0
  33. package/lib/context-manifest.mjs +16 -3
  34. package/lib/control-center.mjs +12 -2
  35. package/lib/dynamic-workflow.mjs +179 -0
  36. package/lib/eval-ablation.mjs +146 -0
  37. package/lib/eval-report.mjs +83 -0
  38. package/lib/eval-telemetry.mjs +64 -0
  39. package/lib/evidence-budget.mjs +84 -0
  40. package/lib/evidence-store.mjs +178 -0
  41. package/lib/hermes-bridge.mjs +45 -1
  42. package/lib/model-config.mjs +9 -1
  43. package/lib/model-policy.mjs +58 -1
  44. package/lib/orchestrator-policy.mjs +100 -7
  45. package/lib/png-diff.mjs +229 -0
  46. package/lib/prompt-cache.mjs +60 -0
  47. package/lib/skill-quality.mjs +72 -0
  48. package/lib/task-engine.mjs +223 -12
  49. package/lib/ui-inspector.mjs +152 -0
  50. package/lib/v11-metrics.mjs +64 -0
  51. package/lib/visual-spec.mjs +159 -0
  52. package/package.json +11 -5
  53. package/scripts/eval-ablation.mjs +47 -0
  54. package/scripts/eval-matrix.mjs +13 -2
  55. package/scripts/validate-v11-suite.mjs +58 -0
  56. package/scripts/validate.mjs +16 -4
@@ -3,6 +3,7 @@ import { mkdir, readFile, readdir, writeFile } from "node:fs/promises"
3
3
  import path from "node:path"
4
4
  import { readLearningState } from "./learning-engine.mjs"
5
5
  import { readRuntimeEvents } from "./runtime-events.mjs"
6
+ import { evidenceStoreStatus } from "./evidence-store.mjs"
6
7
 
7
8
  function escapeHtml(value) {
8
9
  return String(value ?? "")
@@ -64,18 +65,23 @@ async function collectEvalSummary(root) {
64
65
 
65
66
  export async function collectControlCenterData(root = process.cwd()) {
66
67
  root = path.resolve(root)
67
- const [work, learning, evals] = await Promise.all([
68
+ const [work, learning, evals, evidenceStore] = await Promise.all([
68
69
  collectWork(root),
69
70
  readLearningState(root),
70
71
  collectEvalSummary(root),
72
+ evidenceStoreStatus(root),
71
73
  ])
72
74
  return {
73
- schemaVersion: 1,
75
+ schemaVersion: 2,
74
76
  generatedAt: new Date().toISOString(),
75
77
  root,
76
78
  work,
77
79
  learning,
78
80
  evals,
81
+ v11: {
82
+ evidenceStore,
83
+ runtime: "perception-adaptive-execution",
84
+ },
79
85
  }
80
86
  }
81
87
 
@@ -103,6 +109,7 @@ details{margin-top:10px;border-top:1px solid #243049;padding-top:8px}summary{cur
103
109
  <div class="top"><div><h1>UES Control Center</h1><div class="muted" id="root"></div></div><div class="muted" id="generated"></div></div>
104
110
  <div class="section"><h2>Long-horizon work</h2><div class="grid" id="work"></div></div>
105
111
  <div class="section"><h2>Learning loop</h2><div class="grid" id="learning"></div></div>
112
+ <div class="section"><h2>V11 runtime efficiency</h2><div class="grid" id="v11"></div></div>
106
113
  <div class="section"><h2>Recent runtime events</h2><div class="card"><table><thead><tr><th>Work</th><th>Event</th><th>Task</th><th>Time</th></tr></thead><tbody id="events"></tbody></table></div></div>
107
114
  <div class="section"><h2>Recent evaluations</h2><div class="card"><table><thead><tr><th>Suite</th><th>Model</th><th>Baseline</th><th>UES</th></tr></thead><tbody id="evals"></tbody></table></div></div>
108
115
  </div>
@@ -126,6 +133,9 @@ function render(next){
126
133
  if(!data.work.length) work.innerHTML='<div class="card muted">No .ues-work items found.</div>';
127
134
  const learning=document.querySelector("#learning");
128
135
  learning.innerHTML='<div class="card"><h3>Accepted lessons</h3><strong>'+data.learning.accepted.length+'</strong></div><div class="card"><h3>Proposals</h3><strong>'+data.learning.proposals.length+'</strong></div>';
136
+ const v11=document.querySelector("#v11");
137
+ const store=data.v11?.evidenceStore||{};
138
+ v11.innerHTML='<div class="card"><h3>Evidence store</h3><strong>'+esc(store.entries||0)+'</strong><div class="muted">'+esc(store.bytes||0)+' bytes externalized</div></div><div class="card"><h3>Runtime</h3><strong>V11</strong><div class="muted">'+esc(data.v11?.runtime||'adaptive')+'</div></div>';
129
139
  const eventBody=document.querySelector("#events");
130
140
  eventBody.innerHTML="";
131
141
  for(const item of data.work){
@@ -0,0 +1,179 @@
1
+ function taskFiles(task = {}) {
2
+ if (Array.isArray(task.files)) return task.files
3
+ const files = task.files && typeof task.files === "object" ? task.files : {}
4
+ return [...new Set(["create","modify","test","delete"].flatMap((key) => Array.isArray(files[key]) ? files[key] : []))]
5
+ }
6
+
7
+ function writes(task = {}) {
8
+ const files = task.files && typeof task.files === "object" && !Array.isArray(task.files) ? task.files : null
9
+ if (!files) return Array.isArray(task.files) && task.files.length > 0
10
+ return ["create","modify","delete"].some((key) => Array.isArray(files[key]) && files[key].length > 0)
11
+ }
12
+
13
+ function clampInt(value, fallback, min, max) {
14
+ const parsed = Number(value)
15
+ if (!Number.isFinite(parsed)) return fallback
16
+ return Math.max(min, Math.min(max, Math.round(parsed)))
17
+ }
18
+
19
+ export function classifyWorkflowTask(task = {}) {
20
+ const text = [task.title, task.summary, ...(task.acceptance || [])].filter(Boolean).join(" ").toLowerCase()
21
+ const visual = /(visual|screenshot|pixel|figma|layout|giao diện|ảnh mẫu|image reference)/.test(text)
22
+ const deterministic = task.deterministic === true || /(run test|typecheck|lint|format|generate manifest|build index|verify command|compile|unit test)/.test(text)
23
+ const kind = deterministic ? "deterministic" : visual ? "vision" : "llm"
24
+ const fileCount = taskFiles(task).length
25
+ const acceptanceCount = Array.isArray(task.acceptance) ? task.acceptance.length : 0
26
+ const verificationCount = Array.isArray(task.verification) ? task.verification.length : 0
27
+ const estimatedCost = kind === "deterministic"
28
+ ? 1
29
+ : Math.max(
30
+ 2,
31
+ Math.min(
32
+ 12,
33
+ 2 + fileCount + Math.min(3, acceptanceCount) + Math.min(2, verificationCount) + (task.risk === "high" ? 3 : 0),
34
+ ),
35
+ )
36
+ return {
37
+ kind,
38
+ estimatedCost,
39
+ writes: writes(task),
40
+ files: taskFiles(task),
41
+ risk: task.risk || "medium",
42
+ }
43
+ }
44
+
45
+ function conflict(a, b) {
46
+ if (!a.writes && !b.writes) return false
47
+ const aa = new Set(a.files)
48
+ return b.files.some((file) => aa.has(file))
49
+ }
50
+
51
+ function executionMode(classification, options) {
52
+ if (classification.kind === "deterministic") return "deterministic"
53
+ if (classification.kind === "vision") {
54
+ return classification.estimatedCost >= options.minVisionAgentCost ? "agent" : "inline"
55
+ }
56
+ return classification.estimatedCost >= options.minAgentCost ? "agent" : "inline"
57
+ }
58
+
59
+ function kindCapacity(selected, classification, options) {
60
+ if (classification.kind === "vision") {
61
+ return selected.filter((item) => item.classification.kind === "vision" && item.execution === "agent").length < options.maxVisionConcurrent
62
+ }
63
+ if (classification.kind === "llm") {
64
+ return selected.filter((item) => item.classification.kind === "llm" && item.execution === "agent").length < options.maxLLMConcurrent
65
+ }
66
+ return true
67
+ }
68
+
69
+ function readyTasks(remaining, byID, complete) {
70
+ return [...remaining]
71
+ .map((id) => byID.get(id))
72
+ .filter((task) => (task.dependsOn || task.dependencies || []).every((dep) => complete.has(dep)))
73
+ }
74
+
75
+ export function planDynamicWorkflow(tasks = [], inputOptions = {}) {
76
+ const options = {
77
+ maxConcurrent: clampInt(inputOptions.maxConcurrent, 4, 1, 16),
78
+ maxLLMConcurrent: clampInt(inputOptions.maxLLMConcurrent, inputOptions.maxConcurrent || 4, 1, 16),
79
+ maxVisionConcurrent: clampInt(inputOptions.maxVisionConcurrent, 2, 1, 8),
80
+ maxWaveCost: clampInt(inputOptions.maxWaveCost, 24, 1, 128),
81
+ minAgentCost: clampInt(inputOptions.minAgentCost, 5, 2, 12),
82
+ minVisionAgentCost: clampInt(inputOptions.minVisionAgentCost, 4, 2, 12),
83
+ deterministicFirst: inputOptions.deterministicFirst !== false,
84
+ }
85
+
86
+ const byID = new Map(tasks.map((task) => [task.id, task]))
87
+ if (byID.size !== tasks.length || tasks.some((task) => !task?.id)) {
88
+ throw new Error("Workflow tasks require unique non-empty ids")
89
+ }
90
+ for (const task of tasks) {
91
+ for (const dep of task.dependsOn || task.dependencies || []) {
92
+ if (!byID.has(dep)) throw new Error("Workflow contains a missing dependency: " + dep)
93
+ }
94
+ }
95
+
96
+ const remaining = new Set(tasks.map((task) => task.id))
97
+ const complete = new Set()
98
+ const waves = []
99
+ let waveIndex = 0
100
+
101
+ while (remaining.size) {
102
+ const ready = readyTasks(remaining, byID, complete)
103
+ if (!ready.length) throw new Error("Workflow contains a dependency cycle or missing dependency")
104
+
105
+ const deterministicReady = ready.filter((task) => classifyWorkflowTask(task).kind === "deterministic")
106
+ const candidates = options.deterministicFirst && deterministicReady.length
107
+ ? deterministicReady
108
+ : ready
109
+
110
+ const classified = candidates
111
+ .map((task) => {
112
+ const classification = classifyWorkflowTask(task)
113
+ return { task, classification, execution: executionMode(classification, options) }
114
+ })
115
+ .sort((a, b) => {
116
+ const order = { deterministic: 0, inline: 1, agent: 2 }
117
+ return order[a.execution] - order[b.execution] ||
118
+ a.classification.estimatedCost - b.classification.estimatedCost ||
119
+ a.task.id.localeCompare(b.task.id)
120
+ })
121
+
122
+ const selected = []
123
+ let waveCost = 0
124
+ for (const item of classified) {
125
+ if (selected.length >= options.maxConcurrent) break
126
+ if (selected.some((existing) => conflict(item.classification, existing.classification))) continue
127
+ if (!kindCapacity(selected, item.classification, options)) continue
128
+ if (selected.length && waveCost + item.classification.estimatedCost > options.maxWaveCost) continue
129
+ selected.push(item)
130
+ waveCost += item.classification.estimatedCost
131
+ }
132
+
133
+ if (!selected.length) selected.push(classified[0])
134
+
135
+ const tasksInWave = selected.map(({ task, classification, execution }) => ({
136
+ id: task.id,
137
+ kind: classification.kind,
138
+ execution,
139
+ estimatedCost: classification.estimatedCost,
140
+ writes: classification.writes,
141
+ files: classification.files,
142
+ spawnAgent: execution === "agent",
143
+ rationale:
144
+ execution === "deterministic"
145
+ ? "deterministic tool/script work should not consume an agent slot"
146
+ : execution === "inline"
147
+ ? "coordination cost exceeds expected benefit for this bounded unit"
148
+ : classification.kind === "vision"
149
+ ? "visual judgment requires a bounded vision worker"
150
+ : "independent task size justifies isolated agent execution",
151
+ }))
152
+
153
+ waves.push({
154
+ index: waveIndex++,
155
+ tasks: tasksInWave,
156
+ totalEstimatedCost: tasksInWave.reduce((sum, item) => sum + item.estimatedCost, 0),
157
+ agentSlots: tasksInWave.filter((item) => item.spawnAgent).length,
158
+ visionAgentSlots: tasksInWave.filter((item) => item.spawnAgent && item.kind === "vision").length,
159
+ })
160
+
161
+ for (const { task } of selected) {
162
+ remaining.delete(task.id)
163
+ complete.add(task.id)
164
+ }
165
+ }
166
+
167
+ const flat = waves.flatMap((wave) => wave.tasks)
168
+ return {
169
+ schemaVersion: 2,
170
+ options,
171
+ waves,
172
+ taskCount: tasks.length,
173
+ agentTaskCount: flat.filter((task) => task.spawnAgent).length,
174
+ inlineTaskCount: flat.filter((task) => task.execution === "inline").length,
175
+ deterministicTaskCount: flat.filter((task) => task.execution === "deterministic").length,
176
+ visionAgentTaskCount: flat.filter((task) => task.spawnAgent && task.kind === "vision").length,
177
+ estimatedCoordinationSaved: flat.filter((task) => task.execution !== "agent").reduce((sum, task) => sum + task.estimatedCost, 0),
178
+ }
179
+ }
@@ -0,0 +1,146 @@
1
+ function finite(value) {
2
+ if (value === null || value === undefined || value === "") return null
3
+ const number = Number(value)
4
+ return Number.isFinite(number) ? number : null
5
+ }
6
+
7
+ function modeSummary(value) {
8
+ return value?.summary?.modes?.ues || value?.modes?.ues || null
9
+ }
10
+
11
+ function ratio(candidate, reference) {
12
+ return candidate != null && reference != null && reference > 0
13
+ ? candidate / reference
14
+ : null
15
+ }
16
+
17
+ function delta(candidate, reference) {
18
+ return candidate != null && reference != null ? candidate - reference : null
19
+ }
20
+
21
+ export function compareEvalSummaries(reference, candidate, options = {}) {
22
+ const ref = modeSummary(reference)
23
+ const next = modeSummary(candidate)
24
+ if (!ref || !next) throw new Error("reference and candidate must contain a UES mode summary")
25
+
26
+ const passRateTolerance = Math.max(0, Number(options.passRateTolerance || 0))
27
+ const minInitialInputReduction = Math.max(0, Number(options.minInitialInputReduction ?? 0.10))
28
+ const maxTotalTokenRatio = Math.max(0.01, Number(options.maxTotalTokenRatio || 1.05))
29
+ const maxDurationRatio = Math.max(0.01, Number(options.maxDurationRatio || 1.10))
30
+ const minCacheableRatio = options.minCacheableRatio == null ? null : Math.max(0, Math.min(1, Number(options.minCacheableRatio)))
31
+ const minEvidenceReuseRatio = options.minEvidenceReuseRatio == null ? null : Math.max(0, Math.min(1, Number(options.minEvidenceReuseRatio)))
32
+ const maxRepeatedStableRatio = options.maxRepeatedStableRatio == null ? null : Math.max(0, Number(options.maxRepeatedStableRatio))
33
+
34
+ const referencePassRate = finite(ref.passRate)
35
+ const candidatePassRate = finite(next.passRate)
36
+ const referenceInitialInput = finite(ref.avgInitialInputTokens)
37
+ const candidateInitialInput = finite(next.avgInitialInputTokens)
38
+ const referenceTokens = finite(ref.avgTokens)
39
+ const candidateTokens = finite(next.avgTokens)
40
+ const referenceDuration = finite(ref.avgDurationMs)
41
+ const candidateDuration = finite(next.avgDurationMs)
42
+ const referenceCacheableRatio = finite(ref.avgCacheableRatio)
43
+ const candidateCacheableRatio = finite(next.avgCacheableRatio)
44
+ const referenceEvidenceReuseRatio = finite(ref.avgEvidenceReuseRatio)
45
+ const candidateEvidenceReuseRatio = finite(next.avgEvidenceReuseRatio)
46
+ const candidateRepeatedStableChars = finite(next.avgRepeatedStableChars)
47
+ const referenceRepeatedStableRatio = finite(ref.avgRepeatedStableRatio)
48
+ const candidateRepeatedStableRatio = finite(next.avgRepeatedStableRatio)
49
+
50
+ const initialInputRatio = ratio(candidateInitialInput, referenceInitialInput)
51
+ const tokenRatio = ratio(candidateTokens, referenceTokens)
52
+ const durationRatio = ratio(candidateDuration, referenceDuration)
53
+
54
+ const checks = {
55
+ passRatePreserved:
56
+ referencePassRate != null &&
57
+ candidatePassRate != null &&
58
+ candidatePassRate >= referencePassRate - passRateTolerance,
59
+ initialInputReduced:
60
+ initialInputRatio == null
61
+ ? null
62
+ : initialInputRatio <= 1 - minInitialInputReduction,
63
+ totalTokensBounded:
64
+ tokenRatio == null ? null : tokenRatio <= maxTotalTokenRatio,
65
+ durationBounded:
66
+ durationRatio == null ? null : durationRatio <= maxDurationRatio,
67
+ cacheableRatioTarget:
68
+ minCacheableRatio == null ? null :
69
+ candidateCacheableRatio == null ? false :
70
+ candidateCacheableRatio >= minCacheableRatio,
71
+ evidenceReuseTarget:
72
+ minEvidenceReuseRatio == null ? null :
73
+ candidateEvidenceReuseRatio == null ? false :
74
+ candidateEvidenceReuseRatio >= minEvidenceReuseRatio,
75
+ repeatedStableTarget:
76
+ maxRepeatedStableRatio == null ? null :
77
+ candidateRepeatedStableRatio == null ? false :
78
+ candidateRepeatedStableRatio <= maxRepeatedStableRatio,
79
+ }
80
+
81
+ const efficiencyEvidence = [
82
+ checks.initialInputReduced,
83
+ checks.totalTokensBounded,
84
+ checks.durationBounded,
85
+ ].filter((value) => value !== null)
86
+
87
+ return {
88
+ schemaVersion: 1,
89
+ kind: "ues-eval-ablation",
90
+ thresholds: {
91
+ passRateTolerance,
92
+ minInitialInputReduction,
93
+ maxTotalTokenRatio,
94
+ maxDurationRatio,
95
+ minCacheableRatio,
96
+ minEvidenceReuseRatio,
97
+ maxRepeatedStableRatio,
98
+ },
99
+ reference: {
100
+ passRate: referencePassRate,
101
+ avgInitialInputTokens: referenceInitialInput,
102
+ avgTokens: referenceTokens,
103
+ avgDurationMs: referenceDuration,
104
+ avgCacheableRatio: referenceCacheableRatio,
105
+ avgEvidenceReuseRatio: referenceEvidenceReuseRatio,
106
+ avgRepeatedStableRatio: referenceRepeatedStableRatio,
107
+ },
108
+ candidate: {
109
+ passRate: candidatePassRate,
110
+ avgInitialInputTokens: candidateInitialInput,
111
+ avgTokens: candidateTokens,
112
+ avgDurationMs: candidateDuration,
113
+ avgCacheableRatio: candidateCacheableRatio,
114
+ avgEvidenceReuseRatio: candidateEvidenceReuseRatio,
115
+ avgRepeatedStableChars: candidateRepeatedStableChars,
116
+ avgRepeatedStableRatio: candidateRepeatedStableRatio,
117
+ },
118
+ delta: {
119
+ passRate: delta(candidatePassRate, referencePassRate),
120
+ avgInitialInputTokens: delta(candidateInitialInput, referenceInitialInput),
121
+ avgTokens: delta(candidateTokens, referenceTokens),
122
+ avgDurationMs: delta(candidateDuration, referenceDuration),
123
+ avgCacheableRatio: delta(candidateCacheableRatio, referenceCacheableRatio),
124
+ avgEvidenceReuseRatio: delta(candidateEvidenceReuseRatio, referenceEvidenceReuseRatio),
125
+ avgRepeatedStableRatio: delta(candidateRepeatedStableRatio, referenceRepeatedStableRatio),
126
+ },
127
+ ratios: {
128
+ initialInput: initialInputRatio,
129
+ totalTokens: tokenRatio,
130
+ duration: durationRatio,
131
+ },
132
+ checks,
133
+ telemetrySufficient: checks.initialInputReduced !== null,
134
+ optionalTargetsSatisfied:
135
+ (checks.cacheableRatioTarget == null || checks.cacheableRatioTarget === true) &&
136
+ (checks.evidenceReuseTarget == null || checks.evidenceReuseTarget === true) &&
137
+ (checks.repeatedStableTarget == null || checks.repeatedStableTarget === true),
138
+ gateEligible:
139
+ checks.passRatePreserved === true &&
140
+ checks.initialInputReduced === true &&
141
+ efficiencyEvidence.every((value) => value === true) &&
142
+ (checks.cacheableRatioTarget == null || checks.cacheableRatioTarget === true) &&
143
+ (checks.evidenceReuseTarget == null || checks.evidenceReuseTarget === true) &&
144
+ (checks.repeatedStableTarget == null || checks.repeatedStableTarget === true),
145
+ }
146
+ }
@@ -10,8 +10,24 @@ export function summarizeEvalResults(results) {
10
10
  toolSamples: 0,
11
11
  tokens: 0,
12
12
  tokenSamples: 0,
13
+ initialInputTokens: 0,
14
+ initialInputSamples: 0,
13
15
  cost: 0,
14
16
  costSamples: 0,
17
+ cacheableRatio: 0,
18
+ cacheableRatioSamples: 0,
19
+ repeatedStableChars: 0,
20
+ repeatedStableSamples: 0,
21
+ repeatedStableRatio: 0,
22
+ repeatedStableRatioSamples: 0,
23
+ evidenceReuseRatio: 0,
24
+ evidenceReuseSamples: 0,
25
+ visualRepairAttempts: 0,
26
+ visualRepairSamples: 0,
27
+ contextExpansions: 0,
28
+ contextExpansionSamples: 0,
29
+ modelEscalations: 0,
30
+ modelEscalationSamples: 0,
15
31
  }
16
32
  bucket.total += 1
17
33
  if (item.passed) bucket.passed += 1
@@ -25,10 +41,45 @@ export function summarizeEvalResults(results) {
25
41
  bucket.tokens += Number(item.telemetry?.tokens?.total) || 0
26
42
  bucket.tokenSamples += 1
27
43
  }
44
+ const initialInput = Number(item.telemetry?.firstUsage?.input)
45
+ if (Number.isFinite(initialInput)) {
46
+ bucket.initialInputTokens += initialInput
47
+ bucket.initialInputSamples += 1
48
+ }
28
49
  if ((item.telemetry?.costSamples || 0) > 0) {
29
50
  bucket.cost += Number(item.telemetry?.cost) || 0
30
51
  bucket.costSamples += 1
31
52
  }
53
+
54
+ const v11 = item.telemetry?.v11 || {}
55
+ if (Number.isFinite(Number(v11.avgCacheableRatio))) {
56
+ bucket.cacheableRatio += Number(v11.avgCacheableRatio)
57
+ bucket.cacheableRatioSamples += 1
58
+ }
59
+ if (Number.isFinite(Number(v11.repeatedStableChars))) {
60
+ bucket.repeatedStableChars += Number(v11.repeatedStableChars)
61
+ bucket.repeatedStableSamples += 1
62
+ }
63
+ if (Number.isFinite(Number(v11.repeatedStableRatio))) {
64
+ bucket.repeatedStableRatio += Number(v11.repeatedStableRatio)
65
+ bucket.repeatedStableRatioSamples += 1
66
+ }
67
+ if (Number.isFinite(Number(v11.evidenceReuseRatio))) {
68
+ bucket.evidenceReuseRatio += Number(v11.evidenceReuseRatio)
69
+ bucket.evidenceReuseSamples += 1
70
+ }
71
+ if (Number.isFinite(Number(v11.visualRepairAttempts))) {
72
+ bucket.visualRepairAttempts += Number(v11.visualRepairAttempts)
73
+ bucket.visualRepairSamples += 1
74
+ }
75
+ if (Number.isFinite(Number(v11.contextExpansions))) {
76
+ bucket.contextExpansions += Number(v11.contextExpansions)
77
+ bucket.contextExpansionSamples += 1
78
+ }
79
+ if (Number.isFinite(Number(v11.modelEscalations))) {
80
+ bucket.modelEscalations += Number(v11.modelEscalations)
81
+ bucket.modelEscalationSamples += 1
82
+ }
32
83
  }
33
84
 
34
85
  for (const bucket of Object.values(modes)) {
@@ -36,19 +87,51 @@ export function summarizeEvalResults(results) {
36
87
  bucket.avgDurationMs = bucket.total ? bucket.durationMs / bucket.total : 0
37
88
  bucket.avgToolCalls = bucket.toolSamples ? bucket.toolCalls / bucket.toolSamples : null
38
89
  bucket.avgTokens = bucket.tokenSamples ? bucket.tokens / bucket.tokenSamples : null
90
+ bucket.avgInitialInputTokens = bucket.initialInputSamples ? bucket.initialInputTokens / bucket.initialInputSamples : null
39
91
  bucket.avgCost = bucket.costSamples ? bucket.cost / bucket.costSamples : null
92
+ bucket.avgCacheableRatio = bucket.cacheableRatioSamples ? bucket.cacheableRatio / bucket.cacheableRatioSamples : null
93
+ bucket.avgRepeatedStableChars = bucket.repeatedStableSamples ? bucket.repeatedStableChars / bucket.repeatedStableSamples : null
94
+ bucket.avgRepeatedStableRatio = bucket.repeatedStableRatioSamples ? bucket.repeatedStableRatio / bucket.repeatedStableRatioSamples : null
95
+ bucket.avgEvidenceReuseRatio = bucket.evidenceReuseSamples ? bucket.evidenceReuseRatio / bucket.evidenceReuseSamples : null
96
+ bucket.avgVisualRepairAttempts = bucket.visualRepairSamples ? bucket.visualRepairAttempts / bucket.visualRepairSamples : null
97
+ bucket.avgContextExpansions = bucket.contextExpansionSamples ? bucket.contextExpansions / bucket.contextExpansionSamples : null
98
+ bucket.avgModelEscalations = bucket.modelEscalationSamples ? bucket.modelEscalations / bucket.modelEscalationSamples : null
40
99
  bucket.telemetryCoverage = {
41
100
  tools: bucket.total ? bucket.toolSamples / bucket.total : 0,
42
101
  tokens: bucket.total ? bucket.tokenSamples / bucket.total : 0,
102
+ initialInputTokens: bucket.total ? bucket.initialInputSamples / bucket.total : 0,
43
103
  cost: bucket.total ? bucket.costSamples / bucket.total : 0,
104
+ cacheableRatio: bucket.total ? bucket.cacheableRatioSamples / bucket.total : 0,
105
+ repeatedStableChars: bucket.total ? bucket.repeatedStableSamples / bucket.total : 0,
106
+ repeatedStableRatio: bucket.total ? bucket.repeatedStableRatioSamples / bucket.total : 0,
107
+ evidenceReuseRatio: bucket.total ? bucket.evidenceReuseSamples / bucket.total : 0,
108
+ visualRepairAttempts: bucket.total ? bucket.visualRepairSamples / bucket.total : 0,
109
+ contextExpansions: bucket.total ? bucket.contextExpansionSamples / bucket.total : 0,
110
+ modelEscalations: bucket.total ? bucket.modelEscalationSamples / bucket.total : 0,
44
111
  }
45
112
  delete bucket.durationMs
46
113
  delete bucket.toolCalls
47
114
  delete bucket.toolSamples
48
115
  delete bucket.tokens
49
116
  delete bucket.tokenSamples
117
+ delete bucket.initialInputTokens
118
+ delete bucket.initialInputSamples
50
119
  delete bucket.cost
51
120
  delete bucket.costSamples
121
+ delete bucket.cacheableRatio
122
+ delete bucket.cacheableRatioSamples
123
+ delete bucket.repeatedStableChars
124
+ delete bucket.repeatedStableSamples
125
+ delete bucket.repeatedStableRatio
126
+ delete bucket.repeatedStableRatioSamples
127
+ delete bucket.evidenceReuseRatio
128
+ delete bucket.evidenceReuseSamples
129
+ delete bucket.visualRepairAttempts
130
+ delete bucket.visualRepairSamples
131
+ delete bucket.contextExpansions
132
+ delete bucket.contextExpansionSamples
133
+ delete bucket.modelEscalations
134
+ delete bucket.modelEscalationSamples
52
135
  }
53
136
 
54
137
  const byTaskMap = new Map()
@@ -92,6 +92,19 @@ export function parseOpenCodeTelemetry(stdout) {
92
92
  let cost = 0
93
93
  let usageSamples = 0
94
94
  let costSamples = 0
95
+ let firstUsage = null
96
+ const v11 = {
97
+ promptCacheSamples: 0,
98
+ cacheableRatioSum: 0,
99
+ stableChars: 0,
100
+ dynamicChars: 0,
101
+ repeatedStableChars: 0,
102
+ evidenceRefs: new Set(),
103
+ evidenceRefOccurrences: 0,
104
+ visualRepairAttempts: 0,
105
+ contextExpansions: 0,
106
+ modelEscalations: 0,
107
+ }
95
108
 
96
109
  parsed.forEach((event, eventIndex) => {
97
110
  let eventUsage = null
@@ -119,6 +132,39 @@ export function parseOpenCodeTelemetry(stdout) {
119
132
  }
120
133
 
121
134
  if (!eventUsage) eventUsage = usageFromObject(object)
135
+
136
+ const promptCache = object.promptCache && typeof object.promptCache === "object" ? object.promptCache : null
137
+ if (promptCache) {
138
+ const ratio = Number(promptCache.cacheableRatio)
139
+ if (Number.isFinite(ratio)) {
140
+ v11.promptCacheSamples += 1
141
+ v11.cacheableRatioSum += ratio
142
+ }
143
+ const stableChars = Number(promptCache.stableChars)
144
+ if (Number.isFinite(stableChars) && stableChars >= 0) v11.stableChars += stableChars
145
+ const dynamicChars = Number(promptCache.dynamicChars)
146
+ if (Number.isFinite(dynamicChars) && dynamicChars >= 0) v11.dynamicChars += dynamicChars
147
+ const repeated = Number(promptCache.repeatedStableChars)
148
+ if (Number.isFinite(repeated) && repeated >= 0) v11.repeatedStableChars += repeated
149
+ }
150
+
151
+ const scanRefs = (value) => {
152
+ if (typeof value === "string" && /^evidence:sha256:[a-f0-9]{64}$/i.test(value)) {
153
+ v11.evidenceRefOccurrences += 1
154
+ v11.evidenceRefs.add(value.toLowerCase())
155
+ } else if (Array.isArray(value)) {
156
+ for (const item of value) scanRefs(item)
157
+ } else if (value && typeof value === "object") {
158
+ for (const child of Object.values(value)) scanRefs(child)
159
+ }
160
+ }
161
+ scanRefs(object.evidencePointers)
162
+ scanRefs(object.evidence)
163
+
164
+ const eventName = String(object.type || object.event || object.kind || "").toLowerCase()
165
+ if (eventName.includes("visual") && eventName.includes("repair")) v11.visualRepairAttempts += 1
166
+ if (eventName.includes("context") && (eventName.includes("expand") || eventName.includes("recovery"))) v11.contextExpansions += 1
167
+ if (eventName.includes("model") && eventName.includes("escalat")) v11.modelEscalations += 1
122
168
  if (eventCost === null) {
123
169
  for (const key of ["cost", "totalCost", "total_cost"]) {
124
170
  if (typeof object[key] === "number" && Number.isFinite(object[key])) {
@@ -130,6 +176,7 @@ export function parseOpenCodeTelemetry(stdout) {
130
176
  })
131
177
 
132
178
  if (eventUsage) {
179
+ if (!firstUsage && eventUsage.input > 0) firstUsage = { ...eventUsage }
133
180
  usageSamples += 1
134
181
  tokens.input += eventUsage.input
135
182
  tokens.output += eventUsage.output
@@ -151,8 +198,25 @@ export function parseOpenCodeTelemetry(stdout) {
151
198
  skillsLoaded: [...skills].sort(),
152
199
  subagents: [...subagents].sort(),
153
200
  tokens,
201
+ firstUsage,
154
202
  usageSamples,
155
203
  cost,
156
204
  costSamples,
205
+ v11: {
206
+ promptCacheSamples: v11.promptCacheSamples,
207
+ avgCacheableRatio: v11.promptCacheSamples ? v11.cacheableRatioSum / v11.promptCacheSamples : null,
208
+ avgStableChars: v11.promptCacheSamples ? v11.stableChars / v11.promptCacheSamples : null,
209
+ avgDynamicChars: v11.promptCacheSamples ? v11.dynamicChars / v11.promptCacheSamples : null,
210
+ repeatedStableChars: v11.promptCacheSamples ? v11.repeatedStableChars : null,
211
+ repeatedStableRatio: v11.stableChars > 0 ? v11.repeatedStableChars / v11.stableChars : null,
212
+ evidenceRefOccurrences: v11.evidenceRefOccurrences || null,
213
+ uniqueEvidenceRefs: v11.evidenceRefs.size || null,
214
+ evidenceReuseRatio: v11.evidenceRefOccurrences
215
+ ? 1 - (v11.evidenceRefs.size / v11.evidenceRefOccurrences)
216
+ : null,
217
+ visualRepairAttempts: v11.visualRepairAttempts || null,
218
+ contextExpansions: v11.contextExpansions || null,
219
+ modelEscalations: v11.modelEscalations || null,
220
+ },
157
221
  }
158
222
  }
@@ -0,0 +1,84 @@
1
+ const DEFAULT_WEIGHTS = {
2
+ instructions: 0.10,
3
+ task: 0.08,
4
+ declared: 0.34,
5
+ tests: 0.15,
6
+ references: 0.20,
7
+ history: 0.05,
8
+ tools: 0.08,
9
+ }
10
+
11
+ function clamp(value, min, max) {
12
+ return Math.min(max, Math.max(min, value))
13
+ }
14
+
15
+ function normalizeWeights(weights) {
16
+ const total = Object.values(weights).reduce((sum, value) => sum + Math.max(0, Number(value) || 0), 0) || 1
17
+ return Object.fromEntries(Object.entries(weights).map(([key, value]) => [key, Math.max(0, Number(value) || 0) / total]))
18
+ }
19
+
20
+ export function planEvidenceBudget(taskPolicy = {}, task = {}, signals = {}) {
21
+ const base = clamp(
22
+ Number(taskPolicy.contextBudget ?? taskPolicy.profile?.contextBudget ?? signals.contextBudget ?? 20_000),
23
+ 4_000,
24
+ 48_000,
25
+ )
26
+ const text = [task?.title, task?.summary, ...(task?.acceptance || [])].filter(Boolean).join(" ").toLowerCase()
27
+ const visual = signals.visual === true || signals.vision === true || /(screenshot|figma|visual|pixel|layout|giao diện|hình ảnh|ảnh mẫu)/i.test(text)
28
+ const browser = signals.browser === true || /(browser|playwright|e2e|click|navigation|trình duyệt)/i.test(text)
29
+ const debugging = signals.debugging === true || /(fix|bug|error|regression|debug|lỗi)/i.test(text)
30
+ const highRisk = taskPolicy.risk === "high"
31
+
32
+ const weights = { ...DEFAULT_WEIGHTS }
33
+ if (debugging) {
34
+ weights.tests += 0.07
35
+ weights.references -= 0.04
36
+ weights.history += 0.02
37
+ weights.declared -= 0.05
38
+ }
39
+ if (highRisk) {
40
+ weights.tests += 0.06
41
+ weights.references += 0.04
42
+ weights.tools -= 0.03
43
+ weights.declared -= 0.05
44
+ weights.task -= 0.02
45
+ }
46
+ if (visual || browser) {
47
+ weights.tools += 0.08
48
+ weights.references -= 0.04
49
+ weights.declared -= 0.04
50
+ }
51
+
52
+ const normalized = normalizeWeights(weights)
53
+ const buckets = Object.fromEntries(
54
+ Object.entries(normalized).map(([key, weight]) => [key, Math.max(256, Math.round(base * weight))]),
55
+ )
56
+ const allocated = Object.values(buckets).reduce((sum, value) => sum + value, 0)
57
+ const drift = base - allocated
58
+ buckets.declared = Math.max(256, buckets.declared + drift)
59
+
60
+ return {
61
+ schemaVersion: 1,
62
+ total: base,
63
+ unit: "characters",
64
+ buckets,
65
+ signals: { visual, browser, debugging, highRisk },
66
+ expansion: {
67
+ initial: base,
68
+ diagnose: Math.min(48_000, Math.max(base, Math.round(base * 1.35))),
69
+ deepRecovery: Math.min(48_000, Math.max(20_000, Math.round(base * 1.75))),
70
+ },
71
+ }
72
+ }
73
+
74
+ export function bucketLimit(plan, role, remaining = Infinity) {
75
+ const limit = Number(plan?.buckets?.[role] || 0)
76
+ return Math.max(0, Math.min(limit, Number.isFinite(remaining) ? remaining : limit))
77
+ }
78
+
79
+ export function evidenceValueScore({ relevance = 0, freshness = 0, confidence = 0, chars = 1 } = {}) {
80
+ const signal = Math.max(0, Number(relevance)) * 0.55 +
81
+ Math.max(0, Number(freshness)) * 0.20 +
82
+ Math.max(0, Number(confidence)) * 0.25
83
+ return signal / Math.max(1, Number(chars))
84
+ }