opencode-agent-skill 10.0.0 → 12.0.0-beta.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/CHANGELOG.md +85 -0
  2. package/README.md +60 -8
  3. package/bin/ocskill.mjs +354 -6
  4. package/docs/DETERMINISTIC-TOOLS.md +1 -1
  5. package/docs/ENGINEERING-DESIGN.md +4 -4
  6. package/docs/EVALS.md +3 -3
  7. package/docs/GITHUB-RULESET.md +50 -0
  8. package/docs/NPM-PUBLISH.md +4 -4
  9. package/docs/OPENCODE-COMPAT.md +3 -3
  10. package/docs/TRACE-SCHEMA.md +1 -1
  11. package/docs/V11-PERCEPTION-ADAPTIVE-EXECUTION.md +75 -0
  12. package/docs/V11-PERCEPTION-ADAPTIVE.md +220 -0
  13. package/docs/V12-WEAK-MODEL-INTELLIGENCE.md +27 -0
  14. package/evals/repo-scale/tasks.json +62 -0
  15. package/evals/router-triggers.json +82 -0
  16. package/evals/routing.json +76 -0
  17. package/evals/v11/tasks.json +122 -0
  18. package/global-config/agents/merge-arbiter.md +12 -0
  19. package/global-config/agents/visual-verifier.md +12 -0
  20. package/global-config/plugins/ues-router/index.js +272 -2
  21. package/global-config/plugins/ues-router/router.js +27 -3
  22. package/global-config/skills/browser-qa/SKILL.md +14 -0
  23. package/global-config/skills/browser-qa/references/workflow.md +11 -0
  24. package/global-config/skills/browser-security/SKILL.md +12 -0
  25. package/global-config/skills/component-visual-testing/SKILL.md +10 -0
  26. package/global-config/skills/design-source/SKILL.md +10 -0
  27. package/global-config/skills/design-source/references/workflow.md +12 -0
  28. package/global-config/skills/dynamic-workflow/SKILL.md +18 -0
  29. package/global-config/skills/dynamic-workflow/references/workflow.md +19 -0
  30. package/global-config/skills/responsive-verification/SKILL.md +10 -0
  31. package/global-config/skills/skill-authoring/SKILL.md +12 -0
  32. package/global-config/skills/skill-evaluation/SKILL.md +17 -0
  33. package/global-config/skills/visual-fidelity/SKILL.md +14 -0
  34. package/global-config/skills/visual-fidelity/references/workflow.md +14 -0
  35. package/lib/browser-adapter.mjs +82 -0
  36. package/lib/browser-runtime.mjs +193 -0
  37. package/lib/capability-registry.mjs +109 -0
  38. package/lib/context-engine-v11.mjs +150 -0
  39. package/lib/context-manifest.mjs +16 -3
  40. package/lib/context-quality.mjs +59 -0
  41. package/lib/control-center.mjs +12 -2
  42. package/lib/decision-policy.mjs +23 -0
  43. package/lib/dynamic-workflow.mjs +179 -0
  44. package/lib/eval-ablation.mjs +43 -1
  45. package/lib/eval-report.mjs +72 -0
  46. package/lib/eval-telemetry.mjs +61 -0
  47. package/lib/evidence-budget.mjs +84 -0
  48. package/lib/evidence-store.mjs +178 -0
  49. package/lib/hermes-bridge.mjs +45 -1
  50. package/lib/model-config.mjs +21 -1
  51. package/lib/model-performance.mjs +113 -0
  52. package/lib/model-policy.mjs +59 -1
  53. package/lib/orchestrator-policy.mjs +1 -1
  54. package/lib/png-diff.mjs +229 -0
  55. package/lib/prompt-cache.mjs +60 -0
  56. package/lib/repo-scale-fixture.mjs +45 -0
  57. package/lib/skill-quality.mjs +72 -0
  58. package/lib/task-engine.mjs +95 -7
  59. package/lib/ui-inspector.mjs +152 -0
  60. package/lib/v11-metrics.mjs +64 -0
  61. package/lib/visual-spec.mjs +159 -0
  62. package/lib/work-plan-scope.mjs +49 -0
  63. package/package.json +13 -5
  64. package/scripts/check-release-consistency.mjs +228 -0
  65. package/scripts/eval-ablation.mjs +4 -1
  66. package/scripts/validate-repo-scale-suite.mjs +27 -0
  67. package/scripts/validate-v11-suite.mjs +58 -0
  68. package/scripts/validate-v12-foundation.mjs +24 -0
  69. package/scripts/validate.mjs +16 -4
@@ -14,6 +14,20 @@ export function summarizeEvalResults(results) {
14
14
  initialInputSamples: 0,
15
15
  cost: 0,
16
16
  costSamples: 0,
17
+ cacheableRatio: 0,
18
+ cacheableRatioSamples: 0,
19
+ repeatedStableChars: 0,
20
+ repeatedStableSamples: 0,
21
+ repeatedStableRatio: 0,
22
+ repeatedStableRatioSamples: 0,
23
+ evidenceReuseRatio: 0,
24
+ evidenceReuseSamples: 0,
25
+ visualRepairAttempts: 0,
26
+ visualRepairSamples: 0,
27
+ contextExpansions: 0,
28
+ contextExpansionSamples: 0,
29
+ modelEscalations: 0,
30
+ modelEscalationSamples: 0,
17
31
  }
18
32
  bucket.total += 1
19
33
  if (item.passed) bucket.passed += 1
@@ -36,6 +50,36 @@ export function summarizeEvalResults(results) {
36
50
  bucket.cost += Number(item.telemetry?.cost) || 0
37
51
  bucket.costSamples += 1
38
52
  }
53
+
54
+ const v11 = item.telemetry?.v11 || {}
55
+ if (Number.isFinite(Number(v11.avgCacheableRatio))) {
56
+ bucket.cacheableRatio += Number(v11.avgCacheableRatio)
57
+ bucket.cacheableRatioSamples += 1
58
+ }
59
+ if (Number.isFinite(Number(v11.repeatedStableChars))) {
60
+ bucket.repeatedStableChars += Number(v11.repeatedStableChars)
61
+ bucket.repeatedStableSamples += 1
62
+ }
63
+ if (Number.isFinite(Number(v11.repeatedStableRatio))) {
64
+ bucket.repeatedStableRatio += Number(v11.repeatedStableRatio)
65
+ bucket.repeatedStableRatioSamples += 1
66
+ }
67
+ if (Number.isFinite(Number(v11.evidenceReuseRatio))) {
68
+ bucket.evidenceReuseRatio += Number(v11.evidenceReuseRatio)
69
+ bucket.evidenceReuseSamples += 1
70
+ }
71
+ if (Number.isFinite(Number(v11.visualRepairAttempts))) {
72
+ bucket.visualRepairAttempts += Number(v11.visualRepairAttempts)
73
+ bucket.visualRepairSamples += 1
74
+ }
75
+ if (Number.isFinite(Number(v11.contextExpansions))) {
76
+ bucket.contextExpansions += Number(v11.contextExpansions)
77
+ bucket.contextExpansionSamples += 1
78
+ }
79
+ if (Number.isFinite(Number(v11.modelEscalations))) {
80
+ bucket.modelEscalations += Number(v11.modelEscalations)
81
+ bucket.modelEscalationSamples += 1
82
+ }
39
83
  }
40
84
 
41
85
  for (const bucket of Object.values(modes)) {
@@ -45,11 +89,25 @@ export function summarizeEvalResults(results) {
45
89
  bucket.avgTokens = bucket.tokenSamples ? bucket.tokens / bucket.tokenSamples : null
46
90
  bucket.avgInitialInputTokens = bucket.initialInputSamples ? bucket.initialInputTokens / bucket.initialInputSamples : null
47
91
  bucket.avgCost = bucket.costSamples ? bucket.cost / bucket.costSamples : null
92
+ bucket.avgCacheableRatio = bucket.cacheableRatioSamples ? bucket.cacheableRatio / bucket.cacheableRatioSamples : null
93
+ bucket.avgRepeatedStableChars = bucket.repeatedStableSamples ? bucket.repeatedStableChars / bucket.repeatedStableSamples : null
94
+ bucket.avgRepeatedStableRatio = bucket.repeatedStableRatioSamples ? bucket.repeatedStableRatio / bucket.repeatedStableRatioSamples : null
95
+ bucket.avgEvidenceReuseRatio = bucket.evidenceReuseSamples ? bucket.evidenceReuseRatio / bucket.evidenceReuseSamples : null
96
+ bucket.avgVisualRepairAttempts = bucket.visualRepairSamples ? bucket.visualRepairAttempts / bucket.visualRepairSamples : null
97
+ bucket.avgContextExpansions = bucket.contextExpansionSamples ? bucket.contextExpansions / bucket.contextExpansionSamples : null
98
+ bucket.avgModelEscalations = bucket.modelEscalationSamples ? bucket.modelEscalations / bucket.modelEscalationSamples : null
48
99
  bucket.telemetryCoverage = {
49
100
  tools: bucket.total ? bucket.toolSamples / bucket.total : 0,
50
101
  tokens: bucket.total ? bucket.tokenSamples / bucket.total : 0,
51
102
  initialInputTokens: bucket.total ? bucket.initialInputSamples / bucket.total : 0,
52
103
  cost: bucket.total ? bucket.costSamples / bucket.total : 0,
104
+ cacheableRatio: bucket.total ? bucket.cacheableRatioSamples / bucket.total : 0,
105
+ repeatedStableChars: bucket.total ? bucket.repeatedStableSamples / bucket.total : 0,
106
+ repeatedStableRatio: bucket.total ? bucket.repeatedStableRatioSamples / bucket.total : 0,
107
+ evidenceReuseRatio: bucket.total ? bucket.evidenceReuseSamples / bucket.total : 0,
108
+ visualRepairAttempts: bucket.total ? bucket.visualRepairSamples / bucket.total : 0,
109
+ contextExpansions: bucket.total ? bucket.contextExpansionSamples / bucket.total : 0,
110
+ modelEscalations: bucket.total ? bucket.modelEscalationSamples / bucket.total : 0,
53
111
  }
54
112
  delete bucket.durationMs
55
113
  delete bucket.toolCalls
@@ -60,6 +118,20 @@ export function summarizeEvalResults(results) {
60
118
  delete bucket.initialInputSamples
61
119
  delete bucket.cost
62
120
  delete bucket.costSamples
121
+ delete bucket.cacheableRatio
122
+ delete bucket.cacheableRatioSamples
123
+ delete bucket.repeatedStableChars
124
+ delete bucket.repeatedStableSamples
125
+ delete bucket.repeatedStableRatio
126
+ delete bucket.repeatedStableRatioSamples
127
+ delete bucket.evidenceReuseRatio
128
+ delete bucket.evidenceReuseSamples
129
+ delete bucket.visualRepairAttempts
130
+ delete bucket.visualRepairSamples
131
+ delete bucket.contextExpansions
132
+ delete bucket.contextExpansionSamples
133
+ delete bucket.modelEscalations
134
+ delete bucket.modelEscalationSamples
63
135
  }
64
136
 
65
137
  const byTaskMap = new Map()
@@ -93,6 +93,18 @@ export function parseOpenCodeTelemetry(stdout) {
93
93
  let usageSamples = 0
94
94
  let costSamples = 0
95
95
  let firstUsage = null
96
+ const v11 = {
97
+ promptCacheSamples: 0,
98
+ cacheableRatioSum: 0,
99
+ stableChars: 0,
100
+ dynamicChars: 0,
101
+ repeatedStableChars: 0,
102
+ evidenceRefs: new Set(),
103
+ evidenceRefOccurrences: 0,
104
+ visualRepairAttempts: 0,
105
+ contextExpansions: 0,
106
+ modelEscalations: 0,
107
+ }
96
108
 
97
109
  parsed.forEach((event, eventIndex) => {
98
110
  let eventUsage = null
@@ -120,6 +132,39 @@ export function parseOpenCodeTelemetry(stdout) {
120
132
  }
121
133
 
122
134
  if (!eventUsage) eventUsage = usageFromObject(object)
135
+
136
+ const promptCache = object.promptCache && typeof object.promptCache === "object" ? object.promptCache : null
137
+ if (promptCache) {
138
+ const ratio = Number(promptCache.cacheableRatio)
139
+ if (Number.isFinite(ratio)) {
140
+ v11.promptCacheSamples += 1
141
+ v11.cacheableRatioSum += ratio
142
+ }
143
+ const stableChars = Number(promptCache.stableChars)
144
+ if (Number.isFinite(stableChars) && stableChars >= 0) v11.stableChars += stableChars
145
+ const dynamicChars = Number(promptCache.dynamicChars)
146
+ if (Number.isFinite(dynamicChars) && dynamicChars >= 0) v11.dynamicChars += dynamicChars
147
+ const repeated = Number(promptCache.repeatedStableChars)
148
+ if (Number.isFinite(repeated) && repeated >= 0) v11.repeatedStableChars += repeated
149
+ }
150
+
151
+ const scanRefs = (value) => {
152
+ if (typeof value === "string" && /^evidence:sha256:[a-f0-9]{64}$/i.test(value)) {
153
+ v11.evidenceRefOccurrences += 1
154
+ v11.evidenceRefs.add(value.toLowerCase())
155
+ } else if (Array.isArray(value)) {
156
+ for (const item of value) scanRefs(item)
157
+ } else if (value && typeof value === "object") {
158
+ for (const child of Object.values(value)) scanRefs(child)
159
+ }
160
+ }
161
+ scanRefs(object.evidencePointers)
162
+ scanRefs(object.evidence)
163
+
164
+ const eventName = String(object.type || object.event || object.kind || "").toLowerCase()
165
+ if (eventName.includes("visual") && eventName.includes("repair")) v11.visualRepairAttempts += 1
166
+ if (eventName.includes("context") && (eventName.includes("expand") || eventName.includes("recovery"))) v11.contextExpansions += 1
167
+ if (eventName.includes("model") && eventName.includes("escalat")) v11.modelEscalations += 1
123
168
  if (eventCost === null) {
124
169
  for (const key of ["cost", "totalCost", "total_cost"]) {
125
170
  if (typeof object[key] === "number" && Number.isFinite(object[key])) {
@@ -157,5 +202,21 @@ export function parseOpenCodeTelemetry(stdout) {
157
202
  usageSamples,
158
203
  cost,
159
204
  costSamples,
205
+ v11: {
206
+ promptCacheSamples: v11.promptCacheSamples,
207
+ avgCacheableRatio: v11.promptCacheSamples ? v11.cacheableRatioSum / v11.promptCacheSamples : null,
208
+ avgStableChars: v11.promptCacheSamples ? v11.stableChars / v11.promptCacheSamples : null,
209
+ avgDynamicChars: v11.promptCacheSamples ? v11.dynamicChars / v11.promptCacheSamples : null,
210
+ repeatedStableChars: v11.promptCacheSamples ? v11.repeatedStableChars : null,
211
+ repeatedStableRatio: v11.stableChars > 0 ? v11.repeatedStableChars / v11.stableChars : null,
212
+ evidenceRefOccurrences: v11.evidenceRefOccurrences || null,
213
+ uniqueEvidenceRefs: v11.evidenceRefs.size || null,
214
+ evidenceReuseRatio: v11.evidenceRefOccurrences
215
+ ? 1 - (v11.evidenceRefs.size / v11.evidenceRefOccurrences)
216
+ : null,
217
+ visualRepairAttempts: v11.visualRepairAttempts || null,
218
+ contextExpansions: v11.contextExpansions || null,
219
+ modelEscalations: v11.modelEscalations || null,
220
+ },
160
221
  }
161
222
  }
@@ -0,0 +1,84 @@
1
+ const DEFAULT_WEIGHTS = {
2
+ instructions: 0.10,
3
+ task: 0.08,
4
+ declared: 0.34,
5
+ tests: 0.15,
6
+ references: 0.20,
7
+ history: 0.05,
8
+ tools: 0.08,
9
+ }
10
+
11
+ function clamp(value, min, max) {
12
+ return Math.min(max, Math.max(min, value))
13
+ }
14
+
15
+ function normalizeWeights(weights) {
16
+ const total = Object.values(weights).reduce((sum, value) => sum + Math.max(0, Number(value) || 0), 0) || 1
17
+ return Object.fromEntries(Object.entries(weights).map(([key, value]) => [key, Math.max(0, Number(value) || 0) / total]))
18
+ }
19
+
20
+ export function planEvidenceBudget(taskPolicy = {}, task = {}, signals = {}) {
21
+ const base = clamp(
22
+ Number(taskPolicy.contextBudget ?? taskPolicy.profile?.contextBudget ?? signals.contextBudget ?? 20_000),
23
+ 4_000,
24
+ 48_000,
25
+ )
26
+ const text = [task?.title, task?.summary, ...(task?.acceptance || [])].filter(Boolean).join(" ").toLowerCase()
27
+ const visual = signals.visual === true || signals.vision === true || /(screenshot|figma|visual|pixel|layout|giao diện|hình ảnh|ảnh mẫu)/i.test(text)
28
+ const browser = signals.browser === true || /(browser|playwright|e2e|click|navigation|trình duyệt)/i.test(text)
29
+ const debugging = signals.debugging === true || /(fix|bug|error|regression|debug|lỗi)/i.test(text)
30
+ const highRisk = taskPolicy.risk === "high"
31
+
32
+ const weights = { ...DEFAULT_WEIGHTS }
33
+ if (debugging) {
34
+ weights.tests += 0.07
35
+ weights.references -= 0.04
36
+ weights.history += 0.02
37
+ weights.declared -= 0.05
38
+ }
39
+ if (highRisk) {
40
+ weights.tests += 0.06
41
+ weights.references += 0.04
42
+ weights.tools -= 0.03
43
+ weights.declared -= 0.05
44
+ weights.task -= 0.02
45
+ }
46
+ if (visual || browser) {
47
+ weights.tools += 0.08
48
+ weights.references -= 0.04
49
+ weights.declared -= 0.04
50
+ }
51
+
52
+ const normalized = normalizeWeights(weights)
53
+ const buckets = Object.fromEntries(
54
+ Object.entries(normalized).map(([key, weight]) => [key, Math.max(256, Math.round(base * weight))]),
55
+ )
56
+ const allocated = Object.values(buckets).reduce((sum, value) => sum + value, 0)
57
+ const drift = base - allocated
58
+ buckets.declared = Math.max(256, buckets.declared + drift)
59
+
60
+ return {
61
+ schemaVersion: 1,
62
+ total: base,
63
+ unit: "characters",
64
+ buckets,
65
+ signals: { visual, browser, debugging, highRisk },
66
+ expansion: {
67
+ initial: base,
68
+ diagnose: Math.min(48_000, Math.max(base, Math.round(base * 1.35))),
69
+ deepRecovery: Math.min(48_000, Math.max(20_000, Math.round(base * 1.75))),
70
+ },
71
+ }
72
+ }
73
+
74
+ export function bucketLimit(plan, role, remaining = Infinity) {
75
+ const limit = Number(plan?.buckets?.[role] || 0)
76
+ return Math.max(0, Math.min(limit, Number.isFinite(remaining) ? remaining : limit))
77
+ }
78
+
79
+ export function evidenceValueScore({ relevance = 0, freshness = 0, confidence = 0, chars = 1 } = {}) {
80
+ const signal = Math.max(0, Number(relevance)) * 0.55 +
81
+ Math.max(0, Number(freshness)) * 0.20 +
82
+ Math.max(0, Number(confidence)) * 0.25
83
+ return signal / Math.max(1, Number(chars))
84
+ }
@@ -0,0 +1,178 @@
1
+ import { createHash } from "node:crypto"
2
+ import { existsSync } from "node:fs"
3
+ import { mkdir, readFile, readdir, rm, stat, writeFile } from "node:fs/promises"
4
+ import path from "node:path"
5
+
6
+ const STORE_VERSION = 1
7
+ const REF_PREFIX = "evidence:sha256:"
8
+
9
+ function stableValue(value) {
10
+ if (Array.isArray(value)) return value.map(stableValue)
11
+ if (!value || typeof value !== "object" || Buffer.isBuffer(value)) return value
12
+ return Object.fromEntries(
13
+ Object.keys(value).sort().map((key) => [key, stableValue(value[key])]),
14
+ )
15
+ }
16
+
17
+ function toBytes(value, options = {}) {
18
+ if (Buffer.isBuffer(value)) return { bytes: value, encoding: "binary", mediaType: options.mediaType || "application/octet-stream" }
19
+ if (typeof value === "string") return { bytes: Buffer.from(value, "utf8"), encoding: "utf8", mediaType: options.mediaType || "text/plain; charset=utf-8" }
20
+ const text = JSON.stringify(stableValue(value), null, options.pretty === false ? 0 : 2)
21
+ return { bytes: Buffer.from(text, "utf8"), encoding: "utf8", mediaType: options.mediaType || "application/json" }
22
+ }
23
+
24
+ function normalizeHash(ref) {
25
+ const value = String(ref || "").trim()
26
+ const hash = value.startsWith(REF_PREFIX) ? value.slice(REF_PREFIX.length) : value.replace(/^sha256:/, "")
27
+ if (!/^[a-f0-9]{64}$/i.test(hash)) throw new Error("Invalid evidence reference")
28
+ return hash.toLowerCase()
29
+ }
30
+
31
+ export function evidenceReference(hash) {
32
+ return REF_PREFIX + normalizeHash(hash)
33
+ }
34
+
35
+ export function evidenceStoreRoot(root = process.cwd()) {
36
+ return path.join(path.resolve(root), ".ues-cache", "evidence-v1")
37
+ }
38
+
39
+ function evidencePaths(root, hash) {
40
+ const normalized = normalizeHash(hash)
41
+ const dir = path.join(evidenceStoreRoot(root), normalized.slice(0, 2))
42
+ return {
43
+ dir,
44
+ meta: path.join(dir, normalized + ".json"),
45
+ data: path.join(dir, normalized + ".blob"),
46
+ }
47
+ }
48
+
49
+ export async function putEvidence(root, value, options = {}) {
50
+ root = path.resolve(root)
51
+ const encoded = toBytes(value, options)
52
+ const hash = createHash("sha256").update(encoded.bytes).digest("hex")
53
+ const target = evidencePaths(root, hash)
54
+ await mkdir(target.dir, { recursive: true })
55
+
56
+ const now = new Date().toISOString()
57
+ const existing = existsSync(target.meta)
58
+ ? JSON.parse(await readFile(target.meta, "utf8").catch(() => "{}"))
59
+ : null
60
+
61
+ if (!existsSync(target.data)) await writeFile(target.data, encoded.bytes)
62
+
63
+ const preview = encoded.encoding === "utf8"
64
+ ? encoded.bytes.toString("utf8", 0, Math.min(encoded.bytes.length, 600))
65
+ : null
66
+
67
+ const metadata = {
68
+ schemaVersion: STORE_VERSION,
69
+ ref: evidenceReference(hash),
70
+ sha256: hash,
71
+ bytes: encoded.bytes.length,
72
+ encoding: encoded.encoding,
73
+ mediaType: encoded.mediaType,
74
+ kind: options.kind || existing?.kind || "tool-output",
75
+ source: options.source || existing?.source || null,
76
+ summary: options.summary || existing?.summary || null,
77
+ createdAt: existing?.createdAt || now,
78
+ lastSeenAt: now,
79
+ preview,
80
+ }
81
+ await writeFile(target.meta, JSON.stringify(metadata, null, 2) + "\n", "utf8")
82
+ return metadata
83
+ }
84
+
85
+ export async function getEvidence(root, ref, options = {}) {
86
+ const hash = normalizeHash(ref)
87
+ const target = evidencePaths(root, hash)
88
+ const metadata = JSON.parse(await readFile(target.meta, "utf8"))
89
+ const bytes = await readFile(target.data)
90
+ const start = Math.max(0, Number(options.start || 0))
91
+ const maxBytes = Math.max(1, Number(options.maxBytes || options.maxChars || 24_000))
92
+ const slice = bytes.subarray(start, Math.min(bytes.length, start + maxBytes))
93
+ return {
94
+ ...metadata,
95
+ truncated: start + slice.length < bytes.length,
96
+ start,
97
+ returnedBytes: slice.length,
98
+ content: metadata.encoding === "utf8" ? slice.toString("utf8") : slice.toString("base64"),
99
+ }
100
+ }
101
+
102
+ async function metadataFiles(root) {
103
+ const base = evidenceStoreRoot(root)
104
+ const prefixes = await readdir(base, { withFileTypes: true }).catch(() => [])
105
+ const files = []
106
+ for (const prefix of prefixes) {
107
+ if (!prefix.isDirectory()) continue
108
+ const dir = path.join(base, prefix.name)
109
+ for (const entry of await readdir(dir, { withFileTypes: true }).catch(() => [])) {
110
+ if (entry.isFile() && entry.name.endsWith(".json")) files.push(path.join(dir, entry.name))
111
+ }
112
+ }
113
+ return files
114
+ }
115
+
116
+ export async function evidenceStoreStatus(root = process.cwd()) {
117
+ const files = await metadataFiles(root)
118
+ let bytes = 0
119
+ let oldest = null
120
+ let newest = null
121
+ for (const file of files) {
122
+ const meta = JSON.parse(await readFile(file, "utf8").catch(() => "{}"))
123
+ bytes += Number(meta.bytes || 0)
124
+ if (meta.createdAt && (!oldest || meta.createdAt < oldest)) oldest = meta.createdAt
125
+ if (meta.lastSeenAt && (!newest || meta.lastSeenAt > newest)) newest = meta.lastSeenAt
126
+ }
127
+ return {
128
+ schemaVersion: STORE_VERSION,
129
+ root: evidenceStoreRoot(root),
130
+ entries: files.length,
131
+ bytes,
132
+ oldest,
133
+ newest,
134
+ }
135
+ }
136
+
137
+ export async function gcEvidenceStore(root = process.cwd(), options = {}) {
138
+ const files = await metadataFiles(root)
139
+ const maxEntries = Math.max(10, Number(options.maxEntries || 2_000))
140
+ const maxAgeMs = Math.max(0, Number(options.maxAgeDays ?? 30)) * 86_400_000
141
+ const now = Date.now()
142
+ const rows = []
143
+ for (const metaFile of files) {
144
+ const meta = JSON.parse(await readFile(metaFile, "utf8").catch(() => "{}"))
145
+ const time = Date.parse(meta.lastSeenAt || meta.createdAt || 0) || 0
146
+ rows.push({ metaFile, meta, time })
147
+ }
148
+ rows.sort((a, b) => b.time - a.time)
149
+
150
+ const removed = []
151
+ for (let index = 0; index < rows.length; index += 1) {
152
+ const row = rows[index]
153
+ const tooMany = index >= maxEntries
154
+ const tooOld = maxAgeMs > 0 && row.time > 0 && now - row.time > maxAgeMs
155
+ if (!tooMany && !tooOld) continue
156
+ const hash = row.meta.sha256 || path.basename(row.metaFile, ".json")
157
+ const target = evidencePaths(root, hash)
158
+ await rm(target.meta, { force: true })
159
+ await rm(target.data, { force: true })
160
+ removed.push(evidenceReference(hash))
161
+ }
162
+
163
+ return {
164
+ removed,
165
+ removedCount: removed.length,
166
+ status: await evidenceStoreStatus(root),
167
+ }
168
+ }
169
+
170
+ export async function evidenceExists(root, ref) {
171
+ try {
172
+ const target = evidencePaths(root, normalizeHash(ref))
173
+ const info = await stat(target.data)
174
+ return info.isFile()
175
+ } catch {
176
+ return false
177
+ }
178
+ }
@@ -1,11 +1,20 @@
1
1
  import { spawnSync } from "node:child_process"
2
+ import { resolveWindowsCommand } from "./windows-shim.mjs"
3
+
4
+ function runHermesVersion() {
5
+ if (process.platform !== "win32") return spawnSync("hermes", ["--version"], { encoding: "utf8" })
6
+ const resolved = resolveWindowsCommand("hermes")
7
+ if (!resolved) return { status: 127, stdout: "", stderr: "Hermes CLI not found" }
8
+ return spawnSync(resolved.executable, [...resolved.argsPrefix, "--version"], { encoding: "utf8" })
9
+ }
2
10
 
3
11
  export function hermesStatus() {
4
- const result = spawnSync("hermes", ["--version"], { encoding: "utf8", shell: process.platform === "win32" })
12
+ const result = runHermesVersion()
5
13
  return {
6
14
  available: result.status === 0,
7
15
  version: result.status === 0 ? String(result.stdout || result.stderr || "").trim() : null,
8
16
  error: result.status === 0 ? null : String(result.stderr || result.stdout || "Hermes CLI not found").trim(),
17
+ mode: "optional-sidecar",
9
18
  }
10
19
  }
11
20
 
@@ -15,11 +24,46 @@ export function buildHermesDelegationPrompt(contextPack) {
15
24
  "Implement exactly the approved task described below.",
16
25
  "Do not broaden scope, merge, push, publish, deploy, or alter durable UES state.",
17
26
  "Run the declared verification and return a concise structured report.",
27
+ "Large evidence is referenced by evidence:sha256 pointers; fetch only the slice required for the task.",
28
+ "",
29
+ JSON.stringify(contextPack, null, 2),
30
+ ].join("\n")
31
+ }
32
+
33
+ export function buildHermesWorkflowPrompt(contextPack, schedule) {
34
+ return [
35
+ "You are the optional Hermes sidecar for a UES V11 dynamic workflow.",
36
+ "Follow the supplied bounded wave schedule. Deterministic tasks are not delegated to LLM children.",
37
+ "Never exceed declared task ownership or concurrency. Persist outputs/evidence to files instead of conversational summaries.",
38
+ "Do not merge, push, publish, deploy, or mutate durable UES state except through explicitly supplied UES commands.",
39
+ "",
40
+ "SCHEDULE:",
41
+ JSON.stringify(schedule, null, 2),
18
42
  "",
43
+ "CONTEXT:",
19
44
  JSON.stringify(contextPack, null, 2),
20
45
  ].join("\n")
21
46
  }
22
47
 
48
+ export function hermesSidecarPlan(input = {}) {
49
+ return {
50
+ schemaVersion: 2,
51
+ adapter: "hermes",
52
+ optional: true,
53
+ mode: input.mode || "one-shot",
54
+ maxConcurrent: Math.max(1, Math.min(16, Number(input.maxConcurrent || 4))),
55
+ durableStateOwner: "ues",
56
+ evidenceTransport: "content-addressed-pointers",
57
+ allowNestedDelegation: input.allowNestedDelegation === true,
58
+ safety: {
59
+ merge: false,
60
+ push: false,
61
+ publish: false,
62
+ deploy: false,
63
+ destructiveGit: false,
64
+ },
65
+ }
66
+ }
23
67
 
24
68
  export function hermesOneShotArgs(prompt) {
25
69
  const text = String(prompt || "").trim()
@@ -2,6 +2,8 @@ import { existsSync } from "node:fs"
2
2
  import { mkdir, readFile, writeFile } from "node:fs/promises"
3
3
  import path from "node:path"
4
4
  import { defaultModelPolicy, resolveModel } from "./model-policy.mjs"
5
+ import { normalizeCapabilityProfile } from "./capability-registry.mjs"
6
+ import { normalizePerformanceHistory, recordPerformanceOutcome } from "./model-performance.mjs"
5
7
 
6
8
  const TIERS = new Set(["light", "standard", "heavy"])
7
9
 
@@ -23,14 +25,24 @@ function normalize(policy) {
23
25
  if (TIERS.has(tier)) roleTiers[role] = tier
24
26
  }
25
27
 
28
+ const capabilities = {}
29
+ for (const [model, profile] of Object.entries(input.capabilities || {})) {
30
+ if (validModelID(model)) capabilities[model] = normalizeCapabilityProfile(profile)
31
+ }
32
+
33
+ const performance = normalizePerformanceHistory(input.performance || {})
34
+
26
35
  return {
27
- schemaVersion: 1,
36
+ schemaVersion: 3,
28
37
  enabled: input.enabled === true,
29
38
  maxEscalations: Number.isInteger(input.maxEscalations)
30
39
  ? Math.max(0, Math.min(input.maxEscalations, 2))
31
40
  : base.maxEscalations,
32
41
  tiers,
33
42
  roleTiers,
43
+ capabilities,
44
+ performance,
45
+ performanceMinSamples: Number.isInteger(input.performanceMinSamples) ? Math.max(1, Math.min(input.performanceMinSamples, 20)) : base.performanceMinSamples,
34
46
  }
35
47
  }
36
48
 
@@ -56,6 +68,8 @@ export async function writeModelPolicy(configDir, patch = {}) {
56
68
  ...patch,
57
69
  tiers: { ...current.tiers, ...(patch.tiers || {}) },
58
70
  roleTiers: { ...current.roleTiers, ...(patch.roleTiers || {}) },
71
+ capabilities: { ...(current.capabilities || {}), ...(patch.capabilities || {}) },
72
+ performance: patch.performance || current.performance || {},
59
73
  })
60
74
  const file = modelPolicyFile(configDir)
61
75
  await mkdir(path.dirname(file), { recursive: true })
@@ -86,3 +100,9 @@ export function applyConfiguredModel(source, role, policy) {
86
100
  }
87
101
  return lines.join("\n")
88
102
  }
103
+
104
+ export async function recordModelPerformance(configDir, outcome = {}) {
105
+ const current = await readModelPolicy(configDir)
106
+ const performance = recordPerformanceOutcome(current.performance || {}, outcome)
107
+ return writeModelPolicy(configDir, { performance })
108
+ }
@@ -0,0 +1,113 @@
1
+ export const MODEL_TASK_CLASSES = Object.freeze([
2
+ "general", "repo-scale", "debugging", "architecture", "security",
3
+ "migration", "frontend", "backend", "visual", "browser",
4
+ ])
5
+ const KNOWN_TASK_CLASSES = new Set(MODEL_TASK_CLASSES)
6
+
7
+ function boundedNumber(value, fallback = 0, min = 0, max = Number.MAX_SAFE_INTEGER) {
8
+ const number = Number(value)
9
+ if (!Number.isFinite(number)) return fallback
10
+ return Math.max(min, Math.min(max, number))
11
+ }
12
+
13
+ export function inferTaskClass(text = "", facts = {}) {
14
+ const explicit = String(facts.taskClass || "").trim().toLowerCase()
15
+ if (KNOWN_TASK_CLASSES.has(explicit)) return explicit
16
+ const value = String(text || "").toLowerCase()
17
+ if (/(whole repo|entire project|large monorepo|repo[- ]scale|cross[- ]module|toàn bộ dự án|nhiều module)/.test(value)) return "repo-scale"
18
+ if (/(prompt injection|security|auth|authorization|permission|secret|credential|bảo mật|phân quyền)/.test(value)) return "security"
19
+ if (/(migration|schema|database|sql|backfill|migrate)/.test(value)) return "migration"
20
+ if (/(screenshot|visual|figma|pixel|responsive|storybook)/.test(value)) return "visual"
21
+ if (/(browser|playwright|e2e|web page|click flow)/.test(value)) return "browser"
22
+ if (/(root cause|debug|regression|crash|failing|bug|lỗi)/.test(value)) return "debugging"
23
+ if (/(architecture|architect|design decision|system design|kiến trúc)/.test(value)) return "architecture"
24
+ if (/(react|next\.js|vue|svelte|css|frontend|ui\b)/.test(value)) return "frontend"
25
+ if (/(api|service|node|python|java|dotnet|backend|server)/.test(value)) return "backend"
26
+ return "general"
27
+ }
28
+
29
+ export function normalizePerformanceRecord(record = {}) {
30
+ record = record && typeof record === "object" ? record : {}
31
+ const samples = Math.floor(boundedNumber(record.samples, 0, 0))
32
+ const successes = Math.floor(boundedNumber(record.successes, Math.round(samples * boundedNumber(record.passRate, 0, 0, 1)), 0, samples))
33
+ return {
34
+ samples,
35
+ successes,
36
+ passRate: samples ? successes / samples : 0,
37
+ avgRetries: boundedNumber(record.avgRetries, 0, 0, 100),
38
+ avgTokens: boundedNumber(record.avgTokens, 0, 0),
39
+ avgLatencyMs: boundedNumber(record.avgLatencyMs, 0, 0),
40
+ updatedAt: typeof record.updatedAt === "string" ? record.updatedAt : null,
41
+ }
42
+ }
43
+
44
+ export function normalizePerformanceHistory(history = {}) {
45
+ const output = {}
46
+ for (const [model, classes] of Object.entries(history || {})) {
47
+ if (!model || !classes || typeof classes !== "object") continue
48
+ const normalizedClasses = {}
49
+ for (const [taskClass, record] of Object.entries(classes)) {
50
+ if (!KNOWN_TASK_CLASSES.has(taskClass) && taskClass !== "overall") continue
51
+ normalizedClasses[taskClass] = normalizePerformanceRecord(record)
52
+ }
53
+ if (Object.keys(normalizedClasses).length) output[model] = normalizedClasses
54
+ }
55
+ return output
56
+ }
57
+
58
+ function mergeAverage(previousAverage, previousSamples, value) {
59
+ return previousSamples <= 0 ? value : ((previousAverage * previousSamples) + value) / (previousSamples + 1)
60
+ }
61
+
62
+ export function recordPerformanceOutcome(history = {}, outcome = {}) {
63
+ const model = String(outcome.model || "").trim()
64
+ if (!model) throw new Error("model performance outcome requires model")
65
+ const taskClass = inferTaskClass(outcome.text || "", { taskClass: outcome.taskClass })
66
+ const normalized = normalizePerformanceHistory(history)
67
+ const current = normalizePerformanceRecord(normalized[model]?.[taskClass] || {})
68
+ const samples = current.samples
69
+ const passed = outcome.passed === true
70
+ const next = {
71
+ samples: samples + 1,
72
+ successes: current.successes + (passed ? 1 : 0),
73
+ passRate: 0,
74
+ avgRetries: mergeAverage(current.avgRetries, samples, boundedNumber(outcome.retries, 0, 0, 100)),
75
+ avgTokens: mergeAverage(current.avgTokens, samples, boundedNumber(outcome.tokens, 0, 0)),
76
+ avgLatencyMs: mergeAverage(current.avgLatencyMs, samples, boundedNumber(outcome.latencyMs, 0, 0)),
77
+ updatedAt: new Date().toISOString(),
78
+ }
79
+ next.passRate = next.successes / next.samples
80
+ return { ...normalized, [model]: { ...(normalized[model] || {}), [taskClass]: next } }
81
+ }
82
+
83
+ function performanceAdjustment(record, minSamples) {
84
+ const normalized = normalizePerformanceRecord(record)
85
+ if (normalized.samples <= 0) return { adjustment: 0, confidence: 0, record: normalized }
86
+ const confidence = Math.min(1, normalized.samples / Math.max(1, minSamples))
87
+ if (normalized.samples < minSamples) return { adjustment: 0, confidence, record: normalized }
88
+ const correctness = (normalized.passRate - 0.5) * 80
89
+ const retryPenalty = Math.min(20, normalized.avgRetries * 5)
90
+ const latencyPenalty = normalized.avgLatencyMs > 0 ? Math.min(10, Math.max(0, Math.log10(Math.max(1, normalized.avgLatencyMs / 1000)) * 3)) : 0
91
+ return { adjustment: (correctness - retryPenalty - latencyPenalty) * confidence, confidence, record: normalized }
92
+ }
93
+
94
+ export function rerankCapabilitySelection(selection = {}, history = {}, options = {}) {
95
+ const taskClass = inferTaskClass(options.text || "", { taskClass: options.taskClass })
96
+ const minSamples = Math.max(1, Number(options.minSamples || 3))
97
+ const normalized = normalizePerformanceHistory(history)
98
+ const candidates = (selection.candidates || []).map((candidate) => {
99
+ const record = normalized[candidate.id]?.[taskClass] || normalized[candidate.id]?.overall || null
100
+ const evidence = performanceAdjustment(record, minSamples)
101
+ return {
102
+ ...candidate,
103
+ baseScore: Number(candidate.score || 0),
104
+ empiricalTaskClass: taskClass,
105
+ empiricalEvidence: evidence.record,
106
+ empiricalConfidence: Number(evidence.confidence.toFixed(4)),
107
+ adjustedScore: Number((Number(candidate.score || 0) + evidence.adjustment).toFixed(6)),
108
+ }
109
+ })
110
+ const eligible = candidates.filter((candidate) => candidate.eligible)
111
+ .sort((a, b) => b.adjustedScore - a.adjustedScore || b.baseScore - a.baseScore)
112
+ return { ...selection, selected: eligible[0] || null, candidates, taskClass, empirical: true }
113
+ }