opencode-agent-skill 7.7.0 → 10.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/CHANGELOG.md +112 -3
  2. package/README.md +396 -281
  3. package/bin/ocskill.mjs +382 -156
  4. package/docs/DETERMINISTIC-TOOLS.md +25 -8
  5. package/docs/ENGINEERING-DESIGN.md +31 -13
  6. package/docs/EVALS.md +34 -12
  7. package/docs/NPM-PUBLISH.md +6 -6
  8. package/docs/OPENCODE-COMPAT.md +11 -8
  9. package/docs/TRACE-SCHEMA.md +15 -2
  10. package/docs/V8-INTELLIGENCE-RELIABILITY.md +206 -0
  11. package/docs/V9-SPEED-INTELLIGENCE.md +102 -0
  12. package/evals/live/tasks.json +6 -6
  13. package/evals/polyglot/fixtures/polyglot-bench/api/generated/client.ts +2 -0
  14. package/evals/polyglot/fixtures/polyglot-bench/api/openapi.json +25 -0
  15. package/evals/polyglot/fixtures/polyglot-bench/db/migrations/20260920_add_order_key.sql +1 -0
  16. package/evals/polyglot/fixtures/polyglot-bench/dotnet/OrderService.cs +8 -0
  17. package/evals/polyglot/fixtures/polyglot-bench/java/PriceService.java +5 -0
  18. package/evals/polyglot/fixtures/polyglot-bench/monorepo/package.json +6 -0
  19. package/evals/polyglot/fixtures/polyglot-bench/monorepo/packages/api/package.json +4 -0
  20. package/evals/polyglot/fixtures/polyglot-bench/monorepo/packages/web/package.json +7 -0
  21. package/evals/polyglot/fixtures/polyglot-bench/monorepo/pnpm-lock.yaml +5 -0
  22. package/evals/polyglot/fixtures/polyglot-bench/next/app/api/products/route.ts +7 -0
  23. package/evals/polyglot/fixtures/polyglot-bench/python/tenant_auth.py +4 -0
  24. package/evals/polyglot/fixtures/polyglot-bench/react-native/keyboard.ts +3 -0
  25. package/evals/polyglot/graders/polyglot-bench.mjs +101 -0
  26. package/evals/polyglot/tasks.json +54 -0
  27. package/global-config/AGENTS.md +78 -160
  28. package/global-config/agents/integration-verifier.md +1 -1
  29. package/global-config/agents/plan-checker.md +1 -1
  30. package/global-config/commands/run.md +9 -5
  31. package/global-config/plugins/ues-router/capabilities.js +4 -0
  32. package/global-config/plugins/ues-router/index.js +784 -37
  33. package/global-config/plugins/ues-router/router.js +175 -23
  34. package/global-config/plugins/ues-router/runtime-guard.js +265 -0
  35. package/global-config/skills/engineering-orchestrator/references/long-horizon.md +6 -4
  36. package/lib/aci.mjs +128 -0
  37. package/lib/benchmark-confidence.mjs +173 -0
  38. package/lib/cli-utils.mjs +41 -0
  39. package/lib/container-sandbox.mjs +102 -0
  40. package/lib/context-manifest.mjs +300 -22
  41. package/lib/control-center.mjs +36 -3
  42. package/lib/eval-ablation.mjs +104 -0
  43. package/lib/eval-order.mjs +9 -0
  44. package/lib/eval-report.mjs +11 -0
  45. package/lib/eval-telemetry.mjs +8 -2
  46. package/lib/gate-receipt.mjs +52 -0
  47. package/lib/installer.mjs +39 -25
  48. package/lib/learning-engine.mjs +236 -38
  49. package/lib/model-policy.mjs +6 -0
  50. package/lib/opencode-compat.mjs +25 -10
  51. package/lib/orchestrator-policy.mjs +195 -21
  52. package/lib/process-runner.mjs +30 -9
  53. package/lib/runtime-events.mjs +31 -0
  54. package/lib/semantic-index.mjs +318 -0
  55. package/lib/task-engine.mjs +557 -28
  56. package/lib/trajectory.mjs +89 -0
  57. package/lib/windows-shim.mjs +227 -0
  58. package/lib/worktree-sandbox.mjs +85 -3
  59. package/package.json +11 -4
  60. package/scripts/check-release-tag.mjs +22 -0
  61. package/scripts/control-center.mjs +25 -0
  62. package/scripts/eval-ablation.mjs +44 -0
  63. package/scripts/eval-live.mjs +39 -58
  64. package/scripts/eval-matrix.mjs +166 -0
  65. package/scripts/smoke-packed-install.mjs +138 -4
  66. package/scripts/smoke-plain-install.mjs +91 -0
  67. package/scripts/validate-live-suite.mjs +3 -3
  68. package/scripts/validate.mjs +27 -5
@@ -0,0 +1,52 @@
1
+ import { createHash, randomUUID } from "node:crypto"
2
+
3
+ const KINDS = new Set(["plan-verification", "integration-verification"])
4
+ const VERDICTS = new Set(["PASS", "FAIL", "PARTIAL"])
5
+
6
+ function digest(value) {
7
+ return createHash("sha256").update(String(value || "")).digest("hex")
8
+ }
9
+
10
+ export function createGateReceipt(input = {}) {
11
+ const kind = String(input.kind || "")
12
+ const verdict = String(input.verdict || "").toUpperCase()
13
+ const evidence = String(input.evidence || "").trim()
14
+ const report = input.report == null ? null : String(input.report)
15
+ return {
16
+ schemaVersion: 1,
17
+ id: randomUUID(),
18
+ kind,
19
+ slug: String(input.slug || ""),
20
+ verdict,
21
+ verifier: String(input.verifier || ""),
22
+ sessionID: input.sessionID ? String(input.sessionID) : null,
23
+ runId: input.runId ? String(input.runId) : null,
24
+ planHash: input.planHash ? String(input.planHash) : null,
25
+ workspaceFingerprint: input.workspaceFingerprint ? String(input.workspaceFingerprint) : null,
26
+ evidence,
27
+ reportHash: report == null ? null : digest(report),
28
+ createdAt: input.createdAt || new Date().toISOString(),
29
+ }
30
+ }
31
+
32
+ export function validateGateReceipt(receipt) {
33
+ const errors = []
34
+ if (!receipt || typeof receipt !== "object" || Array.isArray(receipt)) {
35
+ return { valid: false, errors: ["receipt must be an object"] }
36
+ }
37
+ if (receipt.schemaVersion !== 1) errors.push("schemaVersion must be 1")
38
+ if (!KINDS.has(receipt.kind)) errors.push("invalid gate receipt kind")
39
+ if (!VERDICTS.has(receipt.verdict)) errors.push("invalid gate receipt verdict")
40
+ if (!String(receipt.id || "").trim()) errors.push("receipt id is required")
41
+ if (!String(receipt.slug || "").trim()) errors.push("slug is required")
42
+ if (!String(receipt.verifier || "").trim()) errors.push("verifier is required")
43
+ if (!String(receipt.evidence || "").trim()) errors.push("evidence is required")
44
+ if (!Number.isFinite(Date.parse(receipt.createdAt || ""))) errors.push("createdAt must be an ISO timestamp")
45
+ if (receipt.kind === "plan-verification" && !String(receipt.planHash || "").trim()) {
46
+ errors.push("plan verification receipt requires planHash")
47
+ }
48
+ if (receipt.kind === "integration-verification" && !String(receipt.workspaceFingerprint || "").trim()) {
49
+ errors.push("integration verification receipt requires workspaceFingerprint")
50
+ }
51
+ return { valid: errors.length === 0, errors }
52
+ }
package/lib/installer.mjs CHANGED
@@ -11,7 +11,8 @@ export { SKILL_PREFIX } from "./ids.mjs"
11
11
 
12
12
  export const PACKAGE_NAME = "opencode-agent-skill"
13
13
  export const LEGACY_PACKAGE_NAME = "@laivannha0202/opencode-agent-skill"
14
- export const MANAGED_MARKER = "<!-- managed-by: @laivannha0202/opencode-agent-skill -->"
14
+ export const MANAGED_MARKER = "<!-- managed-by: opencode-agent-skill -->"
15
+ export const LEGACY_MANAGED_MARKER = "<!-- managed-by: @laivannha0202/opencode-agent-skill -->"
15
16
  export const AGENTS_BEGIN = "<!-- BEGIN OCSKILL UNIVERSAL ENGINEERING SYSTEM -->"
16
17
  export const AGENTS_END = "<!-- END OCSKILL UNIVERSAL ENGINEERING SYSTEM -->"
17
18
 
@@ -26,6 +27,15 @@ function isOwnedPackageName(value) {
26
27
  return !value || value === PACKAGE_NAME || value === LEGACY_PACKAGE_NAME
27
28
  }
28
29
 
30
+ function hasManagedMarker(source) {
31
+ const text = String(source || "")
32
+ return text.includes(MANAGED_MARKER) || text.includes(LEGACY_MANAGED_MARKER)
33
+ }
34
+
35
+ function normalizeManagedMarker(source) {
36
+ return String(source || "").replaceAll(LEGACY_MANAGED_MARKER, MANAGED_MARKER)
37
+ }
38
+
29
39
  export function getConfigDir() {
30
40
  return process.env.OPENCODE_CONFIG_DIR
31
41
  ? path.resolve(process.env.OPENCODE_CONFIG_DIR)
@@ -47,8 +57,10 @@ async function packageVersion() {
47
57
  }
48
58
 
49
59
  function installSkillName(source, name) {
50
- const next = source.replace(/^name:\s*[^\r\n]+/m, `name: ${SKILL_PREFIX}${name}`)
51
- return next.includes(MANAGED_MARKER)
60
+ const next = normalizeManagedMarker(
61
+ source.replace(/^name:\s*[^\r\n]+/m, `name: ${SKILL_PREFIX}${name}`),
62
+ )
63
+ return hasManagedMarker(next)
52
64
  ? next
53
65
  : `${next.trimEnd()}\n\n${MANAGED_MARKER}\n`
54
66
  }
@@ -69,7 +81,7 @@ function removeManagedBlock(source) {
69
81
  async function writeManagedFile(file, content, warnings) {
70
82
  if (existsSync(file)) {
71
83
  const current = await readText(file)
72
- if (!current.includes(MANAGED_MARKER)) {
84
+ if (!hasManagedMarker(current)) {
73
85
  warnings.push(`Skipped existing unmanaged file: ${file}`)
74
86
  return false
75
87
  }
@@ -121,7 +133,7 @@ async function cleanupStaleManaged(configDir, previousState, installed, warnings
121
133
  const dir = path.join(configDir, "skills", id)
122
134
  const markerFile = path.join(dir, "SKILL.md")
123
135
  const current = await readText(markerFile)
124
- if (current.includes(MANAGED_MARKER)) {
136
+ if (hasManagedMarker(current)) {
125
137
  await rm(dir, { recursive: true, force: true })
126
138
  } else if (existsSync(dir)) {
127
139
  warnings.push(`Preserved stale skill without UES marker: ${dir}`)
@@ -132,7 +144,7 @@ async function cleanupStaleManaged(configDir, previousState, installed, warnings
132
144
  if (currentCommands.has(name) || !validManagedMarkdown(name)) continue
133
145
  const file = path.join(configDir, "commands", name)
134
146
  const current = await readText(file)
135
- if (current.includes(MANAGED_MARKER)) {
147
+ if (hasManagedMarker(current)) {
136
148
  await rm(file, { force: true })
137
149
  } else if (existsSync(file)) {
138
150
  warnings.push(`Preserved stale command without UES marker: ${file}`)
@@ -143,7 +155,7 @@ async function cleanupStaleManaged(configDir, previousState, installed, warnings
143
155
  if (currentAgents.has(name) || !validManagedMarkdown(name)) continue
144
156
  const file = path.join(configDir, "agents", name)
145
157
  const current = await readText(file)
146
- if (current.includes(MANAGED_MARKER)) {
158
+ if (hasManagedMarker(current)) {
147
159
  await rm(file, { force: true })
148
160
  } else if (existsSync(file)) {
149
161
  warnings.push(`Preserved stale subagent without UES marker: ${file}`)
@@ -154,7 +166,7 @@ async function cleanupStaleManaged(configDir, previousState, installed, warnings
154
166
  if (currentPlugins.has(name) || !validManagedPlugin(name)) continue
155
167
  const file = path.join(configDir, "plugins", name)
156
168
  const current = await readText(file)
157
- if (current.includes(MANAGED_MARKER)) {
169
+ if (hasManagedMarker(current)) {
158
170
  await rm(path.dirname(file), { recursive: true, force: true })
159
171
  } else if (existsSync(file)) {
160
172
  warnings.push(`Preserved stale plugin without UES marker: ${file}`)
@@ -169,7 +181,7 @@ async function scanUntrackedManaged(configDir) {
169
181
  for (const entry of await readdir(skillsDir, { withFileTypes: true }).catch(() => [])) {
170
182
  if (!entry.isDirectory() || !validSkillID(entry.name)) continue
171
183
  const current = await readText(path.join(skillsDir, entry.name, "SKILL.md"))
172
- if (current.includes(MANAGED_MARKER)) found.skills.push(entry.name)
184
+ if (hasManagedMarker(current)) found.skills.push(entry.name)
173
185
  }
174
186
 
175
187
  for (const [targetDir, key] of [
@@ -179,13 +191,13 @@ async function scanUntrackedManaged(configDir) {
179
191
  for (const entry of await readdir(targetDir, { withFileTypes: true }).catch(() => [])) {
180
192
  if (!entry.isFile() || !validManagedMarkdown(entry.name)) continue
181
193
  const current = await readText(path.join(targetDir, entry.name))
182
- if (current.includes(MANAGED_MARKER)) found[key].push(entry.name)
194
+ if (hasManagedMarker(current)) found[key].push(entry.name)
183
195
  }
184
196
  }
185
197
 
186
198
  const routerFile = path.join(configDir, "plugins", ROUTER_PLUGIN_STATE)
187
199
  const routerCurrent = await readText(routerFile)
188
- if (routerCurrent.includes(MANAGED_MARKER)) found.plugins.push(ROUTER_PLUGIN_STATE)
200
+ if (hasManagedMarker(routerCurrent)) found.plugins.push(ROUTER_PLUGIN_STATE)
189
201
 
190
202
  return found
191
203
  }
@@ -239,7 +251,7 @@ async function installSkillDirectory(sourceDir, targetDir, sourceName, warnings)
239
251
  }
240
252
 
241
253
  const current = await readText(targetSkill)
242
- if (!current.includes(MANAGED_MARKER)) {
254
+ if (!hasManagedMarker(current)) {
243
255
  warnings.push(`Skipped existing unmanaged skill: ${targetSkill}`)
244
256
  return false
245
257
  }
@@ -265,7 +277,7 @@ async function installPluginDirectory(sourceDir, targetDir, warnings) {
265
277
  }
266
278
 
267
279
  const current = await readText(targetEntry)
268
- if (!current.includes(MANAGED_MARKER)) {
280
+ if (!hasManagedMarker(current)) {
269
281
  warnings.push(`Skipped existing unmanaged plugin: ${targetEntry}`)
270
282
  return false
271
283
  }
@@ -276,10 +288,10 @@ async function installPluginDirectory(sourceDir, targetDir, warnings) {
276
288
  await mkdir(path.dirname(targetDir), { recursive: true })
277
289
  await cp(sourceDir, targetDir, { recursive: true })
278
290
 
279
- const copiedEntry = await readFile(targetEntry, "utf8")
291
+ const copiedEntry = normalizeManagedMarker(await readFile(targetEntry, "utf8"))
280
292
  await writeFile(
281
293
  targetEntry,
282
- copiedEntry.includes(MANAGED_MARKER)
294
+ hasManagedMarker(copiedEntry)
283
295
  ? copiedEntry
284
296
  : `${copiedEntry.trimEnd()}\n\n// ${MANAGED_MARKER}\n`,
285
297
  "utf8",
@@ -366,9 +378,10 @@ export async function installResources(options = {}) {
366
378
  const targetName = `${SKILL_PREFIX}${id}.md`
367
379
  const targetFile = path.join(commandsTarget, targetName)
368
380
  const source = await readFile(path.join(sourceCommands, entry.name), "utf8")
369
- const content = source.includes(MANAGED_MARKER)
370
- ? source
371
- : `${source.trimEnd()}\n\n${MANAGED_MARKER}\n`
381
+ const normalizedSource = normalizeManagedMarker(source)
382
+ const content = hasManagedMarker(normalizedSource)
383
+ ? normalizedSource
384
+ : `${normalizedSource.trimEnd()}\n\n${MANAGED_MARKER}\n`
372
385
 
373
386
  if (await writeManagedFile(targetFile, content, warnings)) {
374
387
  installed.commands.push(targetName)
@@ -396,9 +409,10 @@ export async function installResources(options = {}) {
396
409
  id,
397
410
  modelPolicy,
398
411
  )
399
- const content = source.includes(MANAGED_MARKER)
400
- ? source
401
- : `${source.trimEnd()}\n\n${MANAGED_MARKER}\n`
412
+ const normalizedSource = normalizeManagedMarker(source)
413
+ const content = hasManagedMarker(normalizedSource)
414
+ ? normalizedSource
415
+ : `${normalizedSource.trimEnd()}\n\n${MANAGED_MARKER}\n`
402
416
 
403
417
  if (await writeManagedFile(targetFile, content, warnings)) {
404
418
  installed.agents.push(targetName)
@@ -529,7 +543,7 @@ export async function removeResources(options = {}) {
529
543
  const dir = path.join(configDir, "skills", id)
530
544
  const file = path.join(dir, "SKILL.md")
531
545
  const current = await readText(file)
532
- if (current.includes(MANAGED_MARKER)) {
546
+ if (hasManagedMarker(current)) {
533
547
  await rm(dir, { recursive: true, force: true })
534
548
  removed.skills += 1
535
549
  }
@@ -539,7 +553,7 @@ export async function removeResources(options = {}) {
539
553
  if (!validManagedMarkdown(name)) continue
540
554
  const file = path.join(configDir, "commands", name)
541
555
  const current = await readText(file)
542
- if (current.includes(MANAGED_MARKER)) {
556
+ if (hasManagedMarker(current)) {
543
557
  await rm(file, { force: true })
544
558
  removed.commands += 1
545
559
  }
@@ -549,7 +563,7 @@ export async function removeResources(options = {}) {
549
563
  if (!validManagedMarkdown(name)) continue
550
564
  const file = path.join(configDir, "agents", name)
551
565
  const current = await readText(file)
552
- if (current.includes(MANAGED_MARKER)) {
566
+ if (hasManagedMarker(current)) {
553
567
  await rm(file, { force: true })
554
568
  removed.agents += 1
555
569
  }
@@ -559,7 +573,7 @@ export async function removeResources(options = {}) {
559
573
  if (!validManagedPlugin(name)) continue
560
574
  const file = path.join(configDir, "plugins", name)
561
575
  const current = await readText(file)
562
- if (current.includes(MANAGED_MARKER)) {
576
+ if (hasManagedMarker(current)) {
563
577
  await rm(path.dirname(file), { recursive: true, force: true })
564
578
  removed.plugins += 1
565
579
  }
@@ -2,6 +2,7 @@ import { existsSync } from "node:fs"
2
2
  import { mkdir, readFile, readdir, writeFile } from "node:fs/promises"
3
3
  import { createHash } from "node:crypto"
4
4
  import path from "node:path"
5
+ import { pairedBenchmarkConfidence } from "./benchmark-confidence.mjs"
5
6
 
6
7
  const LEARNING_DIR = ".ues-learning"
7
8
  const STATE_FILE = "LEARNINGS.json"
@@ -26,8 +27,46 @@ async function walkJson(dir, limit = 500) {
26
27
  return files
27
28
  }
28
29
 
30
+ function failureReasons(item) {
31
+ const reasons = []
32
+ if (item.timedOut) reasons.push("hard-timeout")
33
+ if (item.idleTimedOut) reasons.push("idle-timeout")
34
+ if (Number(item.agentExit) !== 0) reasons.push("agent-exit")
35
+ if (Number(item.graderExit) !== 0) reasons.push("grader-failure")
36
+ if (item.orchestration?.required && !item.orchestration?.valid) reasons.push("orchestration-failure")
37
+ if ((item.telemetry?.parseErrors || 0) > 0) reasons.push("telemetry-parse-errors")
38
+ return reasons
39
+ }
40
+
41
+ function recommendationFor(key) {
42
+ const recommendations = {
43
+ "hard-timeout": "Split oversized execution units or justify a larger bounded timeout; preserve cancellation and stale-run recovery.",
44
+ "idle-timeout": "Inspect tool/session deadlocks and add observable progress boundaries or recovery paths.",
45
+ "agent-exit": "Classify provider/process failures before retrying; avoid blind model escalation.",
46
+ "grader-failure": "Mine the failed contract for missing context, routing or verification guidance and add a regression before promotion.",
47
+ "orchestration-failure": "Strengthen durable workflow gates/runtime support; direct patches must not count as long-horizon success.",
48
+ "telemetry-parse-errors": "Update telemetry parsing against the observed event shape with fixture-backed tests.",
49
+ }
50
+ return recommendations[key] || "Add a regression, test the candidate rule in shadow evaluation, and promote only on measured improvement."
51
+ }
52
+
29
53
  function proposal(type, key, title, evidence, recommendation) {
30
- return { id: idFor(type, key), type, key, title, evidence, recommendation, status: "proposed" }
54
+ return {
55
+ id: idFor(type, key),
56
+ type,
57
+ key,
58
+ title,
59
+ evidence,
60
+ recommendation,
61
+ candidateRule: recommendation,
62
+ shadowRequired: true,
63
+ status: "proposed",
64
+ }
65
+ }
66
+
67
+ function clampRate(value) {
68
+ const number = Number(value)
69
+ return Number.isFinite(number) && number >= 0 && number <= 1 ? number : null
31
70
  }
32
71
 
33
72
  export async function analyzeEvalTraces(evalDir) {
@@ -36,60 +75,91 @@ export async function analyzeEvalTraces(evalDir) {
36
75
  for (const file of files) {
37
76
  try {
38
77
  const payload = JSON.parse(await readFile(file, "utf8"))
39
- for (const item of payload.results || []) results.push({ ...item, source: file })
78
+ for (const item of payload.results || []) {
79
+ results.push({ ...item, source: file })
80
+ }
40
81
  } catch {}
41
82
  }
42
83
 
43
- const counters = new Map()
44
- const bump = (key, item) => {
45
- const value = counters.get(key) || { count: 0, tasks: new Set(), modes: new Set() }
84
+ const reasonCounters = new Map()
85
+ const clusters = new Map()
86
+
87
+ const bump = (map, key, item) => {
88
+ const value = map.get(key) || {
89
+ count: 0,
90
+ tasks: new Set(),
91
+ modes: new Set(),
92
+ suites: new Set(),
93
+ sources: new Set(),
94
+ }
46
95
  value.count += 1
47
96
  if (item.task) value.tasks.add(item.task)
48
97
  if (item.mode) value.modes.add(item.mode)
49
- counters.set(key, value)
98
+ if (item.suite) value.suites.add(item.suite)
99
+ if (item.source) value.sources.add(item.source)
100
+ map.set(key, value)
50
101
  }
51
102
 
52
103
  for (const item of results) {
53
- if (item.timedOut) bump("hard-timeout", item)
54
- if (item.idleTimedOut) bump("idle-timeout", item)
55
- if (Number(item.agentExit) !== 0) bump("agent-exit", item)
56
- if (Number(item.graderExit) !== 0) bump("grader-failure", item)
57
- if (item.orchestration?.required && !item.orchestration?.valid) bump("orchestration-failure", item)
58
- if ((item.telemetry?.parseErrors || 0) > 0) bump("telemetry-parse-errors", item)
104
+ const reasons = failureReasons(item)
105
+ for (const reason of reasons) bump(reasonCounters, reason, item)
106
+ if (reasons.length) {
107
+ const clusterKey = reasons.sort().join("+")
108
+ bump(clusters, clusterKey, item)
109
+ }
59
110
  }
60
111
 
61
112
  const proposals = []
62
- for (const [key, value] of counters) {
63
- if (value.count < 1) continue
64
- const details = {
113
+ for (const [key, value] of reasonCounters) {
114
+ const evidence = {
65
115
  count: value.count,
66
116
  tasks: [...value.tasks].sort(),
67
117
  modes: [...value.modes].sort(),
118
+ sources: [...value.sources].sort().slice(0, 20),
68
119
  samples: results.length,
69
- }
70
- const recommendations = {
71
- "hard-timeout": "Inspect task decomposition and model/runtime latency; shorten tasks or raise a justified hard timeout.",
72
- "idle-timeout": "Inspect tool/session deadlocks and add progress-producing boundaries or recovery.",
73
- "agent-exit": "Classify process/provider failures before retrying; avoid blind model escalation.",
74
- "grader-failure": "Mine failing task contracts for missing domain guidance or context selection, then add a regression before changing skills.",
75
- "orchestration-failure": "Strengthen durable workflow instructions/gates or runtime support; do not count direct patches as UES success.",
76
- "telemetry-parse-errors": "Update telemetry parsing against the observed OpenCode event shape with a fixture-backed test.",
120
+ confidence: results.length ? value.count / results.length : 0,
77
121
  }
78
122
  proposals.push(proposal(
79
123
  "eval-pattern",
80
124
  key,
81
125
  "Recurring evaluation pattern: " + key,
82
- details,
83
- recommendations[key] || "Review the repeated pattern and add a regression before promoting a reusable lesson.",
126
+ evidence,
127
+ recommendationFor(key),
128
+ ))
129
+ }
130
+
131
+ for (const [key, value] of clusters) {
132
+ if (value.count < 2) continue
133
+ const primary = key.split("+")[0]
134
+ const evidence = {
135
+ count: value.count,
136
+ tasks: [...value.tasks].sort(),
137
+ modes: [...value.modes].sort(),
138
+ sources: [...value.sources].sort().slice(0, 20),
139
+ samples: results.length,
140
+ confidence: results.length ? value.count / results.length : 0,
141
+ signature: key,
142
+ }
143
+ proposals.push(proposal(
144
+ "root-cause-cluster",
145
+ key,
146
+ "Repeated failure cluster: " + key,
147
+ evidence,
148
+ recommendationFor(primary),
84
149
  ))
85
150
  }
86
151
 
87
152
  return {
88
- schemaVersion: 1,
153
+ schemaVersion: 2,
89
154
  analyzedAt: new Date().toISOString(),
90
155
  files: files.length,
91
156
  results: results.length,
92
- proposals: proposals.sort((a, b) => b.evidence.count - a.evidence.count || a.key.localeCompare(b.key)),
157
+ clusters: [...clusters.entries()]
158
+ .map(([key, value]) => ({ key, count: value.count, tasks: [...value.tasks].sort() }))
159
+ .sort((a, b) => b.count - a.count || a.key.localeCompare(b.key)),
160
+ proposals: proposals.sort((a, b) =>
161
+ b.evidence.count - a.evidence.count || a.key.localeCompare(b.key),
162
+ ),
93
163
  }
94
164
  }
95
165
 
@@ -99,17 +169,19 @@ export function learningFile(root) {
99
169
 
100
170
  export async function readLearningState(root) {
101
171
  const file = learningFile(root)
102
- if (!existsSync(file)) return { schemaVersion: 1, updatedAt: null, proposals: [], accepted: [] }
172
+ if (!existsSync(file)) {
173
+ return { schemaVersion: 2, updatedAt: null, proposals: [], accepted: [] }
174
+ }
103
175
  try {
104
176
  const parsed = JSON.parse(await readFile(file, "utf8"))
105
177
  return {
106
- schemaVersion: 1,
178
+ schemaVersion: 2,
107
179
  updatedAt: parsed.updatedAt || null,
108
180
  proposals: Array.isArray(parsed.proposals) ? parsed.proposals : [],
109
181
  accepted: Array.isArray(parsed.accepted) ? parsed.accepted : [],
110
182
  }
111
183
  } catch {
112
- return { schemaVersion: 1, updatedAt: null, proposals: [], accepted: [], invalid: true }
184
+ return { schemaVersion: 2, updatedAt: null, proposals: [], accepted: [], invalid: true }
113
185
  }
114
186
  }
115
187
 
@@ -118,9 +190,16 @@ export async function saveLearningAnalysis(root, analysis) {
118
190
  await mkdir(dir, { recursive: true })
119
191
  const current = await readLearningState(root)
120
192
  const acceptedIDs = new Set(current.accepted.map((item) => item.id))
121
- const proposals = analysis.proposals.filter((item) => !acceptedIDs.has(item.id))
193
+ const existing = new Map(current.proposals.map((item) => [item.id, item]))
194
+ const proposals = analysis.proposals
195
+ .filter((item) => !acceptedIDs.has(item.id))
196
+ .map((item) => ({
197
+ ...item,
198
+ firstProposedAt: existing.get(item.id)?.firstProposedAt || new Date().toISOString(),
199
+ lastAnalyzedAt: analysis.analyzedAt,
200
+ }))
122
201
  const next = {
123
- schemaVersion: 1,
202
+ schemaVersion: 2,
124
203
  updatedAt: new Date().toISOString(),
125
204
  proposals,
126
205
  accepted: current.accepted,
@@ -129,13 +208,19 @@ export async function saveLearningAnalysis(root, analysis) {
129
208
  return next
130
209
  }
131
210
 
211
+ // Backward-compatible manual acceptance. Shadow-required proposals remain excluded
212
+ // from retrieval until they are promoted with measured benchmark improvement.
132
213
  export async function acceptLearning(root, id) {
133
214
  const current = await readLearningState(root)
134
215
  const item = current.proposals.find((entry) => entry.id === id)
135
216
  if (!item) throw new Error("unknown learning proposal: " + id)
136
- const accepted = { ...item, status: "accepted", acceptedAt: new Date().toISOString() }
217
+ const accepted = {
218
+ ...item,
219
+ status: item.shadowRequired ? "accepted-awaiting-shadow" : "accepted",
220
+ acceptedAt: new Date().toISOString(),
221
+ }
137
222
  const next = {
138
- schemaVersion: 1,
223
+ schemaVersion: 2,
139
224
  updatedAt: accepted.acceptedAt,
140
225
  proposals: current.proposals.filter((entry) => entry.id !== id),
141
226
  accepted: [...current.accepted.filter((entry) => entry.id !== id), accepted],
@@ -145,14 +230,127 @@ export async function acceptLearning(root, id) {
145
230
  return accepted
146
231
  }
147
232
 
233
+ async function benchmarkValidationFromArtifact(root, reportPath) {
234
+ const value = String(reportPath || "").trim()
235
+ if (!value) {
236
+ throw new Error("learning promotion requires --report pointing to a benchmark matrix artifact")
237
+ }
238
+
239
+ root = path.resolve(root)
240
+ const file = path.isAbsolute(value) ? value : path.resolve(root, value)
241
+ const raw = await readFile(file, "utf8").catch(() => null)
242
+ if (!raw) throw new Error("learning promotion benchmark report could not be read: " + value)
243
+
244
+ let report
245
+ try {
246
+ report = JSON.parse(raw)
247
+ } catch {
248
+ throw new Error("learning promotion benchmark report is not valid JSON: " + value)
249
+ }
250
+
251
+ if (report.kind !== "ues-benchmark-matrix" || report.schemaVersion !== 1) {
252
+ throw new Error("learning promotion requires a ues-benchmark-matrix schemaVersion 1 artifact")
253
+ }
254
+ if (report.coverageComplete !== true) {
255
+ throw new Error("learning promotion requires a benchmark artifact with complete baseline/UES coverage")
256
+ }
257
+
258
+ const baseline = report.summary?.modes?.baseline
259
+ const candidate = report.summary?.modes?.ues
260
+ const baselinePassRate = clampRate(baseline?.passRate)
261
+ const candidatePassRate = clampRate(candidate?.passRate)
262
+ const baselineTotal = Number.parseInt(String(baseline?.total ?? "0"), 10)
263
+ const candidateTotal = Number.parseInt(String(candidate?.total ?? "0"), 10)
264
+ const samples = Math.min(baselineTotal, candidateTotal)
265
+
266
+ if (baselinePassRate === null || candidatePassRate === null || samples < 1) {
267
+ throw new Error("learning promotion benchmark artifact is missing valid pass-rate/sample evidence")
268
+ }
269
+ if (baselineTotal !== candidateTotal) {
270
+ throw new Error("learning promotion requires equal baseline and UES sample counts")
271
+ }
272
+ if (candidatePassRate <= baselinePassRate) {
273
+ throw new Error("learning promotion requires measured shadow benchmark improvement")
274
+ }
275
+ if (!Array.isArray(report.pairedResults) || report.pairedResults.length < 2) {
276
+ throw new Error("learning promotion requires embedded paired baseline/UES benchmark evidence")
277
+ }
278
+ const confidence = pairedBenchmarkConfidence(report.pairedResults)
279
+ if (confidence.promotionEligible !== true) {
280
+ throw new Error("learning promotion requires statistically supported improvement without suite regression")
281
+ }
282
+
283
+ return {
284
+ baselinePassRate,
285
+ candidatePassRate,
286
+ delta: candidatePassRate - baselinePassRate,
287
+ samples,
288
+ report: path.relative(root, file).replaceAll("\\", "/"),
289
+ reportHash: createHash("sha256").update(raw).digest("hex"),
290
+ model: report.model || null,
291
+ suites: Array.isArray(report.suites) ? report.suites : [],
292
+ finishedAt: report.finishedAt || null,
293
+ confidence,
294
+ }
295
+ }
296
+
297
+ export async function promoteLearning(root, id, validation = {}) {
298
+ const current = await readLearningState(root)
299
+ const item = current.accepted.find((entry) => entry.id === id)
300
+ if (!item) {
301
+ if (current.proposals.some((entry) => entry.id === id)) {
302
+ throw new Error("learning proposal must be explicitly accepted before shadow promotion: " + id)
303
+ }
304
+ throw new Error("unknown learning proposal: " + id)
305
+ }
306
+
307
+ const shadowValidation = await benchmarkValidationFromArtifact(root, validation.report)
308
+ const promotedAt = new Date().toISOString()
309
+ const promoted = {
310
+ ...item,
311
+ status: "promoted",
312
+ promotedAt,
313
+ shadowValidation,
314
+ }
315
+ const next = {
316
+ schemaVersion: 2,
317
+ updatedAt: promotedAt,
318
+ proposals: current.proposals.filter((entry) => entry.id !== id),
319
+ accepted: [...current.accepted.filter((entry) => entry.id !== id), promoted],
320
+ }
321
+ await mkdir(path.dirname(learningFile(root)), { recursive: true })
322
+ await writeFile(learningFile(root), JSON.stringify(next, null, 2) + "\n", "utf8")
323
+ return promoted
324
+ }
325
+
326
+ function retrievalTerms(text) {
327
+ return [...new Set(
328
+ String(text || "")
329
+ .toLowerCase()
330
+ .split(/[^\p{L}\p{N}_-]+/u)
331
+ .filter((term) => term.length >= 4),
332
+ )]
333
+ }
334
+
148
335
  export async function relevantAcceptedLearnings(root, text, limit = 5) {
149
336
  const state = await readLearningState(root)
150
- const terms = new Set(String(text || "").toLowerCase().split(/[^a-z0-9_-]+/).filter((term) => term.length >= 4))
337
+ const terms = retrievalTerms(text)
151
338
  return state.accepted
339
+ .filter((item) => !item.shadowRequired || item.status === "promoted" || item.shadowValidation?.delta > 0)
152
340
  .map((item) => {
153
- const haystack = JSON.stringify(item).toLowerCase()
154
- const score = [...terms].reduce((sum, term) => sum + (haystack.includes(term) ? 1 : 0), 0)
155
- return { item, score }
341
+ const title = String(item.title || "").toLowerCase()
342
+ const rule = String(item.candidateRule || item.recommendation || "").toLowerCase()
343
+ const tasks = JSON.stringify(item.evidence?.tasks || []).toLowerCase()
344
+ const score = terms.reduce((sum, term) => {
345
+ if (title.includes(term)) sum += 4
346
+ if (rule.includes(term)) sum += 3
347
+ if (tasks.includes(term)) sum += 2
348
+ return sum
349
+ }, 0)
350
+ const validationBoost = item.shadowValidation?.delta > 0
351
+ ? Math.min(5, item.shadowValidation.delta * 20)
352
+ : 0
353
+ return { item, score: score + validationBoost }
156
354
  })
157
355
  .filter((entry) => entry.score > 0)
158
356
  .sort((a, b) => b.score - a.score)
@@ -58,11 +58,17 @@ export function resolveAdaptiveModel(role, attempt, taskPolicy = {}, config = {}
58
58
  ...config,
59
59
  roleTiers: { ...(config.roleTiers || {}), [role]: baseTier },
60
60
  })
61
+ const normalizedAttempt = Math.max(1, Number(attempt || 1))
61
62
  return {
62
63
  ...resolved,
64
+ recoveryStage:
65
+ normalizedAttempt <= 1 ? "initial" :
66
+ normalizedAttempt === 2 ? "diagnose" :
67
+ "deep-recovery",
63
68
  policy: {
64
69
  mode: taskPolicy.mode || null,
65
70
  risk: taskPolicy.risk || null,
71
+ executionProfile: taskPolicy.executionProfile || null,
66
72
  score: Number(taskPolicy.score || 0),
67
73
  maxAttempts: Number(taskPolicy.maxAttempts || 0) || null,
68
74
  contextBudget: Number(taskPolicy.contextBudget || 0) || null,