opencode-agent-skill 7.7.0 → 10.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/CHANGELOG.md +112 -3
  2. package/README.md +396 -281
  3. package/bin/ocskill.mjs +382 -156
  4. package/docs/DETERMINISTIC-TOOLS.md +25 -8
  5. package/docs/ENGINEERING-DESIGN.md +31 -13
  6. package/docs/EVALS.md +34 -12
  7. package/docs/NPM-PUBLISH.md +6 -6
  8. package/docs/OPENCODE-COMPAT.md +11 -8
  9. package/docs/TRACE-SCHEMA.md +15 -2
  10. package/docs/V8-INTELLIGENCE-RELIABILITY.md +206 -0
  11. package/docs/V9-SPEED-INTELLIGENCE.md +102 -0
  12. package/evals/live/tasks.json +6 -6
  13. package/evals/polyglot/fixtures/polyglot-bench/api/generated/client.ts +2 -0
  14. package/evals/polyglot/fixtures/polyglot-bench/api/openapi.json +25 -0
  15. package/evals/polyglot/fixtures/polyglot-bench/db/migrations/20260920_add_order_key.sql +1 -0
  16. package/evals/polyglot/fixtures/polyglot-bench/dotnet/OrderService.cs +8 -0
  17. package/evals/polyglot/fixtures/polyglot-bench/java/PriceService.java +5 -0
  18. package/evals/polyglot/fixtures/polyglot-bench/monorepo/package.json +6 -0
  19. package/evals/polyglot/fixtures/polyglot-bench/monorepo/packages/api/package.json +4 -0
  20. package/evals/polyglot/fixtures/polyglot-bench/monorepo/packages/web/package.json +7 -0
  21. package/evals/polyglot/fixtures/polyglot-bench/monorepo/pnpm-lock.yaml +5 -0
  22. package/evals/polyglot/fixtures/polyglot-bench/next/app/api/products/route.ts +7 -0
  23. package/evals/polyglot/fixtures/polyglot-bench/python/tenant_auth.py +4 -0
  24. package/evals/polyglot/fixtures/polyglot-bench/react-native/keyboard.ts +3 -0
  25. package/evals/polyglot/graders/polyglot-bench.mjs +101 -0
  26. package/evals/polyglot/tasks.json +54 -0
  27. package/global-config/AGENTS.md +78 -160
  28. package/global-config/agents/integration-verifier.md +1 -1
  29. package/global-config/agents/plan-checker.md +1 -1
  30. package/global-config/commands/run.md +9 -5
  31. package/global-config/plugins/ues-router/capabilities.js +4 -0
  32. package/global-config/plugins/ues-router/index.js +784 -37
  33. package/global-config/plugins/ues-router/router.js +175 -23
  34. package/global-config/plugins/ues-router/runtime-guard.js +265 -0
  35. package/global-config/skills/engineering-orchestrator/references/long-horizon.md +6 -4
  36. package/lib/aci.mjs +128 -0
  37. package/lib/benchmark-confidence.mjs +173 -0
  38. package/lib/cli-utils.mjs +41 -0
  39. package/lib/container-sandbox.mjs +102 -0
  40. package/lib/context-manifest.mjs +300 -22
  41. package/lib/control-center.mjs +36 -3
  42. package/lib/eval-ablation.mjs +104 -0
  43. package/lib/eval-order.mjs +9 -0
  44. package/lib/eval-report.mjs +11 -0
  45. package/lib/eval-telemetry.mjs +8 -2
  46. package/lib/gate-receipt.mjs +52 -0
  47. package/lib/installer.mjs +39 -25
  48. package/lib/learning-engine.mjs +236 -38
  49. package/lib/model-policy.mjs +6 -0
  50. package/lib/opencode-compat.mjs +25 -10
  51. package/lib/orchestrator-policy.mjs +195 -21
  52. package/lib/process-runner.mjs +30 -9
  53. package/lib/runtime-events.mjs +31 -0
  54. package/lib/semantic-index.mjs +318 -0
  55. package/lib/task-engine.mjs +557 -28
  56. package/lib/trajectory.mjs +89 -0
  57. package/lib/windows-shim.mjs +227 -0
  58. package/lib/worktree-sandbox.mjs +85 -3
  59. package/package.json +11 -4
  60. package/scripts/check-release-tag.mjs +22 -0
  61. package/scripts/control-center.mjs +25 -0
  62. package/scripts/eval-ablation.mjs +44 -0
  63. package/scripts/eval-live.mjs +39 -58
  64. package/scripts/eval-matrix.mjs +166 -0
  65. package/scripts/smoke-packed-install.mjs +138 -4
  66. package/scripts/smoke-plain-install.mjs +91 -0
  67. package/scripts/validate-live-suite.mjs +3 -3
  68. package/scripts/validate.mjs +27 -5
@@ -1,24 +1,39 @@
1
1
  import { spawnSync } from "node:child_process"
2
+ import { resolveWindowsCommand } from "./windows-shim.mjs"
2
3
 
3
4
  export function parseOpenCodeMajor(value) {
4
5
  const match = String(value || "").match(/(\d+)\.(\d+)\.(\d+)/)
5
6
  return match ? Number(match[1]) : null
6
7
  }
7
8
 
9
+
10
+ function runOpenCodeVersion() {
11
+ if (process.platform !== "win32") {
12
+ return spawnSync("opencode", ["--version"], { encoding: "utf8" })
13
+ }
14
+
15
+ const resolved = resolveWindowsCommand("opencode")
16
+ if (!resolved) {
17
+ return {
18
+ status: 127,
19
+ stdout: "",
20
+ stderr: "no safely executable OpenCode command found",
21
+ }
22
+ }
23
+
24
+ return spawnSync(
25
+ resolved.executable,
26
+ [...resolved.argsPrefix, "--version"],
27
+ { encoding: "utf8" },
28
+ )
29
+ }
30
+
8
31
  export function detectOpenCodeMajor() {
9
32
  const forced = process.env.UES_OPENCODE_MAJOR
10
33
  if (forced && /^\d+$/.test(forced)) return Number(forced)
11
34
 
12
- const candidates = process.platform === "win32"
13
- ? [
14
- () => spawnSync(process.env.ComSpec || "cmd.exe", ["/d", "/s", "/c", "opencode --version"], { encoding: "utf8" }),
15
- () => spawnSync("opencode", ["--version"], { encoding: "utf8" }),
16
- ]
17
- : [() => spawnSync("opencode", ["--version"], { encoding: "utf8" })]
18
-
19
- for (const run of candidates) {
20
- const result = run()
21
- if (result.status !== 0) continue
35
+ const result = runOpenCodeVersion()
36
+ if (result.status === 0) {
22
37
  const major = parseOpenCodeMajor(result.stdout || result.stderr)
23
38
  if (major !== null) return major
24
39
  }
@@ -1,35 +1,209 @@
1
- const HIGH_RISK = /(auth|security|permission|payment|migration|schema|database|production|deploy|public api|breaking|secret|credential)/i
2
- const LONG = /(whole repo|whole repository|entire repo|entire project|long[- ]running|multi[- ]file|cross[- ]module|resume|migration|refactor all)/i
1
+ const HIGH_RISK = /(auth|security|permission|payment|migration|schema|database|production|deploy|public api|breaking|secret|credential|phân quyền|bảo mật|thanh toán|cơ sở dữ liệu|triển khai|migrate|migration)/i
2
+ const LONG = /(whole repo|whole repository|whole project|entire repo|entire project|large refactor|major refactor|long[- ](?:running|horizon)|durable state|dependency graph|integration verification|resume|migration|refactor all|toàn bộ repo|toàn bộ repository|toàn bộ dự án|toàn bộ project|refactor lớn|tác vụ dài|nhiều file|nhiều module|tiếp tục công việc|refactor toàn bộ|xác minh tích hợp|kiểm tra tích hợp|chia (?:công việc|task|tác vụ).*(?:dependency|phụ thuộc))/i
3
+ const DEBUG = /(fix|bug|debug|crash|regression|failure|error|broken|sửa lỗi|lỗi|điều tra lỗi|không chạy)/i
4
+ const CONTRACT = /(public api|api contract|openapi|response schema|request schema|breaking api|hợp đồng api|api công khai)/i
5
+ const DATA = /(database|sql|migration|schema|transaction|index|cơ sở dữ liệu|dữ liệu|migrate)/i
6
+
7
+ function signal(name, matched, weight) {
8
+ return matched ? { name, weight } : null
9
+ }
10
+
11
+ function profileFor(mode, risk) {
12
+ if (mode === "inline" && risk === "low") {
13
+ return {
14
+ name: "fast",
15
+ maxSkills: 2,
16
+ contextBudget: 8_000,
17
+ contextStrategy: "incremental-semantic",
18
+ skillLoading: "direct-only",
19
+ durableState: false,
20
+ worktree: "off",
21
+ critic: "off",
22
+ verification: "targeted",
23
+ fullCI: false,
24
+ containerVerification: "off",
25
+ }
26
+ }
27
+ if (mode === "long-horizon" || risk === "high") {
28
+ return {
29
+ name: "deep",
30
+ maxSkills: 5,
31
+ contextBudget: 48_000,
32
+ contextStrategy: "semantic+graph+git",
33
+ skillLoading: "orchestrated",
34
+ durableState: true,
35
+ worktree: "auto-writers",
36
+ critic: "required-for-high-risk-or-final",
37
+ verification: "targeted+integration",
38
+ fullCI: true,
39
+ containerVerification: risk === "high" ? "preferred-if-capable" : "optional",
40
+ }
41
+ }
42
+ return {
43
+ name: "standard",
44
+ maxSkills: 4,
45
+ contextBudget: 20_000,
46
+ contextStrategy: "incremental-semantic+git",
47
+ skillLoading: "selective",
48
+ durableState: false,
49
+ worktree: "auto-on-conflict",
50
+ critic: "on-failure-or-elevated-risk",
51
+ verification: "targeted+affected",
52
+ fullCI: false,
53
+ containerVerification: "off",
54
+ }
55
+ }
56
+
57
+ function boundedInt(value, fallback, min, max) {
58
+ const parsed = Number(value)
59
+ if (!Number.isFinite(parsed)) return fallback
60
+ return Math.min(max, Math.max(min, Math.round(parsed)))
61
+ }
62
+
63
+ export function recoveryPolicyForAttempt(taskPolicy = {}, attempt = 1) {
64
+ const normalizedAttempt = boundedInt(attempt, 1, 1, 99)
65
+ const baseBudget = boundedInt(
66
+ taskPolicy.contextBudget ?? taskPolicy.profile?.contextBudget,
67
+ 20_000,
68
+ 4_000,
69
+ 48_000,
70
+ )
71
+ const baseSkills = boundedInt(
72
+ taskPolicy.maxSkills ?? taskPolicy.profile?.maxSkills,
73
+ 4,
74
+ 1,
75
+ 5,
76
+ )
77
+ const baseStrategy = taskPolicy.profile?.contextStrategy || "incremental-semantic+git"
78
+
79
+ if (normalizedAttempt <= 1) {
80
+ return {
81
+ schemaVersion: 1,
82
+ stage: "initial",
83
+ attempt: normalizedAttempt,
84
+ contextBudget: baseBudget,
85
+ maxSkills: baseSkills,
86
+ contextStrategy: baseStrategy,
87
+ requireDiagnosis: false,
88
+ requireCritic: false,
89
+ modelEscalation: false,
90
+ directives: [],
91
+ }
92
+ }
93
+
94
+ if (normalizedAttempt === 2) {
95
+ return {
96
+ schemaVersion: 1,
97
+ stage: "diagnose",
98
+ attempt: normalizedAttempt,
99
+ contextBudget: Math.min(48_000, Math.max(baseBudget, Math.round(baseBudget * 1.35))),
100
+ maxSkills: Math.min(5, baseSkills + 1),
101
+ contextStrategy:
102
+ taskPolicy.executionProfile === "fast"
103
+ ? "incremental-semantic+git"
104
+ : baseStrategy,
105
+ requireDiagnosis: true,
106
+ requireCritic: false,
107
+ modelEscalation: true,
108
+ directives: [
109
+ "reproduce or capture the exact previous failure before editing",
110
+ "inspect the direct caller, nearest test and failure-adjacent evidence",
111
+ "do not stack another speculative patch on top of the failed attempt",
112
+ ],
113
+ }
114
+ }
115
+
116
+ return {
117
+ schemaVersion: 1,
118
+ stage: "deep-recovery",
119
+ attempt: normalizedAttempt,
120
+ contextBudget: Math.min(48_000, Math.max(20_000, Math.round(baseBudget * 1.75))),
121
+ maxSkills: Math.min(5, baseSkills + 2),
122
+ contextStrategy: "semantic+graph+git",
123
+ requireDiagnosis: true,
124
+ requireCritic: true,
125
+ modelEscalation: true,
126
+ directives: [
127
+ "re-investigate from fresh evidence and explicitly reject the failed hypothesis",
128
+ "expand to callers, dependencies, tests and boundary contracts before editing",
129
+ "challenge the architecture or coupling if repeated fixes expose a wider problem",
130
+ "run an independent critic or review pass before accepting the recovery",
131
+ ],
132
+ }
133
+ }
3
134
 
4
135
  export function classifyEngineeringTask(text, facts = {}) {
5
136
  const value = String(text || "")
6
- let score = 0
7
- if (value.length > 250) score += 1
8
- if (value.length > 700) score += 1
9
- if (HIGH_RISK.test(value)) score += 2
10
- if (LONG.test(value)) score += 2
11
- if (Number(facts.changedFiles || 0) > 5) score += 1
12
- if (Number(facts.changedFiles || 0) > 12) score += 1
13
- if (facts.hasMigration || facts.hasPublicContract) score += 2
14
-
15
- const risk = HIGH_RISK.test(value) || facts.hasMigration || facts.hasPublicContract
16
- ? "high"
17
- : score >= 3 ? "medium" : "low"
18
-
19
- const mode = score >= 4 ? "long-horizon" : score >= 2 ? "standard" : "inline"
20
- const modelTier = risk === "high" || score >= 4 ? "heavy" : score >= 2 ? "standard" : "light"
137
+ const declaredHighRisk =
138
+ ["high", "critical"].includes(String(facts.risk || "").toLowerCase()) ||
139
+ /\b(?:risk|rủi ro)\s*[:=\/-]?\s*(?:high|critical|cao|nghiêm trọng)\b/i.test(value) ||
140
+ /\b(?:high|critical)[-\s]?(?:risk|rủi ro)\b/i.test(value)
141
+ const declaredLongHorizon =
142
+ facts.longHorizon === true ||
143
+ ["long", "long-horizon", "deep"].includes(String(facts.mode || "").toLowerCase())
144
+ const compoundLongRisk =
145
+ /\blong\s*[/|,+]\s*(?:high|critical)[-\s]?risk\b/i.test(value) ||
146
+ /\b(?:high|critical)[-\s]?risk\s*[/|,+]\s*long\b/i.test(value)
147
+ const explicitLongHorizon = declaredLongHorizon || compoundLongRisk || LONG.test(value)
148
+ const signals = [
149
+ signal("long-request-text", value.length > 700, 1),
150
+ signal("medium-request-text", value.length > 250, 1),
151
+ signal("high-risk-domain", HIGH_RISK.test(value), 2),
152
+ signal("declared-high-risk", declaredHighRisk, 2),
153
+ signal("explicit-long-horizon", explicitLongHorizon, 2),
154
+ signal("debugging", DEBUG.test(value), 1),
155
+ signal("public-contract", CONTRACT.test(value) || facts.hasPublicContract, 2),
156
+ signal("data-migration", (DATA.test(value) && /migration|schema|migrate|di trú|chuyển đổi/i.test(value)) || facts.hasMigration, 2),
157
+ signal("many-changed-files", Number(facts.changedFiles || 0) > 5, 1),
158
+ signal("very-many-changed-files", Number(facts.changedFiles || 0) > 12, 1),
159
+ signal("large-repository", Number(facts.repoFiles || 0) > 1500, 1),
160
+ signal("monorepo", facts.monorepo === true, 1),
161
+ ].filter(Boolean)
162
+
163
+ const score = signals.reduce((sum, item) => sum + item.weight, 0)
164
+ const highRisk = declaredHighRisk || HIGH_RISK.test(value) || Boolean(facts.hasMigration) || Boolean(facts.hasPublicContract)
165
+ const risk = highRisk ? "high" : score >= 3 ? "medium" : "low"
166
+ const mode = explicitLongHorizon || score >= 4 ? "long-horizon" : score >= 2 ? "standard" : "inline"
167
+ const modelTier = risk === "high" || mode === "long-horizon" ? "heavy" : score >= 2 ? "standard" : "light"
21
168
  const maxAttempts = risk === "high" ? 2 : 3
22
- const contextBudget = mode === "long-horizon" ? 48_000 : mode === "standard" ? 32_000 : 16_000
169
+ const profile = profileFor(mode, risk)
170
+
171
+ const domains = []
172
+ if (/(auth|permission|tenant|token|session|phân quyền|xác thực)/i.test(value)) domains.push("auth-security")
173
+ if (/(payment|checkout|refund|webhook|thanh toán|hoàn tiền)/i.test(value)) domains.push("payment")
174
+ if (DATA.test(value)) domains.push("database")
175
+ if (CONTRACT.test(value)) domains.push("api-contract")
176
+ if (/(react native|expo|android|ios|gradle|xcode)/i.test(value)) domains.push("react-native")
177
+ else if (/(next\.js|nextjs|app router|server component)/i.test(value)) domains.push("nextjs")
178
+ else if (/\breact\b|useeffect|usestate|component/i.test(value)) domains.push("react")
179
+ if (/(docker|kubernetes|terraform|github actions|ci\/cd|deploy|triển khai)/i.test(value)) domains.push("devops")
23
180
 
24
181
  return {
25
- schemaVersion: 1,
182
+ schemaVersion: 4,
26
183
  score,
184
+ signals,
27
185
  risk,
28
186
  mode,
187
+ executionProfile: profile.name,
29
188
  modelTier,
30
189
  maxAttempts,
31
- contextBudget,
32
- requirePlanCheck: mode === "long-horizon" || risk === "high",
190
+ contextBudget: profile.contextBudget,
191
+ maxSkills: profile.maxSkills,
192
+ requirePlanCheck: profile.durableState || risk === "high",
33
193
  requireIntegrationVerification: mode !== "inline" || risk === "high",
194
+ requireFreshEvidence: true,
195
+ profile,
196
+ domains: [...new Set(domains)],
197
+ recovery: {
198
+ escalateAfterFailure: true,
199
+ diagnosisBeforePatch: true,
200
+ deepRecoveryFromAttempt: 3,
201
+ },
202
+ antiHallucination: {
203
+ evidenceFirst: true,
204
+ noCompletionWithoutVerification: mode !== "inline" || risk === "high",
205
+ noUnsupportedSemanticClaims: true,
206
+ failClosedOnMissingCapability: risk === "high",
207
+ },
34
208
  }
35
209
  }
@@ -6,16 +6,17 @@ function appendBounded(current, chunk, limit) {
6
6
  return next.slice(next.length - limit)
7
7
  }
8
8
 
9
- function killTree(child) {
9
+ function killTree(child, force = false) {
10
10
  if (!child?.pid) return
11
11
  if (process.platform === "win32") {
12
12
  spawnSync("taskkill", ["/PID", String(child.pid), "/T", "/F"], { stdio: "ignore" })
13
13
  return
14
14
  }
15
+ const signal = force ? "SIGKILL" : "SIGTERM"
15
16
  try {
16
- process.kill(-child.pid, "SIGTERM")
17
+ process.kill(-child.pid, signal)
17
18
  } catch {
18
- try { child.kill("SIGTERM") } catch {}
19
+ try { child.kill(signal) } catch {}
19
20
  }
20
21
  }
21
22
 
@@ -34,6 +35,7 @@ export function runProcess(executable, args = [], options = {}) {
34
35
  let idleTimedOut = false
35
36
  let aborted = false
36
37
  let finished = false
38
+ let escalationTimer = null
37
39
 
38
40
  const child = spawn(executable, args, {
39
41
  cwd: options.cwd,
@@ -46,6 +48,7 @@ export function runProcess(executable, args = [], options = {}) {
46
48
  const cleanup = () => {
47
49
  clearInterval(heartbeatTimer)
48
50
  clearInterval(watchdogTimer)
51
+ if (escalationTimer) clearTimeout(escalationTimer)
49
52
  if (options.signal) options.signal.removeEventListener("abort", onAbort)
50
53
  }
51
54
 
@@ -54,7 +57,9 @@ export function runProcess(executable, args = [], options = {}) {
54
57
  finished = true
55
58
  cleanup()
56
59
  resolve({
57
- status: Number.isInteger(status) ? status : (spawnError || signal || timedOut || idleTimedOut || aborted) ? 1 : 0,
60
+ status: (spawnError || signal || timedOut || idleTimedOut || aborted)
61
+ ? (Number.isInteger(status) && status !== 0 ? status : 1)
62
+ : (Number.isInteger(status) ? status : 0),
58
63
  signal: signal || null,
59
64
  stdout,
60
65
  stderr: spawnError ? appendBounded(stderr, String(spawnError.message || spawnError), maxBuffer) : stderr,
@@ -89,24 +94,40 @@ export function runProcess(executable, args = [], options = {}) {
89
94
  }, heartbeatMs)
90
95
  : null
91
96
 
97
+ const requestStop = () => {
98
+ killTree(child)
99
+ if (process.platform !== "win32" && !escalationTimer) {
100
+ const graceMs = Math.max(100, Number(options.killGraceMs ?? 2_000))
101
+ escalationTimer = setTimeout(() => {
102
+ if (!finished) killTree(child, true)
103
+ }, graceMs)
104
+ escalationTimer.unref?.()
105
+ }
106
+ }
107
+
92
108
  const watchdogTimer = (timeoutMs > 0 || idleTimeoutMs > 0)
93
109
  ? setInterval(() => {
94
110
  const now = Date.now()
95
111
  if (timeoutMs > 0 && now - startedAt >= timeoutMs) {
96
- timedOut = true
97
- killTree(child)
112
+ if (!timedOut) {
113
+ timedOut = true
114
+ requestStop()
115
+ }
98
116
  return
99
117
  }
100
118
  if (idleTimeoutMs > 0 && now - lastOutputAt >= idleTimeoutMs) {
101
- idleTimedOut = true
102
- killTree(child)
119
+ if (!idleTimedOut) {
120
+ idleTimedOut = true
121
+ requestStop()
122
+ }
103
123
  }
104
124
  }, 500)
105
125
  : null
106
126
 
107
127
  function onAbort() {
128
+ if (aborted) return
108
129
  aborted = true
109
- killTree(child)
130
+ requestStop()
110
131
  }
111
132
 
112
133
  if (options.signal) {
@@ -0,0 +1,31 @@
1
+ import { randomUUID } from "node:crypto"
2
+ import { appendFile, readFile } from "node:fs/promises"
3
+
4
+ export async function appendRuntimeEvent(file, type, data = {}) {
5
+ const event = {
6
+ schemaVersion: 1,
7
+ id: randomUUID(),
8
+ type: String(type || "unknown"),
9
+ at: new Date().toISOString(),
10
+ ...data,
11
+ }
12
+ await appendFile(file, JSON.stringify(event) + "\n", "utf8")
13
+ return event
14
+ }
15
+
16
+ export async function readRuntimeEvents(file, options = {}) {
17
+ const requested = Number(options.limit)
18
+ const limit = Number.isFinite(requested) && requested > 0
19
+ ? Math.max(1, Math.min(Math.trunc(requested), 5000))
20
+ : 200
21
+ const raw = await readFile(file, "utf8").catch((error) => {
22
+ if (error?.code === "ENOENT") return ""
23
+ throw error
24
+ })
25
+ const lines = raw.split(/\r?\n/).filter(Boolean)
26
+ const events = []
27
+ for (const line of lines.slice(-limit)) {
28
+ try { events.push(JSON.parse(line)) } catch {}
29
+ }
30
+ return events
31
+ }