opencode-agent-skill 9.0.0 → 10.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +49 -0
- package/README.md +696 -675
- package/bin/ocskill.mjs +24 -0
- package/global-config/AGENTS.md +78 -163
- package/global-config/plugins/ues-router/index.js +413 -59
- package/global-config/plugins/ues-router/router.js +35 -0
- package/global-config/plugins/ues-router/runtime-guard.js +265 -0
- package/lib/benchmark-confidence.mjs +49 -11
- package/lib/eval-ablation.mjs +104 -0
- package/lib/eval-report.mjs +11 -0
- package/lib/eval-telemetry.mjs +3 -0
- package/lib/model-policy.mjs +6 -0
- package/lib/orchestrator-policy.mjs +99 -6
- package/lib/task-engine.mjs +146 -9
- package/package.json +4 -3
- package/scripts/eval-ablation.mjs +44 -0
- package/scripts/eval-matrix.mjs +13 -2
|
@@ -13,8 +13,9 @@ function profileFor(mode, risk) {
|
|
|
13
13
|
return {
|
|
14
14
|
name: "fast",
|
|
15
15
|
maxSkills: 2,
|
|
16
|
-
contextBudget:
|
|
16
|
+
contextBudget: 8_000,
|
|
17
17
|
contextStrategy: "incremental-semantic",
|
|
18
|
+
skillLoading: "direct-only",
|
|
18
19
|
durableState: false,
|
|
19
20
|
worktree: "off",
|
|
20
21
|
critic: "off",
|
|
@@ -29,6 +30,7 @@ function profileFor(mode, risk) {
|
|
|
29
30
|
maxSkills: 5,
|
|
30
31
|
contextBudget: 48_000,
|
|
31
32
|
contextStrategy: "semantic+graph+git",
|
|
33
|
+
skillLoading: "orchestrated",
|
|
32
34
|
durableState: true,
|
|
33
35
|
worktree: "auto-writers",
|
|
34
36
|
critic: "required-for-high-risk-or-final",
|
|
@@ -40,8 +42,9 @@ function profileFor(mode, risk) {
|
|
|
40
42
|
return {
|
|
41
43
|
name: "standard",
|
|
42
44
|
maxSkills: 4,
|
|
43
|
-
contextBudget:
|
|
45
|
+
contextBudget: 20_000,
|
|
44
46
|
contextStrategy: "incremental-semantic+git",
|
|
47
|
+
skillLoading: "selective",
|
|
45
48
|
durableState: false,
|
|
46
49
|
worktree: "auto-on-conflict",
|
|
47
50
|
critic: "on-failure-or-elevated-risk",
|
|
@@ -51,12 +54,97 @@ function profileFor(mode, risk) {
|
|
|
51
54
|
}
|
|
52
55
|
}
|
|
53
56
|
|
|
57
|
+
function boundedInt(value, fallback, min, max) {
|
|
58
|
+
const parsed = Number(value)
|
|
59
|
+
if (!Number.isFinite(parsed)) return fallback
|
|
60
|
+
return Math.min(max, Math.max(min, Math.round(parsed)))
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
export function recoveryPolicyForAttempt(taskPolicy = {}, attempt = 1) {
|
|
64
|
+
const normalizedAttempt = boundedInt(attempt, 1, 1, 99)
|
|
65
|
+
const baseBudget = boundedInt(
|
|
66
|
+
taskPolicy.contextBudget ?? taskPolicy.profile?.contextBudget,
|
|
67
|
+
20_000,
|
|
68
|
+
4_000,
|
|
69
|
+
48_000,
|
|
70
|
+
)
|
|
71
|
+
const baseSkills = boundedInt(
|
|
72
|
+
taskPolicy.maxSkills ?? taskPolicy.profile?.maxSkills,
|
|
73
|
+
4,
|
|
74
|
+
1,
|
|
75
|
+
5,
|
|
76
|
+
)
|
|
77
|
+
const baseStrategy = taskPolicy.profile?.contextStrategy || "incremental-semantic+git"
|
|
78
|
+
|
|
79
|
+
if (normalizedAttempt <= 1) {
|
|
80
|
+
return {
|
|
81
|
+
schemaVersion: 1,
|
|
82
|
+
stage: "initial",
|
|
83
|
+
attempt: normalizedAttempt,
|
|
84
|
+
contextBudget: baseBudget,
|
|
85
|
+
maxSkills: baseSkills,
|
|
86
|
+
contextStrategy: baseStrategy,
|
|
87
|
+
requireDiagnosis: false,
|
|
88
|
+
requireCritic: false,
|
|
89
|
+
modelEscalation: false,
|
|
90
|
+
directives: [],
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
if (normalizedAttempt === 2) {
|
|
95
|
+
return {
|
|
96
|
+
schemaVersion: 1,
|
|
97
|
+
stage: "diagnose",
|
|
98
|
+
attempt: normalizedAttempt,
|
|
99
|
+
contextBudget: Math.min(48_000, Math.max(baseBudget, Math.round(baseBudget * 1.35))),
|
|
100
|
+
maxSkills: Math.min(5, baseSkills + 1),
|
|
101
|
+
contextStrategy:
|
|
102
|
+
taskPolicy.executionProfile === "fast"
|
|
103
|
+
? "incremental-semantic+git"
|
|
104
|
+
: baseStrategy,
|
|
105
|
+
requireDiagnosis: true,
|
|
106
|
+
requireCritic: false,
|
|
107
|
+
modelEscalation: true,
|
|
108
|
+
directives: [
|
|
109
|
+
"reproduce or capture the exact previous failure before editing",
|
|
110
|
+
"inspect the direct caller, nearest test and failure-adjacent evidence",
|
|
111
|
+
"do not stack another speculative patch on top of the failed attempt",
|
|
112
|
+
],
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
return {
|
|
117
|
+
schemaVersion: 1,
|
|
118
|
+
stage: "deep-recovery",
|
|
119
|
+
attempt: normalizedAttempt,
|
|
120
|
+
contextBudget: Math.min(48_000, Math.max(20_000, Math.round(baseBudget * 1.75))),
|
|
121
|
+
maxSkills: Math.min(5, baseSkills + 2),
|
|
122
|
+
contextStrategy: "semantic+graph+git",
|
|
123
|
+
requireDiagnosis: true,
|
|
124
|
+
requireCritic: true,
|
|
125
|
+
modelEscalation: true,
|
|
126
|
+
directives: [
|
|
127
|
+
"re-investigate from fresh evidence and explicitly reject the failed hypothesis",
|
|
128
|
+
"expand to callers, dependencies, tests and boundary contracts before editing",
|
|
129
|
+
"challenge the architecture or coupling if repeated fixes expose a wider problem",
|
|
130
|
+
"run an independent critic or review pass before accepting the recovery",
|
|
131
|
+
],
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
|
|
54
135
|
export function classifyEngineeringTask(text, facts = {}) {
|
|
55
136
|
const value = String(text || "")
|
|
56
137
|
const declaredHighRisk =
|
|
57
138
|
["high", "critical"].includes(String(facts.risk || "").toLowerCase()) ||
|
|
58
|
-
/\b(?:risk|rủi ro)\s*[
|
|
59
|
-
|
|
139
|
+
/\b(?:risk|rủi ro)\s*[:=\/-]?\s*(?:high|critical|cao|nghiêm trọng)\b/i.test(value) ||
|
|
140
|
+
/\b(?:high|critical)[-\s]?(?:risk|rủi ro)\b/i.test(value)
|
|
141
|
+
const declaredLongHorizon =
|
|
142
|
+
facts.longHorizon === true ||
|
|
143
|
+
["long", "long-horizon", "deep"].includes(String(facts.mode || "").toLowerCase())
|
|
144
|
+
const compoundLongRisk =
|
|
145
|
+
/\blong\s*[/|,+]\s*(?:high|critical)[-\s]?risk\b/i.test(value) ||
|
|
146
|
+
/\b(?:high|critical)[-\s]?risk\s*[/|,+]\s*long\b/i.test(value)
|
|
147
|
+
const explicitLongHorizon = declaredLongHorizon || compoundLongRisk || LONG.test(value)
|
|
60
148
|
const signals = [
|
|
61
149
|
signal("long-request-text", value.length > 700, 1),
|
|
62
150
|
signal("medium-request-text", value.length > 250, 1),
|
|
@@ -91,8 +179,9 @@ export function classifyEngineeringTask(text, facts = {}) {
|
|
|
91
179
|
if (/(docker|kubernetes|terraform|github actions|ci\/cd|deploy|triển khai)/i.test(value)) domains.push("devops")
|
|
92
180
|
|
|
93
181
|
return {
|
|
94
|
-
schemaVersion:
|
|
182
|
+
schemaVersion: 4,
|
|
95
183
|
score,
|
|
184
|
+
signals,
|
|
96
185
|
risk,
|
|
97
186
|
mode,
|
|
98
187
|
executionProfile: profile.name,
|
|
@@ -105,7 +194,11 @@ export function classifyEngineeringTask(text, facts = {}) {
|
|
|
105
194
|
requireFreshEvidence: true,
|
|
106
195
|
profile,
|
|
107
196
|
domains: [...new Set(domains)],
|
|
108
|
-
|
|
197
|
+
recovery: {
|
|
198
|
+
escalateAfterFailure: true,
|
|
199
|
+
diagnosisBeforePatch: true,
|
|
200
|
+
deepRecoveryFromAttempt: 3,
|
|
201
|
+
},
|
|
109
202
|
antiHallucination: {
|
|
110
203
|
evidenceFirst: true,
|
|
111
204
|
noCompletionWithoutVerification: mode !== "inline" || risk === "high",
|
package/lib/task-engine.mjs
CHANGED
|
@@ -9,10 +9,10 @@ import { relevantAcceptedLearnings } from "./learning-engine.mjs"
|
|
|
9
9
|
import { validateVerificationReceipt } from "./evidence-receipt.mjs"
|
|
10
10
|
import { appendRuntimeEvent, readRuntimeEvents } from "./runtime-events.mjs"
|
|
11
11
|
import { createGateReceipt, validateGateReceipt } from "./gate-receipt.mjs"
|
|
12
|
-
import { classifyEngineeringTask } from "./orchestrator-policy.mjs"
|
|
12
|
+
import { classifyEngineeringTask, recoveryPolicyForAttempt } from "./orchestrator-policy.mjs"
|
|
13
13
|
|
|
14
14
|
const WORK_DIR = ".ues-work"
|
|
15
|
-
const STATE_SCHEMA =
|
|
15
|
+
const STATE_SCHEMA = 4
|
|
16
16
|
const LOCK_TIMEOUT_MS = 30_000
|
|
17
17
|
const LOCK_STALE_MS = 120_000
|
|
18
18
|
const LOCK_HEARTBEAT_MS = 15_000
|
|
@@ -284,6 +284,7 @@ export async function initWork(root, slug, goal) {
|
|
|
284
284
|
tasks: {},
|
|
285
285
|
decisions: [],
|
|
286
286
|
blockers: [],
|
|
287
|
+
checkpoint: null,
|
|
287
288
|
nextAction: "Complete SPEC.md, create PLAN.json, then run ocskill work plan.",
|
|
288
289
|
})
|
|
289
290
|
await writeJson(paths.evidence, {
|
|
@@ -387,6 +388,7 @@ export async function importPlan(root, slug, planInput) {
|
|
|
387
388
|
planHash: hash,
|
|
388
389
|
planApproval: { status: "pending", planHash: hash, at: updatedAt, evidence: null },
|
|
389
390
|
integrationVerification: null,
|
|
391
|
+
checkpoint: null,
|
|
390
392
|
evidencePolicy: evidencePolicyForPlan(plan),
|
|
391
393
|
tasks: stateTasks,
|
|
392
394
|
nextAction: "Run ues-plan-checker, then record PASS with 'ocskill work approve-plan'.",
|
|
@@ -489,7 +491,7 @@ export function readyTasks(plan, state) {
|
|
|
489
491
|
const ready = []
|
|
490
492
|
for (const task of plan.tasks) {
|
|
491
493
|
const status = state.tasks?.[task.id]?.status
|
|
492
|
-
if (!["pending", "failed"].includes(status)) continue
|
|
494
|
+
if (!["pending", "failed", "retryable"].includes(status)) continue
|
|
493
495
|
if (!(task.dependsOn || []).every((dep) => completed.has(dep))) continue
|
|
494
496
|
const files = taskFiles(task)
|
|
495
497
|
if (files.length === 0 && runningFiles.size > 0) continue
|
|
@@ -501,7 +503,7 @@ export function readyTasks(plan, state) {
|
|
|
501
503
|
|
|
502
504
|
export async function workStatus(root, slug) {
|
|
503
505
|
const loaded = await loadWork(root, slug)
|
|
504
|
-
const counts = { pending: 0, running: 0, completed: 0, failed: 0 }
|
|
506
|
+
const counts = { pending: 0, running: 0, completed: 0, failed: 0, retryable: 0 }
|
|
505
507
|
for (const value of Object.values(loaded.state.tasks || {})) {
|
|
506
508
|
if (Object.hasOwn(counts, value.status)) counts[value.status] += 1
|
|
507
509
|
}
|
|
@@ -541,6 +543,7 @@ export async function workStatus(root, slug) {
|
|
|
541
543
|
gateReceipts: (loaded.evidence.gateReceipts || []).length,
|
|
542
544
|
},
|
|
543
545
|
nextAction: loaded.state.nextAction,
|
|
546
|
+
checkpoint: loaded.state.checkpoint || null,
|
|
544
547
|
updatedAt: loaded.state.updatedAt,
|
|
545
548
|
}
|
|
546
549
|
}
|
|
@@ -673,7 +676,7 @@ export async function recoverTask(root, slug, taskID, options = {}) {
|
|
|
673
676
|
const timestamp = now()
|
|
674
677
|
const previousOwner = record.owner ? { ...record.owner } : null
|
|
675
678
|
const previousRunId = record.runId || null
|
|
676
|
-
record.status = "
|
|
679
|
+
record.status = "retryable"
|
|
677
680
|
record.lastFailure = String(options.reason || "stale executor lease recovered after interruption")
|
|
678
681
|
record.lastOwner = previousOwner
|
|
679
682
|
record.lastRunId = previousRunId
|
|
@@ -709,7 +712,7 @@ export async function recoverStaleTasks(root, slug, options = {}) {
|
|
|
709
712
|
if (!stale) continue
|
|
710
713
|
const previousOwner = record.owner ? { ...record.owner } : null
|
|
711
714
|
const previousRunId = record.runId || null
|
|
712
|
-
record.status = "
|
|
715
|
+
record.status = "retryable"
|
|
713
716
|
record.lastFailure = "stale executor lease recovered after interruption"
|
|
714
717
|
record.lastOwner = previousOwner
|
|
715
718
|
record.lastRunId = previousRunId
|
|
@@ -1082,12 +1085,22 @@ async function buildContextPack(loaded, taskID) {
|
|
|
1082
1085
|
changedFiles: taskFiles(task).length,
|
|
1083
1086
|
risk: task.risk,
|
|
1084
1087
|
})
|
|
1088
|
+
const attempt = Math.max(1, Number(loaded.state.tasks?.[taskID]?.attempts || 1))
|
|
1089
|
+
const recovery = recoveryPolicyForAttempt(contextPolicy, attempt)
|
|
1090
|
+
const effectiveContextPolicy = {
|
|
1091
|
+
...contextPolicy,
|
|
1092
|
+
contextBudget: recovery.contextBudget,
|
|
1093
|
+
maxSkills: recovery.maxSkills,
|
|
1094
|
+
effectiveContextBudget: recovery.contextBudget,
|
|
1095
|
+
effectiveMaxSkills: recovery.maxSkills,
|
|
1096
|
+
recovery,
|
|
1097
|
+
}
|
|
1085
1098
|
const contextManifest = await buildContextManifest(
|
|
1086
1099
|
loaded.paths.root,
|
|
1087
1100
|
task,
|
|
1088
1101
|
{
|
|
1089
|
-
budget:
|
|
1090
|
-
strategy:
|
|
1102
|
+
budget: recovery.contextBudget,
|
|
1103
|
+
strategy: recovery.contextStrategy,
|
|
1091
1104
|
},
|
|
1092
1105
|
).catch(() => null)
|
|
1093
1106
|
const learnings = await relevantAcceptedLearnings(
|
|
@@ -1097,7 +1110,8 @@ async function buildContextPack(loaded, taskID) {
|
|
|
1097
1110
|
|
|
1098
1111
|
return {
|
|
1099
1112
|
schemaVersion: STATE_SCHEMA,
|
|
1100
|
-
contextPolicy,
|
|
1113
|
+
contextPolicy: effectiveContextPolicy,
|
|
1114
|
+
attempt,
|
|
1101
1115
|
slug: loaded.state.slug,
|
|
1102
1116
|
task,
|
|
1103
1117
|
taskBrief: path.relative(loaded.paths.root, path.join(loaded.paths.tasks, `${taskID}.md`)).replaceAll("\\", "/"),
|
|
@@ -1111,6 +1125,7 @@ async function buildContextPack(loaded, taskID) {
|
|
|
1111
1125
|
workingState: {
|
|
1112
1126
|
status: loaded.state.status,
|
|
1113
1127
|
task: loaded.state.tasks?.[taskID],
|
|
1128
|
+
checkpoint: loaded.state.checkpoint || null,
|
|
1114
1129
|
},
|
|
1115
1130
|
}
|
|
1116
1131
|
}
|
|
@@ -1151,6 +1166,128 @@ export async function contextPack(root, slug, taskID) {
|
|
|
1151
1166
|
return buildContextPack(await loadWork(root, slug), taskID)
|
|
1152
1167
|
}
|
|
1153
1168
|
|
|
1169
|
+
function checkpointEvidencePointers(evidence = {}) {
|
|
1170
|
+
const pointers = []
|
|
1171
|
+
for (const entry of (evidence.entries || []).slice(-12)) {
|
|
1172
|
+
pointers.push({
|
|
1173
|
+
kind: "evidence-entry",
|
|
1174
|
+
task: entry.task || null,
|
|
1175
|
+
at: entry.at || entry.recordedAt || null,
|
|
1176
|
+
runId: entry.runId || null,
|
|
1177
|
+
})
|
|
1178
|
+
}
|
|
1179
|
+
for (const receipt of (evidence.receipts || []).slice(-12)) {
|
|
1180
|
+
pointers.push({
|
|
1181
|
+
kind: "verification-receipt",
|
|
1182
|
+
id: receipt.id || null,
|
|
1183
|
+
task: receipt.task || null,
|
|
1184
|
+
runId: receipt.runId || null,
|
|
1185
|
+
verdict: receipt.verdict || null,
|
|
1186
|
+
})
|
|
1187
|
+
}
|
|
1188
|
+
for (const receipt of (evidence.gateReceipts || []).slice(-8)) {
|
|
1189
|
+
pointers.push({
|
|
1190
|
+
kind: "gate-receipt",
|
|
1191
|
+
id: receipt.id || null,
|
|
1192
|
+
receiptKind: receipt.kind || null,
|
|
1193
|
+
verdict: receipt.verdict || null,
|
|
1194
|
+
})
|
|
1195
|
+
}
|
|
1196
|
+
return pointers.slice(-24)
|
|
1197
|
+
}
|
|
1198
|
+
|
|
1199
|
+
function checkpointNextAction(loaded, taskID, runId) {
|
|
1200
|
+
const record = taskID ? loaded.state.tasks?.[taskID] : null
|
|
1201
|
+
if (record?.status === "running") {
|
|
1202
|
+
return {
|
|
1203
|
+
type: "continue-task",
|
|
1204
|
+
taskId: taskID,
|
|
1205
|
+
runId: runId || record.runId || null,
|
|
1206
|
+
instruction: "Continue the active scoped task from durable evidence; execute the next required tool/action before narrative summary.",
|
|
1207
|
+
}
|
|
1208
|
+
}
|
|
1209
|
+
|
|
1210
|
+
const ready = readyTasks(loaded.plan, loaded.state)
|
|
1211
|
+
if (ready.length) {
|
|
1212
|
+
const nextTask = ready[0]
|
|
1213
|
+
return {
|
|
1214
|
+
type: "dispatch-task",
|
|
1215
|
+
taskId: nextTask,
|
|
1216
|
+
attempt: Number(loaded.state.tasks?.[nextTask]?.attempts || 0) + 1,
|
|
1217
|
+
instruction: "Dispatch the next ready task in a fresh executor from .ues-work.",
|
|
1218
|
+
}
|
|
1219
|
+
}
|
|
1220
|
+
|
|
1221
|
+
return {
|
|
1222
|
+
type: "inspect-state",
|
|
1223
|
+
instruction: loaded.state.nextAction || "Inspect durable work state and choose the next deterministic action.",
|
|
1224
|
+
}
|
|
1225
|
+
}
|
|
1226
|
+
|
|
1227
|
+
export async function checkpointWork(root, slug, options = {}) {
|
|
1228
|
+
return withWorkLock(root, slug, async () => {
|
|
1229
|
+
const loaded = await loadWork(root, slug)
|
|
1230
|
+
const taskID = String(options.taskID || "").trim() ||
|
|
1231
|
+
Object.entries(loaded.state.tasks || {}).find(([, value]) => value.status === "running")?.[0] ||
|
|
1232
|
+
null
|
|
1233
|
+
const record = taskID ? loaded.state.tasks?.[taskID] : null
|
|
1234
|
+
if (taskID && !record) throw new Error(`task '${taskID}' is not tracked`)
|
|
1235
|
+
if (record?.status === "running") assertRunFence(record, { runId: options.runId || null })
|
|
1236
|
+
|
|
1237
|
+
const timestamp = now()
|
|
1238
|
+
const checkpoint = {
|
|
1239
|
+
schemaVersion: 1,
|
|
1240
|
+
reason: String(options.reason || "runtime-checkpoint"),
|
|
1241
|
+
createdAt: timestamp,
|
|
1242
|
+
currentTaskId: taskID,
|
|
1243
|
+
runId: record?.runId || options.runId || null,
|
|
1244
|
+
planHash: loaded.plan ? planHash(loaded.plan) : loaded.state.planHash || null,
|
|
1245
|
+
nextAction: checkpointNextAction(loaded, taskID, record?.runId || options.runId || null),
|
|
1246
|
+
workspaceFingerprint: workspaceFingerprint(loaded.paths.root),
|
|
1247
|
+
evidencePointers: checkpointEvidencePointers(loaded.evidence),
|
|
1248
|
+
resumedAt: null,
|
|
1249
|
+
}
|
|
1250
|
+
|
|
1251
|
+
loaded.state.schemaVersion = STATE_SCHEMA
|
|
1252
|
+
loaded.state.checkpoint = checkpoint
|
|
1253
|
+
loaded.state.updatedAt = timestamp
|
|
1254
|
+
await writeJson(loaded.paths.state, loaded.state)
|
|
1255
|
+
await journal(loaded.paths, "work.checkpoint", {
|
|
1256
|
+
task: taskID,
|
|
1257
|
+
runId: checkpoint.runId,
|
|
1258
|
+
planHash: checkpoint.planHash,
|
|
1259
|
+
reason: checkpoint.reason,
|
|
1260
|
+
nextAction: checkpoint.nextAction,
|
|
1261
|
+
})
|
|
1262
|
+
return checkpoint
|
|
1263
|
+
})
|
|
1264
|
+
}
|
|
1265
|
+
|
|
1266
|
+
export async function markCheckpointResumed(root, slug, options = {}) {
|
|
1267
|
+
return withWorkLock(root, slug, async () => {
|
|
1268
|
+
const loaded = await loadWork(root, slug)
|
|
1269
|
+
const checkpoint = loaded.state.checkpoint
|
|
1270
|
+
if (!checkpoint) return null
|
|
1271
|
+
if (options.taskID && checkpoint.currentTaskId && checkpoint.currentTaskId !== options.taskID) {
|
|
1272
|
+
throw new Error("checkpoint task mismatch")
|
|
1273
|
+
}
|
|
1274
|
+
if (checkpoint.runId && options.runId && checkpoint.runId !== options.runId) {
|
|
1275
|
+
throw new Error("checkpoint runId mismatch")
|
|
1276
|
+
}
|
|
1277
|
+
const timestamp = now()
|
|
1278
|
+
checkpoint.resumedAt = timestamp
|
|
1279
|
+
checkpoint.resumeReason = String(options.reason || "deterministic-action-observed")
|
|
1280
|
+
loaded.state.updatedAt = timestamp
|
|
1281
|
+
await writeJson(loaded.paths.state, loaded.state)
|
|
1282
|
+
await journal(loaded.paths, "work.checkpoint-resumed", {
|
|
1283
|
+
task: checkpoint.currentTaskId || null,
|
|
1284
|
+
runId: checkpoint.runId || null,
|
|
1285
|
+
reason: checkpoint.resumeReason,
|
|
1286
|
+
})
|
|
1287
|
+
return checkpoint
|
|
1288
|
+
})
|
|
1289
|
+
}
|
|
1290
|
+
|
|
1154
1291
|
export async function runtimeEvents(root, slug, options = {}) {
|
|
1155
1292
|
const paths = workPaths(root, slug)
|
|
1156
1293
|
return readRuntimeEvents(paths.events, options)
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "opencode-agent-skill",
|
|
3
|
-
"version": "
|
|
4
|
-
"description": "Evidence-first engineering runtime with adaptive
|
|
3
|
+
"version": "10.0.0",
|
|
4
|
+
"description": "Evidence-first engineering runtime with adaptive context, selective routing, weak-model recovery, bounded ACI and benchmark-gated verification for OpenCode",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
7
7
|
"ocskill": "bin/ocskill.mjs"
|
|
@@ -45,7 +45,8 @@
|
|
|
45
45
|
"release:check-tag": "node scripts/check-release-tag.mjs",
|
|
46
46
|
"evals:polyglot:validate": "node scripts/validate-live-suite.mjs --suite polyglot",
|
|
47
47
|
"evals:polyglot": "node scripts/eval-live.mjs --suite polyglot",
|
|
48
|
-
"evals:matrix:gate": "node scripts/eval-matrix.mjs --require-confidence"
|
|
48
|
+
"evals:matrix:gate": "node scripts/eval-matrix.mjs --require-confidence",
|
|
49
|
+
"evals:ablation": "node scripts/eval-ablation.mjs"
|
|
49
50
|
},
|
|
50
51
|
"keywords": [
|
|
51
52
|
"opencode",
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { readFile } from "node:fs/promises"
|
|
3
|
+
import path from "node:path"
|
|
4
|
+
import { compareEvalSummaries } from "../lib/eval-ablation.mjs"
|
|
5
|
+
|
|
6
|
+
const args = process.argv.slice(2)
|
|
7
|
+
|
|
8
|
+
function option(name, fallback) {
|
|
9
|
+
const index = args.indexOf(name)
|
|
10
|
+
return index >= 0 && index + 1 < args.length ? args[index + 1] : fallback
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
function positional() {
|
|
14
|
+
const values = []
|
|
15
|
+
for (let index = 0; index < args.length; index += 1) {
|
|
16
|
+
if (args[index].startsWith("--")) {
|
|
17
|
+
if (["--pass-rate-tolerance", "--min-initial-reduction", "--max-token-ratio", "--max-duration-ratio"].includes(args[index])) index += 1
|
|
18
|
+
continue
|
|
19
|
+
}
|
|
20
|
+
values.push(args[index])
|
|
21
|
+
}
|
|
22
|
+
return values
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
const [referenceFile, candidateFile] = positional()
|
|
26
|
+
if (!referenceFile || !candidateFile) {
|
|
27
|
+
console.error("Usage: node scripts/eval-ablation.mjs <reference-summary.json> <candidate-summary.json> [--require-gate] [--min-initial-reduction 0.10]")
|
|
28
|
+
process.exit(2)
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
const [reference, candidate] = await Promise.all([
|
|
32
|
+
readFile(path.resolve(referenceFile), "utf8").then(JSON.parse),
|
|
33
|
+
readFile(path.resolve(candidateFile), "utf8").then(JSON.parse),
|
|
34
|
+
])
|
|
35
|
+
|
|
36
|
+
const report = compareEvalSummaries(reference, candidate, {
|
|
37
|
+
passRateTolerance: Number(option("--pass-rate-tolerance", "0")),
|
|
38
|
+
minInitialInputReduction: Number(option("--min-initial-reduction", "0.10")),
|
|
39
|
+
maxTotalTokenRatio: Number(option("--max-token-ratio", "1.05")),
|
|
40
|
+
maxDurationRatio: Number(option("--max-duration-ratio", "1.10")),
|
|
41
|
+
})
|
|
42
|
+
|
|
43
|
+
console.log(JSON.stringify(report, null, 2))
|
|
44
|
+
if (args.includes("--require-gate") && !report.gateEligible) process.exitCode = 1
|
package/scripts/eval-matrix.mjs
CHANGED
|
@@ -32,6 +32,8 @@ const onlyLive = has("--standard-only")
|
|
|
32
32
|
const onlyPolyglot = has("--polyglot-only")
|
|
33
33
|
const withoutPolyglot = has("--without-polyglot")
|
|
34
34
|
const requireConfidence = has("--require-confidence")
|
|
35
|
+
const maxInitialInputRatio = Number(argValue("--max-initial-input-ratio", "1.5"))
|
|
36
|
+
const maxTokenRatio = Number(argValue("--max-token-ratio", "1.75"))
|
|
35
37
|
|
|
36
38
|
if (!model) {
|
|
37
39
|
console.error("Usage: node scripts/eval-matrix.mjs --model provider/model [--trials 3] [--auth current|env-only] [--variant high] [--without-polyglot|--long-only|--standard-only|--polyglot-only]")
|
|
@@ -110,7 +112,10 @@ const baselineCount = results.filter((item) => item.mode === "baseline").length
|
|
|
110
112
|
const uesCount = results.filter((item) => item.mode === "ues").length
|
|
111
113
|
const coverageComplete = baselineCount === expectedPerMode && uesCount === expectedPerMode
|
|
112
114
|
const summary = summarizeEvalResults(results)
|
|
113
|
-
const confidence = pairedBenchmarkConfidence(results
|
|
115
|
+
const confidence = pairedBenchmarkConfidence(results, {
|
|
116
|
+
maxInitialInputRatio: Number.isFinite(maxInitialInputRatio) ? maxInitialInputRatio : 1.5,
|
|
117
|
+
maxTokenRatio: Number.isFinite(maxTokenRatio) ? maxTokenRatio : 1.75,
|
|
118
|
+
})
|
|
114
119
|
|
|
115
120
|
const report = {
|
|
116
121
|
schemaVersion: 1,
|
|
@@ -134,7 +139,11 @@ const report = {
|
|
|
134
139
|
mode: item.mode,
|
|
135
140
|
passed: item.passed === true,
|
|
136
141
|
durationMs: item.durationMs ?? null,
|
|
137
|
-
telemetry:
|
|
142
|
+
telemetry: {
|
|
143
|
+
...(item.telemetry?.costSamples > 0 ? { cost: item.telemetry.cost } : {}),
|
|
144
|
+
...(item.telemetry?.firstUsage ? { firstUsage: item.telemetry.firstUsage } : {}),
|
|
145
|
+
...(item.telemetry?.tokens ? { tokens: item.telemetry.tokens } : {}),
|
|
146
|
+
},
|
|
138
147
|
})),
|
|
139
148
|
summary,
|
|
140
149
|
confidence,
|
|
@@ -149,6 +158,8 @@ console.log("- UES: " + (summary.modes.ues?.passed || 0) + "/" + (summary.modes.
|
|
|
149
158
|
console.log("- pass-rate delta: " + (summary.passRateDelta == null ? "n/a" : (summary.passRateDelta * 100).toFixed(1) + " pp"))
|
|
150
159
|
console.log("- coverage: " + (coverageComplete ? "COMPLETE" : "INCOMPLETE"))
|
|
151
160
|
console.log("- paired confidence: " + (confidence.promotionEligible ? "SUPPORTED" : "NOT YET SUPPORTED") + " (pairs=" + confidence.pairs + ", p=" + confidence.pValue.toFixed(4) + ")")
|
|
161
|
+
console.log("- initial-input ratio UES/baseline: " + (confidence.initialInput.ratio == null ? "n/a" : confidence.initialInput.ratio.toFixed(3)) + " (max " + confidence.initialInput.maxRatio + ")")
|
|
162
|
+
console.log("- total-token ratio UES/baseline: " + (confidence.tokens.ratio == null ? "n/a" : confidence.tokens.ratio.toFixed(3)) + " (max " + confidence.tokens.maxRatio + ")")
|
|
152
163
|
console.log("- confidence gate: " + (requireConfidence ? "REQUIRED" : "report-only"))
|
|
153
164
|
console.log("- report: " + reportFile)
|
|
154
165
|
|