opencode-agent-skill 9.0.0 → 10.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +49 -0
- package/README.md +696 -675
- package/bin/ocskill.mjs +24 -0
- package/global-config/AGENTS.md +78 -163
- package/global-config/plugins/ues-router/index.js +413 -59
- package/global-config/plugins/ues-router/router.js +35 -0
- package/global-config/plugins/ues-router/runtime-guard.js +265 -0
- package/lib/benchmark-confidence.mjs +49 -11
- package/lib/eval-ablation.mjs +104 -0
- package/lib/eval-report.mjs +11 -0
- package/lib/eval-telemetry.mjs +3 -0
- package/lib/model-policy.mjs +6 -0
- package/lib/orchestrator-policy.mjs +99 -6
- package/lib/task-engine.mjs +146 -9
- package/package.json +4 -3
- package/scripts/eval-ablation.mjs +44 -0
- package/scripts/eval-matrix.mjs +13 -2
|
@@ -0,0 +1,265 @@
|
|
|
1
|
+
import { createHash } from "node:crypto"
|
|
2
|
+
|
|
3
|
+
const EXPLORATION_TOOL = /(?:^|[._-])(read|grep|glob|repo[-_.]?graph|semantic[-_.]?search|aci[-_.]?search)(?:$|[._-])/i
|
|
4
|
+
const VERIFY_OR_WRITE_TOOL = /(?:^|[._-])(edit|write|patch|bash|shell|verify|test|apply)(?:$|[._-])/i
|
|
5
|
+
|
|
6
|
+
function canonical(value) {
|
|
7
|
+
if (Array.isArray(value)) return value.map(canonical)
|
|
8
|
+
if (!value || typeof value !== "object") return value
|
|
9
|
+
return Object.fromEntries(
|
|
10
|
+
Object.keys(value).sort().map((key) => [key, canonical(value[key])]),
|
|
11
|
+
)
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
export function stableRuntimeHash(value) {
|
|
15
|
+
const text = typeof value === "string" ? value : JSON.stringify(canonical(value))
|
|
16
|
+
return createHash("sha256").update(text || "").digest("hex")
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
export function isExplorationTool(tool) {
|
|
20
|
+
return EXPLORATION_TOOL.test(String(tool || ""))
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
export function budgetToolResult(tool, result, options = {}) {
|
|
24
|
+
const name = String(tool || "")
|
|
25
|
+
const aggressive = /grep|glob|repo[-_.]?graph|semantic[-_.]?search/i.test(name)
|
|
26
|
+
const maxChars = Math.max(2_000, Number(options.maxChars || (aggressive ? 12_000 : 24_000)))
|
|
27
|
+
const maxLines = Math.max(40, Number(options.maxLines || (aggressive ? 160 : 360)))
|
|
28
|
+
|
|
29
|
+
if (!aggressive && !/read/i.test(name)) return result
|
|
30
|
+
|
|
31
|
+
const original = typeof result === "string" ? result : String(result?.output || "")
|
|
32
|
+
const lines = original.split(/\r?\n/)
|
|
33
|
+
if (original.length <= maxChars && lines.length <= maxLines) return result
|
|
34
|
+
|
|
35
|
+
const digest = stableRuntimeHash(original)
|
|
36
|
+
const headLineCount = Math.max(1, Math.floor(maxLines * 0.7))
|
|
37
|
+
const tailLineCount = Math.max(1, maxLines - headLineCount)
|
|
38
|
+
let bounded = [
|
|
39
|
+
...lines.slice(0, headLineCount),
|
|
40
|
+
`...[UES tool output truncated: ${lines.length} lines / ${original.length} chars, sha256=${digest.slice(0, 16)}]...`,
|
|
41
|
+
...lines.slice(-tailLineCount),
|
|
42
|
+
].join("\n")
|
|
43
|
+
|
|
44
|
+
if (bounded.length > maxChars) {
|
|
45
|
+
const marker = `\n...[UES char budget applied; sha256=${digest.slice(0, 16)}]...\n`
|
|
46
|
+
const room = Math.max(0, maxChars - marker.length)
|
|
47
|
+
const head = Math.floor(room * 0.7)
|
|
48
|
+
bounded = bounded.slice(0, head) + marker + bounded.slice(-(room - head))
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
if (typeof result === "string") return bounded
|
|
52
|
+
return {
|
|
53
|
+
...result,
|
|
54
|
+
output: bounded,
|
|
55
|
+
metadata: {
|
|
56
|
+
...(result?.metadata || {}),
|
|
57
|
+
uesTruncated: true,
|
|
58
|
+
uesOriginalChars: original.length,
|
|
59
|
+
uesOriginalLines: lines.length,
|
|
60
|
+
uesOutputDigest: digest,
|
|
61
|
+
},
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
export function classifyProviderFailure(error = {}) {
|
|
66
|
+
const status = Number(error?.status ?? error?.cause?.status)
|
|
67
|
+
const type = String(error?.type || error?.code || "")
|
|
68
|
+
const message = String(error?.message || error || "")
|
|
69
|
+
const value = `${type} ${message}`.toLowerCase()
|
|
70
|
+
|
|
71
|
+
if (status === 401 || status === 403 || /auth|unauthori[sz]ed|invalid api key|credential/.test(value)) return "AUTH"
|
|
72
|
+
if (/context.{0,20}(large|length|window|overflow)|too many tokens|max(?:imum)? context/.test(value)) return "CONTEXT_TOO_LARGE"
|
|
73
|
+
if (status === 429 || /rate.?limit|too many requests/.test(value)) return "RATE_LIMIT"
|
|
74
|
+
if (/quota|credit|billing|insufficient balance/.test(value)) return "QUOTA"
|
|
75
|
+
if (/no token|no output|empty response|empty completion|returned no content|no content|stalled|no-progress/.test(value)) return "NO_TOKEN"
|
|
76
|
+
if (/timeout|timed out|deadline|abort(?:ed)?/.test(value)) return "TIMEOUT"
|
|
77
|
+
if (Number.isFinite(status) && status >= 500 && status < 600) return "PROVIDER_5XX"
|
|
78
|
+
if (/provider|upstream|gateway|service unavailable/.test(value)) return "PROVIDER_5XX"
|
|
79
|
+
return "OTHER"
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
export function progressWatchdogDecision(snapshot = {}, now = Date.now(), stallMs = 60_000) {
|
|
83
|
+
const limit = Math.max(30_000, Math.min(Number(stallMs || 60_000), 5 * 60_000))
|
|
84
|
+
const lastProgressAt = Number(snapshot.lastProgressAt || 0)
|
|
85
|
+
const activeToolCalls = Math.max(0, Number(snapshot.activeToolCalls || 0))
|
|
86
|
+
const idleMs = Math.max(0, Number(now) - lastProgressAt)
|
|
87
|
+
|
|
88
|
+
if (activeToolCalls > 0) {
|
|
89
|
+
return { stalled: false, idleMs, limitMs: limit, reason: "tool-active" }
|
|
90
|
+
}
|
|
91
|
+
return {
|
|
92
|
+
stalled: idleMs >= limit,
|
|
93
|
+
idleMs,
|
|
94
|
+
limitMs: limit,
|
|
95
|
+
reason: idleMs >= limit ? "no-progress" : "within-grace",
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
export function providerRecoveryPlan(kind, physicalAttempt = 1, options = {}) {
|
|
100
|
+
const attempt = Math.max(1, Number(physicalAttempt || 1))
|
|
101
|
+
const hasEscalationModel = Boolean(options.hasEscalationModel)
|
|
102
|
+
|
|
103
|
+
if (kind === "AUTH") return { action: "fail-fast", retry: false, reason: "authentication failure requires user/config repair" }
|
|
104
|
+
if (kind === "CONTEXT_TOO_LARGE") return { action: "compact-context", retry: false, reason: "reduce/compact context instead of repeating the same request" }
|
|
105
|
+
if (kind === "QUOTA") {
|
|
106
|
+
return hasEscalationModel
|
|
107
|
+
? { action: "fresh-session-escalated-model", retry: true, reason: "quota exhausted on current provider/model" }
|
|
108
|
+
: { action: "fail-retryable", retry: false, reason: "quota exhausted and no configured fallback model exists" }
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
if (["NO_TOKEN", "TIMEOUT", "RATE_LIMIT", "PROVIDER_5XX"].includes(kind)) {
|
|
112
|
+
if (attempt <= 1) {
|
|
113
|
+
return { action: "fresh-session-same-model", retry: true, reason: "retry transport/provider stall once with fresh session state" }
|
|
114
|
+
}
|
|
115
|
+
if (hasEscalationModel) {
|
|
116
|
+
return { action: "fresh-session-escalated-model", retry: true, reason: "repeated provider failure triggers configured model/provider escalation" }
|
|
117
|
+
}
|
|
118
|
+
return { action: "fail-retryable", retry: false, reason: "repeated provider failure without configured fallback" }
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
return { action: "fail-task", retry: false, reason: "non-provider failure should be diagnosed by the task recovery policy" }
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
function createSessionState(now = Date.now()) {
|
|
125
|
+
return {
|
|
126
|
+
signatures: new Map(),
|
|
127
|
+
lastWorkspaceSignal: null,
|
|
128
|
+
lastEvidenceKey: null,
|
|
129
|
+
noProgressCalls: 0,
|
|
130
|
+
loopBlocked: false,
|
|
131
|
+
lastProgressAt: now,
|
|
132
|
+
activeCalls: new Set(),
|
|
133
|
+
compactionAt: null,
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
export function createRuntimeGuard(options = {}) {
|
|
138
|
+
const duplicateLimit = Math.max(2, Number(options.duplicateLimit || 3))
|
|
139
|
+
const loopLimit = Math.max(4, Number(options.loopLimit || 6))
|
|
140
|
+
const sessions = new Map()
|
|
141
|
+
|
|
142
|
+
function stateFor(sessionID, at = Date.now()) {
|
|
143
|
+
const key = String(sessionID || "global")
|
|
144
|
+
if (!sessions.has(key)) sessions.set(key, createSessionState(at))
|
|
145
|
+
return sessions.get(key)
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
function resetForWorkspace(state, workspaceSignal) {
|
|
149
|
+
if (state.lastWorkspaceSignal === null || state.lastWorkspaceSignal === workspaceSignal) return
|
|
150
|
+
state.signatures.clear()
|
|
151
|
+
state.noProgressCalls = 0
|
|
152
|
+
state.loopBlocked = false
|
|
153
|
+
state.lastEvidenceKey = null
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
return {
|
|
157
|
+
before(input = {}) {
|
|
158
|
+
const at = Number(input.now || Date.now())
|
|
159
|
+
const state = stateFor(input.sessionID, at)
|
|
160
|
+
const workspaceSignal = String(input.workspaceSignal || "")
|
|
161
|
+
resetForWorkspace(state, workspaceSignal)
|
|
162
|
+
state.lastWorkspaceSignal = workspaceSignal
|
|
163
|
+
|
|
164
|
+
const tool = String(input.tool || "")
|
|
165
|
+
if (isExplorationTool(tool)) {
|
|
166
|
+
const signature = stableRuntimeHash({ tool, input: input.input || {}, cwd: input.cwd || "", workspaceSignal })
|
|
167
|
+
const seen = state.signatures.get(signature) || { count: 0, lastResultHash: null }
|
|
168
|
+
seen.count += 1
|
|
169
|
+
state.signatures.set(signature, seen)
|
|
170
|
+
|
|
171
|
+
if (state.loopBlocked) {
|
|
172
|
+
return {
|
|
173
|
+
blocked: true,
|
|
174
|
+
code: "UES_LOOP_DETECTED",
|
|
175
|
+
message: `UES loop guard blocked repeated exploration after ${state.noProgressCalls} no-progress calls. Choose a deterministic next action: edit the scoped target, run focused verification, inspect a direct caller/failing stack, change hypothesis, or escalate recovery.`,
|
|
176
|
+
signature,
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
if (seen.count > duplicateLimit) {
|
|
180
|
+
return {
|
|
181
|
+
blocked: true,
|
|
182
|
+
code: "UES_DUPLICATE_TOOL",
|
|
183
|
+
message: `UES duplicate-tool guard blocked ${tool}: equivalent arguments were already executed ${seen.count - 1} times with no workspace change. Use the previous evidence or change the query/scope before retrying.`,
|
|
184
|
+
signature,
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
state.lastProgressAt = at
|
|
190
|
+
if (input.callID) state.activeCalls.add(String(input.callID))
|
|
191
|
+
return { blocked: false }
|
|
192
|
+
},
|
|
193
|
+
|
|
194
|
+
after(input = {}) {
|
|
195
|
+
const at = Number(input.now || Date.now())
|
|
196
|
+
const state = stateFor(input.sessionID, at)
|
|
197
|
+
const workspaceSignal = String(input.workspaceSignal || "")
|
|
198
|
+
const tool = String(input.tool || "")
|
|
199
|
+
if (input.callID) state.activeCalls.delete(String(input.callID))
|
|
200
|
+
resetForWorkspace(state, workspaceSignal)
|
|
201
|
+
|
|
202
|
+
if (input.status === "completed" && isExplorationTool(tool)) {
|
|
203
|
+
const resultHash = stableRuntimeHash(input.result || "")
|
|
204
|
+
const evidenceKey = stableRuntimeHash({ workspaceSignal, resultHash })
|
|
205
|
+
if (state.lastEvidenceKey === evidenceKey) {
|
|
206
|
+
state.noProgressCalls += 1
|
|
207
|
+
} else {
|
|
208
|
+
state.lastEvidenceKey = evidenceKey
|
|
209
|
+
state.noProgressCalls = 0
|
|
210
|
+
}
|
|
211
|
+
if (state.noProgressCalls >= loopLimit) state.loopBlocked = true
|
|
212
|
+
|
|
213
|
+
const signature = stableRuntimeHash({ tool, input: input.input || {}, cwd: input.cwd || "", workspaceSignal })
|
|
214
|
+
const seen = state.signatures.get(signature)
|
|
215
|
+
if (seen) seen.lastResultHash = resultHash
|
|
216
|
+
} else if (
|
|
217
|
+
input.status === "completed" &&
|
|
218
|
+
(VERIFY_OR_WRITE_TOOL.test(tool) || state.lastWorkspaceSignal !== workspaceSignal)
|
|
219
|
+
) {
|
|
220
|
+
state.noProgressCalls = 0
|
|
221
|
+
state.loopBlocked = false
|
|
222
|
+
state.lastEvidenceKey = null
|
|
223
|
+
state.signatures.clear()
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
state.lastWorkspaceSignal = workspaceSignal
|
|
227
|
+
state.lastProgressAt = at
|
|
228
|
+
return {
|
|
229
|
+
loopBlocked: state.loopBlocked,
|
|
230
|
+
noProgressCalls: state.noProgressCalls,
|
|
231
|
+
lastProgressAt: state.lastProgressAt,
|
|
232
|
+
}
|
|
233
|
+
},
|
|
234
|
+
|
|
235
|
+
touch(sessionID, at = Date.now()) {
|
|
236
|
+
const state = stateFor(sessionID, at)
|
|
237
|
+
state.lastProgressAt = Number(at)
|
|
238
|
+
},
|
|
239
|
+
|
|
240
|
+
compacted(sessionID, at = Date.now()) {
|
|
241
|
+
const state = stateFor(sessionID, at)
|
|
242
|
+
state.compactionAt = Number(at)
|
|
243
|
+
state.lastProgressAt = Number(at)
|
|
244
|
+
state.signatures.clear()
|
|
245
|
+
state.noProgressCalls = 0
|
|
246
|
+
state.loopBlocked = false
|
|
247
|
+
state.lastEvidenceKey = null
|
|
248
|
+
},
|
|
249
|
+
|
|
250
|
+
snapshot(sessionID) {
|
|
251
|
+
const state = stateFor(sessionID)
|
|
252
|
+
return {
|
|
253
|
+
noProgressCalls: state.noProgressCalls,
|
|
254
|
+
loopBlocked: state.loopBlocked,
|
|
255
|
+
lastProgressAt: state.lastProgressAt,
|
|
256
|
+
activeToolCalls: state.activeCalls.size,
|
|
257
|
+
compactionAt: state.compactionAt,
|
|
258
|
+
}
|
|
259
|
+
},
|
|
260
|
+
|
|
261
|
+
clear(sessionID) {
|
|
262
|
+
sessions.delete(String(sessionID || "global"))
|
|
263
|
+
},
|
|
264
|
+
}
|
|
265
|
+
}
|
|
@@ -24,6 +24,14 @@ function mean(values) {
|
|
|
24
24
|
return usable.length ? usable.reduce((sum, value) => sum + value, 0) / usable.length : null
|
|
25
25
|
}
|
|
26
26
|
|
|
27
|
+
function positiveMean(values) {
|
|
28
|
+
return mean(values.map(Number).filter((value) => Number.isFinite(value) && value > 0))
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
function ratio(candidate, reference) {
|
|
32
|
+
return reference && candidate ? candidate / reference : null
|
|
33
|
+
}
|
|
34
|
+
|
|
27
35
|
export function pairedBenchmarkConfidence(results, options = {}) {
|
|
28
36
|
const byKey = new Map()
|
|
29
37
|
for (const item of results || []) {
|
|
@@ -71,25 +79,41 @@ export function pairedBenchmarkConfidence(results, options = {}) {
|
|
|
71
79
|
item.delta = item.uesPassRate - item.baselinePassRate
|
|
72
80
|
}
|
|
73
81
|
|
|
74
|
-
const
|
|
75
|
-
const
|
|
76
|
-
const
|
|
77
|
-
const uesDuration = mean(uesDurations)
|
|
78
|
-
const durationRatio = baselineDuration && uesDuration ? uesDuration / baselineDuration : null
|
|
82
|
+
const baselineDuration = mean(pairs.map((pair) => Number(pair.baseline.durationMs)))
|
|
83
|
+
const uesDuration = mean(pairs.map((pair) => Number(pair.ues.durationMs)))
|
|
84
|
+
const durationRatio = ratio(uesDuration, baselineDuration)
|
|
79
85
|
|
|
80
|
-
const
|
|
81
|
-
const
|
|
82
|
-
const
|
|
83
|
-
|
|
84
|
-
const
|
|
86
|
+
const baselineCost = positiveMean(pairs.map((pair) => pair.baseline.telemetry?.cost))
|
|
87
|
+
const uesCost = positiveMean(pairs.map((pair) => pair.ues.telemetry?.cost))
|
|
88
|
+
const costRatio = ratio(uesCost, baselineCost)
|
|
89
|
+
|
|
90
|
+
const baselineInitialInput = positiveMean(
|
|
91
|
+
pairs.map((pair) => pair.baseline.telemetry?.firstUsage?.input),
|
|
92
|
+
)
|
|
93
|
+
const uesInitialInput = positiveMean(
|
|
94
|
+
pairs.map((pair) => pair.ues.telemetry?.firstUsage?.input),
|
|
95
|
+
)
|
|
96
|
+
const initialInputRatio = ratio(uesInitialInput, baselineInitialInput)
|
|
97
|
+
|
|
98
|
+
const baselineTokens = positiveMean(
|
|
99
|
+
pairs.map((pair) => pair.baseline.telemetry?.tokens?.total),
|
|
100
|
+
)
|
|
101
|
+
const uesTokens = positiveMean(
|
|
102
|
+
pairs.map((pair) => pair.ues.telemetry?.tokens?.total),
|
|
103
|
+
)
|
|
104
|
+
const tokenRatio = ratio(uesTokens, baselineTokens)
|
|
85
105
|
|
|
86
106
|
const minPairs = Math.max(1, Number(options.minPairs || 20))
|
|
87
107
|
const alpha = Math.min(0.5, Math.max(0.0001, Number(options.alpha || 0.05)))
|
|
88
108
|
const minDelta = Math.max(0, Number(options.minDelta || 0))
|
|
89
109
|
const suiteRegressionTolerance = Math.max(0, Number(options.suiteRegressionTolerance || 0))
|
|
90
110
|
const maxDurationRatio = Math.max(1, Number(options.maxDurationRatio || 1.75))
|
|
111
|
+
const maxInitialInputRatio = Math.max(1, Number(options.maxInitialInputRatio || 1.5))
|
|
112
|
+
const maxTokenRatio = Math.max(1, Number(options.maxTokenRatio || 1.75))
|
|
91
113
|
const noSuiteRegression = Object.values(suites).every((item) => item.delta >= -suiteRegressionTolerance)
|
|
92
114
|
const speedAcceptable = durationRatio == null || durationRatio <= maxDurationRatio
|
|
115
|
+
const initialInputAcceptable = initialInputRatio == null || initialInputRatio <= maxInitialInputRatio
|
|
116
|
+
const tokensAcceptable = tokenRatio == null || tokenRatio <= maxTokenRatio
|
|
93
117
|
|
|
94
118
|
const checks = {
|
|
95
119
|
pairedCoverage: total >= minPairs,
|
|
@@ -98,9 +122,11 @@ export function pairedBenchmarkConfidence(results, options = {}) {
|
|
|
98
122
|
statisticallySupported: pValue <= alpha,
|
|
99
123
|
noSuiteRegression,
|
|
100
124
|
speedAcceptable,
|
|
125
|
+
initialInputAcceptable,
|
|
126
|
+
tokensAcceptable,
|
|
101
127
|
}
|
|
102
128
|
return {
|
|
103
|
-
schemaVersion:
|
|
129
|
+
schemaVersion: 2,
|
|
104
130
|
kind: "ues-paired-benchmark-confidence",
|
|
105
131
|
pairs: total,
|
|
106
132
|
bothPass,
|
|
@@ -123,6 +149,18 @@ export function pairedBenchmarkConfidence(results, options = {}) {
|
|
|
123
149
|
ratio: durationRatio,
|
|
124
150
|
maxRatio: maxDurationRatio,
|
|
125
151
|
},
|
|
152
|
+
initialInput: {
|
|
153
|
+
baselineMeanTokens: baselineInitialInput,
|
|
154
|
+
uesMeanTokens: uesInitialInput,
|
|
155
|
+
ratio: initialInputRatio,
|
|
156
|
+
maxRatio: maxInitialInputRatio,
|
|
157
|
+
},
|
|
158
|
+
tokens: {
|
|
159
|
+
baselineMean: baselineTokens,
|
|
160
|
+
uesMean: uesTokens,
|
|
161
|
+
ratio: tokenRatio,
|
|
162
|
+
maxRatio: maxTokenRatio,
|
|
163
|
+
},
|
|
126
164
|
cost: {
|
|
127
165
|
baselineMean: baselineCost,
|
|
128
166
|
uesMean: uesCost,
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
function finite(value) {
|
|
2
|
+
if (value === null || value === undefined || value === "") return null
|
|
3
|
+
const number = Number(value)
|
|
4
|
+
return Number.isFinite(number) ? number : null
|
|
5
|
+
}
|
|
6
|
+
|
|
7
|
+
function modeSummary(value) {
|
|
8
|
+
return value?.summary?.modes?.ues || value?.modes?.ues || null
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
function ratio(candidate, reference) {
|
|
12
|
+
return candidate != null && reference != null && reference > 0
|
|
13
|
+
? candidate / reference
|
|
14
|
+
: null
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
function delta(candidate, reference) {
|
|
18
|
+
return candidate != null && reference != null ? candidate - reference : null
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
export function compareEvalSummaries(reference, candidate, options = {}) {
|
|
22
|
+
const ref = modeSummary(reference)
|
|
23
|
+
const next = modeSummary(candidate)
|
|
24
|
+
if (!ref || !next) throw new Error("reference and candidate must contain a UES mode summary")
|
|
25
|
+
|
|
26
|
+
const passRateTolerance = Math.max(0, Number(options.passRateTolerance || 0))
|
|
27
|
+
const minInitialInputReduction = Math.max(0, Number(options.minInitialInputReduction ?? 0.10))
|
|
28
|
+
const maxTotalTokenRatio = Math.max(0.01, Number(options.maxTotalTokenRatio || 1.05))
|
|
29
|
+
const maxDurationRatio = Math.max(0.01, Number(options.maxDurationRatio || 1.10))
|
|
30
|
+
|
|
31
|
+
const referencePassRate = finite(ref.passRate)
|
|
32
|
+
const candidatePassRate = finite(next.passRate)
|
|
33
|
+
const referenceInitialInput = finite(ref.avgInitialInputTokens)
|
|
34
|
+
const candidateInitialInput = finite(next.avgInitialInputTokens)
|
|
35
|
+
const referenceTokens = finite(ref.avgTokens)
|
|
36
|
+
const candidateTokens = finite(next.avgTokens)
|
|
37
|
+
const referenceDuration = finite(ref.avgDurationMs)
|
|
38
|
+
const candidateDuration = finite(next.avgDurationMs)
|
|
39
|
+
|
|
40
|
+
const initialInputRatio = ratio(candidateInitialInput, referenceInitialInput)
|
|
41
|
+
const tokenRatio = ratio(candidateTokens, referenceTokens)
|
|
42
|
+
const durationRatio = ratio(candidateDuration, referenceDuration)
|
|
43
|
+
|
|
44
|
+
const checks = {
|
|
45
|
+
passRatePreserved:
|
|
46
|
+
referencePassRate != null &&
|
|
47
|
+
candidatePassRate != null &&
|
|
48
|
+
candidatePassRate >= referencePassRate - passRateTolerance,
|
|
49
|
+
initialInputReduced:
|
|
50
|
+
initialInputRatio == null
|
|
51
|
+
? null
|
|
52
|
+
: initialInputRatio <= 1 - minInitialInputReduction,
|
|
53
|
+
totalTokensBounded:
|
|
54
|
+
tokenRatio == null ? null : tokenRatio <= maxTotalTokenRatio,
|
|
55
|
+
durationBounded:
|
|
56
|
+
durationRatio == null ? null : durationRatio <= maxDurationRatio,
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
const efficiencyEvidence = [
|
|
60
|
+
checks.initialInputReduced,
|
|
61
|
+
checks.totalTokensBounded,
|
|
62
|
+
checks.durationBounded,
|
|
63
|
+
].filter((value) => value !== null)
|
|
64
|
+
|
|
65
|
+
return {
|
|
66
|
+
schemaVersion: 1,
|
|
67
|
+
kind: "ues-eval-ablation",
|
|
68
|
+
thresholds: {
|
|
69
|
+
passRateTolerance,
|
|
70
|
+
minInitialInputReduction,
|
|
71
|
+
maxTotalTokenRatio,
|
|
72
|
+
maxDurationRatio,
|
|
73
|
+
},
|
|
74
|
+
reference: {
|
|
75
|
+
passRate: referencePassRate,
|
|
76
|
+
avgInitialInputTokens: referenceInitialInput,
|
|
77
|
+
avgTokens: referenceTokens,
|
|
78
|
+
avgDurationMs: referenceDuration,
|
|
79
|
+
},
|
|
80
|
+
candidate: {
|
|
81
|
+
passRate: candidatePassRate,
|
|
82
|
+
avgInitialInputTokens: candidateInitialInput,
|
|
83
|
+
avgTokens: candidateTokens,
|
|
84
|
+
avgDurationMs: candidateDuration,
|
|
85
|
+
},
|
|
86
|
+
delta: {
|
|
87
|
+
passRate: delta(candidatePassRate, referencePassRate),
|
|
88
|
+
avgInitialInputTokens: delta(candidateInitialInput, referenceInitialInput),
|
|
89
|
+
avgTokens: delta(candidateTokens, referenceTokens),
|
|
90
|
+
avgDurationMs: delta(candidateDuration, referenceDuration),
|
|
91
|
+
},
|
|
92
|
+
ratios: {
|
|
93
|
+
initialInput: initialInputRatio,
|
|
94
|
+
totalTokens: tokenRatio,
|
|
95
|
+
duration: durationRatio,
|
|
96
|
+
},
|
|
97
|
+
checks,
|
|
98
|
+
telemetrySufficient: checks.initialInputReduced !== null,
|
|
99
|
+
gateEligible:
|
|
100
|
+
checks.passRatePreserved === true &&
|
|
101
|
+
checks.initialInputReduced === true &&
|
|
102
|
+
efficiencyEvidence.every((value) => value === true),
|
|
103
|
+
}
|
|
104
|
+
}
|
package/lib/eval-report.mjs
CHANGED
|
@@ -10,6 +10,8 @@ export function summarizeEvalResults(results) {
|
|
|
10
10
|
toolSamples: 0,
|
|
11
11
|
tokens: 0,
|
|
12
12
|
tokenSamples: 0,
|
|
13
|
+
initialInputTokens: 0,
|
|
14
|
+
initialInputSamples: 0,
|
|
13
15
|
cost: 0,
|
|
14
16
|
costSamples: 0,
|
|
15
17
|
}
|
|
@@ -25,6 +27,11 @@ export function summarizeEvalResults(results) {
|
|
|
25
27
|
bucket.tokens += Number(item.telemetry?.tokens?.total) || 0
|
|
26
28
|
bucket.tokenSamples += 1
|
|
27
29
|
}
|
|
30
|
+
const initialInput = Number(item.telemetry?.firstUsage?.input)
|
|
31
|
+
if (Number.isFinite(initialInput)) {
|
|
32
|
+
bucket.initialInputTokens += initialInput
|
|
33
|
+
bucket.initialInputSamples += 1
|
|
34
|
+
}
|
|
28
35
|
if ((item.telemetry?.costSamples || 0) > 0) {
|
|
29
36
|
bucket.cost += Number(item.telemetry?.cost) || 0
|
|
30
37
|
bucket.costSamples += 1
|
|
@@ -36,10 +43,12 @@ export function summarizeEvalResults(results) {
|
|
|
36
43
|
bucket.avgDurationMs = bucket.total ? bucket.durationMs / bucket.total : 0
|
|
37
44
|
bucket.avgToolCalls = bucket.toolSamples ? bucket.toolCalls / bucket.toolSamples : null
|
|
38
45
|
bucket.avgTokens = bucket.tokenSamples ? bucket.tokens / bucket.tokenSamples : null
|
|
46
|
+
bucket.avgInitialInputTokens = bucket.initialInputSamples ? bucket.initialInputTokens / bucket.initialInputSamples : null
|
|
39
47
|
bucket.avgCost = bucket.costSamples ? bucket.cost / bucket.costSamples : null
|
|
40
48
|
bucket.telemetryCoverage = {
|
|
41
49
|
tools: bucket.total ? bucket.toolSamples / bucket.total : 0,
|
|
42
50
|
tokens: bucket.total ? bucket.tokenSamples / bucket.total : 0,
|
|
51
|
+
initialInputTokens: bucket.total ? bucket.initialInputSamples / bucket.total : 0,
|
|
43
52
|
cost: bucket.total ? bucket.costSamples / bucket.total : 0,
|
|
44
53
|
}
|
|
45
54
|
delete bucket.durationMs
|
|
@@ -47,6 +56,8 @@ export function summarizeEvalResults(results) {
|
|
|
47
56
|
delete bucket.toolSamples
|
|
48
57
|
delete bucket.tokens
|
|
49
58
|
delete bucket.tokenSamples
|
|
59
|
+
delete bucket.initialInputTokens
|
|
60
|
+
delete bucket.initialInputSamples
|
|
50
61
|
delete bucket.cost
|
|
51
62
|
delete bucket.costSamples
|
|
52
63
|
}
|
package/lib/eval-telemetry.mjs
CHANGED
|
@@ -92,6 +92,7 @@ export function parseOpenCodeTelemetry(stdout) {
|
|
|
92
92
|
let cost = 0
|
|
93
93
|
let usageSamples = 0
|
|
94
94
|
let costSamples = 0
|
|
95
|
+
let firstUsage = null
|
|
95
96
|
|
|
96
97
|
parsed.forEach((event, eventIndex) => {
|
|
97
98
|
let eventUsage = null
|
|
@@ -130,6 +131,7 @@ export function parseOpenCodeTelemetry(stdout) {
|
|
|
130
131
|
})
|
|
131
132
|
|
|
132
133
|
if (eventUsage) {
|
|
134
|
+
if (!firstUsage && eventUsage.input > 0) firstUsage = { ...eventUsage }
|
|
133
135
|
usageSamples += 1
|
|
134
136
|
tokens.input += eventUsage.input
|
|
135
137
|
tokens.output += eventUsage.output
|
|
@@ -151,6 +153,7 @@ export function parseOpenCodeTelemetry(stdout) {
|
|
|
151
153
|
skillsLoaded: [...skills].sort(),
|
|
152
154
|
subagents: [...subagents].sort(),
|
|
153
155
|
tokens,
|
|
156
|
+
firstUsage,
|
|
154
157
|
usageSamples,
|
|
155
158
|
cost,
|
|
156
159
|
costSamples,
|
package/lib/model-policy.mjs
CHANGED
|
@@ -58,11 +58,17 @@ export function resolveAdaptiveModel(role, attempt, taskPolicy = {}, config = {}
|
|
|
58
58
|
...config,
|
|
59
59
|
roleTiers: { ...(config.roleTiers || {}), [role]: baseTier },
|
|
60
60
|
})
|
|
61
|
+
const normalizedAttempt = Math.max(1, Number(attempt || 1))
|
|
61
62
|
return {
|
|
62
63
|
...resolved,
|
|
64
|
+
recoveryStage:
|
|
65
|
+
normalizedAttempt <= 1 ? "initial" :
|
|
66
|
+
normalizedAttempt === 2 ? "diagnose" :
|
|
67
|
+
"deep-recovery",
|
|
63
68
|
policy: {
|
|
64
69
|
mode: taskPolicy.mode || null,
|
|
65
70
|
risk: taskPolicy.risk || null,
|
|
71
|
+
executionProfile: taskPolicy.executionProfile || null,
|
|
66
72
|
score: Number(taskPolicy.score || 0),
|
|
67
73
|
maxAttempts: Number(taskPolicy.maxAttempts || 0) || null,
|
|
68
74
|
contextBudget: Number(taskPolicy.contextBudget || 0) || null,
|