opencode-agent-skill 9.0.0 → 10.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,265 @@
1
+ import { createHash } from "node:crypto"
2
+
3
+ const EXPLORATION_TOOL = /(?:^|[._-])(read|grep|glob|repo[-_.]?graph|semantic[-_.]?search|aci[-_.]?search)(?:$|[._-])/i
4
+ const VERIFY_OR_WRITE_TOOL = /(?:^|[._-])(edit|write|patch|bash|shell|verify|test|apply)(?:$|[._-])/i
5
+
6
+ function canonical(value) {
7
+ if (Array.isArray(value)) return value.map(canonical)
8
+ if (!value || typeof value !== "object") return value
9
+ return Object.fromEntries(
10
+ Object.keys(value).sort().map((key) => [key, canonical(value[key])]),
11
+ )
12
+ }
13
+
14
+ export function stableRuntimeHash(value) {
15
+ const text = typeof value === "string" ? value : JSON.stringify(canonical(value))
16
+ return createHash("sha256").update(text || "").digest("hex")
17
+ }
18
+
19
+ export function isExplorationTool(tool) {
20
+ return EXPLORATION_TOOL.test(String(tool || ""))
21
+ }
22
+
23
+ export function budgetToolResult(tool, result, options = {}) {
24
+ const name = String(tool || "")
25
+ const aggressive = /grep|glob|repo[-_.]?graph|semantic[-_.]?search/i.test(name)
26
+ const maxChars = Math.max(2_000, Number(options.maxChars || (aggressive ? 12_000 : 24_000)))
27
+ const maxLines = Math.max(40, Number(options.maxLines || (aggressive ? 160 : 360)))
28
+
29
+ if (!aggressive && !/read/i.test(name)) return result
30
+
31
+ const original = typeof result === "string" ? result : String(result?.output || "")
32
+ const lines = original.split(/\r?\n/)
33
+ if (original.length <= maxChars && lines.length <= maxLines) return result
34
+
35
+ const digest = stableRuntimeHash(original)
36
+ const headLineCount = Math.max(1, Math.floor(maxLines * 0.7))
37
+ const tailLineCount = Math.max(1, maxLines - headLineCount)
38
+ let bounded = [
39
+ ...lines.slice(0, headLineCount),
40
+ `...[UES tool output truncated: ${lines.length} lines / ${original.length} chars, sha256=${digest.slice(0, 16)}]...`,
41
+ ...lines.slice(-tailLineCount),
42
+ ].join("\n")
43
+
44
+ if (bounded.length > maxChars) {
45
+ const marker = `\n...[UES char budget applied; sha256=${digest.slice(0, 16)}]...\n`
46
+ const room = Math.max(0, maxChars - marker.length)
47
+ const head = Math.floor(room * 0.7)
48
+ bounded = bounded.slice(0, head) + marker + bounded.slice(-(room - head))
49
+ }
50
+
51
+ if (typeof result === "string") return bounded
52
+ return {
53
+ ...result,
54
+ output: bounded,
55
+ metadata: {
56
+ ...(result?.metadata || {}),
57
+ uesTruncated: true,
58
+ uesOriginalChars: original.length,
59
+ uesOriginalLines: lines.length,
60
+ uesOutputDigest: digest,
61
+ },
62
+ }
63
+ }
64
+
65
+ export function classifyProviderFailure(error = {}) {
66
+ const status = Number(error?.status ?? error?.cause?.status)
67
+ const type = String(error?.type || error?.code || "")
68
+ const message = String(error?.message || error || "")
69
+ const value = `${type} ${message}`.toLowerCase()
70
+
71
+ if (status === 401 || status === 403 || /auth|unauthori[sz]ed|invalid api key|credential/.test(value)) return "AUTH"
72
+ if (/context.{0,20}(large|length|window|overflow)|too many tokens|max(?:imum)? context/.test(value)) return "CONTEXT_TOO_LARGE"
73
+ if (status === 429 || /rate.?limit|too many requests/.test(value)) return "RATE_LIMIT"
74
+ if (/quota|credit|billing|insufficient balance/.test(value)) return "QUOTA"
75
+ if (/no token|no output|empty response|empty completion|returned no content|no content|stalled|no-progress/.test(value)) return "NO_TOKEN"
76
+ if (/timeout|timed out|deadline|abort(?:ed)?/.test(value)) return "TIMEOUT"
77
+ if (Number.isFinite(status) && status >= 500 && status < 600) return "PROVIDER_5XX"
78
+ if (/provider|upstream|gateway|service unavailable/.test(value)) return "PROVIDER_5XX"
79
+ return "OTHER"
80
+ }
81
+
82
+ export function progressWatchdogDecision(snapshot = {}, now = Date.now(), stallMs = 60_000) {
83
+ const limit = Math.max(30_000, Math.min(Number(stallMs || 60_000), 5 * 60_000))
84
+ const lastProgressAt = Number(snapshot.lastProgressAt || 0)
85
+ const activeToolCalls = Math.max(0, Number(snapshot.activeToolCalls || 0))
86
+ const idleMs = Math.max(0, Number(now) - lastProgressAt)
87
+
88
+ if (activeToolCalls > 0) {
89
+ return { stalled: false, idleMs, limitMs: limit, reason: "tool-active" }
90
+ }
91
+ return {
92
+ stalled: idleMs >= limit,
93
+ idleMs,
94
+ limitMs: limit,
95
+ reason: idleMs >= limit ? "no-progress" : "within-grace",
96
+ }
97
+ }
98
+
99
+ export function providerRecoveryPlan(kind, physicalAttempt = 1, options = {}) {
100
+ const attempt = Math.max(1, Number(physicalAttempt || 1))
101
+ const hasEscalationModel = Boolean(options.hasEscalationModel)
102
+
103
+ if (kind === "AUTH") return { action: "fail-fast", retry: false, reason: "authentication failure requires user/config repair" }
104
+ if (kind === "CONTEXT_TOO_LARGE") return { action: "compact-context", retry: false, reason: "reduce/compact context instead of repeating the same request" }
105
+ if (kind === "QUOTA") {
106
+ return hasEscalationModel
107
+ ? { action: "fresh-session-escalated-model", retry: true, reason: "quota exhausted on current provider/model" }
108
+ : { action: "fail-retryable", retry: false, reason: "quota exhausted and no configured fallback model exists" }
109
+ }
110
+
111
+ if (["NO_TOKEN", "TIMEOUT", "RATE_LIMIT", "PROVIDER_5XX"].includes(kind)) {
112
+ if (attempt <= 1) {
113
+ return { action: "fresh-session-same-model", retry: true, reason: "retry transport/provider stall once with fresh session state" }
114
+ }
115
+ if (hasEscalationModel) {
116
+ return { action: "fresh-session-escalated-model", retry: true, reason: "repeated provider failure triggers configured model/provider escalation" }
117
+ }
118
+ return { action: "fail-retryable", retry: false, reason: "repeated provider failure without configured fallback" }
119
+ }
120
+
121
+ return { action: "fail-task", retry: false, reason: "non-provider failure should be diagnosed by the task recovery policy" }
122
+ }
123
+
124
+ function createSessionState(now = Date.now()) {
125
+ return {
126
+ signatures: new Map(),
127
+ lastWorkspaceSignal: null,
128
+ lastEvidenceKey: null,
129
+ noProgressCalls: 0,
130
+ loopBlocked: false,
131
+ lastProgressAt: now,
132
+ activeCalls: new Set(),
133
+ compactionAt: null,
134
+ }
135
+ }
136
+
137
+ export function createRuntimeGuard(options = {}) {
138
+ const duplicateLimit = Math.max(2, Number(options.duplicateLimit || 3))
139
+ const loopLimit = Math.max(4, Number(options.loopLimit || 6))
140
+ const sessions = new Map()
141
+
142
+ function stateFor(sessionID, at = Date.now()) {
143
+ const key = String(sessionID || "global")
144
+ if (!sessions.has(key)) sessions.set(key, createSessionState(at))
145
+ return sessions.get(key)
146
+ }
147
+
148
+ function resetForWorkspace(state, workspaceSignal) {
149
+ if (state.lastWorkspaceSignal === null || state.lastWorkspaceSignal === workspaceSignal) return
150
+ state.signatures.clear()
151
+ state.noProgressCalls = 0
152
+ state.loopBlocked = false
153
+ state.lastEvidenceKey = null
154
+ }
155
+
156
+ return {
157
+ before(input = {}) {
158
+ const at = Number(input.now || Date.now())
159
+ const state = stateFor(input.sessionID, at)
160
+ const workspaceSignal = String(input.workspaceSignal || "")
161
+ resetForWorkspace(state, workspaceSignal)
162
+ state.lastWorkspaceSignal = workspaceSignal
163
+
164
+ const tool = String(input.tool || "")
165
+ if (isExplorationTool(tool)) {
166
+ const signature = stableRuntimeHash({ tool, input: input.input || {}, cwd: input.cwd || "", workspaceSignal })
167
+ const seen = state.signatures.get(signature) || { count: 0, lastResultHash: null }
168
+ seen.count += 1
169
+ state.signatures.set(signature, seen)
170
+
171
+ if (state.loopBlocked) {
172
+ return {
173
+ blocked: true,
174
+ code: "UES_LOOP_DETECTED",
175
+ message: `UES loop guard blocked repeated exploration after ${state.noProgressCalls} no-progress calls. Choose a deterministic next action: edit the scoped target, run focused verification, inspect a direct caller/failing stack, change hypothesis, or escalate recovery.`,
176
+ signature,
177
+ }
178
+ }
179
+ if (seen.count > duplicateLimit) {
180
+ return {
181
+ blocked: true,
182
+ code: "UES_DUPLICATE_TOOL",
183
+ message: `UES duplicate-tool guard blocked ${tool}: equivalent arguments were already executed ${seen.count - 1} times with no workspace change. Use the previous evidence or change the query/scope before retrying.`,
184
+ signature,
185
+ }
186
+ }
187
+ }
188
+
189
+ state.lastProgressAt = at
190
+ if (input.callID) state.activeCalls.add(String(input.callID))
191
+ return { blocked: false }
192
+ },
193
+
194
+ after(input = {}) {
195
+ const at = Number(input.now || Date.now())
196
+ const state = stateFor(input.sessionID, at)
197
+ const workspaceSignal = String(input.workspaceSignal || "")
198
+ const tool = String(input.tool || "")
199
+ if (input.callID) state.activeCalls.delete(String(input.callID))
200
+ resetForWorkspace(state, workspaceSignal)
201
+
202
+ if (input.status === "completed" && isExplorationTool(tool)) {
203
+ const resultHash = stableRuntimeHash(input.result || "")
204
+ const evidenceKey = stableRuntimeHash({ workspaceSignal, resultHash })
205
+ if (state.lastEvidenceKey === evidenceKey) {
206
+ state.noProgressCalls += 1
207
+ } else {
208
+ state.lastEvidenceKey = evidenceKey
209
+ state.noProgressCalls = 0
210
+ }
211
+ if (state.noProgressCalls >= loopLimit) state.loopBlocked = true
212
+
213
+ const signature = stableRuntimeHash({ tool, input: input.input || {}, cwd: input.cwd || "", workspaceSignal })
214
+ const seen = state.signatures.get(signature)
215
+ if (seen) seen.lastResultHash = resultHash
216
+ } else if (
217
+ input.status === "completed" &&
218
+ (VERIFY_OR_WRITE_TOOL.test(tool) || state.lastWorkspaceSignal !== workspaceSignal)
219
+ ) {
220
+ state.noProgressCalls = 0
221
+ state.loopBlocked = false
222
+ state.lastEvidenceKey = null
223
+ state.signatures.clear()
224
+ }
225
+
226
+ state.lastWorkspaceSignal = workspaceSignal
227
+ state.lastProgressAt = at
228
+ return {
229
+ loopBlocked: state.loopBlocked,
230
+ noProgressCalls: state.noProgressCalls,
231
+ lastProgressAt: state.lastProgressAt,
232
+ }
233
+ },
234
+
235
+ touch(sessionID, at = Date.now()) {
236
+ const state = stateFor(sessionID, at)
237
+ state.lastProgressAt = Number(at)
238
+ },
239
+
240
+ compacted(sessionID, at = Date.now()) {
241
+ const state = stateFor(sessionID, at)
242
+ state.compactionAt = Number(at)
243
+ state.lastProgressAt = Number(at)
244
+ state.signatures.clear()
245
+ state.noProgressCalls = 0
246
+ state.loopBlocked = false
247
+ state.lastEvidenceKey = null
248
+ },
249
+
250
+ snapshot(sessionID) {
251
+ const state = stateFor(sessionID)
252
+ return {
253
+ noProgressCalls: state.noProgressCalls,
254
+ loopBlocked: state.loopBlocked,
255
+ lastProgressAt: state.lastProgressAt,
256
+ activeToolCalls: state.activeCalls.size,
257
+ compactionAt: state.compactionAt,
258
+ }
259
+ },
260
+
261
+ clear(sessionID) {
262
+ sessions.delete(String(sessionID || "global"))
263
+ },
264
+ }
265
+ }
@@ -24,6 +24,14 @@ function mean(values) {
24
24
  return usable.length ? usable.reduce((sum, value) => sum + value, 0) / usable.length : null
25
25
  }
26
26
 
27
+ function positiveMean(values) {
28
+ return mean(values.map(Number).filter((value) => Number.isFinite(value) && value > 0))
29
+ }
30
+
31
+ function ratio(candidate, reference) {
32
+ return reference && candidate ? candidate / reference : null
33
+ }
34
+
27
35
  export function pairedBenchmarkConfidence(results, options = {}) {
28
36
  const byKey = new Map()
29
37
  for (const item of results || []) {
@@ -71,25 +79,41 @@ export function pairedBenchmarkConfidence(results, options = {}) {
71
79
  item.delta = item.uesPassRate - item.baselinePassRate
72
80
  }
73
81
 
74
- const baselineDurations = pairs.map((pair) => Number(pair.baseline.durationMs))
75
- const uesDurations = pairs.map((pair) => Number(pair.ues.durationMs))
76
- const baselineDuration = mean(baselineDurations)
77
- const uesDuration = mean(uesDurations)
78
- const durationRatio = baselineDuration && uesDuration ? uesDuration / baselineDuration : null
82
+ const baselineDuration = mean(pairs.map((pair) => Number(pair.baseline.durationMs)))
83
+ const uesDuration = mean(pairs.map((pair) => Number(pair.ues.durationMs)))
84
+ const durationRatio = ratio(uesDuration, baselineDuration)
79
85
 
80
- const baselineCosts = pairs.map((pair) => Number(pair.baseline.telemetry?.cost)).filter((value) => value > 0)
81
- const uesCosts = pairs.map((pair) => Number(pair.ues.telemetry?.cost)).filter((value) => value > 0)
82
- const baselineCost = mean(baselineCosts)
83
- const uesCost = mean(uesCosts)
84
- const costRatio = baselineCost && uesCost ? uesCost / baselineCost : null
86
+ const baselineCost = positiveMean(pairs.map((pair) => pair.baseline.telemetry?.cost))
87
+ const uesCost = positiveMean(pairs.map((pair) => pair.ues.telemetry?.cost))
88
+ const costRatio = ratio(uesCost, baselineCost)
89
+
90
+ const baselineInitialInput = positiveMean(
91
+ pairs.map((pair) => pair.baseline.telemetry?.firstUsage?.input),
92
+ )
93
+ const uesInitialInput = positiveMean(
94
+ pairs.map((pair) => pair.ues.telemetry?.firstUsage?.input),
95
+ )
96
+ const initialInputRatio = ratio(uesInitialInput, baselineInitialInput)
97
+
98
+ const baselineTokens = positiveMean(
99
+ pairs.map((pair) => pair.baseline.telemetry?.tokens?.total),
100
+ )
101
+ const uesTokens = positiveMean(
102
+ pairs.map((pair) => pair.ues.telemetry?.tokens?.total),
103
+ )
104
+ const tokenRatio = ratio(uesTokens, baselineTokens)
85
105
 
86
106
  const minPairs = Math.max(1, Number(options.minPairs || 20))
87
107
  const alpha = Math.min(0.5, Math.max(0.0001, Number(options.alpha || 0.05)))
88
108
  const minDelta = Math.max(0, Number(options.minDelta || 0))
89
109
  const suiteRegressionTolerance = Math.max(0, Number(options.suiteRegressionTolerance || 0))
90
110
  const maxDurationRatio = Math.max(1, Number(options.maxDurationRatio || 1.75))
111
+ const maxInitialInputRatio = Math.max(1, Number(options.maxInitialInputRatio || 1.5))
112
+ const maxTokenRatio = Math.max(1, Number(options.maxTokenRatio || 1.75))
91
113
  const noSuiteRegression = Object.values(suites).every((item) => item.delta >= -suiteRegressionTolerance)
92
114
  const speedAcceptable = durationRatio == null || durationRatio <= maxDurationRatio
115
+ const initialInputAcceptable = initialInputRatio == null || initialInputRatio <= maxInitialInputRatio
116
+ const tokensAcceptable = tokenRatio == null || tokenRatio <= maxTokenRatio
93
117
 
94
118
  const checks = {
95
119
  pairedCoverage: total >= minPairs,
@@ -98,9 +122,11 @@ export function pairedBenchmarkConfidence(results, options = {}) {
98
122
  statisticallySupported: pValue <= alpha,
99
123
  noSuiteRegression,
100
124
  speedAcceptable,
125
+ initialInputAcceptable,
126
+ tokensAcceptable,
101
127
  }
102
128
  return {
103
- schemaVersion: 1,
129
+ schemaVersion: 2,
104
130
  kind: "ues-paired-benchmark-confidence",
105
131
  pairs: total,
106
132
  bothPass,
@@ -123,6 +149,18 @@ export function pairedBenchmarkConfidence(results, options = {}) {
123
149
  ratio: durationRatio,
124
150
  maxRatio: maxDurationRatio,
125
151
  },
152
+ initialInput: {
153
+ baselineMeanTokens: baselineInitialInput,
154
+ uesMeanTokens: uesInitialInput,
155
+ ratio: initialInputRatio,
156
+ maxRatio: maxInitialInputRatio,
157
+ },
158
+ tokens: {
159
+ baselineMean: baselineTokens,
160
+ uesMean: uesTokens,
161
+ ratio: tokenRatio,
162
+ maxRatio: maxTokenRatio,
163
+ },
126
164
  cost: {
127
165
  baselineMean: baselineCost,
128
166
  uesMean: uesCost,
@@ -0,0 +1,104 @@
1
+ function finite(value) {
2
+ if (value === null || value === undefined || value === "") return null
3
+ const number = Number(value)
4
+ return Number.isFinite(number) ? number : null
5
+ }
6
+
7
+ function modeSummary(value) {
8
+ return value?.summary?.modes?.ues || value?.modes?.ues || null
9
+ }
10
+
11
+ function ratio(candidate, reference) {
12
+ return candidate != null && reference != null && reference > 0
13
+ ? candidate / reference
14
+ : null
15
+ }
16
+
17
+ function delta(candidate, reference) {
18
+ return candidate != null && reference != null ? candidate - reference : null
19
+ }
20
+
21
+ export function compareEvalSummaries(reference, candidate, options = {}) {
22
+ const ref = modeSummary(reference)
23
+ const next = modeSummary(candidate)
24
+ if (!ref || !next) throw new Error("reference and candidate must contain a UES mode summary")
25
+
26
+ const passRateTolerance = Math.max(0, Number(options.passRateTolerance || 0))
27
+ const minInitialInputReduction = Math.max(0, Number(options.minInitialInputReduction ?? 0.10))
28
+ const maxTotalTokenRatio = Math.max(0.01, Number(options.maxTotalTokenRatio || 1.05))
29
+ const maxDurationRatio = Math.max(0.01, Number(options.maxDurationRatio || 1.10))
30
+
31
+ const referencePassRate = finite(ref.passRate)
32
+ const candidatePassRate = finite(next.passRate)
33
+ const referenceInitialInput = finite(ref.avgInitialInputTokens)
34
+ const candidateInitialInput = finite(next.avgInitialInputTokens)
35
+ const referenceTokens = finite(ref.avgTokens)
36
+ const candidateTokens = finite(next.avgTokens)
37
+ const referenceDuration = finite(ref.avgDurationMs)
38
+ const candidateDuration = finite(next.avgDurationMs)
39
+
40
+ const initialInputRatio = ratio(candidateInitialInput, referenceInitialInput)
41
+ const tokenRatio = ratio(candidateTokens, referenceTokens)
42
+ const durationRatio = ratio(candidateDuration, referenceDuration)
43
+
44
+ const checks = {
45
+ passRatePreserved:
46
+ referencePassRate != null &&
47
+ candidatePassRate != null &&
48
+ candidatePassRate >= referencePassRate - passRateTolerance,
49
+ initialInputReduced:
50
+ initialInputRatio == null
51
+ ? null
52
+ : initialInputRatio <= 1 - minInitialInputReduction,
53
+ totalTokensBounded:
54
+ tokenRatio == null ? null : tokenRatio <= maxTotalTokenRatio,
55
+ durationBounded:
56
+ durationRatio == null ? null : durationRatio <= maxDurationRatio,
57
+ }
58
+
59
+ const efficiencyEvidence = [
60
+ checks.initialInputReduced,
61
+ checks.totalTokensBounded,
62
+ checks.durationBounded,
63
+ ].filter((value) => value !== null)
64
+
65
+ return {
66
+ schemaVersion: 1,
67
+ kind: "ues-eval-ablation",
68
+ thresholds: {
69
+ passRateTolerance,
70
+ minInitialInputReduction,
71
+ maxTotalTokenRatio,
72
+ maxDurationRatio,
73
+ },
74
+ reference: {
75
+ passRate: referencePassRate,
76
+ avgInitialInputTokens: referenceInitialInput,
77
+ avgTokens: referenceTokens,
78
+ avgDurationMs: referenceDuration,
79
+ },
80
+ candidate: {
81
+ passRate: candidatePassRate,
82
+ avgInitialInputTokens: candidateInitialInput,
83
+ avgTokens: candidateTokens,
84
+ avgDurationMs: candidateDuration,
85
+ },
86
+ delta: {
87
+ passRate: delta(candidatePassRate, referencePassRate),
88
+ avgInitialInputTokens: delta(candidateInitialInput, referenceInitialInput),
89
+ avgTokens: delta(candidateTokens, referenceTokens),
90
+ avgDurationMs: delta(candidateDuration, referenceDuration),
91
+ },
92
+ ratios: {
93
+ initialInput: initialInputRatio,
94
+ totalTokens: tokenRatio,
95
+ duration: durationRatio,
96
+ },
97
+ checks,
98
+ telemetrySufficient: checks.initialInputReduced !== null,
99
+ gateEligible:
100
+ checks.passRatePreserved === true &&
101
+ checks.initialInputReduced === true &&
102
+ efficiencyEvidence.every((value) => value === true),
103
+ }
104
+ }
@@ -10,6 +10,8 @@ export function summarizeEvalResults(results) {
10
10
  toolSamples: 0,
11
11
  tokens: 0,
12
12
  tokenSamples: 0,
13
+ initialInputTokens: 0,
14
+ initialInputSamples: 0,
13
15
  cost: 0,
14
16
  costSamples: 0,
15
17
  }
@@ -25,6 +27,11 @@ export function summarizeEvalResults(results) {
25
27
  bucket.tokens += Number(item.telemetry?.tokens?.total) || 0
26
28
  bucket.tokenSamples += 1
27
29
  }
30
+ const initialInput = Number(item.telemetry?.firstUsage?.input)
31
+ if (Number.isFinite(initialInput)) {
32
+ bucket.initialInputTokens += initialInput
33
+ bucket.initialInputSamples += 1
34
+ }
28
35
  if ((item.telemetry?.costSamples || 0) > 0) {
29
36
  bucket.cost += Number(item.telemetry?.cost) || 0
30
37
  bucket.costSamples += 1
@@ -36,10 +43,12 @@ export function summarizeEvalResults(results) {
36
43
  bucket.avgDurationMs = bucket.total ? bucket.durationMs / bucket.total : 0
37
44
  bucket.avgToolCalls = bucket.toolSamples ? bucket.toolCalls / bucket.toolSamples : null
38
45
  bucket.avgTokens = bucket.tokenSamples ? bucket.tokens / bucket.tokenSamples : null
46
+ bucket.avgInitialInputTokens = bucket.initialInputSamples ? bucket.initialInputTokens / bucket.initialInputSamples : null
39
47
  bucket.avgCost = bucket.costSamples ? bucket.cost / bucket.costSamples : null
40
48
  bucket.telemetryCoverage = {
41
49
  tools: bucket.total ? bucket.toolSamples / bucket.total : 0,
42
50
  tokens: bucket.total ? bucket.tokenSamples / bucket.total : 0,
51
+ initialInputTokens: bucket.total ? bucket.initialInputSamples / bucket.total : 0,
43
52
  cost: bucket.total ? bucket.costSamples / bucket.total : 0,
44
53
  }
45
54
  delete bucket.durationMs
@@ -47,6 +56,8 @@ export function summarizeEvalResults(results) {
47
56
  delete bucket.toolSamples
48
57
  delete bucket.tokens
49
58
  delete bucket.tokenSamples
59
+ delete bucket.initialInputTokens
60
+ delete bucket.initialInputSamples
50
61
  delete bucket.cost
51
62
  delete bucket.costSamples
52
63
  }
@@ -92,6 +92,7 @@ export function parseOpenCodeTelemetry(stdout) {
92
92
  let cost = 0
93
93
  let usageSamples = 0
94
94
  let costSamples = 0
95
+ let firstUsage = null
95
96
 
96
97
  parsed.forEach((event, eventIndex) => {
97
98
  let eventUsage = null
@@ -130,6 +131,7 @@ export function parseOpenCodeTelemetry(stdout) {
130
131
  })
131
132
 
132
133
  if (eventUsage) {
134
+ if (!firstUsage && eventUsage.input > 0) firstUsage = { ...eventUsage }
133
135
  usageSamples += 1
134
136
  tokens.input += eventUsage.input
135
137
  tokens.output += eventUsage.output
@@ -151,6 +153,7 @@ export function parseOpenCodeTelemetry(stdout) {
151
153
  skillsLoaded: [...skills].sort(),
152
154
  subagents: [...subagents].sort(),
153
155
  tokens,
156
+ firstUsage,
154
157
  usageSamples,
155
158
  cost,
156
159
  costSamples,
@@ -58,11 +58,17 @@ export function resolveAdaptiveModel(role, attempt, taskPolicy = {}, config = {}
58
58
  ...config,
59
59
  roleTiers: { ...(config.roleTiers || {}), [role]: baseTier },
60
60
  })
61
+ const normalizedAttempt = Math.max(1, Number(attempt || 1))
61
62
  return {
62
63
  ...resolved,
64
+ recoveryStage:
65
+ normalizedAttempt <= 1 ? "initial" :
66
+ normalizedAttempt === 2 ? "diagnose" :
67
+ "deep-recovery",
63
68
  policy: {
64
69
  mode: taskPolicy.mode || null,
65
70
  risk: taskPolicy.risk || null,
71
+ executionProfile: taskPolicy.executionProfile || null,
66
72
  score: Number(taskPolicy.score || 0),
67
73
  maxAttempts: Number(taskPolicy.maxAttempts || 0) || null,
68
74
  contextBudget: Number(taskPolicy.contextBudget || 0) || null,