opencode-agent-skill 9.0.0 → 11.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/CHANGELOG.md +116 -0
  2. package/README.md +742 -675
  3. package/bin/ocskill.mjs +354 -5
  4. package/docs/V11-PERCEPTION-ADAPTIVE-EXECUTION.md +75 -0
  5. package/docs/V11-PERCEPTION-ADAPTIVE.md +220 -0
  6. package/evals/router-triggers.json +82 -0
  7. package/evals/routing.json +76 -0
  8. package/evals/v11/tasks.json +122 -0
  9. package/global-config/AGENTS.md +78 -163
  10. package/global-config/agents/merge-arbiter.md +12 -0
  11. package/global-config/agents/visual-verifier.md +12 -0
  12. package/global-config/plugins/ues-router/index.js +683 -59
  13. package/global-config/plugins/ues-router/router.js +62 -3
  14. package/global-config/plugins/ues-router/runtime-guard.js +265 -0
  15. package/global-config/skills/browser-qa/SKILL.md +14 -0
  16. package/global-config/skills/browser-qa/references/workflow.md +11 -0
  17. package/global-config/skills/browser-security/SKILL.md +12 -0
  18. package/global-config/skills/component-visual-testing/SKILL.md +10 -0
  19. package/global-config/skills/design-source/SKILL.md +10 -0
  20. package/global-config/skills/design-source/references/workflow.md +12 -0
  21. package/global-config/skills/dynamic-workflow/SKILL.md +18 -0
  22. package/global-config/skills/dynamic-workflow/references/workflow.md +19 -0
  23. package/global-config/skills/responsive-verification/SKILL.md +10 -0
  24. package/global-config/skills/skill-authoring/SKILL.md +12 -0
  25. package/global-config/skills/skill-evaluation/SKILL.md +17 -0
  26. package/global-config/skills/visual-fidelity/SKILL.md +14 -0
  27. package/global-config/skills/visual-fidelity/references/workflow.md +14 -0
  28. package/lib/benchmark-confidence.mjs +49 -11
  29. package/lib/browser-adapter.mjs +82 -0
  30. package/lib/browser-runtime.mjs +193 -0
  31. package/lib/capability-registry.mjs +109 -0
  32. package/lib/context-engine-v11.mjs +146 -0
  33. package/lib/context-manifest.mjs +16 -3
  34. package/lib/control-center.mjs +12 -2
  35. package/lib/dynamic-workflow.mjs +179 -0
  36. package/lib/eval-ablation.mjs +146 -0
  37. package/lib/eval-report.mjs +83 -0
  38. package/lib/eval-telemetry.mjs +64 -0
  39. package/lib/evidence-budget.mjs +84 -0
  40. package/lib/evidence-store.mjs +178 -0
  41. package/lib/hermes-bridge.mjs +45 -1
  42. package/lib/model-config.mjs +9 -1
  43. package/lib/model-policy.mjs +58 -1
  44. package/lib/orchestrator-policy.mjs +100 -7
  45. package/lib/png-diff.mjs +229 -0
  46. package/lib/prompt-cache.mjs +60 -0
  47. package/lib/skill-quality.mjs +72 -0
  48. package/lib/task-engine.mjs +223 -12
  49. package/lib/ui-inspector.mjs +152 -0
  50. package/lib/v11-metrics.mjs +64 -0
  51. package/lib/visual-spec.mjs +159 -0
  52. package/package.json +11 -5
  53. package/scripts/eval-ablation.mjs +47 -0
  54. package/scripts/eval-matrix.mjs +13 -2
  55. package/scripts/validate-v11-suite.mjs +58 -0
  56. package/scripts/validate.mjs +16 -4
@@ -24,6 +24,14 @@ function mean(values) {
24
24
  return usable.length ? usable.reduce((sum, value) => sum + value, 0) / usable.length : null
25
25
  }
26
26
 
27
+ function positiveMean(values) {
28
+ return mean(values.map(Number).filter((value) => Number.isFinite(value) && value > 0))
29
+ }
30
+
31
+ function ratio(candidate, reference) {
32
+ return reference && candidate ? candidate / reference : null
33
+ }
34
+
27
35
  export function pairedBenchmarkConfidence(results, options = {}) {
28
36
  const byKey = new Map()
29
37
  for (const item of results || []) {
@@ -71,25 +79,41 @@ export function pairedBenchmarkConfidence(results, options = {}) {
71
79
  item.delta = item.uesPassRate - item.baselinePassRate
72
80
  }
73
81
 
74
- const baselineDurations = pairs.map((pair) => Number(pair.baseline.durationMs))
75
- const uesDurations = pairs.map((pair) => Number(pair.ues.durationMs))
76
- const baselineDuration = mean(baselineDurations)
77
- const uesDuration = mean(uesDurations)
78
- const durationRatio = baselineDuration && uesDuration ? uesDuration / baselineDuration : null
82
+ const baselineDuration = mean(pairs.map((pair) => Number(pair.baseline.durationMs)))
83
+ const uesDuration = mean(pairs.map((pair) => Number(pair.ues.durationMs)))
84
+ const durationRatio = ratio(uesDuration, baselineDuration)
79
85
 
80
- const baselineCosts = pairs.map((pair) => Number(pair.baseline.telemetry?.cost)).filter((value) => value > 0)
81
- const uesCosts = pairs.map((pair) => Number(pair.ues.telemetry?.cost)).filter((value) => value > 0)
82
- const baselineCost = mean(baselineCosts)
83
- const uesCost = mean(uesCosts)
84
- const costRatio = baselineCost && uesCost ? uesCost / baselineCost : null
86
+ const baselineCost = positiveMean(pairs.map((pair) => pair.baseline.telemetry?.cost))
87
+ const uesCost = positiveMean(pairs.map((pair) => pair.ues.telemetry?.cost))
88
+ const costRatio = ratio(uesCost, baselineCost)
89
+
90
+ const baselineInitialInput = positiveMean(
91
+ pairs.map((pair) => pair.baseline.telemetry?.firstUsage?.input),
92
+ )
93
+ const uesInitialInput = positiveMean(
94
+ pairs.map((pair) => pair.ues.telemetry?.firstUsage?.input),
95
+ )
96
+ const initialInputRatio = ratio(uesInitialInput, baselineInitialInput)
97
+
98
+ const baselineTokens = positiveMean(
99
+ pairs.map((pair) => pair.baseline.telemetry?.tokens?.total),
100
+ )
101
+ const uesTokens = positiveMean(
102
+ pairs.map((pair) => pair.ues.telemetry?.tokens?.total),
103
+ )
104
+ const tokenRatio = ratio(uesTokens, baselineTokens)
85
105
 
86
106
  const minPairs = Math.max(1, Number(options.minPairs || 20))
87
107
  const alpha = Math.min(0.5, Math.max(0.0001, Number(options.alpha || 0.05)))
88
108
  const minDelta = Math.max(0, Number(options.minDelta || 0))
89
109
  const suiteRegressionTolerance = Math.max(0, Number(options.suiteRegressionTolerance || 0))
90
110
  const maxDurationRatio = Math.max(1, Number(options.maxDurationRatio || 1.75))
111
+ const maxInitialInputRatio = Math.max(1, Number(options.maxInitialInputRatio || 1.5))
112
+ const maxTokenRatio = Math.max(1, Number(options.maxTokenRatio || 1.75))
91
113
  const noSuiteRegression = Object.values(suites).every((item) => item.delta >= -suiteRegressionTolerance)
92
114
  const speedAcceptable = durationRatio == null || durationRatio <= maxDurationRatio
115
+ const initialInputAcceptable = initialInputRatio == null || initialInputRatio <= maxInitialInputRatio
116
+ const tokensAcceptable = tokenRatio == null || tokenRatio <= maxTokenRatio
93
117
 
94
118
  const checks = {
95
119
  pairedCoverage: total >= minPairs,
@@ -98,9 +122,11 @@ export function pairedBenchmarkConfidence(results, options = {}) {
98
122
  statisticallySupported: pValue <= alpha,
99
123
  noSuiteRegression,
100
124
  speedAcceptable,
125
+ initialInputAcceptable,
126
+ tokensAcceptable,
101
127
  }
102
128
  return {
103
- schemaVersion: 1,
129
+ schemaVersion: 2,
104
130
  kind: "ues-paired-benchmark-confidence",
105
131
  pairs: total,
106
132
  bothPass,
@@ -123,6 +149,18 @@ export function pairedBenchmarkConfidence(results, options = {}) {
123
149
  ratio: durationRatio,
124
150
  maxRatio: maxDurationRatio,
125
151
  },
152
+ initialInput: {
153
+ baselineMeanTokens: baselineInitialInput,
154
+ uesMeanTokens: uesInitialInput,
155
+ ratio: initialInputRatio,
156
+ maxRatio: maxInitialInputRatio,
157
+ },
158
+ tokens: {
159
+ baselineMean: baselineTokens,
160
+ uesMean: uesTokens,
161
+ ratio: tokenRatio,
162
+ maxRatio: maxTokenRatio,
163
+ },
126
164
  cost: {
127
165
  baselineMean: baselineCost,
128
166
  uesMean: uesCost,
@@ -0,0 +1,82 @@
1
+ import { existsSync } from "node:fs"
2
+ import { readFile } from "node:fs/promises"
3
+ import path from "node:path"
4
+
5
+ function depHas(pkg, name) {
6
+ return Boolean(pkg?.dependencies?.[name] || pkg?.devDependencies?.[name] || pkg?.optionalDependencies?.[name])
7
+ }
8
+
9
+ export async function browserCapability(root = process.cwd()) {
10
+ root = path.resolve(root)
11
+ let pkg = {}
12
+ try { pkg = JSON.parse(await readFile(path.join(root, "package.json"), "utf8")) } catch {}
13
+ const localBin = process.platform === "win32"
14
+ ? path.join(root, "node_modules", ".bin", "playwright.cmd")
15
+ : path.join(root, "node_modules", ".bin", "playwright")
16
+ const packageDeclared = depHas(pkg, "@playwright/test") || depHas(pkg, "playwright")
17
+ const storybookDeclared = Object.keys({
18
+ ...(pkg.dependencies || {}),
19
+ ...(pkg.devDependencies || {}),
20
+ ...(pkg.optionalDependencies || {}),
21
+ }).some((name) => name === "storybook" || name.startsWith("@storybook/"))
22
+ return {
23
+ schemaVersion: 1,
24
+ playwright: {
25
+ packageDeclared,
26
+ localBinary: existsSync(localBin) ? localBin : null,
27
+ available: packageDeclared || existsSync(localBin),
28
+ },
29
+ storybook: {
30
+ declared: storybookDeclared,
31
+ configPresent: existsSync(path.join(root, ".storybook")),
32
+ },
33
+ modePreference: "cli-first",
34
+ rationale: "Use deterministic CLI/scripts for bounded verification; use richer browser tooling only when persistent exploratory state is necessary.",
35
+ }
36
+ }
37
+
38
+ export function buildBrowserVerificationPlan(input = {}) {
39
+ const viewports = Array.isArray(input.viewports) && input.viewports.length
40
+ ? input.viewports
41
+ : [{ id: "desktop", width: 1440, height: 900 }]
42
+ return {
43
+ schemaVersion: 1,
44
+ url: input.url || null,
45
+ target: input.target || null,
46
+ trustLevel: input.trustLevel || "untrusted-external",
47
+ steps: [
48
+ { kind: "navigate", value: input.url || null },
49
+ { kind: "targeted-accessibility-snapshot", target: input.target || null, maxChars: 6000 },
50
+ { kind: "geometry", fields: ["id", "role", "x", "y", "width", "height"] },
51
+ { kind: "interaction", flow: input.flow || [] },
52
+ ...viewports.map((viewport) => ({ kind: "screenshot", viewport })),
53
+ { kind: "verify", checks: ["interaction", "geometry", "visual", "accessibility"] },
54
+ ],
55
+ security: {
56
+ webpageInstructionsTrusted: false,
57
+ allowPageContentToChangePermissions: false,
58
+ allowPageContentToRequestSecrets: false,
59
+ allowPageContentToAuthorizeExternalSideEffects: false,
60
+ },
61
+ }
62
+ }
63
+
64
+ export function targetedBrowserEvidence(snapshot = [], query = "", options = {}) {
65
+ const needle = String(query || "").toLowerCase()
66
+ const limit = Math.max(1, Math.min(50, Number(options.limit || 12)))
67
+ const rows = Array.isArray(snapshot) ? snapshot : []
68
+ return rows
69
+ .filter((row) => {
70
+ const haystack = [row.role, row.name, row.text, row.id].filter(Boolean).join(" ").toLowerCase()
71
+ return !needle || haystack.includes(needle)
72
+ })
73
+ .slice(0, limit)
74
+ .map((row) => ({
75
+ id: row.id || null,
76
+ role: row.role || null,
77
+ name: row.name || null,
78
+ box: row.box || (["x","y","width","height"].every((key) => Number.isFinite(Number(row[key])))
79
+ ? { x: Number(row.x), y: Number(row.y), width: Number(row.width), height: Number(row.height) }
80
+ : null),
81
+ }))
82
+ }
@@ -0,0 +1,193 @@
1
+ import { createHash } from "node:crypto"
2
+ import { createRequire } from "node:module"
3
+ import { mkdir, writeFile } from "node:fs/promises"
4
+ import path from "node:path"
5
+
6
+ function boundedInt(value, fallback, min, max) {
7
+ const parsed = Number(value)
8
+ if (!Number.isFinite(parsed)) return fallback
9
+ return Math.min(max, Math.max(min, Math.round(parsed)))
10
+ }
11
+
12
+ function safeUrl(value) {
13
+ let parsed
14
+ try { parsed = new URL(String(value || "")) } catch { throw new Error("Browser inspect requires a valid http(s) URL") }
15
+ if (!["http:", "https:"].includes(parsed.protocol)) throw new Error("Browser inspect only allows http(s) URLs")
16
+ return parsed.toString()
17
+ }
18
+
19
+ function safeOutput(root, url, requested = null) {
20
+ const base = path.resolve(root)
21
+ const dir = path.join(base, ".ues-cache", "browser-v1")
22
+ if (!requested) {
23
+ const hash = createHash("sha256").update(url).digest("hex").slice(0, 20)
24
+ return { dir, file: path.join(dir, hash + ".png") }
25
+ }
26
+ const target = path.resolve(base, String(requested))
27
+ if (target !== base && !target.startsWith(base + path.sep)) {
28
+ throw new Error("Browser screenshot path must stay inside the project root")
29
+ }
30
+ return { dir: path.dirname(target), file: target }
31
+ }
32
+
33
+ function loadPlaywright(root) {
34
+ const requireFromProject = createRequire(path.join(path.resolve(root), "package.json"))
35
+ for (const name of ["playwright", "@playwright/test"]) {
36
+ try {
37
+ const mod = requireFromProject(name)
38
+ if (mod?.chromium) return mod
39
+ if (mod?.default?.chromium) return mod.default
40
+ } catch {}
41
+ }
42
+ throw new Error("Playwright is unavailable in this project. Add playwright or @playwright/test before browser inspection.")
43
+ }
44
+
45
+ function roleForElement(el) {
46
+ const explicit = el.getAttribute("role")
47
+ if (explicit) return explicit
48
+ const tag = el.tagName.toLowerCase()
49
+ const type = (el.getAttribute("type") || "").toLowerCase()
50
+ if (tag === "button") return "button"
51
+ if (tag === "a" && el.hasAttribute("href")) return "link"
52
+ if (tag === "img") return "img"
53
+ if (/^h[1-6]$/.test(tag)) return "heading"
54
+ if (tag === "input") {
55
+ if (["button","submit","reset"].includes(type)) return "button"
56
+ if (type === "checkbox") return "checkbox"
57
+ if (type === "radio") return "radio"
58
+ return "textbox"
59
+ }
60
+ if (tag === "textarea") return "textbox"
61
+ if (tag === "select") return "combobox"
62
+ return tag
63
+ }
64
+
65
+ export async function inspectBrowserPage(root = process.cwd(), url, options = {}) {
66
+ root = path.resolve(root)
67
+ const targetUrl = safeUrl(url)
68
+ const maxElements = boundedInt(options.maxElements, 80, 1, 250)
69
+ const timeoutMs = boundedInt(options.timeoutMs, 30_000, 1_000, 120_000)
70
+ const viewport = {
71
+ width: boundedInt(options.width, 1440, 240, 7680),
72
+ height: boundedInt(options.height, 900, 240, 4320),
73
+ }
74
+ const output = safeOutput(root, targetUrl, options.screenshot)
75
+ await mkdir(output.dir, { recursive: true })
76
+
77
+ const playwright = loadPlaywright(root)
78
+ const browser = await playwright.chromium.launch({ headless: true })
79
+ let page
80
+ try {
81
+ page = await browser.newPage({ viewport })
82
+ await page.goto(targetUrl, { waitUntil: options.waitUntil || "domcontentloaded", timeout: timeoutMs })
83
+ if (options.waitMs) await page.waitForTimeout(boundedInt(options.waitMs, 0, 0, 15_000))
84
+
85
+ const selector = options.selector ? String(options.selector) : null
86
+ const raw = await page.evaluate(({ maxElements, selector }) => {
87
+ const clean = (value, max = 180) => String(value || "").replace(/\s+/g, " ").trim().slice(0, max)
88
+ const candidateSelector = selector || [
89
+ "button","a[href]","input","textarea","select","img",
90
+ "h1","h2","h3","h4","h5","h6",
91
+ "[role]","[tabindex]","[data-testid]"
92
+ ].join(",")
93
+ let nodes = []
94
+ try { nodes = Array.from(document.querySelectorAll(candidateSelector)) } catch { nodes = [] }
95
+ return nodes.slice(0, maxElements).map((el, index) => {
96
+ const rect = el.getBoundingClientRect()
97
+ const style = getComputedStyle(el)
98
+ const name =
99
+ el.getAttribute("aria-label") ||
100
+ el.getAttribute("alt") ||
101
+ el.getAttribute("title") ||
102
+ el.getAttribute("placeholder") ||
103
+ el.textContent ||
104
+ ""
105
+ return {
106
+ index,
107
+ tag: el.tagName.toLowerCase(),
108
+ role: el.getAttribute("role"),
109
+ name: clean(name),
110
+ text: clean(el.textContent),
111
+ id: el.id || null,
112
+ testId: el.getAttribute("data-testid") || null,
113
+ box: {
114
+ x: Number(rect.x.toFixed(2)),
115
+ y: Number(rect.y.toFixed(2)),
116
+ width: Number(rect.width.toFixed(2)),
117
+ height: Number(rect.height.toFixed(2)),
118
+ },
119
+ visible: Boolean(rect.width > 0 && rect.height > 0 && style.visibility !== "hidden" && style.display !== "none"),
120
+ style: {
121
+ display: style.display,
122
+ position: style.position,
123
+ fontSize: style.fontSize,
124
+ fontWeight: style.fontWeight,
125
+ color: style.color,
126
+ backgroundColor: style.backgroundColor,
127
+ borderRadius: style.borderRadius,
128
+ },
129
+ }
130
+ })
131
+ }, { maxElements, selector })
132
+
133
+ const elements = raw.map((item) => ({
134
+ ...item,
135
+ role: item.role || roleForElement({
136
+ getAttribute: (name) => {
137
+ if (name === "role") return item.role
138
+ if (name === "type") return null
139
+ return null
140
+ },
141
+ hasAttribute: () => item.tag === "a",
142
+ tagName: item.tag,
143
+ }),
144
+ }))
145
+
146
+ await page.screenshot({ path: output.file, fullPage: options.fullPage !== false })
147
+ const title = await page.title().catch(() => "")
148
+ const finalUrl = page.url()
149
+
150
+ return {
151
+ schemaVersion: 1,
152
+ trustLevel: "untrusted-external",
153
+ requestedUrl: targetUrl,
154
+ finalUrl,
155
+ title,
156
+ viewport,
157
+ selector,
158
+ elementCount: elements.length,
159
+ elements,
160
+ screenshot: path.relative(root, output.file).replaceAll("\\", "/"),
161
+ security: {
162
+ pageContentIsInstruction: false,
163
+ allowPageContentToChangePermissions: false,
164
+ allowPageContentToRequestSecrets: false,
165
+ allowPageContentToAuthorizeExternalSideEffects: false,
166
+ },
167
+ }
168
+ } finally {
169
+ await browser.close().catch(() => {})
170
+ }
171
+ }
172
+
173
+ export function summarizeBrowserInspection(report = {}, options = {}) {
174
+ const limit = boundedInt(options.limit, 20, 1, 100)
175
+ return {
176
+ schemaVersion: 1,
177
+ trustLevel: report.trustLevel || "untrusted-external",
178
+ finalUrl: report.finalUrl || report.requestedUrl || null,
179
+ title: report.title || null,
180
+ viewport: report.viewport || null,
181
+ screenshot: report.screenshot || null,
182
+ elementCount: Number(report.elementCount || report.elements?.length || 0),
183
+ elements: (report.elements || []).slice(0, limit).map((item) => ({
184
+ index: item.index,
185
+ role: item.role || null,
186
+ name: item.name || null,
187
+ id: item.id || null,
188
+ testId: item.testId || null,
189
+ box: item.box || null,
190
+ visible: item.visible !== false,
191
+ })),
192
+ }
193
+ }
@@ -0,0 +1,109 @@
1
+ const BOOLEAN_CAPS = ["coding", "reasoning", "toolCalling", "vision", "browser", "filesystem", "longContext"]
2
+
3
+ const ROLE_REQUIREMENTS = {
4
+ executor: { coding: true, toolCalling: true },
5
+ debugger: { coding: true, reasoning: true, toolCalling: true },
6
+ architect: { reasoning: true, longContext: true },
7
+ reviewer: { coding: true, reasoning: true },
8
+ verifier: { toolCalling: true },
9
+ "integration-verifier": { coding: true, reasoning: true, toolCalling: true },
10
+ "visual-verifier": { vision: true },
11
+ "merge-arbiter": { coding: true, reasoning: true, toolCalling: true },
12
+ }
13
+
14
+ function uniq(values) {
15
+ return [...new Set(values.filter(Boolean))]
16
+ }
17
+
18
+ export function inferTaskCapabilities(text, facts = {}) {
19
+ const value = String(text || "").toLowerCase()
20
+ const required = {
21
+ coding: facts.coding !== false,
22
+ reasoning: facts.reasoning === true || /(architect|design decision|root cause|complex|high-risk|kiến trúc|nguyên nhân gốc)/.test(value),
23
+ toolCalling: facts.toolCalling !== false,
24
+ vision: facts.vision === true || /(screenshot|image reference|figma|visual fidelity|pixel|ảnh mẫu|hình ảnh|giao diện giống)/.test(value),
25
+ browser: facts.browser === true || /(playwright|browser|e2e|web page|click|form flow|trình duyệt)/.test(value),
26
+ filesystem: facts.filesystem !== false,
27
+ longContext: facts.longContext === true || /(whole repo|entire project|large refactor|long-horizon|toàn bộ dự án|tác vụ dài)/.test(value),
28
+ }
29
+ const tags = []
30
+ if (required.vision) tags.push("vision")
31
+ if (required.browser) tags.push("browser")
32
+ if (/(figma|design token|design source|ảnh mẫu)/.test(value)) tags.push("design-source")
33
+ if (/(responsive|breakpoint|mobile|tablet|viewport)/.test(value)) tags.push("responsive")
34
+ if (/(storybook|visual regression|snapshot)/.test(value)) tags.push("component-visual-testing")
35
+ if (/(untrusted page|prompt injection|web content injection)/.test(value)) tags.push("browser-security")
36
+ return { schemaVersion: 1, required, tags: uniq(tags) }
37
+ }
38
+
39
+ export function normalizeCapabilityProfile(profile = {}) {
40
+ const normalized = {}
41
+ for (const key of BOOLEAN_CAPS) normalized[key] = profile[key] === true
42
+ return {
43
+ ...normalized,
44
+ latencyClass: ["fast", "medium", "slow"].includes(profile.latencyClass) ? profile.latencyClass : "medium",
45
+ costClass: ["low", "medium", "high"].includes(profile.costClass) ? profile.costClass : "medium",
46
+ quality: Number.isFinite(Number(profile.quality)) ? Math.max(0, Math.min(1, Number(profile.quality))) : 0.5,
47
+ }
48
+ }
49
+
50
+ function classPenalty(value, order) {
51
+ const index = order.indexOf(value)
52
+ return index < 0 ? 1 : index
53
+ }
54
+
55
+ function tierPenalty(tier, preferredTier) {
56
+ const order = ["light", "standard", "heavy"]
57
+ const current = Math.max(0, order.indexOf(tier))
58
+ const preferred = Math.max(0, order.indexOf(preferredTier))
59
+ return Math.max(0, current - preferred)
60
+ }
61
+
62
+ export function selectCapabilityCandidate(requirements = {}, candidates = [], options = {}) {
63
+ const required = { ...(ROLE_REQUIREMENTS[options.role] || {}), ...(requirements.required || requirements) }
64
+ const rows = candidates.map((candidate) => {
65
+ const profile = normalizeCapabilityProfile(candidate.capabilities || {})
66
+ const missing = Object.entries(required)
67
+ .filter(([key, needed]) => needed === true && BOOLEAN_CAPS.includes(key) && !profile[key])
68
+ .map(([key]) => key)
69
+ const score =
70
+ profile.quality * 100 -
71
+ classPenalty(profile.costClass, ["low", "medium", "high"]) * 8 -
72
+ classPenalty(profile.latencyClass, ["fast", "medium", "slow"]) * 5 -
73
+ tierPenalty(candidate.tier, options.preferredTier || candidate.tier) * 25
74
+ return { ...candidate, capabilities: profile, missing, eligible: missing.length === 0, score }
75
+ })
76
+ const eligible = rows.filter((item) => item.eligible).sort((a, b) => b.score - a.score)
77
+ return {
78
+ selected: eligible[0] || null,
79
+ candidates: rows,
80
+ required,
81
+ fallbackNeeded: eligible.length === 0,
82
+ }
83
+ }
84
+
85
+ export function modelCandidatesFromPolicy(policy = {}) {
86
+ const seen = new Set()
87
+ const output = []
88
+ for (const tier of ["light", "standard", "heavy"]) {
89
+ const model = policy.tiers?.[tier]
90
+ if (!model || seen.has(model)) continue
91
+ seen.add(model)
92
+ output.push({
93
+ id: model,
94
+ tier,
95
+ capabilities: policy.capabilities?.[model] || {
96
+ coding: true,
97
+ reasoning: tier !== "light",
98
+ toolCalling: true,
99
+ filesystem: true,
100
+ longContext: tier === "heavy",
101
+ },
102
+ })
103
+ }
104
+ return output
105
+ }
106
+
107
+ export function roleCapabilityRequirements(role) {
108
+ return { ...(ROLE_REQUIREMENTS[role] || {}) }
109
+ }
@@ -0,0 +1,146 @@
1
+ import { buildContextManifest } from "./context-manifest.mjs"
2
+ import { planEvidenceBudget, evidenceValueScore } from "./evidence-budget.mjs"
3
+ import { putEvidence } from "./evidence-store.mjs"
4
+ import { inferTaskCapabilities } from "./capability-registry.mjs"
5
+ import { buildPromptEnvelope, comparePromptEnvelopes } from "./prompt-cache.mjs"
6
+
7
+ function taskText(task = {}) {
8
+ return [
9
+ task.title,
10
+ task.summary,
11
+ ...(task.acceptance || []),
12
+ ...(task.verification || []),
13
+ ].filter(Boolean).join(" ")
14
+ }
15
+
16
+ function rankExcerpt(item = {}, task = {}) {
17
+ const text = taskText(task).toLowerCase()
18
+ const pathValue = String(item.path || "").toLowerCase()
19
+ const declared = item.role === "declared" ? 1 : 0
20
+ const test = item.role === "test" ? 0.9 : 0
21
+ const instruction = item.role === "instruction" ? 0.85 : 0
22
+ const pathMatch = text && pathValue
23
+ ? text.split(/[^a-z0-9_$.-]+/i).filter((term) => term.length >= 4 && pathValue.includes(term.toLowerCase())).length
24
+ : 0
25
+ const relevance = Math.min(1, declared + test + instruction + pathMatch * 0.15)
26
+ return evidenceValueScore({
27
+ relevance,
28
+ freshness: 1,
29
+ confidence: item.role === "reference" ? 0.7 : 1,
30
+ chars: String(item.text || "").length || 1,
31
+ })
32
+ }
33
+
34
+ export async function externalizeContextExcerpts(root, manifest, options = {}) {
35
+ if (!manifest) return { manifest: null, externalized: [], externalizedBytes: 0 }
36
+ const threshold = Math.max(512, Number(options.threshold || 2_500))
37
+ const keepInline = Math.max(256, Number(options.inlineChars || 1_200))
38
+ const externalized = []
39
+ let externalizedBytes = 0
40
+ const excerpts = []
41
+
42
+ for (const item of manifest.excerpts || []) {
43
+ const text = String(item.text || "")
44
+ if (text.length <= threshold) {
45
+ excerpts.push(item)
46
+ continue
47
+ }
48
+
49
+ const stored = await putEvidence(root, text, {
50
+ kind: "context-excerpt",
51
+ source: item.path || null,
52
+ summary: `Externalized ${item.role || "reference"} context excerpt for ${item.path || "unknown"}`,
53
+ })
54
+ externalized.push({
55
+ ref: stored.ref,
56
+ path: item.path || null,
57
+ role: item.role || null,
58
+ bytes: stored.bytes,
59
+ score: Number(rankExcerpt(item, options.task).toFixed(8)),
60
+ })
61
+ externalizedBytes += stored.bytes
62
+ excerpts.push({
63
+ ...item,
64
+ text: text.slice(0, keepInline) + "\n...[externalized: " + stored.ref + "]",
65
+ evidenceRef: stored.ref,
66
+ originalChars: text.length,
67
+ externalized: true,
68
+ })
69
+ }
70
+
71
+ return {
72
+ manifest: {
73
+ ...manifest,
74
+ schemaVersion: Math.max(5, Number(manifest.schemaVersion || 0)),
75
+ excerpts,
76
+ evidencePointers: externalized,
77
+ },
78
+ externalized,
79
+ externalizedBytes,
80
+ }
81
+ }
82
+
83
+ export async function buildAdaptiveTaskContext(root, task, options = {}) {
84
+ const policy = options.policy || {}
85
+ const capabilities = options.capabilities || inferTaskCapabilities(taskText(task), options.facts || {})
86
+ const evidenceBudget = options.evidenceBudget || planEvidenceBudget(policy, task, capabilities.required)
87
+ const manifest = await buildContextManifest(root, task, {
88
+ budget: evidenceBudget.total,
89
+ evidenceBudget,
90
+ strategy: options.strategy || policy?.profile?.contextStrategy || "incremental-semantic+git",
91
+ semanticMaxFiles: options.semanticMaxFiles,
92
+ maxFiles: options.maxFiles,
93
+ })
94
+
95
+ const externalized = await externalizeContextExcerpts(root, manifest, {
96
+ task,
97
+ threshold: options.externalizeThreshold,
98
+ inlineChars: options.inlineChars,
99
+ })
100
+
101
+ const promptEnvelope = buildPromptEnvelope({
102
+ invariants: options.invariants || "evidence-first; scoped edits; fresh verification; no unsupported completion claims",
103
+ role: options.role || "executor",
104
+ skills: options.skills || policy.domains || [],
105
+ projectFacts: {
106
+ strategy: externalized.manifest?.strategy || null,
107
+ instructions: externalized.manifest?.instructions || [],
108
+ ...(options.projectFacts || {}),
109
+ },
110
+ task,
111
+ evidence: [
112
+ ...(externalized.externalized || []).map((item) => item.ref),
113
+ ...(externalized.manifest?.rankedReferences || []).slice(0, 12).map((item) => item.path),
114
+ ...(options.evidence || []),
115
+ ],
116
+ recentFailure: options.recentFailure || null,
117
+ nextAction: options.nextAction || null,
118
+ recentMessages: options.recentMessages || [],
119
+ })
120
+
121
+ const cache = options.previousPromptEnvelope
122
+ ? comparePromptEnvelopes(options.previousPromptEnvelope, promptEnvelope)
123
+ : null
124
+
125
+ return {
126
+ schemaVersion: 1,
127
+ contextSchemaVersion: 6,
128
+ capabilities,
129
+ evidenceBudget,
130
+ contextManifest: externalized.manifest,
131
+ evidenceStore: {
132
+ refs: externalized.externalized.length,
133
+ externalizedBytes: externalized.externalizedBytes,
134
+ entries: externalized.externalized,
135
+ },
136
+ promptEnvelope,
137
+ promptCache: {
138
+ stablePrefixHash: promptEnvelope.stablePrefixHash,
139
+ dynamicHash: promptEnvelope.dynamicHash,
140
+ stableChars: promptEnvelope.stableChars,
141
+ dynamicChars: promptEnvelope.dynamicChars,
142
+ cacheableRatio: promptEnvelope.cacheableRatio,
143
+ ...(cache || {}),
144
+ },
145
+ }
146
+ }
@@ -243,7 +243,8 @@ async function rankedReferences(root, nodes, terms, declared, changed, limit = 2
243
243
 
244
244
  export async function buildContextManifest(root, task, options = {}) {
245
245
  root = path.resolve(root)
246
- const budget = Math.max(4_000, Number(options.budget ?? 24_000))
246
+ const budget = Math.max(4_000, Number(options.evidenceBudget?.total ?? options.budget ?? 24_000))
247
+ const evidenceBudget = options.evidenceBudget || null
247
248
  const declared = taskFiles(task)
248
249
  const terms = taskTerms(task)
249
250
  const changed = gitChangedFiles(root)
@@ -331,20 +332,30 @@ export async function buildContextManifest(root, task, options = {}) {
331
332
  }
332
333
 
333
334
  let remaining = budget
335
+ const categoryRemaining = {
336
+ declared: evidenceBudget?.buckets?.declared ?? Math.round(budget * 0.42),
337
+ test: evidenceBudget?.buckets?.tests ?? Math.round(budget * 0.20),
338
+ instruction: evidenceBudget?.buckets?.instructions ?? Math.round(budget * 0.12),
339
+ reference: evidenceBudget?.buckets?.references ?? Math.round(budget * 0.26),
340
+ }
334
341
  const excerpts = []
335
342
  for (const file of priority) {
336
343
  if (remaining <= 0) break
337
344
  const isDeclared = declared.includes(file)
338
345
  const isTest = tests.includes(file)
339
346
  const isInstruction = instructions.includes(file)
347
+ const role = isDeclared ? "declared" : isTest ? "test" : isInstruction ? "instruction" : "reference"
340
348
  const desired = isDeclared ? 6_000 : isTest ? 4_000 : isInstruction ? 3_000 : 2_500
341
- const perFile = Math.min(desired, remaining)
349
+ const category = Math.max(0, Number(categoryRemaining[role] || 0))
350
+ if (category <= 0) continue
351
+ const perFile = Math.min(desired, remaining, category)
342
352
  const item = await excerpt(root, file, perFile, terms)
343
353
  if (!item) continue
344
354
  remaining -= item.text.length
355
+ categoryRemaining[role] = Math.max(0, categoryRemaining[role] - item.text.length)
345
356
  excerpts.push({
346
357
  ...item,
347
- role: isDeclared ? "declared" : isTest ? "test" : isInstruction ? "instruction" : "reference",
358
+ role,
348
359
  })
349
360
  }
350
361
 
@@ -373,6 +384,8 @@ export async function buildContextManifest(root, task, options = {}) {
373
384
  hotspots: graph.hotspots.slice(0, 12),
374
385
  },
375
386
  budget,
387
+ evidenceBudget,
388
+ categoryRemaining,
376
389
  used: budget - remaining,
377
390
  }
378
391
  }