opencode-agent-skill 9.0.0 → 11.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +116 -0
- package/README.md +742 -675
- package/bin/ocskill.mjs +354 -5
- package/docs/V11-PERCEPTION-ADAPTIVE-EXECUTION.md +75 -0
- package/docs/V11-PERCEPTION-ADAPTIVE.md +220 -0
- package/evals/router-triggers.json +82 -0
- package/evals/routing.json +76 -0
- package/evals/v11/tasks.json +122 -0
- package/global-config/AGENTS.md +78 -163
- package/global-config/agents/merge-arbiter.md +12 -0
- package/global-config/agents/visual-verifier.md +12 -0
- package/global-config/plugins/ues-router/index.js +683 -59
- package/global-config/plugins/ues-router/router.js +62 -3
- package/global-config/plugins/ues-router/runtime-guard.js +265 -0
- package/global-config/skills/browser-qa/SKILL.md +14 -0
- package/global-config/skills/browser-qa/references/workflow.md +11 -0
- package/global-config/skills/browser-security/SKILL.md +12 -0
- package/global-config/skills/component-visual-testing/SKILL.md +10 -0
- package/global-config/skills/design-source/SKILL.md +10 -0
- package/global-config/skills/design-source/references/workflow.md +12 -0
- package/global-config/skills/dynamic-workflow/SKILL.md +18 -0
- package/global-config/skills/dynamic-workflow/references/workflow.md +19 -0
- package/global-config/skills/responsive-verification/SKILL.md +10 -0
- package/global-config/skills/skill-authoring/SKILL.md +12 -0
- package/global-config/skills/skill-evaluation/SKILL.md +17 -0
- package/global-config/skills/visual-fidelity/SKILL.md +14 -0
- package/global-config/skills/visual-fidelity/references/workflow.md +14 -0
- package/lib/benchmark-confidence.mjs +49 -11
- package/lib/browser-adapter.mjs +82 -0
- package/lib/browser-runtime.mjs +193 -0
- package/lib/capability-registry.mjs +109 -0
- package/lib/context-engine-v11.mjs +146 -0
- package/lib/context-manifest.mjs +16 -3
- package/lib/control-center.mjs +12 -2
- package/lib/dynamic-workflow.mjs +179 -0
- package/lib/eval-ablation.mjs +146 -0
- package/lib/eval-report.mjs +83 -0
- package/lib/eval-telemetry.mjs +64 -0
- package/lib/evidence-budget.mjs +84 -0
- package/lib/evidence-store.mjs +178 -0
- package/lib/hermes-bridge.mjs +45 -1
- package/lib/model-config.mjs +9 -1
- package/lib/model-policy.mjs +58 -1
- package/lib/orchestrator-policy.mjs +100 -7
- package/lib/png-diff.mjs +229 -0
- package/lib/prompt-cache.mjs +60 -0
- package/lib/skill-quality.mjs +72 -0
- package/lib/task-engine.mjs +223 -12
- package/lib/ui-inspector.mjs +152 -0
- package/lib/v11-metrics.mjs +64 -0
- package/lib/visual-spec.mjs +159 -0
- package/package.json +11 -5
- package/scripts/eval-ablation.mjs +47 -0
- package/scripts/eval-matrix.mjs +13 -2
- package/scripts/validate-v11-suite.mjs +58 -0
- package/scripts/validate.mjs +16 -4
|
@@ -24,6 +24,14 @@ function mean(values) {
|
|
|
24
24
|
return usable.length ? usable.reduce((sum, value) => sum + value, 0) / usable.length : null
|
|
25
25
|
}
|
|
26
26
|
|
|
27
|
+
function positiveMean(values) {
|
|
28
|
+
return mean(values.map(Number).filter((value) => Number.isFinite(value) && value > 0))
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
function ratio(candidate, reference) {
|
|
32
|
+
return reference && candidate ? candidate / reference : null
|
|
33
|
+
}
|
|
34
|
+
|
|
27
35
|
export function pairedBenchmarkConfidence(results, options = {}) {
|
|
28
36
|
const byKey = new Map()
|
|
29
37
|
for (const item of results || []) {
|
|
@@ -71,25 +79,41 @@ export function pairedBenchmarkConfidence(results, options = {}) {
|
|
|
71
79
|
item.delta = item.uesPassRate - item.baselinePassRate
|
|
72
80
|
}
|
|
73
81
|
|
|
74
|
-
const
|
|
75
|
-
const
|
|
76
|
-
const
|
|
77
|
-
const uesDuration = mean(uesDurations)
|
|
78
|
-
const durationRatio = baselineDuration && uesDuration ? uesDuration / baselineDuration : null
|
|
82
|
+
const baselineDuration = mean(pairs.map((pair) => Number(pair.baseline.durationMs)))
|
|
83
|
+
const uesDuration = mean(pairs.map((pair) => Number(pair.ues.durationMs)))
|
|
84
|
+
const durationRatio = ratio(uesDuration, baselineDuration)
|
|
79
85
|
|
|
80
|
-
const
|
|
81
|
-
const
|
|
82
|
-
const
|
|
83
|
-
|
|
84
|
-
const
|
|
86
|
+
const baselineCost = positiveMean(pairs.map((pair) => pair.baseline.telemetry?.cost))
|
|
87
|
+
const uesCost = positiveMean(pairs.map((pair) => pair.ues.telemetry?.cost))
|
|
88
|
+
const costRatio = ratio(uesCost, baselineCost)
|
|
89
|
+
|
|
90
|
+
const baselineInitialInput = positiveMean(
|
|
91
|
+
pairs.map((pair) => pair.baseline.telemetry?.firstUsage?.input),
|
|
92
|
+
)
|
|
93
|
+
const uesInitialInput = positiveMean(
|
|
94
|
+
pairs.map((pair) => pair.ues.telemetry?.firstUsage?.input),
|
|
95
|
+
)
|
|
96
|
+
const initialInputRatio = ratio(uesInitialInput, baselineInitialInput)
|
|
97
|
+
|
|
98
|
+
const baselineTokens = positiveMean(
|
|
99
|
+
pairs.map((pair) => pair.baseline.telemetry?.tokens?.total),
|
|
100
|
+
)
|
|
101
|
+
const uesTokens = positiveMean(
|
|
102
|
+
pairs.map((pair) => pair.ues.telemetry?.tokens?.total),
|
|
103
|
+
)
|
|
104
|
+
const tokenRatio = ratio(uesTokens, baselineTokens)
|
|
85
105
|
|
|
86
106
|
const minPairs = Math.max(1, Number(options.minPairs || 20))
|
|
87
107
|
const alpha = Math.min(0.5, Math.max(0.0001, Number(options.alpha || 0.05)))
|
|
88
108
|
const minDelta = Math.max(0, Number(options.minDelta || 0))
|
|
89
109
|
const suiteRegressionTolerance = Math.max(0, Number(options.suiteRegressionTolerance || 0))
|
|
90
110
|
const maxDurationRatio = Math.max(1, Number(options.maxDurationRatio || 1.75))
|
|
111
|
+
const maxInitialInputRatio = Math.max(1, Number(options.maxInitialInputRatio || 1.5))
|
|
112
|
+
const maxTokenRatio = Math.max(1, Number(options.maxTokenRatio || 1.75))
|
|
91
113
|
const noSuiteRegression = Object.values(suites).every((item) => item.delta >= -suiteRegressionTolerance)
|
|
92
114
|
const speedAcceptable = durationRatio == null || durationRatio <= maxDurationRatio
|
|
115
|
+
const initialInputAcceptable = initialInputRatio == null || initialInputRatio <= maxInitialInputRatio
|
|
116
|
+
const tokensAcceptable = tokenRatio == null || tokenRatio <= maxTokenRatio
|
|
93
117
|
|
|
94
118
|
const checks = {
|
|
95
119
|
pairedCoverage: total >= minPairs,
|
|
@@ -98,9 +122,11 @@ export function pairedBenchmarkConfidence(results, options = {}) {
|
|
|
98
122
|
statisticallySupported: pValue <= alpha,
|
|
99
123
|
noSuiteRegression,
|
|
100
124
|
speedAcceptable,
|
|
125
|
+
initialInputAcceptable,
|
|
126
|
+
tokensAcceptable,
|
|
101
127
|
}
|
|
102
128
|
return {
|
|
103
|
-
schemaVersion:
|
|
129
|
+
schemaVersion: 2,
|
|
104
130
|
kind: "ues-paired-benchmark-confidence",
|
|
105
131
|
pairs: total,
|
|
106
132
|
bothPass,
|
|
@@ -123,6 +149,18 @@ export function pairedBenchmarkConfidence(results, options = {}) {
|
|
|
123
149
|
ratio: durationRatio,
|
|
124
150
|
maxRatio: maxDurationRatio,
|
|
125
151
|
},
|
|
152
|
+
initialInput: {
|
|
153
|
+
baselineMeanTokens: baselineInitialInput,
|
|
154
|
+
uesMeanTokens: uesInitialInput,
|
|
155
|
+
ratio: initialInputRatio,
|
|
156
|
+
maxRatio: maxInitialInputRatio,
|
|
157
|
+
},
|
|
158
|
+
tokens: {
|
|
159
|
+
baselineMean: baselineTokens,
|
|
160
|
+
uesMean: uesTokens,
|
|
161
|
+
ratio: tokenRatio,
|
|
162
|
+
maxRatio: maxTokenRatio,
|
|
163
|
+
},
|
|
126
164
|
cost: {
|
|
127
165
|
baselineMean: baselineCost,
|
|
128
166
|
uesMean: uesCost,
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
import { existsSync } from "node:fs"
|
|
2
|
+
import { readFile } from "node:fs/promises"
|
|
3
|
+
import path from "node:path"
|
|
4
|
+
|
|
5
|
+
function depHas(pkg, name) {
|
|
6
|
+
return Boolean(pkg?.dependencies?.[name] || pkg?.devDependencies?.[name] || pkg?.optionalDependencies?.[name])
|
|
7
|
+
}
|
|
8
|
+
|
|
9
|
+
export async function browserCapability(root = process.cwd()) {
|
|
10
|
+
root = path.resolve(root)
|
|
11
|
+
let pkg = {}
|
|
12
|
+
try { pkg = JSON.parse(await readFile(path.join(root, "package.json"), "utf8")) } catch {}
|
|
13
|
+
const localBin = process.platform === "win32"
|
|
14
|
+
? path.join(root, "node_modules", ".bin", "playwright.cmd")
|
|
15
|
+
: path.join(root, "node_modules", ".bin", "playwright")
|
|
16
|
+
const packageDeclared = depHas(pkg, "@playwright/test") || depHas(pkg, "playwright")
|
|
17
|
+
const storybookDeclared = Object.keys({
|
|
18
|
+
...(pkg.dependencies || {}),
|
|
19
|
+
...(pkg.devDependencies || {}),
|
|
20
|
+
...(pkg.optionalDependencies || {}),
|
|
21
|
+
}).some((name) => name === "storybook" || name.startsWith("@storybook/"))
|
|
22
|
+
return {
|
|
23
|
+
schemaVersion: 1,
|
|
24
|
+
playwright: {
|
|
25
|
+
packageDeclared,
|
|
26
|
+
localBinary: existsSync(localBin) ? localBin : null,
|
|
27
|
+
available: packageDeclared || existsSync(localBin),
|
|
28
|
+
},
|
|
29
|
+
storybook: {
|
|
30
|
+
declared: storybookDeclared,
|
|
31
|
+
configPresent: existsSync(path.join(root, ".storybook")),
|
|
32
|
+
},
|
|
33
|
+
modePreference: "cli-first",
|
|
34
|
+
rationale: "Use deterministic CLI/scripts for bounded verification; use richer browser tooling only when persistent exploratory state is necessary.",
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
export function buildBrowserVerificationPlan(input = {}) {
|
|
39
|
+
const viewports = Array.isArray(input.viewports) && input.viewports.length
|
|
40
|
+
? input.viewports
|
|
41
|
+
: [{ id: "desktop", width: 1440, height: 900 }]
|
|
42
|
+
return {
|
|
43
|
+
schemaVersion: 1,
|
|
44
|
+
url: input.url || null,
|
|
45
|
+
target: input.target || null,
|
|
46
|
+
trustLevel: input.trustLevel || "untrusted-external",
|
|
47
|
+
steps: [
|
|
48
|
+
{ kind: "navigate", value: input.url || null },
|
|
49
|
+
{ kind: "targeted-accessibility-snapshot", target: input.target || null, maxChars: 6000 },
|
|
50
|
+
{ kind: "geometry", fields: ["id", "role", "x", "y", "width", "height"] },
|
|
51
|
+
{ kind: "interaction", flow: input.flow || [] },
|
|
52
|
+
...viewports.map((viewport) => ({ kind: "screenshot", viewport })),
|
|
53
|
+
{ kind: "verify", checks: ["interaction", "geometry", "visual", "accessibility"] },
|
|
54
|
+
],
|
|
55
|
+
security: {
|
|
56
|
+
webpageInstructionsTrusted: false,
|
|
57
|
+
allowPageContentToChangePermissions: false,
|
|
58
|
+
allowPageContentToRequestSecrets: false,
|
|
59
|
+
allowPageContentToAuthorizeExternalSideEffects: false,
|
|
60
|
+
},
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
export function targetedBrowserEvidence(snapshot = [], query = "", options = {}) {
|
|
65
|
+
const needle = String(query || "").toLowerCase()
|
|
66
|
+
const limit = Math.max(1, Math.min(50, Number(options.limit || 12)))
|
|
67
|
+
const rows = Array.isArray(snapshot) ? snapshot : []
|
|
68
|
+
return rows
|
|
69
|
+
.filter((row) => {
|
|
70
|
+
const haystack = [row.role, row.name, row.text, row.id].filter(Boolean).join(" ").toLowerCase()
|
|
71
|
+
return !needle || haystack.includes(needle)
|
|
72
|
+
})
|
|
73
|
+
.slice(0, limit)
|
|
74
|
+
.map((row) => ({
|
|
75
|
+
id: row.id || null,
|
|
76
|
+
role: row.role || null,
|
|
77
|
+
name: row.name || null,
|
|
78
|
+
box: row.box || (["x","y","width","height"].every((key) => Number.isFinite(Number(row[key])))
|
|
79
|
+
? { x: Number(row.x), y: Number(row.y), width: Number(row.width), height: Number(row.height) }
|
|
80
|
+
: null),
|
|
81
|
+
}))
|
|
82
|
+
}
|
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
import { createHash } from "node:crypto"
|
|
2
|
+
import { createRequire } from "node:module"
|
|
3
|
+
import { mkdir, writeFile } from "node:fs/promises"
|
|
4
|
+
import path from "node:path"
|
|
5
|
+
|
|
6
|
+
function boundedInt(value, fallback, min, max) {
|
|
7
|
+
const parsed = Number(value)
|
|
8
|
+
if (!Number.isFinite(parsed)) return fallback
|
|
9
|
+
return Math.min(max, Math.max(min, Math.round(parsed)))
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
function safeUrl(value) {
|
|
13
|
+
let parsed
|
|
14
|
+
try { parsed = new URL(String(value || "")) } catch { throw new Error("Browser inspect requires a valid http(s) URL") }
|
|
15
|
+
if (!["http:", "https:"].includes(parsed.protocol)) throw new Error("Browser inspect only allows http(s) URLs")
|
|
16
|
+
return parsed.toString()
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
function safeOutput(root, url, requested = null) {
|
|
20
|
+
const base = path.resolve(root)
|
|
21
|
+
const dir = path.join(base, ".ues-cache", "browser-v1")
|
|
22
|
+
if (!requested) {
|
|
23
|
+
const hash = createHash("sha256").update(url).digest("hex").slice(0, 20)
|
|
24
|
+
return { dir, file: path.join(dir, hash + ".png") }
|
|
25
|
+
}
|
|
26
|
+
const target = path.resolve(base, String(requested))
|
|
27
|
+
if (target !== base && !target.startsWith(base + path.sep)) {
|
|
28
|
+
throw new Error("Browser screenshot path must stay inside the project root")
|
|
29
|
+
}
|
|
30
|
+
return { dir: path.dirname(target), file: target }
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
function loadPlaywright(root) {
|
|
34
|
+
const requireFromProject = createRequire(path.join(path.resolve(root), "package.json"))
|
|
35
|
+
for (const name of ["playwright", "@playwright/test"]) {
|
|
36
|
+
try {
|
|
37
|
+
const mod = requireFromProject(name)
|
|
38
|
+
if (mod?.chromium) return mod
|
|
39
|
+
if (mod?.default?.chromium) return mod.default
|
|
40
|
+
} catch {}
|
|
41
|
+
}
|
|
42
|
+
throw new Error("Playwright is unavailable in this project. Add playwright or @playwright/test before browser inspection.")
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
function roleForElement(el) {
|
|
46
|
+
const explicit = el.getAttribute("role")
|
|
47
|
+
if (explicit) return explicit
|
|
48
|
+
const tag = el.tagName.toLowerCase()
|
|
49
|
+
const type = (el.getAttribute("type") || "").toLowerCase()
|
|
50
|
+
if (tag === "button") return "button"
|
|
51
|
+
if (tag === "a" && el.hasAttribute("href")) return "link"
|
|
52
|
+
if (tag === "img") return "img"
|
|
53
|
+
if (/^h[1-6]$/.test(tag)) return "heading"
|
|
54
|
+
if (tag === "input") {
|
|
55
|
+
if (["button","submit","reset"].includes(type)) return "button"
|
|
56
|
+
if (type === "checkbox") return "checkbox"
|
|
57
|
+
if (type === "radio") return "radio"
|
|
58
|
+
return "textbox"
|
|
59
|
+
}
|
|
60
|
+
if (tag === "textarea") return "textbox"
|
|
61
|
+
if (tag === "select") return "combobox"
|
|
62
|
+
return tag
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
export async function inspectBrowserPage(root = process.cwd(), url, options = {}) {
|
|
66
|
+
root = path.resolve(root)
|
|
67
|
+
const targetUrl = safeUrl(url)
|
|
68
|
+
const maxElements = boundedInt(options.maxElements, 80, 1, 250)
|
|
69
|
+
const timeoutMs = boundedInt(options.timeoutMs, 30_000, 1_000, 120_000)
|
|
70
|
+
const viewport = {
|
|
71
|
+
width: boundedInt(options.width, 1440, 240, 7680),
|
|
72
|
+
height: boundedInt(options.height, 900, 240, 4320),
|
|
73
|
+
}
|
|
74
|
+
const output = safeOutput(root, targetUrl, options.screenshot)
|
|
75
|
+
await mkdir(output.dir, { recursive: true })
|
|
76
|
+
|
|
77
|
+
const playwright = loadPlaywright(root)
|
|
78
|
+
const browser = await playwright.chromium.launch({ headless: true })
|
|
79
|
+
let page
|
|
80
|
+
try {
|
|
81
|
+
page = await browser.newPage({ viewport })
|
|
82
|
+
await page.goto(targetUrl, { waitUntil: options.waitUntil || "domcontentloaded", timeout: timeoutMs })
|
|
83
|
+
if (options.waitMs) await page.waitForTimeout(boundedInt(options.waitMs, 0, 0, 15_000))
|
|
84
|
+
|
|
85
|
+
const selector = options.selector ? String(options.selector) : null
|
|
86
|
+
const raw = await page.evaluate(({ maxElements, selector }) => {
|
|
87
|
+
const clean = (value, max = 180) => String(value || "").replace(/\s+/g, " ").trim().slice(0, max)
|
|
88
|
+
const candidateSelector = selector || [
|
|
89
|
+
"button","a[href]","input","textarea","select","img",
|
|
90
|
+
"h1","h2","h3","h4","h5","h6",
|
|
91
|
+
"[role]","[tabindex]","[data-testid]"
|
|
92
|
+
].join(",")
|
|
93
|
+
let nodes = []
|
|
94
|
+
try { nodes = Array.from(document.querySelectorAll(candidateSelector)) } catch { nodes = [] }
|
|
95
|
+
return nodes.slice(0, maxElements).map((el, index) => {
|
|
96
|
+
const rect = el.getBoundingClientRect()
|
|
97
|
+
const style = getComputedStyle(el)
|
|
98
|
+
const name =
|
|
99
|
+
el.getAttribute("aria-label") ||
|
|
100
|
+
el.getAttribute("alt") ||
|
|
101
|
+
el.getAttribute("title") ||
|
|
102
|
+
el.getAttribute("placeholder") ||
|
|
103
|
+
el.textContent ||
|
|
104
|
+
""
|
|
105
|
+
return {
|
|
106
|
+
index,
|
|
107
|
+
tag: el.tagName.toLowerCase(),
|
|
108
|
+
role: el.getAttribute("role"),
|
|
109
|
+
name: clean(name),
|
|
110
|
+
text: clean(el.textContent),
|
|
111
|
+
id: el.id || null,
|
|
112
|
+
testId: el.getAttribute("data-testid") || null,
|
|
113
|
+
box: {
|
|
114
|
+
x: Number(rect.x.toFixed(2)),
|
|
115
|
+
y: Number(rect.y.toFixed(2)),
|
|
116
|
+
width: Number(rect.width.toFixed(2)),
|
|
117
|
+
height: Number(rect.height.toFixed(2)),
|
|
118
|
+
},
|
|
119
|
+
visible: Boolean(rect.width > 0 && rect.height > 0 && style.visibility !== "hidden" && style.display !== "none"),
|
|
120
|
+
style: {
|
|
121
|
+
display: style.display,
|
|
122
|
+
position: style.position,
|
|
123
|
+
fontSize: style.fontSize,
|
|
124
|
+
fontWeight: style.fontWeight,
|
|
125
|
+
color: style.color,
|
|
126
|
+
backgroundColor: style.backgroundColor,
|
|
127
|
+
borderRadius: style.borderRadius,
|
|
128
|
+
},
|
|
129
|
+
}
|
|
130
|
+
})
|
|
131
|
+
}, { maxElements, selector })
|
|
132
|
+
|
|
133
|
+
const elements = raw.map((item) => ({
|
|
134
|
+
...item,
|
|
135
|
+
role: item.role || roleForElement({
|
|
136
|
+
getAttribute: (name) => {
|
|
137
|
+
if (name === "role") return item.role
|
|
138
|
+
if (name === "type") return null
|
|
139
|
+
return null
|
|
140
|
+
},
|
|
141
|
+
hasAttribute: () => item.tag === "a",
|
|
142
|
+
tagName: item.tag,
|
|
143
|
+
}),
|
|
144
|
+
}))
|
|
145
|
+
|
|
146
|
+
await page.screenshot({ path: output.file, fullPage: options.fullPage !== false })
|
|
147
|
+
const title = await page.title().catch(() => "")
|
|
148
|
+
const finalUrl = page.url()
|
|
149
|
+
|
|
150
|
+
return {
|
|
151
|
+
schemaVersion: 1,
|
|
152
|
+
trustLevel: "untrusted-external",
|
|
153
|
+
requestedUrl: targetUrl,
|
|
154
|
+
finalUrl,
|
|
155
|
+
title,
|
|
156
|
+
viewport,
|
|
157
|
+
selector,
|
|
158
|
+
elementCount: elements.length,
|
|
159
|
+
elements,
|
|
160
|
+
screenshot: path.relative(root, output.file).replaceAll("\\", "/"),
|
|
161
|
+
security: {
|
|
162
|
+
pageContentIsInstruction: false,
|
|
163
|
+
allowPageContentToChangePermissions: false,
|
|
164
|
+
allowPageContentToRequestSecrets: false,
|
|
165
|
+
allowPageContentToAuthorizeExternalSideEffects: false,
|
|
166
|
+
},
|
|
167
|
+
}
|
|
168
|
+
} finally {
|
|
169
|
+
await browser.close().catch(() => {})
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
export function summarizeBrowserInspection(report = {}, options = {}) {
|
|
174
|
+
const limit = boundedInt(options.limit, 20, 1, 100)
|
|
175
|
+
return {
|
|
176
|
+
schemaVersion: 1,
|
|
177
|
+
trustLevel: report.trustLevel || "untrusted-external",
|
|
178
|
+
finalUrl: report.finalUrl || report.requestedUrl || null,
|
|
179
|
+
title: report.title || null,
|
|
180
|
+
viewport: report.viewport || null,
|
|
181
|
+
screenshot: report.screenshot || null,
|
|
182
|
+
elementCount: Number(report.elementCount || report.elements?.length || 0),
|
|
183
|
+
elements: (report.elements || []).slice(0, limit).map((item) => ({
|
|
184
|
+
index: item.index,
|
|
185
|
+
role: item.role || null,
|
|
186
|
+
name: item.name || null,
|
|
187
|
+
id: item.id || null,
|
|
188
|
+
testId: item.testId || null,
|
|
189
|
+
box: item.box || null,
|
|
190
|
+
visible: item.visible !== false,
|
|
191
|
+
})),
|
|
192
|
+
}
|
|
193
|
+
}
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
const BOOLEAN_CAPS = ["coding", "reasoning", "toolCalling", "vision", "browser", "filesystem", "longContext"]
|
|
2
|
+
|
|
3
|
+
const ROLE_REQUIREMENTS = {
|
|
4
|
+
executor: { coding: true, toolCalling: true },
|
|
5
|
+
debugger: { coding: true, reasoning: true, toolCalling: true },
|
|
6
|
+
architect: { reasoning: true, longContext: true },
|
|
7
|
+
reviewer: { coding: true, reasoning: true },
|
|
8
|
+
verifier: { toolCalling: true },
|
|
9
|
+
"integration-verifier": { coding: true, reasoning: true, toolCalling: true },
|
|
10
|
+
"visual-verifier": { vision: true },
|
|
11
|
+
"merge-arbiter": { coding: true, reasoning: true, toolCalling: true },
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
function uniq(values) {
|
|
15
|
+
return [...new Set(values.filter(Boolean))]
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
export function inferTaskCapabilities(text, facts = {}) {
|
|
19
|
+
const value = String(text || "").toLowerCase()
|
|
20
|
+
const required = {
|
|
21
|
+
coding: facts.coding !== false,
|
|
22
|
+
reasoning: facts.reasoning === true || /(architect|design decision|root cause|complex|high-risk|kiến trúc|nguyên nhân gốc)/.test(value),
|
|
23
|
+
toolCalling: facts.toolCalling !== false,
|
|
24
|
+
vision: facts.vision === true || /(screenshot|image reference|figma|visual fidelity|pixel|ảnh mẫu|hình ảnh|giao diện giống)/.test(value),
|
|
25
|
+
browser: facts.browser === true || /(playwright|browser|e2e|web page|click|form flow|trình duyệt)/.test(value),
|
|
26
|
+
filesystem: facts.filesystem !== false,
|
|
27
|
+
longContext: facts.longContext === true || /(whole repo|entire project|large refactor|long-horizon|toàn bộ dự án|tác vụ dài)/.test(value),
|
|
28
|
+
}
|
|
29
|
+
const tags = []
|
|
30
|
+
if (required.vision) tags.push("vision")
|
|
31
|
+
if (required.browser) tags.push("browser")
|
|
32
|
+
if (/(figma|design token|design source|ảnh mẫu)/.test(value)) tags.push("design-source")
|
|
33
|
+
if (/(responsive|breakpoint|mobile|tablet|viewport)/.test(value)) tags.push("responsive")
|
|
34
|
+
if (/(storybook|visual regression|snapshot)/.test(value)) tags.push("component-visual-testing")
|
|
35
|
+
if (/(untrusted page|prompt injection|web content injection)/.test(value)) tags.push("browser-security")
|
|
36
|
+
return { schemaVersion: 1, required, tags: uniq(tags) }
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export function normalizeCapabilityProfile(profile = {}) {
|
|
40
|
+
const normalized = {}
|
|
41
|
+
for (const key of BOOLEAN_CAPS) normalized[key] = profile[key] === true
|
|
42
|
+
return {
|
|
43
|
+
...normalized,
|
|
44
|
+
latencyClass: ["fast", "medium", "slow"].includes(profile.latencyClass) ? profile.latencyClass : "medium",
|
|
45
|
+
costClass: ["low", "medium", "high"].includes(profile.costClass) ? profile.costClass : "medium",
|
|
46
|
+
quality: Number.isFinite(Number(profile.quality)) ? Math.max(0, Math.min(1, Number(profile.quality))) : 0.5,
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
function classPenalty(value, order) {
|
|
51
|
+
const index = order.indexOf(value)
|
|
52
|
+
return index < 0 ? 1 : index
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
function tierPenalty(tier, preferredTier) {
|
|
56
|
+
const order = ["light", "standard", "heavy"]
|
|
57
|
+
const current = Math.max(0, order.indexOf(tier))
|
|
58
|
+
const preferred = Math.max(0, order.indexOf(preferredTier))
|
|
59
|
+
return Math.max(0, current - preferred)
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export function selectCapabilityCandidate(requirements = {}, candidates = [], options = {}) {
|
|
63
|
+
const required = { ...(ROLE_REQUIREMENTS[options.role] || {}), ...(requirements.required || requirements) }
|
|
64
|
+
const rows = candidates.map((candidate) => {
|
|
65
|
+
const profile = normalizeCapabilityProfile(candidate.capabilities || {})
|
|
66
|
+
const missing = Object.entries(required)
|
|
67
|
+
.filter(([key, needed]) => needed === true && BOOLEAN_CAPS.includes(key) && !profile[key])
|
|
68
|
+
.map(([key]) => key)
|
|
69
|
+
const score =
|
|
70
|
+
profile.quality * 100 -
|
|
71
|
+
classPenalty(profile.costClass, ["low", "medium", "high"]) * 8 -
|
|
72
|
+
classPenalty(profile.latencyClass, ["fast", "medium", "slow"]) * 5 -
|
|
73
|
+
tierPenalty(candidate.tier, options.preferredTier || candidate.tier) * 25
|
|
74
|
+
return { ...candidate, capabilities: profile, missing, eligible: missing.length === 0, score }
|
|
75
|
+
})
|
|
76
|
+
const eligible = rows.filter((item) => item.eligible).sort((a, b) => b.score - a.score)
|
|
77
|
+
return {
|
|
78
|
+
selected: eligible[0] || null,
|
|
79
|
+
candidates: rows,
|
|
80
|
+
required,
|
|
81
|
+
fallbackNeeded: eligible.length === 0,
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
export function modelCandidatesFromPolicy(policy = {}) {
|
|
86
|
+
const seen = new Set()
|
|
87
|
+
const output = []
|
|
88
|
+
for (const tier of ["light", "standard", "heavy"]) {
|
|
89
|
+
const model = policy.tiers?.[tier]
|
|
90
|
+
if (!model || seen.has(model)) continue
|
|
91
|
+
seen.add(model)
|
|
92
|
+
output.push({
|
|
93
|
+
id: model,
|
|
94
|
+
tier,
|
|
95
|
+
capabilities: policy.capabilities?.[model] || {
|
|
96
|
+
coding: true,
|
|
97
|
+
reasoning: tier !== "light",
|
|
98
|
+
toolCalling: true,
|
|
99
|
+
filesystem: true,
|
|
100
|
+
longContext: tier === "heavy",
|
|
101
|
+
},
|
|
102
|
+
})
|
|
103
|
+
}
|
|
104
|
+
return output
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
export function roleCapabilityRequirements(role) {
|
|
108
|
+
return { ...(ROLE_REQUIREMENTS[role] || {}) }
|
|
109
|
+
}
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
import { buildContextManifest } from "./context-manifest.mjs"
|
|
2
|
+
import { planEvidenceBudget, evidenceValueScore } from "./evidence-budget.mjs"
|
|
3
|
+
import { putEvidence } from "./evidence-store.mjs"
|
|
4
|
+
import { inferTaskCapabilities } from "./capability-registry.mjs"
|
|
5
|
+
import { buildPromptEnvelope, comparePromptEnvelopes } from "./prompt-cache.mjs"
|
|
6
|
+
|
|
7
|
+
function taskText(task = {}) {
|
|
8
|
+
return [
|
|
9
|
+
task.title,
|
|
10
|
+
task.summary,
|
|
11
|
+
...(task.acceptance || []),
|
|
12
|
+
...(task.verification || []),
|
|
13
|
+
].filter(Boolean).join(" ")
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
function rankExcerpt(item = {}, task = {}) {
|
|
17
|
+
const text = taskText(task).toLowerCase()
|
|
18
|
+
const pathValue = String(item.path || "").toLowerCase()
|
|
19
|
+
const declared = item.role === "declared" ? 1 : 0
|
|
20
|
+
const test = item.role === "test" ? 0.9 : 0
|
|
21
|
+
const instruction = item.role === "instruction" ? 0.85 : 0
|
|
22
|
+
const pathMatch = text && pathValue
|
|
23
|
+
? text.split(/[^a-z0-9_$.-]+/i).filter((term) => term.length >= 4 && pathValue.includes(term.toLowerCase())).length
|
|
24
|
+
: 0
|
|
25
|
+
const relevance = Math.min(1, declared + test + instruction + pathMatch * 0.15)
|
|
26
|
+
return evidenceValueScore({
|
|
27
|
+
relevance,
|
|
28
|
+
freshness: 1,
|
|
29
|
+
confidence: item.role === "reference" ? 0.7 : 1,
|
|
30
|
+
chars: String(item.text || "").length || 1,
|
|
31
|
+
})
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
export async function externalizeContextExcerpts(root, manifest, options = {}) {
|
|
35
|
+
if (!manifest) return { manifest: null, externalized: [], externalizedBytes: 0 }
|
|
36
|
+
const threshold = Math.max(512, Number(options.threshold || 2_500))
|
|
37
|
+
const keepInline = Math.max(256, Number(options.inlineChars || 1_200))
|
|
38
|
+
const externalized = []
|
|
39
|
+
let externalizedBytes = 0
|
|
40
|
+
const excerpts = []
|
|
41
|
+
|
|
42
|
+
for (const item of manifest.excerpts || []) {
|
|
43
|
+
const text = String(item.text || "")
|
|
44
|
+
if (text.length <= threshold) {
|
|
45
|
+
excerpts.push(item)
|
|
46
|
+
continue
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
const stored = await putEvidence(root, text, {
|
|
50
|
+
kind: "context-excerpt",
|
|
51
|
+
source: item.path || null,
|
|
52
|
+
summary: `Externalized ${item.role || "reference"} context excerpt for ${item.path || "unknown"}`,
|
|
53
|
+
})
|
|
54
|
+
externalized.push({
|
|
55
|
+
ref: stored.ref,
|
|
56
|
+
path: item.path || null,
|
|
57
|
+
role: item.role || null,
|
|
58
|
+
bytes: stored.bytes,
|
|
59
|
+
score: Number(rankExcerpt(item, options.task).toFixed(8)),
|
|
60
|
+
})
|
|
61
|
+
externalizedBytes += stored.bytes
|
|
62
|
+
excerpts.push({
|
|
63
|
+
...item,
|
|
64
|
+
text: text.slice(0, keepInline) + "\n...[externalized: " + stored.ref + "]",
|
|
65
|
+
evidenceRef: stored.ref,
|
|
66
|
+
originalChars: text.length,
|
|
67
|
+
externalized: true,
|
|
68
|
+
})
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
return {
|
|
72
|
+
manifest: {
|
|
73
|
+
...manifest,
|
|
74
|
+
schemaVersion: Math.max(5, Number(manifest.schemaVersion || 0)),
|
|
75
|
+
excerpts,
|
|
76
|
+
evidencePointers: externalized,
|
|
77
|
+
},
|
|
78
|
+
externalized,
|
|
79
|
+
externalizedBytes,
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
export async function buildAdaptiveTaskContext(root, task, options = {}) {
|
|
84
|
+
const policy = options.policy || {}
|
|
85
|
+
const capabilities = options.capabilities || inferTaskCapabilities(taskText(task), options.facts || {})
|
|
86
|
+
const evidenceBudget = options.evidenceBudget || planEvidenceBudget(policy, task, capabilities.required)
|
|
87
|
+
const manifest = await buildContextManifest(root, task, {
|
|
88
|
+
budget: evidenceBudget.total,
|
|
89
|
+
evidenceBudget,
|
|
90
|
+
strategy: options.strategy || policy?.profile?.contextStrategy || "incremental-semantic+git",
|
|
91
|
+
semanticMaxFiles: options.semanticMaxFiles,
|
|
92
|
+
maxFiles: options.maxFiles,
|
|
93
|
+
})
|
|
94
|
+
|
|
95
|
+
const externalized = await externalizeContextExcerpts(root, manifest, {
|
|
96
|
+
task,
|
|
97
|
+
threshold: options.externalizeThreshold,
|
|
98
|
+
inlineChars: options.inlineChars,
|
|
99
|
+
})
|
|
100
|
+
|
|
101
|
+
const promptEnvelope = buildPromptEnvelope({
|
|
102
|
+
invariants: options.invariants || "evidence-first; scoped edits; fresh verification; no unsupported completion claims",
|
|
103
|
+
role: options.role || "executor",
|
|
104
|
+
skills: options.skills || policy.domains || [],
|
|
105
|
+
projectFacts: {
|
|
106
|
+
strategy: externalized.manifest?.strategy || null,
|
|
107
|
+
instructions: externalized.manifest?.instructions || [],
|
|
108
|
+
...(options.projectFacts || {}),
|
|
109
|
+
},
|
|
110
|
+
task,
|
|
111
|
+
evidence: [
|
|
112
|
+
...(externalized.externalized || []).map((item) => item.ref),
|
|
113
|
+
...(externalized.manifest?.rankedReferences || []).slice(0, 12).map((item) => item.path),
|
|
114
|
+
...(options.evidence || []),
|
|
115
|
+
],
|
|
116
|
+
recentFailure: options.recentFailure || null,
|
|
117
|
+
nextAction: options.nextAction || null,
|
|
118
|
+
recentMessages: options.recentMessages || [],
|
|
119
|
+
})
|
|
120
|
+
|
|
121
|
+
const cache = options.previousPromptEnvelope
|
|
122
|
+
? comparePromptEnvelopes(options.previousPromptEnvelope, promptEnvelope)
|
|
123
|
+
: null
|
|
124
|
+
|
|
125
|
+
return {
|
|
126
|
+
schemaVersion: 1,
|
|
127
|
+
contextSchemaVersion: 6,
|
|
128
|
+
capabilities,
|
|
129
|
+
evidenceBudget,
|
|
130
|
+
contextManifest: externalized.manifest,
|
|
131
|
+
evidenceStore: {
|
|
132
|
+
refs: externalized.externalized.length,
|
|
133
|
+
externalizedBytes: externalized.externalizedBytes,
|
|
134
|
+
entries: externalized.externalized,
|
|
135
|
+
},
|
|
136
|
+
promptEnvelope,
|
|
137
|
+
promptCache: {
|
|
138
|
+
stablePrefixHash: promptEnvelope.stablePrefixHash,
|
|
139
|
+
dynamicHash: promptEnvelope.dynamicHash,
|
|
140
|
+
stableChars: promptEnvelope.stableChars,
|
|
141
|
+
dynamicChars: promptEnvelope.dynamicChars,
|
|
142
|
+
cacheableRatio: promptEnvelope.cacheableRatio,
|
|
143
|
+
...(cache || {}),
|
|
144
|
+
},
|
|
145
|
+
}
|
|
146
|
+
}
|
package/lib/context-manifest.mjs
CHANGED
|
@@ -243,7 +243,8 @@ async function rankedReferences(root, nodes, terms, declared, changed, limit = 2
|
|
|
243
243
|
|
|
244
244
|
export async function buildContextManifest(root, task, options = {}) {
|
|
245
245
|
root = path.resolve(root)
|
|
246
|
-
const budget = Math.max(4_000, Number(options.budget ?? 24_000))
|
|
246
|
+
const budget = Math.max(4_000, Number(options.evidenceBudget?.total ?? options.budget ?? 24_000))
|
|
247
|
+
const evidenceBudget = options.evidenceBudget || null
|
|
247
248
|
const declared = taskFiles(task)
|
|
248
249
|
const terms = taskTerms(task)
|
|
249
250
|
const changed = gitChangedFiles(root)
|
|
@@ -331,20 +332,30 @@ export async function buildContextManifest(root, task, options = {}) {
|
|
|
331
332
|
}
|
|
332
333
|
|
|
333
334
|
let remaining = budget
|
|
335
|
+
const categoryRemaining = {
|
|
336
|
+
declared: evidenceBudget?.buckets?.declared ?? Math.round(budget * 0.42),
|
|
337
|
+
test: evidenceBudget?.buckets?.tests ?? Math.round(budget * 0.20),
|
|
338
|
+
instruction: evidenceBudget?.buckets?.instructions ?? Math.round(budget * 0.12),
|
|
339
|
+
reference: evidenceBudget?.buckets?.references ?? Math.round(budget * 0.26),
|
|
340
|
+
}
|
|
334
341
|
const excerpts = []
|
|
335
342
|
for (const file of priority) {
|
|
336
343
|
if (remaining <= 0) break
|
|
337
344
|
const isDeclared = declared.includes(file)
|
|
338
345
|
const isTest = tests.includes(file)
|
|
339
346
|
const isInstruction = instructions.includes(file)
|
|
347
|
+
const role = isDeclared ? "declared" : isTest ? "test" : isInstruction ? "instruction" : "reference"
|
|
340
348
|
const desired = isDeclared ? 6_000 : isTest ? 4_000 : isInstruction ? 3_000 : 2_500
|
|
341
|
-
const
|
|
349
|
+
const category = Math.max(0, Number(categoryRemaining[role] || 0))
|
|
350
|
+
if (category <= 0) continue
|
|
351
|
+
const perFile = Math.min(desired, remaining, category)
|
|
342
352
|
const item = await excerpt(root, file, perFile, terms)
|
|
343
353
|
if (!item) continue
|
|
344
354
|
remaining -= item.text.length
|
|
355
|
+
categoryRemaining[role] = Math.max(0, categoryRemaining[role] - item.text.length)
|
|
345
356
|
excerpts.push({
|
|
346
357
|
...item,
|
|
347
|
-
role
|
|
358
|
+
role,
|
|
348
359
|
})
|
|
349
360
|
}
|
|
350
361
|
|
|
@@ -373,6 +384,8 @@ export async function buildContextManifest(root, task, options = {}) {
|
|
|
373
384
|
hotspots: graph.hotspots.slice(0, 12),
|
|
374
385
|
},
|
|
375
386
|
budget,
|
|
387
|
+
evidenceBudget,
|
|
388
|
+
categoryRemaining,
|
|
376
389
|
used: budget - remaining,
|
|
377
390
|
}
|
|
378
391
|
}
|