opencode-agent-skill 9.0.0 → 11.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +116 -0
- package/README.md +742 -675
- package/bin/ocskill.mjs +354 -5
- package/docs/V11-PERCEPTION-ADAPTIVE-EXECUTION.md +75 -0
- package/docs/V11-PERCEPTION-ADAPTIVE.md +220 -0
- package/evals/router-triggers.json +82 -0
- package/evals/routing.json +76 -0
- package/evals/v11/tasks.json +122 -0
- package/global-config/AGENTS.md +78 -163
- package/global-config/agents/merge-arbiter.md +12 -0
- package/global-config/agents/visual-verifier.md +12 -0
- package/global-config/plugins/ues-router/index.js +683 -59
- package/global-config/plugins/ues-router/router.js +62 -3
- package/global-config/plugins/ues-router/runtime-guard.js +265 -0
- package/global-config/skills/browser-qa/SKILL.md +14 -0
- package/global-config/skills/browser-qa/references/workflow.md +11 -0
- package/global-config/skills/browser-security/SKILL.md +12 -0
- package/global-config/skills/component-visual-testing/SKILL.md +10 -0
- package/global-config/skills/design-source/SKILL.md +10 -0
- package/global-config/skills/design-source/references/workflow.md +12 -0
- package/global-config/skills/dynamic-workflow/SKILL.md +18 -0
- package/global-config/skills/dynamic-workflow/references/workflow.md +19 -0
- package/global-config/skills/responsive-verification/SKILL.md +10 -0
- package/global-config/skills/skill-authoring/SKILL.md +12 -0
- package/global-config/skills/skill-evaluation/SKILL.md +17 -0
- package/global-config/skills/visual-fidelity/SKILL.md +14 -0
- package/global-config/skills/visual-fidelity/references/workflow.md +14 -0
- package/lib/benchmark-confidence.mjs +49 -11
- package/lib/browser-adapter.mjs +82 -0
- package/lib/browser-runtime.mjs +193 -0
- package/lib/capability-registry.mjs +109 -0
- package/lib/context-engine-v11.mjs +146 -0
- package/lib/context-manifest.mjs +16 -3
- package/lib/control-center.mjs +12 -2
- package/lib/dynamic-workflow.mjs +179 -0
- package/lib/eval-ablation.mjs +146 -0
- package/lib/eval-report.mjs +83 -0
- package/lib/eval-telemetry.mjs +64 -0
- package/lib/evidence-budget.mjs +84 -0
- package/lib/evidence-store.mjs +178 -0
- package/lib/hermes-bridge.mjs +45 -1
- package/lib/model-config.mjs +9 -1
- package/lib/model-policy.mjs +58 -1
- package/lib/orchestrator-policy.mjs +100 -7
- package/lib/png-diff.mjs +229 -0
- package/lib/prompt-cache.mjs +60 -0
- package/lib/skill-quality.mjs +72 -0
- package/lib/task-engine.mjs +223 -12
- package/lib/ui-inspector.mjs +152 -0
- package/lib/v11-metrics.mjs +64 -0
- package/lib/visual-spec.mjs +159 -0
- package/package.json +11 -5
- package/scripts/eval-ablation.mjs +47 -0
- package/scripts/eval-matrix.mjs +13 -2
- package/scripts/validate-v11-suite.mjs +58 -0
- package/scripts/validate.mjs +16 -4
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
function number(value, fallback = 0) {
|
|
2
|
+
const parsed = Number(value)
|
|
3
|
+
return Number.isFinite(parsed) ? parsed : fallback
|
|
4
|
+
}
|
|
5
|
+
|
|
6
|
+
function range(value, tolerance = 0) {
|
|
7
|
+
if (Array.isArray(value) && value.length >= 2) return [number(value[0]), number(value[1])]
|
|
8
|
+
const exact = number(value)
|
|
9
|
+
return [exact - tolerance, exact + tolerance]
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
function within(actual, expected, tolerance) {
|
|
13
|
+
const [min, max] = range(expected, tolerance)
|
|
14
|
+
return actual >= Math.min(min, max) && actual <= Math.max(min, max)
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
function normalizeElement(item = {}) {
|
|
18
|
+
return {
|
|
19
|
+
id: String(item.id || "").trim(),
|
|
20
|
+
role: item.role || null,
|
|
21
|
+
region: item.region || null,
|
|
22
|
+
expected: {
|
|
23
|
+
...(item.expected || {}),
|
|
24
|
+
...(item.x !== undefined ? { x: item.x } : {}),
|
|
25
|
+
...(item.y !== undefined ? { y: item.y } : {}),
|
|
26
|
+
...(item.width !== undefined ? { width: item.width } : {}),
|
|
27
|
+
...(item.height !== undefined ? { height: item.height } : {}),
|
|
28
|
+
},
|
|
29
|
+
tolerance: {
|
|
30
|
+
position: number(item.tolerance?.position, 8),
|
|
31
|
+
size: number(item.tolerance?.size, 8),
|
|
32
|
+
},
|
|
33
|
+
required: item.required !== false,
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export function normalizeVisualSpec(spec = {}) {
|
|
38
|
+
const viewport = {
|
|
39
|
+
width: Math.max(1, number(spec.viewport?.width, 1440)),
|
|
40
|
+
height: Math.max(1, number(spec.viewport?.height, 900)),
|
|
41
|
+
deviceScaleFactor: Math.max(0.1, number(spec.viewport?.deviceScaleFactor, 1)),
|
|
42
|
+
}
|
|
43
|
+
return {
|
|
44
|
+
schemaVersion: 1,
|
|
45
|
+
name: spec.name || null,
|
|
46
|
+
viewport,
|
|
47
|
+
elements: (spec.elements || []).map(normalizeElement).filter((item) => item.id),
|
|
48
|
+
regions: Array.isArray(spec.regions) ? spec.regions : [],
|
|
49
|
+
tokens: spec.tokens || {},
|
|
50
|
+
metadata: spec.metadata || {},
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export function validateVisualSpec(spec = {}) {
|
|
55
|
+
const normalized = normalizeVisualSpec(spec)
|
|
56
|
+
const errors = []
|
|
57
|
+
const ids = new Set()
|
|
58
|
+
for (const item of normalized.elements) {
|
|
59
|
+
if (ids.has(item.id)) errors.push("duplicate element id: " + item.id)
|
|
60
|
+
ids.add(item.id)
|
|
61
|
+
for (const key of ["x", "y", "width", "height"]) {
|
|
62
|
+
if (item.expected[key] === undefined) continue
|
|
63
|
+
const values = Array.isArray(item.expected[key]) ? item.expected[key] : [item.expected[key]]
|
|
64
|
+
if (values.some((value) => !Number.isFinite(Number(value)))) errors.push(item.id + ": invalid " + key)
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
return { valid: errors.length === 0, errors, spec: normalized }
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
function actualMap(actual = []) {
|
|
71
|
+
const values = Array.isArray(actual) ? actual : Object.values(actual || {})
|
|
72
|
+
return new Map(values.filter((item) => item?.id).map((item) => [String(item.id), item]))
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
export function createGeometryReceipt(specInput, actualInput, options = {}) {
|
|
76
|
+
const checked = validateVisualSpec(specInput)
|
|
77
|
+
if (!checked.valid) throw new Error("Invalid visual spec: " + checked.errors.join("; "))
|
|
78
|
+
const actual = actualMap(actualInput)
|
|
79
|
+
const elements = []
|
|
80
|
+
let failed = 0
|
|
81
|
+
|
|
82
|
+
for (const expected of checked.spec.elements) {
|
|
83
|
+
const found = actual.get(expected.id)
|
|
84
|
+
if (!found) {
|
|
85
|
+
const pass = !expected.required
|
|
86
|
+
if (!pass) failed += 1
|
|
87
|
+
elements.push({ id: expected.id, verdict: pass ? "PASS" : "FAIL", reason: "missing", expected: expected.expected, actual: null })
|
|
88
|
+
continue
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
const failures = []
|
|
92
|
+
const checks = {}
|
|
93
|
+
for (const key of ["x", "y"]) {
|
|
94
|
+
if (expected.expected[key] === undefined) continue
|
|
95
|
+
const ok = within(number(found[key]), expected.expected[key], expected.tolerance.position)
|
|
96
|
+
checks[key] = ok
|
|
97
|
+
if (!ok) failures.push(key)
|
|
98
|
+
}
|
|
99
|
+
for (const key of ["width", "height"]) {
|
|
100
|
+
if (expected.expected[key] === undefined) continue
|
|
101
|
+
const ok = within(number(found[key]), expected.expected[key], expected.tolerance.size)
|
|
102
|
+
checks[key] = ok
|
|
103
|
+
if (!ok) failures.push(key)
|
|
104
|
+
}
|
|
105
|
+
const pass = failures.length === 0
|
|
106
|
+
if (!pass) failed += 1
|
|
107
|
+
elements.push({
|
|
108
|
+
id: expected.id,
|
|
109
|
+
verdict: pass ? "PASS" : "FAIL",
|
|
110
|
+
failures,
|
|
111
|
+
checks,
|
|
112
|
+
expected: expected.expected,
|
|
113
|
+
actual: {
|
|
114
|
+
x: number(found.x),
|
|
115
|
+
y: number(found.y),
|
|
116
|
+
width: number(found.width),
|
|
117
|
+
height: number(found.height),
|
|
118
|
+
},
|
|
119
|
+
})
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
return {
|
|
123
|
+
schemaVersion: 1,
|
|
124
|
+
kind: "visual-geometry",
|
|
125
|
+
verdict: failed === 0 ? "PASS" : "FAIL",
|
|
126
|
+
viewport: checked.spec.viewport,
|
|
127
|
+
total: elements.length,
|
|
128
|
+
passed: elements.length - failed,
|
|
129
|
+
failed,
|
|
130
|
+
elements,
|
|
131
|
+
generatedAt: options.generatedAt || new Date().toISOString(),
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
export function buildVisualRepairPlan(receipt, pixelDiff = null) {
|
|
136
|
+
const failedElements = (receipt?.elements || []).filter((item) => item.verdict !== "PASS")
|
|
137
|
+
return {
|
|
138
|
+
schemaVersion: 1,
|
|
139
|
+
action: failedElements.length || (pixelDiff && pixelDiff.differentPixels > 0) ? "repair" : "accept",
|
|
140
|
+
failedElementIDs: failedElements.map((item) => item.id),
|
|
141
|
+
geometryFailures: failedElements.map((item) => ({ id: item.id, failures: item.failures || [item.reason] })),
|
|
142
|
+
pixelRegion: pixelDiff?.bounds || null,
|
|
143
|
+
directives: [
|
|
144
|
+
...(failedElements.length ? ["edit only the owning component/style for failed geometry unless evidence shows a shared token defect"] : []),
|
|
145
|
+
...(pixelDiff?.bounds ? ["inspect and, if vision is available, crop only the largest changed pixel region before another edit"] : []),
|
|
146
|
+
"re-render the affected viewport and produce a fresh receipt before declaring success",
|
|
147
|
+
],
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
export function responsiveViewportMatrix(input = null) {
|
|
152
|
+
if (Array.isArray(input) && input.length) return input
|
|
153
|
+
return [
|
|
154
|
+
{ id: "mobile", width: 390, height: 844 },
|
|
155
|
+
{ id: "tablet", width: 768, height: 1024 },
|
|
156
|
+
{ id: "desktop", width: 1440, height: 900 },
|
|
157
|
+
{ id: "wide", width: 1920, height: 1080 },
|
|
158
|
+
]
|
|
159
|
+
}
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "opencode-agent-skill",
|
|
3
|
-
"version": "
|
|
4
|
-
"description": "
|
|
3
|
+
"version": "11.0.0",
|
|
4
|
+
"description": "Perception-aware evidence-first engineering runtime with adaptive context, capability routing, visual/browser verification, weak-model recovery and benchmark-gated execution for OpenCode",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
7
7
|
"ocskill": "bin/ocskill.mjs"
|
|
@@ -23,7 +23,7 @@
|
|
|
23
23
|
"evals": "node scripts/eval-skills.mjs",
|
|
24
24
|
"evals:live:validate": "node scripts/validate-live-suite.mjs",
|
|
25
25
|
"test": "node --test test/*.test.mjs",
|
|
26
|
-
"ci": "npm run syntax && npm run validate && npm run evals && npm run evals:router && npm run evals:live:validate && npm run evals:long:validate && npm run evals:polyglot:validate && npm test && npm pack --dry-run && npm run smoke:pack && npm run smoke:plain-install",
|
|
26
|
+
"ci": "npm run syntax && npm run validate && npm run evals && npm run evals:router && npm run evals:v11:validate && npm run evals:live:validate && npm run evals:long:validate && npm run evals:polyglot:validate && npm test && npm pack --dry-run && npm run smoke:pack && npm run smoke:plain-install",
|
|
27
27
|
"postinstall": "node scripts/install.mjs",
|
|
28
28
|
"preuninstall": "node scripts/uninstall.mjs",
|
|
29
29
|
"prepublishOnly": "npm run ci",
|
|
@@ -45,7 +45,10 @@
|
|
|
45
45
|
"release:check-tag": "node scripts/check-release-tag.mjs",
|
|
46
46
|
"evals:polyglot:validate": "node scripts/validate-live-suite.mjs --suite polyglot",
|
|
47
47
|
"evals:polyglot": "node scripts/eval-live.mjs --suite polyglot",
|
|
48
|
-
"evals:matrix:gate": "node scripts/eval-matrix.mjs --require-confidence"
|
|
48
|
+
"evals:matrix:gate": "node scripts/eval-matrix.mjs --require-confidence",
|
|
49
|
+
"evals:ablation": "node scripts/eval-ablation.mjs",
|
|
50
|
+
"evals:v11": "node --test test/*-v11.test.mjs",
|
|
51
|
+
"evals:v11:validate": "node scripts/validate-v11-suite.mjs"
|
|
49
52
|
},
|
|
50
53
|
"keywords": [
|
|
51
54
|
"opencode",
|
|
@@ -55,7 +58,10 @@
|
|
|
55
58
|
"engineering-workflow",
|
|
56
59
|
"code-review",
|
|
57
60
|
"debugging",
|
|
58
|
-
"verification"
|
|
61
|
+
"verification",
|
|
62
|
+
"visual-testing",
|
|
63
|
+
"browser-automation",
|
|
64
|
+
"multimodal-agents"
|
|
59
65
|
],
|
|
60
66
|
"engines": {
|
|
61
67
|
"node": ">=20"
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { readFile } from "node:fs/promises"
|
|
3
|
+
import path from "node:path"
|
|
4
|
+
import { compareEvalSummaries } from "../lib/eval-ablation.mjs"
|
|
5
|
+
|
|
6
|
+
const args = process.argv.slice(2)
|
|
7
|
+
|
|
8
|
+
function option(name, fallback) {
|
|
9
|
+
const index = args.indexOf(name)
|
|
10
|
+
return index >= 0 && index + 1 < args.length ? args[index + 1] : fallback
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
function positional() {
|
|
14
|
+
const values = []
|
|
15
|
+
for (let index = 0; index < args.length; index += 1) {
|
|
16
|
+
if (args[index].startsWith("--")) {
|
|
17
|
+
if (["--pass-rate-tolerance", "--min-initial-reduction", "--max-token-ratio", "--max-duration-ratio", "--min-cacheable-ratio", "--min-evidence-reuse-ratio", "--max-repeated-stable-ratio"].includes(args[index])) index += 1
|
|
18
|
+
continue
|
|
19
|
+
}
|
|
20
|
+
values.push(args[index])
|
|
21
|
+
}
|
|
22
|
+
return values
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
const [referenceFile, candidateFile] = positional()
|
|
26
|
+
if (!referenceFile || !candidateFile) {
|
|
27
|
+
console.error("Usage: node scripts/eval-ablation.mjs <reference-summary.json> <candidate-summary.json> [--require-gate] [--min-initial-reduction 0.10]")
|
|
28
|
+
process.exit(2)
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
const [reference, candidate] = await Promise.all([
|
|
32
|
+
readFile(path.resolve(referenceFile), "utf8").then(JSON.parse),
|
|
33
|
+
readFile(path.resolve(candidateFile), "utf8").then(JSON.parse),
|
|
34
|
+
])
|
|
35
|
+
|
|
36
|
+
const report = compareEvalSummaries(reference, candidate, {
|
|
37
|
+
passRateTolerance: Number(option("--pass-rate-tolerance", "0")),
|
|
38
|
+
minInitialInputReduction: Number(option("--min-initial-reduction", "0.10")),
|
|
39
|
+
maxTotalTokenRatio: Number(option("--max-token-ratio", "1.05")),
|
|
40
|
+
maxDurationRatio: Number(option("--max-duration-ratio", "1.10")),
|
|
41
|
+
minCacheableRatio: option("--min-cacheable-ratio", null) == null ? null : Number(option("--min-cacheable-ratio", null)),
|
|
42
|
+
minEvidenceReuseRatio: option("--min-evidence-reuse-ratio", null) == null ? null : Number(option("--min-evidence-reuse-ratio", null)),
|
|
43
|
+
maxRepeatedStableRatio: option("--max-repeated-stable-ratio", null) == null ? null : Number(option("--max-repeated-stable-ratio", null)),
|
|
44
|
+
})
|
|
45
|
+
|
|
46
|
+
console.log(JSON.stringify(report, null, 2))
|
|
47
|
+
if (args.includes("--require-gate") && !report.gateEligible) process.exitCode = 1
|
package/scripts/eval-matrix.mjs
CHANGED
|
@@ -32,6 +32,8 @@ const onlyLive = has("--standard-only")
|
|
|
32
32
|
const onlyPolyglot = has("--polyglot-only")
|
|
33
33
|
const withoutPolyglot = has("--without-polyglot")
|
|
34
34
|
const requireConfidence = has("--require-confidence")
|
|
35
|
+
const maxInitialInputRatio = Number(argValue("--max-initial-input-ratio", "1.5"))
|
|
36
|
+
const maxTokenRatio = Number(argValue("--max-token-ratio", "1.75"))
|
|
35
37
|
|
|
36
38
|
if (!model) {
|
|
37
39
|
console.error("Usage: node scripts/eval-matrix.mjs --model provider/model [--trials 3] [--auth current|env-only] [--variant high] [--without-polyglot|--long-only|--standard-only|--polyglot-only]")
|
|
@@ -110,7 +112,10 @@ const baselineCount = results.filter((item) => item.mode === "baseline").length
|
|
|
110
112
|
const uesCount = results.filter((item) => item.mode === "ues").length
|
|
111
113
|
const coverageComplete = baselineCount === expectedPerMode && uesCount === expectedPerMode
|
|
112
114
|
const summary = summarizeEvalResults(results)
|
|
113
|
-
const confidence = pairedBenchmarkConfidence(results
|
|
115
|
+
const confidence = pairedBenchmarkConfidence(results, {
|
|
116
|
+
maxInitialInputRatio: Number.isFinite(maxInitialInputRatio) ? maxInitialInputRatio : 1.5,
|
|
117
|
+
maxTokenRatio: Number.isFinite(maxTokenRatio) ? maxTokenRatio : 1.75,
|
|
118
|
+
})
|
|
114
119
|
|
|
115
120
|
const report = {
|
|
116
121
|
schemaVersion: 1,
|
|
@@ -134,7 +139,11 @@ const report = {
|
|
|
134
139
|
mode: item.mode,
|
|
135
140
|
passed: item.passed === true,
|
|
136
141
|
durationMs: item.durationMs ?? null,
|
|
137
|
-
telemetry:
|
|
142
|
+
telemetry: {
|
|
143
|
+
...(item.telemetry?.costSamples > 0 ? { cost: item.telemetry.cost } : {}),
|
|
144
|
+
...(item.telemetry?.firstUsage ? { firstUsage: item.telemetry.firstUsage } : {}),
|
|
145
|
+
...(item.telemetry?.tokens ? { tokens: item.telemetry.tokens } : {}),
|
|
146
|
+
},
|
|
138
147
|
})),
|
|
139
148
|
summary,
|
|
140
149
|
confidence,
|
|
@@ -149,6 +158,8 @@ console.log("- UES: " + (summary.modes.ues?.passed || 0) + "/" + (summary.modes.
|
|
|
149
158
|
console.log("- pass-rate delta: " + (summary.passRateDelta == null ? "n/a" : (summary.passRateDelta * 100).toFixed(1) + " pp"))
|
|
150
159
|
console.log("- coverage: " + (coverageComplete ? "COMPLETE" : "INCOMPLETE"))
|
|
151
160
|
console.log("- paired confidence: " + (confidence.promotionEligible ? "SUPPORTED" : "NOT YET SUPPORTED") + " (pairs=" + confidence.pairs + ", p=" + confidence.pValue.toFixed(4) + ")")
|
|
161
|
+
console.log("- initial-input ratio UES/baseline: " + (confidence.initialInput.ratio == null ? "n/a" : confidence.initialInput.ratio.toFixed(3)) + " (max " + confidence.initialInput.maxRatio + ")")
|
|
162
|
+
console.log("- total-token ratio UES/baseline: " + (confidence.tokens.ratio == null ? "n/a" : confidence.tokens.ratio.toFixed(3)) + " (max " + confidence.tokens.maxRatio + ")")
|
|
152
163
|
console.log("- confidence gate: " + (requireConfidence ? "REQUIRED" : "report-only"))
|
|
153
164
|
console.log("- report: " + reportFile)
|
|
154
165
|
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
import { existsSync } from "node:fs"
|
|
2
|
+
import { readFile } from "node:fs/promises"
|
|
3
|
+
import path from "node:path"
|
|
4
|
+
import { fileURLToPath } from "node:url"
|
|
5
|
+
|
|
6
|
+
const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..")
|
|
7
|
+
const file = path.join(root, "evals", "v11", "tasks.json")
|
|
8
|
+
const suite = JSON.parse(await readFile(file, "utf8"))
|
|
9
|
+
const errors = []
|
|
10
|
+
|
|
11
|
+
const REQUIRED_V11_FILES = [
|
|
12
|
+
"lib/context-engine-v11.mjs",
|
|
13
|
+
"lib/evidence-store.mjs",
|
|
14
|
+
"lib/evidence-budget.mjs",
|
|
15
|
+
"lib/prompt-cache.mjs",
|
|
16
|
+
"lib/capability-registry.mjs",
|
|
17
|
+
"lib/browser-adapter.mjs",
|
|
18
|
+
"lib/browser-runtime.mjs",
|
|
19
|
+
"lib/png-diff.mjs",
|
|
20
|
+
"lib/visual-spec.mjs",
|
|
21
|
+
"lib/dynamic-workflow.mjs",
|
|
22
|
+
"lib/skill-quality.mjs",
|
|
23
|
+
"lib/ui-inspector.mjs",
|
|
24
|
+
"lib/v11-metrics.mjs",
|
|
25
|
+
"global-config/agents/visual-verifier.md",
|
|
26
|
+
"global-config/agents/merge-arbiter.md",
|
|
27
|
+
]
|
|
28
|
+
for (const relative of REQUIRED_V11_FILES) {
|
|
29
|
+
if (!existsSync(path.join(root, relative))) errors.push("missing V11 runtime file: " + relative)
|
|
30
|
+
}
|
|
31
|
+
const ids = new Set()
|
|
32
|
+
const categories = new Set()
|
|
33
|
+
|
|
34
|
+
if (!Array.isArray(suite.tasks) || suite.tasks.length < 10) errors.push("V11 suite must contain at least 10 tasks")
|
|
35
|
+
|
|
36
|
+
for (const task of suite.tasks || []) {
|
|
37
|
+
if (!task.id || ids.has(task.id)) errors.push("missing/duplicate id: " + (task.id || "<missing>"))
|
|
38
|
+
ids.add(task.id)
|
|
39
|
+
if (!task.category) errors.push((task.id || "<missing>") + ": missing category")
|
|
40
|
+
categories.add(task.category)
|
|
41
|
+
if (!task.objective || task.objective.length < 30) errors.push((task.id || "<missing>") + ": objective is too weak")
|
|
42
|
+
if (!Array.isArray(task.requiredFiles) || !task.requiredFiles.length) errors.push((task.id || "<missing>") + ": requiredFiles missing")
|
|
43
|
+
for (const relative of task.requiredFiles || []) {
|
|
44
|
+
if (!existsSync(path.join(root, relative))) errors.push((task.id || "<missing>") + ": missing required file " + relative)
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
for (const category of ["context","routing","visual","browser","workflow","sidecar"]) {
|
|
49
|
+
if (!categories.has(category)) errors.push("V11 suite missing category: " + category)
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
if (errors.length) {
|
|
53
|
+
console.error("V11 suite validation failed:")
|
|
54
|
+
for (const error of errors) console.error("- " + error)
|
|
55
|
+
process.exit(1)
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
console.log("Validated " + suite.tasks.length + " V11 contract tasks across " + categories.size + " categories and " + REQUIRED_V11_FILES.length + " required runtime files.")
|
package/scripts/validate.mjs
CHANGED
|
@@ -87,9 +87,12 @@ for (const entry of await readdir(agentsRoot, { withFileTypes: true })) {
|
|
|
87
87
|
if (!/mode:\s*subagent/.test(source)) errors.push(`${entry.name}: agent must use mode: subagent`)
|
|
88
88
|
}
|
|
89
89
|
|
|
90
|
-
|
|
90
|
+
const V11_REQUIRED_SKILLS = ["visual-fidelity","browser-qa","design-source","responsive-verification","component-visual-testing","browser-security","skill-authoring","skill-evaluation","dynamic-workflow"]
|
|
91
|
+
for (const id of V11_REQUIRED_SKILLS) if (!ids.has(id)) errors.push(`missing V11 skill ${id}`)
|
|
92
|
+
|
|
93
|
+
if (ids.size < 48) errors.push(`expected at least 48 skills for V11, found ${ids.size}`)
|
|
91
94
|
if (commands < 11) errors.push(`expected at least 11 commands, found ${commands}`)
|
|
92
|
-
if (agents <
|
|
95
|
+
if (agents < 12) errors.push(`expected at least 12 subagents for V11, found ${agents}`)
|
|
93
96
|
|
|
94
97
|
const routerIndex = path.join(pluginsRoot, "ues-router", "index.js")
|
|
95
98
|
const routerCore = path.join(pluginsRoot, "ues-router", "router.js")
|
|
@@ -115,6 +118,15 @@ if (existsSync(routerIndex)) {
|
|
|
115
118
|
if (!source.includes('name: "task_policy"')) errors.push("v8 router plugin must expose adaptive task policy")
|
|
116
119
|
if (!source.includes('name: "cancel_task"')) errors.push("v8 router plugin must expose executor cancellation")
|
|
117
120
|
if (!source.includes('name: "recover_task"')) errors.push("v8 router plugin must expose task-scoped recovery")
|
|
121
|
+
if (!source.includes('name: "capability_requirements"')) errors.push("v11 router plugin must expose capability requirements")
|
|
122
|
+
if (!source.includes('name: "evidence_get"')) errors.push("v11 router plugin must expose bounded evidence retrieval")
|
|
123
|
+
if (!source.includes('name: "browser_plan"')) errors.push("v11 router plugin must expose browser verification planning")
|
|
124
|
+
if (!source.includes('name: "browser_inspect"')) errors.push("v11 router plugin must expose bounded browser inspection")
|
|
125
|
+
if (!source.includes('name: "visual_geometry"')) errors.push("v11 router plugin must expose visual geometry receipts")
|
|
126
|
+
if (!source.includes('name: "visual_compare"')) errors.push("v11 router plugin must expose deterministic PNG comparison")
|
|
127
|
+
if (!source.includes('name: "workflow_plan"')) errors.push("v11 router plugin must expose dynamic workflow planning")
|
|
128
|
+
if (!source.includes('name: "ui_tokens"')) errors.push("v11 router plugin must expose compact design-token extraction")
|
|
129
|
+
if (!source.includes('name: "ui_layout"')) errors.push("v11 router plugin must expose responsive UI layout verification")
|
|
118
130
|
if (!source.includes("ctx.session.interrupt")) errors.push("v8 task dispatch must interrupt timed-out executors")
|
|
119
131
|
if (!source.includes('"work", "heartbeat"')) errors.push("v8 task dispatch must refresh task leases")
|
|
120
132
|
}
|
|
@@ -138,11 +150,11 @@ if (existsSync(plainInstallSmoke)) {
|
|
|
138
150
|
}
|
|
139
151
|
}
|
|
140
152
|
|
|
141
|
-
for (const name of ["process-runner.mjs","evidence-receipt.mjs","gate-receipt.mjs","runtime-events.mjs","context-manifest.mjs","orchestrator-policy.mjs","worktree-sandbox.mjs","learning-engine.mjs","hermes-bridge.mjs","control-center.mjs"]) {
|
|
153
|
+
for (const name of ["process-runner.mjs","evidence-receipt.mjs","gate-receipt.mjs","runtime-events.mjs","context-manifest.mjs","orchestrator-policy.mjs","worktree-sandbox.mjs","learning-engine.mjs","hermes-bridge.mjs","control-center.mjs","evidence-store.mjs","evidence-budget.mjs","prompt-cache.mjs","capability-registry.mjs","visual-spec.mjs","png-diff.mjs","browser-adapter.mjs","browser-runtime.mjs","dynamic-workflow.mjs","skill-quality.mjs","ui-inspector.mjs","context-engine-v11.mjs","v11-metrics.mjs"]) {
|
|
142
154
|
if (!existsSync(path.join(root, "lib", name))) errors.push(`missing V8 core module ${name}`)
|
|
143
155
|
}
|
|
144
156
|
|
|
145
|
-
for (const name of ["codebase-mapper.md","plan-checker.md","executor.md","integration-verifier.md"]) {
|
|
157
|
+
for (const name of ["codebase-mapper.md","plan-checker.md","executor.md","integration-verifier.md","visual-verifier.md","merge-arbiter.md"]) {
|
|
146
158
|
if (!existsSync(path.join(agentsRoot, name))) errors.push(`missing V6 subagent ${name}`)
|
|
147
159
|
}
|
|
148
160
|
|