opencode-agent-skill 10.0.0 → 11.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/CHANGELOG.md +67 -0
  2. package/README.md +49 -3
  3. package/bin/ocskill.mjs +330 -5
  4. package/docs/V11-PERCEPTION-ADAPTIVE-EXECUTION.md +75 -0
  5. package/docs/V11-PERCEPTION-ADAPTIVE.md +220 -0
  6. package/evals/router-triggers.json +82 -0
  7. package/evals/routing.json +76 -0
  8. package/evals/v11/tasks.json +122 -0
  9. package/global-config/agents/merge-arbiter.md +12 -0
  10. package/global-config/agents/visual-verifier.md +12 -0
  11. package/global-config/plugins/ues-router/index.js +272 -2
  12. package/global-config/plugins/ues-router/router.js +27 -3
  13. package/global-config/skills/browser-qa/SKILL.md +14 -0
  14. package/global-config/skills/browser-qa/references/workflow.md +11 -0
  15. package/global-config/skills/browser-security/SKILL.md +12 -0
  16. package/global-config/skills/component-visual-testing/SKILL.md +10 -0
  17. package/global-config/skills/design-source/SKILL.md +10 -0
  18. package/global-config/skills/design-source/references/workflow.md +12 -0
  19. package/global-config/skills/dynamic-workflow/SKILL.md +18 -0
  20. package/global-config/skills/dynamic-workflow/references/workflow.md +19 -0
  21. package/global-config/skills/responsive-verification/SKILL.md +10 -0
  22. package/global-config/skills/skill-authoring/SKILL.md +12 -0
  23. package/global-config/skills/skill-evaluation/SKILL.md +17 -0
  24. package/global-config/skills/visual-fidelity/SKILL.md +14 -0
  25. package/global-config/skills/visual-fidelity/references/workflow.md +14 -0
  26. package/lib/browser-adapter.mjs +82 -0
  27. package/lib/browser-runtime.mjs +193 -0
  28. package/lib/capability-registry.mjs +109 -0
  29. package/lib/context-engine-v11.mjs +146 -0
  30. package/lib/context-manifest.mjs +16 -3
  31. package/lib/control-center.mjs +12 -2
  32. package/lib/dynamic-workflow.mjs +179 -0
  33. package/lib/eval-ablation.mjs +43 -1
  34. package/lib/eval-report.mjs +72 -0
  35. package/lib/eval-telemetry.mjs +61 -0
  36. package/lib/evidence-budget.mjs +84 -0
  37. package/lib/evidence-store.mjs +178 -0
  38. package/lib/hermes-bridge.mjs +45 -1
  39. package/lib/model-config.mjs +9 -1
  40. package/lib/model-policy.mjs +52 -1
  41. package/lib/orchestrator-policy.mjs +1 -1
  42. package/lib/png-diff.mjs +229 -0
  43. package/lib/prompt-cache.mjs +60 -0
  44. package/lib/skill-quality.mjs +72 -0
  45. package/lib/task-engine.mjs +78 -4
  46. package/lib/ui-inspector.mjs +152 -0
  47. package/lib/v11-metrics.mjs +64 -0
  48. package/lib/visual-spec.mjs +159 -0
  49. package/package.json +10 -5
  50. package/scripts/eval-ablation.mjs +4 -1
  51. package/scripts/validate-v11-suite.mjs +58 -0
  52. package/scripts/validate.mjs +16 -4
@@ -0,0 +1,72 @@
1
+ import { existsSync } from "node:fs"
2
+ import { readFile, readdir } from "node:fs/promises"
3
+ import path from "node:path"
4
+
5
+ function words(value) {
6
+ return new Set(String(value || "").toLowerCase().match(/[a-z0-9-]{3,}/g) || [])
7
+ }
8
+
9
+ function similarity(a, b) {
10
+ const left = words(a)
11
+ const right = words(b)
12
+ const union = new Set([...left, ...right])
13
+ let intersection = 0
14
+ for (const item of left) if (right.has(item)) intersection += 1
15
+ return union.size ? intersection / union.size : 0
16
+ }
17
+
18
+ function frontmatter(source) {
19
+ const name = source.match(/^name:\s*([^\r\n]+)/m)?.[1]?.trim() || null
20
+ const description = source.match(/^description:\s*([^\r\n]+)/m)?.[1]?.trim() || null
21
+ return { name, description }
22
+ }
23
+
24
+ export async function lintSkillCatalog(root = process.cwd(), options = {}) {
25
+ const skillsRoot = path.join(path.resolve(root), "global-config", "skills")
26
+ const entries = await readdir(skillsRoot, { withFileTypes: true })
27
+ const skills = []
28
+ const errors = []
29
+ const warnings = []
30
+
31
+ for (const entry of entries) {
32
+ if (!entry.isDirectory()) continue
33
+ const file = path.join(skillsRoot, entry.name, "SKILL.md")
34
+ if (!existsSync(file)) {
35
+ errors.push({ skill: entry.name, issue: "missing-skill-md" })
36
+ continue
37
+ }
38
+ const source = await readFile(file, "utf8")
39
+ const meta = frontmatter(source)
40
+ const body = source.replace(/^---[\s\S]*?---\s*/m, "")
41
+ const row = {
42
+ id: entry.name,
43
+ description: meta.description,
44
+ chars: source.length,
45
+ bodyChars: body.length,
46
+ estimatedTokens: Math.ceil(source.length / 4),
47
+ }
48
+ skills.push(row)
49
+ if (meta.name !== entry.name) errors.push({ skill: entry.name, issue: "frontmatter-name-mismatch" })
50
+ if (!meta.description) errors.push({ skill: entry.name, issue: "missing-description" })
51
+ if (row.estimatedTokens > Number(options.maxSkillTokens || 1800)) warnings.push({ skill: entry.name, issue: "large-entrypoint", estimatedTokens: row.estimatedTokens })
52
+ }
53
+
54
+ const collisions = []
55
+ const threshold = Number(options.collisionThreshold || 0.58)
56
+ for (let i = 0; i < skills.length; i += 1) {
57
+ for (let j = i + 1; j < skills.length; j += 1) {
58
+ const score = similarity(skills[i].description, skills[j].description)
59
+ if (score >= threshold) collisions.push({ a: skills[i].id, b: skills[j].id, score: Number(score.toFixed(3)) })
60
+ }
61
+ }
62
+
63
+ return {
64
+ schemaVersion: 1,
65
+ valid: errors.length === 0,
66
+ skillCount: skills.length,
67
+ errors,
68
+ warnings,
69
+ collisions: collisions.sort((a, b) => b.score - a.score),
70
+ skills: skills.sort((a, b) => a.id.localeCompare(b.id)),
71
+ }
72
+ }
@@ -10,6 +10,11 @@ import { validateVerificationReceipt } from "./evidence-receipt.mjs"
10
10
  import { appendRuntimeEvent, readRuntimeEvents } from "./runtime-events.mjs"
11
11
  import { createGateReceipt, validateGateReceipt } from "./gate-receipt.mjs"
12
12
  import { classifyEngineeringTask, recoveryPolicyForAttempt } from "./orchestrator-policy.mjs"
13
+ import { planEvidenceBudget } from "./evidence-budget.mjs"
14
+ import { putEvidence } from "./evidence-store.mjs"
15
+ import { inferTaskCapabilities } from "./capability-registry.mjs"
16
+ import { buildPromptEnvelope } from "./prompt-cache.mjs"
17
+ import { externalizeContextExcerpts } from "./context-engine-v11.mjs"
13
18
 
14
19
  const WORK_DIR = ".ues-work"
15
20
  const STATE_SCHEMA = 4
@@ -1068,11 +1073,19 @@ async function buildContextPack(loaded, taskID) {
1068
1073
  const task = taskByID(loaded.plan, taskID)
1069
1074
  const spec = await readFile(loaded.paths.spec, "utf8").catch(() => "")
1070
1075
  const dependencyReports = {}
1076
+ const dependencyReportRefs = {}
1071
1077
 
1072
1078
  for (const dep of task.dependsOn || []) {
1073
1079
  const file = path.join(loaded.paths.reports, `${dep}.md`)
1074
1080
  if (!existsSync(file)) continue
1075
- dependencyReports[dep] = (await readFile(file, "utf8")).slice(0, 12000)
1081
+ const report = await readFile(file, "utf8")
1082
+ dependencyReports[dep] = report.slice(0, 6000)
1083
+ const stored = await putEvidence(loaded.paths.root, report, {
1084
+ kind: "dependency-report",
1085
+ source: path.relative(loaded.paths.root, file).replaceAll("\\", "/"),
1086
+ summary: `Dependency report for ${dep}`,
1087
+ }).catch(() => null)
1088
+ if (stored?.ref) dependencyReportRefs[dep] = stored.ref
1076
1089
  }
1077
1090
 
1078
1091
  const taskText = [
@@ -1095,27 +1108,88 @@ async function buildContextPack(loaded, taskID) {
1095
1108
  effectiveMaxSkills: recovery.maxSkills,
1096
1109
  recovery,
1097
1110
  }
1098
- const contextManifest = await buildContextManifest(
1111
+ const capabilities = inferTaskCapabilities(taskText, {
1112
+ longContext: contextPolicy.mode === "long-horizon",
1113
+ coding: true,
1114
+ toolCalling: true,
1115
+ filesystem: true,
1116
+ })
1117
+ const evidenceBudget = planEvidenceBudget(effectiveContextPolicy, task, capabilities.required)
1118
+ const rawContextManifest = await buildContextManifest(
1099
1119
  loaded.paths.root,
1100
1120
  task,
1101
1121
  {
1102
- budget: recovery.contextBudget,
1122
+ budget: evidenceBudget.total,
1103
1123
  strategy: recovery.contextStrategy,
1124
+ evidenceBudget,
1104
1125
  },
1105
1126
  ).catch(() => null)
1127
+ const externalizedContext = rawContextManifest
1128
+ ? await externalizeContextExcerpts(loaded.paths.root, rawContextManifest, {
1129
+ task,
1130
+ threshold: Math.max(1800, Math.round(evidenceBudget.total * 0.12)),
1131
+ inlineChars: Math.max(600, Math.min(1600, Math.round(evidenceBudget.total * 0.08))),
1132
+ }).catch(() => ({ manifest: rawContextManifest, externalized: [], externalizedBytes: 0 }))
1133
+ : { manifest: null, externalized: [], externalizedBytes: 0 }
1134
+ const contextManifest = externalizedContext.manifest
1106
1135
  const learnings = await relevantAcceptedLearnings(
1107
1136
  loaded.paths.root,
1108
1137
  [task.title, task.summary, ...(task.acceptance || [])].join(" "),
1109
1138
  ).catch(() => [])
1110
1139
 
1140
+ const specStored = spec
1141
+ ? await putEvidence(loaded.paths.root, spec, {
1142
+ kind: "work-spec",
1143
+ source: path.relative(loaded.paths.root, loaded.paths.spec).replaceAll("\\", "/"),
1144
+ summary: "Durable work specification",
1145
+ }).catch(() => null)
1146
+ : null
1147
+ const promptEnvelope = buildPromptEnvelope({
1148
+ role: "executor",
1149
+ invariants: "evidence-first; scoped edits; fresh verification; no unsupported completion claims",
1150
+ skills: contextPolicy.domains || [],
1151
+ projectFacts: {
1152
+ instructions: contextManifest?.instructions || [],
1153
+ strategy: recovery.contextStrategy,
1154
+ },
1155
+ task,
1156
+ evidence: [
1157
+ ...(specStored?.ref ? [specStored.ref] : []),
1158
+ ...Object.values(dependencyReportRefs),
1159
+ ...(contextManifest?.evidencePointers || []).map((item) => item.ref),
1160
+ ...(contextManifest?.rankedReferences || []).slice(0, 12).map((item) => item.path),
1161
+ ],
1162
+ recentFailure: recovery.stage === "initial" ? null : loaded.state.tasks?.[taskID]?.lastError || null,
1163
+ nextAction: loaded.state.checkpoint?.nextAction || loaded.state.nextAction || null,
1164
+ })
1165
+
1111
1166
  return {
1112
1167
  schemaVersion: STATE_SCHEMA,
1168
+ contextSchemaVersion: 6,
1113
1169
  contextPolicy: effectiveContextPolicy,
1170
+ capabilities,
1171
+ evidenceBudget,
1172
+ promptCache: {
1173
+ stablePrefixHash: promptEnvelope.stablePrefixHash,
1174
+ dynamicHash: promptEnvelope.dynamicHash,
1175
+ stableChars: promptEnvelope.stableChars,
1176
+ dynamicChars: promptEnvelope.dynamicChars,
1177
+ cacheableRatio: promptEnvelope.cacheableRatio,
1178
+ },
1114
1179
  attempt,
1115
1180
  slug: loaded.state.slug,
1116
1181
  task,
1117
1182
  taskBrief: path.relative(loaded.paths.root, path.join(loaded.paths.tasks, `${taskID}.md`)).replaceAll("\\", "/"),
1118
- spec: spec.slice(0, 24000),
1183
+ spec: spec.slice(0, Math.min(12000, evidenceBudget.buckets.instructions + evidenceBudget.buckets.task)),
1184
+ evidencePointers: {
1185
+ spec: specStored?.ref || null,
1186
+ dependencyReports: dependencyReportRefs,
1187
+ context: (contextManifest?.evidencePointers || []).map((item) => item.ref),
1188
+ },
1189
+ evidenceStore: {
1190
+ refs: externalizedContext.externalized.length,
1191
+ externalizedBytes: externalizedContext.externalizedBytes,
1192
+ },
1119
1193
  dependencyReports,
1120
1194
  decisions: loaded.state.decisions || [],
1121
1195
  blockers: loaded.state.blockers || [],
@@ -0,0 +1,152 @@
1
+ function uniq(values) {
2
+ return [...new Set(values.filter((value) => value !== null && value !== undefined && value !== ""))]
3
+ }
4
+
5
+ function round(value, digits = 3) {
6
+ const factor = 10 ** digits
7
+ return Math.round(Number(value) * factor) / factor
8
+ }
9
+
10
+ function normalizeBox(item = {}) {
11
+ const box = item.box && typeof item.box === "object" ? item.box : item
12
+ const x = Number(box.x)
13
+ const y = Number(box.y)
14
+ const width = Number(box.width)
15
+ const height = Number(box.height)
16
+ if (![x,y,width,height].every(Number.isFinite)) return null
17
+ return {
18
+ id: String(item.id || item.name || "").trim() || null,
19
+ role: item.role || null,
20
+ x, y, width, height,
21
+ right: x + width,
22
+ bottom: y + height,
23
+ }
24
+ }
25
+
26
+ function containsBox(a, b) {
27
+ return a.x <= b.x && a.y <= b.y && a.right >= b.right && a.bottom >= b.bottom
28
+ }
29
+
30
+ function intersectionArea(a, b) {
31
+ const width = Math.max(0, Math.min(a.right, b.right) - Math.max(a.x, b.x))
32
+ const height = Math.max(0, Math.min(a.bottom, b.bottom) - Math.max(a.y, b.y))
33
+ return width * height
34
+ }
35
+
36
+ export function inspectResponsiveLayout(items = [], viewport = {}, options = {}) {
37
+ const width = Math.max(1, Number(viewport.width || 0))
38
+ const height = Math.max(1, Number(viewport.height || 0))
39
+ const boxes = items.map(normalizeBox).filter(Boolean)
40
+ const issues = []
41
+ const minTouch = Math.max(1, Number(options.minTouchTarget || 44))
42
+ const touchRoles = new Set(["button","link","checkbox","radio","switch","tab","menuitem"])
43
+
44
+ for (const box of boxes) {
45
+ if (box.x < 0 || box.y < 0 || box.right > width || box.bottom > height) {
46
+ issues.push({
47
+ kind: "viewport-overflow",
48
+ id: box.id,
49
+ box,
50
+ viewport: { width, height },
51
+ })
52
+ }
53
+ if (touchRoles.has(String(box.role || "").toLowerCase()) && (box.width < minTouch || box.height < minTouch)) {
54
+ issues.push({
55
+ kind: "small-touch-target",
56
+ id: box.id,
57
+ role: box.role,
58
+ actual: { width: box.width, height: box.height },
59
+ minimum: minTouch,
60
+ })
61
+ }
62
+ }
63
+
64
+ const maxPairChecks = Math.max(0, Math.min(10000, Number(options.maxPairChecks || 3000)))
65
+ let checks = 0
66
+ for (let i = 0; i < boxes.length; i += 1) {
67
+ for (let j = i + 1; j < boxes.length && checks < maxPairChecks; j += 1) {
68
+ checks += 1
69
+ const a = boxes[i]
70
+ const b = boxes[j]
71
+ if (containsBox(a,b) || containsBox(b,a)) continue
72
+ const area = intersectionArea(a,b)
73
+ if (!area) continue
74
+ const smaller = Math.max(1, Math.min(a.width*a.height, b.width*b.height))
75
+ const ratio = area / smaller
76
+ if (ratio >= Number(options.overlapRatio ?? 0.15)) {
77
+ issues.push({
78
+ kind: "element-overlap",
79
+ a: a.id,
80
+ b: b.id,
81
+ intersectionArea: area,
82
+ smallerElementRatio: round(ratio),
83
+ })
84
+ }
85
+ }
86
+ }
87
+
88
+ return {
89
+ schemaVersion: 1,
90
+ viewport: { width, height },
91
+ elementCount: boxes.length,
92
+ pairChecks: checks,
93
+ verdict: issues.length ? "FAIL" : "PASS",
94
+ issues,
95
+ }
96
+ }
97
+
98
+ function normalizeCssValue(value) {
99
+ return String(value || "").trim().replace(/\s+/g, " ")
100
+ }
101
+
102
+ export function extractDesignTokens(css = "") {
103
+ const source = String(css || "")
104
+ const variables = {}
105
+ for (const match of source.matchAll(/--([a-zA-Z0-9_-]+)\s*:\s*([^;}{]+)\s*;/g)) {
106
+ variables["--" + match[1]] = normalizeCssValue(match[2])
107
+ }
108
+
109
+ const colors = uniq([
110
+ ...source.matchAll(/#[0-9a-fA-F]{3,8}\b/g),
111
+ ...source.matchAll(/\b(?:rgb|rgba|hsl|hsla)\([^)]*\)/g),
112
+ ].map((match) => match[0].toLowerCase())).slice(0, 128)
113
+
114
+ const lengths = uniq([...source.matchAll(/-?\d*\.?\d+(?:px|rem|em)\b/g)].map((m)=>m[0]))
115
+ const px = lengths.filter((value)=>value.endsWith("px")).map((value)=>Number.parseFloat(value)).filter((value)=>Number.isFinite(value) && value >= 0)
116
+ const radii = uniq([...source.matchAll(/border-radius\s*:\s*([^;}{]+)/g)].map((m)=>normalizeCssValue(m[1]))).slice(0,64)
117
+ const fontSizes = uniq([...source.matchAll(/font-size\s*:\s*([^;}{]+)/g)].map((m)=>normalizeCssValue(m[1]))).slice(0,64)
118
+ const shadows = uniq([...source.matchAll(/box-shadow\s*:\s*([^;}{]+)/g)].map((m)=>normalizeCssValue(m[1]))).slice(0,64)
119
+
120
+ const spacingCandidates = uniq(px.filter((value)=>value <= 128).sort((a,b)=>a-b)).slice(0,32)
121
+
122
+ return {
123
+ schemaVersion: 1,
124
+ variables,
125
+ colors,
126
+ spacingCandidates,
127
+ radii,
128
+ fontSizes,
129
+ shadows,
130
+ stats: {
131
+ variableCount: Object.keys(variables).length,
132
+ colorCount: colors.length,
133
+ lengthCount: lengths.length,
134
+ },
135
+ }
136
+ }
137
+
138
+ export function designTokenEvidence(tokens = {}) {
139
+ const variableEntries = Object.entries(tokens.variables || {})
140
+ return {
141
+ schemaVersion: 1,
142
+ summary: {
143
+ variables: variableEntries.slice(0,40),
144
+ colors: (tokens.colors || []).slice(0,24),
145
+ spacingCandidates: (tokens.spacingCandidates || []).slice(0,20),
146
+ radii: (tokens.radii || []).slice(0,16),
147
+ fontSizes: (tokens.fontSizes || []).slice(0,16),
148
+ shadows: (tokens.shadows || []).slice(0,12),
149
+ },
150
+ stats: tokens.stats || {},
151
+ }
152
+ }
@@ -0,0 +1,64 @@
1
+ function finite(value) {
2
+ const n = Number(value)
3
+ return Number.isFinite(n) ? n : 0
4
+ }
5
+
6
+ function ratio(numerator, denominator) {
7
+ return denominator > 0 ? numerator / denominator : null
8
+ }
9
+
10
+ export function summarizeV11Efficiency(samples = []) {
11
+ const totals = {
12
+ samples: samples.length,
13
+ stableChars: 0,
14
+ dynamicChars: 0,
15
+ repeatedStableChars: 0,
16
+ externalizedEvidenceBytes: 0,
17
+ evidenceRefs: 0,
18
+ visualRepairAttempts: 0,
19
+ contextExpansions: 0,
20
+ modelEscalations: 0,
21
+ duplicateToolBlocks: 0,
22
+ loopBlocks: 0,
23
+ }
24
+
25
+ for (const sample of samples) {
26
+ const cache = sample.promptCache || sample.contextPack?.promptCache || {}
27
+ totals.stableChars += finite(cache.stableChars)
28
+ totals.dynamicChars += finite(cache.dynamicChars)
29
+ totals.repeatedStableChars += finite(sample.repeatedStableChars ?? cache.repeatedStableChars)
30
+
31
+ const evidence = sample.evidenceStore || sample.evidence || {}
32
+ totals.externalizedEvidenceBytes += finite(evidence.externalizedBytes ?? evidence.bytes)
33
+ totals.evidenceRefs += finite(evidence.refs ?? sample.evidenceRefs)
34
+
35
+ totals.visualRepairAttempts += finite(sample.visualRepairAttempts)
36
+ totals.contextExpansions += finite(sample.contextExpansions)
37
+ totals.modelEscalations += finite(sample.modelEscalations)
38
+ totals.duplicateToolBlocks += finite(sample.duplicateToolBlocks)
39
+ totals.loopBlocks += finite(sample.loopBlocks)
40
+ }
41
+
42
+ const inputChars = totals.stableChars + totals.dynamicChars
43
+ return {
44
+ schemaVersion: 1,
45
+ totals,
46
+ cacheablePrefixRatio: ratio(totals.stableChars, inputChars),
47
+ repeatedStableRatio: ratio(totals.repeatedStableChars, totals.stableChars),
48
+ externalizedEvidenceBytesPerSample: ratio(totals.externalizedEvidenceBytes, totals.samples),
49
+ evidenceRefsPerSample: ratio(totals.evidenceRefs, totals.samples),
50
+ visualRepairsPerSample: ratio(totals.visualRepairAttempts, totals.samples),
51
+ contextExpansionsPerSample: ratio(totals.contextExpansions, totals.samples),
52
+ modelEscalationsPerSample: ratio(totals.modelEscalations, totals.samples),
53
+ }
54
+ }
55
+
56
+ export function verifiedSuccessPer100kTokens(results = []) {
57
+ let success = 0
58
+ let tokens = 0
59
+ for (const item of results) {
60
+ if (item.passed === true && item.verified !== false) success += 1
61
+ tokens += finite(item.tokens ?? item.telemetry?.tokens?.total)
62
+ }
63
+ return tokens > 0 ? success / (tokens / 100_000) : null
64
+ }
@@ -0,0 +1,159 @@
1
+ function number(value, fallback = 0) {
2
+ const parsed = Number(value)
3
+ return Number.isFinite(parsed) ? parsed : fallback
4
+ }
5
+
6
+ function range(value, tolerance = 0) {
7
+ if (Array.isArray(value) && value.length >= 2) return [number(value[0]), number(value[1])]
8
+ const exact = number(value)
9
+ return [exact - tolerance, exact + tolerance]
10
+ }
11
+
12
+ function within(actual, expected, tolerance) {
13
+ const [min, max] = range(expected, tolerance)
14
+ return actual >= Math.min(min, max) && actual <= Math.max(min, max)
15
+ }
16
+
17
+ function normalizeElement(item = {}) {
18
+ return {
19
+ id: String(item.id || "").trim(),
20
+ role: item.role || null,
21
+ region: item.region || null,
22
+ expected: {
23
+ ...(item.expected || {}),
24
+ ...(item.x !== undefined ? { x: item.x } : {}),
25
+ ...(item.y !== undefined ? { y: item.y } : {}),
26
+ ...(item.width !== undefined ? { width: item.width } : {}),
27
+ ...(item.height !== undefined ? { height: item.height } : {}),
28
+ },
29
+ tolerance: {
30
+ position: number(item.tolerance?.position, 8),
31
+ size: number(item.tolerance?.size, 8),
32
+ },
33
+ required: item.required !== false,
34
+ }
35
+ }
36
+
37
+ export function normalizeVisualSpec(spec = {}) {
38
+ const viewport = {
39
+ width: Math.max(1, number(spec.viewport?.width, 1440)),
40
+ height: Math.max(1, number(spec.viewport?.height, 900)),
41
+ deviceScaleFactor: Math.max(0.1, number(spec.viewport?.deviceScaleFactor, 1)),
42
+ }
43
+ return {
44
+ schemaVersion: 1,
45
+ name: spec.name || null,
46
+ viewport,
47
+ elements: (spec.elements || []).map(normalizeElement).filter((item) => item.id),
48
+ regions: Array.isArray(spec.regions) ? spec.regions : [],
49
+ tokens: spec.tokens || {},
50
+ metadata: spec.metadata || {},
51
+ }
52
+ }
53
+
54
+ export function validateVisualSpec(spec = {}) {
55
+ const normalized = normalizeVisualSpec(spec)
56
+ const errors = []
57
+ const ids = new Set()
58
+ for (const item of normalized.elements) {
59
+ if (ids.has(item.id)) errors.push("duplicate element id: " + item.id)
60
+ ids.add(item.id)
61
+ for (const key of ["x", "y", "width", "height"]) {
62
+ if (item.expected[key] === undefined) continue
63
+ const values = Array.isArray(item.expected[key]) ? item.expected[key] : [item.expected[key]]
64
+ if (values.some((value) => !Number.isFinite(Number(value)))) errors.push(item.id + ": invalid " + key)
65
+ }
66
+ }
67
+ return { valid: errors.length === 0, errors, spec: normalized }
68
+ }
69
+
70
+ function actualMap(actual = []) {
71
+ const values = Array.isArray(actual) ? actual : Object.values(actual || {})
72
+ return new Map(values.filter((item) => item?.id).map((item) => [String(item.id), item]))
73
+ }
74
+
75
+ export function createGeometryReceipt(specInput, actualInput, options = {}) {
76
+ const checked = validateVisualSpec(specInput)
77
+ if (!checked.valid) throw new Error("Invalid visual spec: " + checked.errors.join("; "))
78
+ const actual = actualMap(actualInput)
79
+ const elements = []
80
+ let failed = 0
81
+
82
+ for (const expected of checked.spec.elements) {
83
+ const found = actual.get(expected.id)
84
+ if (!found) {
85
+ const pass = !expected.required
86
+ if (!pass) failed += 1
87
+ elements.push({ id: expected.id, verdict: pass ? "PASS" : "FAIL", reason: "missing", expected: expected.expected, actual: null })
88
+ continue
89
+ }
90
+
91
+ const failures = []
92
+ const checks = {}
93
+ for (const key of ["x", "y"]) {
94
+ if (expected.expected[key] === undefined) continue
95
+ const ok = within(number(found[key]), expected.expected[key], expected.tolerance.position)
96
+ checks[key] = ok
97
+ if (!ok) failures.push(key)
98
+ }
99
+ for (const key of ["width", "height"]) {
100
+ if (expected.expected[key] === undefined) continue
101
+ const ok = within(number(found[key]), expected.expected[key], expected.tolerance.size)
102
+ checks[key] = ok
103
+ if (!ok) failures.push(key)
104
+ }
105
+ const pass = failures.length === 0
106
+ if (!pass) failed += 1
107
+ elements.push({
108
+ id: expected.id,
109
+ verdict: pass ? "PASS" : "FAIL",
110
+ failures,
111
+ checks,
112
+ expected: expected.expected,
113
+ actual: {
114
+ x: number(found.x),
115
+ y: number(found.y),
116
+ width: number(found.width),
117
+ height: number(found.height),
118
+ },
119
+ })
120
+ }
121
+
122
+ return {
123
+ schemaVersion: 1,
124
+ kind: "visual-geometry",
125
+ verdict: failed === 0 ? "PASS" : "FAIL",
126
+ viewport: checked.spec.viewport,
127
+ total: elements.length,
128
+ passed: elements.length - failed,
129
+ failed,
130
+ elements,
131
+ generatedAt: options.generatedAt || new Date().toISOString(),
132
+ }
133
+ }
134
+
135
+ export function buildVisualRepairPlan(receipt, pixelDiff = null) {
136
+ const failedElements = (receipt?.elements || []).filter((item) => item.verdict !== "PASS")
137
+ return {
138
+ schemaVersion: 1,
139
+ action: failedElements.length || (pixelDiff && pixelDiff.differentPixels > 0) ? "repair" : "accept",
140
+ failedElementIDs: failedElements.map((item) => item.id),
141
+ geometryFailures: failedElements.map((item) => ({ id: item.id, failures: item.failures || [item.reason] })),
142
+ pixelRegion: pixelDiff?.bounds || null,
143
+ directives: [
144
+ ...(failedElements.length ? ["edit only the owning component/style for failed geometry unless evidence shows a shared token defect"] : []),
145
+ ...(pixelDiff?.bounds ? ["inspect and, if vision is available, crop only the largest changed pixel region before another edit"] : []),
146
+ "re-render the affected viewport and produce a fresh receipt before declaring success",
147
+ ],
148
+ }
149
+ }
150
+
151
+ export function responsiveViewportMatrix(input = null) {
152
+ if (Array.isArray(input) && input.length) return input
153
+ return [
154
+ { id: "mobile", width: 390, height: 844 },
155
+ { id: "tablet", width: 768, height: 1024 },
156
+ { id: "desktop", width: 1440, height: 900 },
157
+ { id: "wide", width: 1920, height: 1080 },
158
+ ]
159
+ }
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "opencode-agent-skill",
3
- "version": "10.0.0",
4
- "description": "Evidence-first engineering runtime with adaptive context, selective routing, weak-model recovery, bounded ACI and benchmark-gated verification for OpenCode",
3
+ "version": "11.0.0",
4
+ "description": "Perception-aware evidence-first engineering runtime with adaptive context, capability routing, visual/browser verification, weak-model recovery and benchmark-gated execution for OpenCode",
5
5
  "type": "module",
6
6
  "bin": {
7
7
  "ocskill": "bin/ocskill.mjs"
@@ -23,7 +23,7 @@
23
23
  "evals": "node scripts/eval-skills.mjs",
24
24
  "evals:live:validate": "node scripts/validate-live-suite.mjs",
25
25
  "test": "node --test test/*.test.mjs",
26
- "ci": "npm run syntax && npm run validate && npm run evals && npm run evals:router && npm run evals:live:validate && npm run evals:long:validate && npm run evals:polyglot:validate && npm test && npm pack --dry-run && npm run smoke:pack && npm run smoke:plain-install",
26
+ "ci": "npm run syntax && npm run validate && npm run evals && npm run evals:router && npm run evals:v11:validate && npm run evals:live:validate && npm run evals:long:validate && npm run evals:polyglot:validate && npm test && npm pack --dry-run && npm run smoke:pack && npm run smoke:plain-install",
27
27
  "postinstall": "node scripts/install.mjs",
28
28
  "preuninstall": "node scripts/uninstall.mjs",
29
29
  "prepublishOnly": "npm run ci",
@@ -46,7 +46,9 @@
46
46
  "evals:polyglot:validate": "node scripts/validate-live-suite.mjs --suite polyglot",
47
47
  "evals:polyglot": "node scripts/eval-live.mjs --suite polyglot",
48
48
  "evals:matrix:gate": "node scripts/eval-matrix.mjs --require-confidence",
49
- "evals:ablation": "node scripts/eval-ablation.mjs"
49
+ "evals:ablation": "node scripts/eval-ablation.mjs",
50
+ "evals:v11": "node --test test/*-v11.test.mjs",
51
+ "evals:v11:validate": "node scripts/validate-v11-suite.mjs"
50
52
  },
51
53
  "keywords": [
52
54
  "opencode",
@@ -56,7 +58,10 @@
56
58
  "engineering-workflow",
57
59
  "code-review",
58
60
  "debugging",
59
- "verification"
61
+ "verification",
62
+ "visual-testing",
63
+ "browser-automation",
64
+ "multimodal-agents"
60
65
  ],
61
66
  "engines": {
62
67
  "node": ">=20"
@@ -14,7 +14,7 @@ function positional() {
14
14
  const values = []
15
15
  for (let index = 0; index < args.length; index += 1) {
16
16
  if (args[index].startsWith("--")) {
17
- if (["--pass-rate-tolerance", "--min-initial-reduction", "--max-token-ratio", "--max-duration-ratio"].includes(args[index])) index += 1
17
+ if (["--pass-rate-tolerance", "--min-initial-reduction", "--max-token-ratio", "--max-duration-ratio", "--min-cacheable-ratio", "--min-evidence-reuse-ratio", "--max-repeated-stable-ratio"].includes(args[index])) index += 1
18
18
  continue
19
19
  }
20
20
  values.push(args[index])
@@ -38,6 +38,9 @@ const report = compareEvalSummaries(reference, candidate, {
38
38
  minInitialInputReduction: Number(option("--min-initial-reduction", "0.10")),
39
39
  maxTotalTokenRatio: Number(option("--max-token-ratio", "1.05")),
40
40
  maxDurationRatio: Number(option("--max-duration-ratio", "1.10")),
41
+ minCacheableRatio: option("--min-cacheable-ratio", null) == null ? null : Number(option("--min-cacheable-ratio", null)),
42
+ minEvidenceReuseRatio: option("--min-evidence-reuse-ratio", null) == null ? null : Number(option("--min-evidence-reuse-ratio", null)),
43
+ maxRepeatedStableRatio: option("--max-repeated-stable-ratio", null) == null ? null : Number(option("--max-repeated-stable-ratio", null)),
41
44
  })
42
45
 
43
46
  console.log(JSON.stringify(report, null, 2))