opencode-agent-skill 9.0.0 → 11.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +116 -0
- package/README.md +742 -675
- package/bin/ocskill.mjs +354 -5
- package/docs/V11-PERCEPTION-ADAPTIVE-EXECUTION.md +75 -0
- package/docs/V11-PERCEPTION-ADAPTIVE.md +220 -0
- package/evals/router-triggers.json +82 -0
- package/evals/routing.json +76 -0
- package/evals/v11/tasks.json +122 -0
- package/global-config/AGENTS.md +78 -163
- package/global-config/agents/merge-arbiter.md +12 -0
- package/global-config/agents/visual-verifier.md +12 -0
- package/global-config/plugins/ues-router/index.js +683 -59
- package/global-config/plugins/ues-router/router.js +62 -3
- package/global-config/plugins/ues-router/runtime-guard.js +265 -0
- package/global-config/skills/browser-qa/SKILL.md +14 -0
- package/global-config/skills/browser-qa/references/workflow.md +11 -0
- package/global-config/skills/browser-security/SKILL.md +12 -0
- package/global-config/skills/component-visual-testing/SKILL.md +10 -0
- package/global-config/skills/design-source/SKILL.md +10 -0
- package/global-config/skills/design-source/references/workflow.md +12 -0
- package/global-config/skills/dynamic-workflow/SKILL.md +18 -0
- package/global-config/skills/dynamic-workflow/references/workflow.md +19 -0
- package/global-config/skills/responsive-verification/SKILL.md +10 -0
- package/global-config/skills/skill-authoring/SKILL.md +12 -0
- package/global-config/skills/skill-evaluation/SKILL.md +17 -0
- package/global-config/skills/visual-fidelity/SKILL.md +14 -0
- package/global-config/skills/visual-fidelity/references/workflow.md +14 -0
- package/lib/benchmark-confidence.mjs +49 -11
- package/lib/browser-adapter.mjs +82 -0
- package/lib/browser-runtime.mjs +193 -0
- package/lib/capability-registry.mjs +109 -0
- package/lib/context-engine-v11.mjs +146 -0
- package/lib/context-manifest.mjs +16 -3
- package/lib/control-center.mjs +12 -2
- package/lib/dynamic-workflow.mjs +179 -0
- package/lib/eval-ablation.mjs +146 -0
- package/lib/eval-report.mjs +83 -0
- package/lib/eval-telemetry.mjs +64 -0
- package/lib/evidence-budget.mjs +84 -0
- package/lib/evidence-store.mjs +178 -0
- package/lib/hermes-bridge.mjs +45 -1
- package/lib/model-config.mjs +9 -1
- package/lib/model-policy.mjs +58 -1
- package/lib/orchestrator-policy.mjs +100 -7
- package/lib/png-diff.mjs +229 -0
- package/lib/prompt-cache.mjs +60 -0
- package/lib/skill-quality.mjs +72 -0
- package/lib/task-engine.mjs +223 -12
- package/lib/ui-inspector.mjs +152 -0
- package/lib/v11-metrics.mjs +64 -0
- package/lib/visual-spec.mjs +159 -0
- package/package.json +11 -5
- package/scripts/eval-ablation.mjs +47 -0
- package/scripts/eval-matrix.mjs +13 -2
- package/scripts/validate-v11-suite.mjs +58 -0
- package/scripts/validate.mjs +16 -4
package/bin/ocskill.mjs
CHANGED
|
@@ -54,16 +54,27 @@ import {
|
|
|
54
54
|
createPlanVerificationReceipt,
|
|
55
55
|
createIntegrationVerificationReceipt,
|
|
56
56
|
runtimeEvents,
|
|
57
|
+
checkpointWork,
|
|
58
|
+
markCheckpointResumed,
|
|
57
59
|
} from "../lib/task-engine.mjs"
|
|
58
60
|
import { reviewScope } from "../lib/review-scope.mjs"
|
|
59
61
|
import { buildVerificationPlan } from "../lib/verification-plan.mjs"
|
|
60
|
-
import { resolveAdaptiveModel, resolveModel } from "../lib/model-policy.mjs"
|
|
62
|
+
import { resolveAdaptiveModel, resolveCapabilityModel, resolveModel } from "../lib/model-policy.mjs"
|
|
61
63
|
import { createVerificationReceipt } from "../lib/evidence-receipt.mjs"
|
|
62
64
|
import { classifyEngineeringTask } from "../lib/orchestrator-policy.mjs"
|
|
63
65
|
import { createTaskSandbox, integrateTaskSandbox, listTaskSandboxes, removeTaskSandbox } from "../lib/worktree-sandbox.mjs"
|
|
64
66
|
import { analyzeEvalTraces, saveLearningAnalysis, readLearningState, acceptLearning, promoteLearning } from "../lib/learning-engine.mjs"
|
|
65
|
-
import { hermesStatus, buildHermesDelegationPrompt, hermesOneShotArgs } from "../lib/hermes-bridge.mjs"
|
|
67
|
+
import { hermesStatus, buildHermesDelegationPrompt, buildHermesWorkflowPrompt, hermesOneShotArgs, hermesSidecarPlan } from "../lib/hermes-bridge.mjs"
|
|
66
68
|
import { readModelPolicy, validateModelID, writeModelPolicy } from "../lib/model-config.mjs"
|
|
69
|
+
import { evidenceStoreStatus, gcEvidenceStore, getEvidence, putEvidence } from "../lib/evidence-store.mjs"
|
|
70
|
+
import { inferTaskCapabilities } from "../lib/capability-registry.mjs"
|
|
71
|
+
import { browserCapability, buildBrowserVerificationPlan } from "../lib/browser-adapter.mjs"
|
|
72
|
+
import { inspectBrowserPage, summarizeBrowserInspection } from "../lib/browser-runtime.mjs"
|
|
73
|
+
import { comparePngFiles, cropPngFile } from "../lib/png-diff.mjs"
|
|
74
|
+
import { createGeometryReceipt, responsiveViewportMatrix, validateVisualSpec } from "../lib/visual-spec.mjs"
|
|
75
|
+
import { planDynamicWorkflow } from "../lib/dynamic-workflow.mjs"
|
|
76
|
+
import { lintSkillCatalog } from "../lib/skill-quality.mjs"
|
|
77
|
+
import { designTokenEvidence, extractDesignTokens, inspectResponsiveLayout } from "../lib/ui-inspector.mjs"
|
|
67
78
|
import {
|
|
68
79
|
clipOutput,
|
|
69
80
|
errorMessage,
|
|
@@ -115,7 +126,14 @@ Usage:
|
|
|
115
126
|
ocskill sandbox <action> ... Create, integrate and clean isolated Git worktree sandboxes
|
|
116
127
|
Also supports capability/exec for fail-closed container verification
|
|
117
128
|
ocskill learn <action> ... Analyze eval traces and promote benchmark-validated lessons
|
|
118
|
-
ocskill hermes <action> ... Optional Hermes
|
|
129
|
+
ocskill hermes <action> ... Optional Hermes sidecar/status/task/workflow planning
|
|
130
|
+
ocskill store <status|put|get|gc> ... Content-addressed evidence storage and bounded retrieval
|
|
131
|
+
ocskill capabilities <text> Infer required execution/model capabilities
|
|
132
|
+
ocskill visual <action> ... Geometry receipts, PNG diff/crop and viewport matrix
|
|
133
|
+
ocskill browser <action> ... Browser capability, plan and bounded Playwright inspection
|
|
134
|
+
ocskill ui <tokens|layout> ... Extract design tokens or verify responsive geometry
|
|
135
|
+
ocskill workflow-plan <plan> Cost-aware deterministic/LLM/vision wave schedule
|
|
136
|
+
ocskill skills lint [dir] Lint skill size, metadata and routing-description collisions
|
|
119
137
|
ocskill dashboard [dir] [--serve] [--port N]
|
|
120
138
|
Generate/serve the local UES Control Center
|
|
121
139
|
ocskill models <status|on|off|set|role> ...
|
|
@@ -152,6 +170,7 @@ Usage:
|
|
|
152
170
|
ocskill models on|off
|
|
153
171
|
ocskill models set <light|standard|heavy> <provider/model[#variant]>
|
|
154
172
|
ocskill models role <role> <light|standard|heavy>
|
|
173
|
+
ocskill models capability <provider/model> [--vision on|off] [--browser on|off] [--reasoning on|off] [--long-context on|off] [--cost low|medium|high] [--latency fast|medium|slow] [--quality 0..1]
|
|
155
174
|
|
|
156
175
|
--force backs up and replaces/removes state owned by another package.
|
|
157
176
|
`)
|
|
@@ -589,6 +608,28 @@ async function workControl() {
|
|
|
589
608
|
}))
|
|
590
609
|
return
|
|
591
610
|
}
|
|
611
|
+
if (action === "checkpoint") {
|
|
612
|
+
const taskID = args[3]
|
|
613
|
+
const root = positionalArg(args, 4) || process.cwd()
|
|
614
|
+
if (!taskID) throw new Error("Usage: ocskill work checkpoint <slug> <task-id> [dir] --run-id <id> [--reason <text>]")
|
|
615
|
+
printJson(await checkpointWork(root, slug, {
|
|
616
|
+
taskID,
|
|
617
|
+
runId: optionValue(args, "--run-id"),
|
|
618
|
+
reason: optionValue(args, "--reason") || "pre-compaction",
|
|
619
|
+
}))
|
|
620
|
+
return
|
|
621
|
+
}
|
|
622
|
+
if (action === "checkpoint-resumed") {
|
|
623
|
+
const taskID = args[3]
|
|
624
|
+
const root = positionalArg(args, 4) || process.cwd()
|
|
625
|
+
if (!taskID) throw new Error("Usage: ocskill work checkpoint-resumed <slug> <task-id> [dir] --run-id <id> [--reason <text>]")
|
|
626
|
+
printJson(await markCheckpointResumed(root, slug, {
|
|
627
|
+
taskID,
|
|
628
|
+
runId: optionValue(args, "--run-id"),
|
|
629
|
+
reason: optionValue(args, "--reason"),
|
|
630
|
+
}))
|
|
631
|
+
return
|
|
632
|
+
}
|
|
592
633
|
if (action === "events") {
|
|
593
634
|
const root = positionalArg(args, 3) || process.cwd()
|
|
594
635
|
const limit = optionInt(args, "--limit", 200)
|
|
@@ -717,7 +758,8 @@ async function modelPolicy() {
|
|
|
717
758
|
const normalizedAttempt = Number.isInteger(attempt) && attempt > 0 ? attempt : 1
|
|
718
759
|
const taskText = optionValue(args, "--text")
|
|
719
760
|
if (taskText) {
|
|
720
|
-
|
|
761
|
+
const taskPolicy = classifyEngineeringTask(taskText)
|
|
762
|
+
printJson(resolveCapabilityModel(role, normalizedAttempt, taskText, taskPolicy, policy))
|
|
721
763
|
return
|
|
722
764
|
}
|
|
723
765
|
printJson(resolveModel(role, normalizedAttempt, policy))
|
|
@@ -770,6 +812,48 @@ async function modelsControl() {
|
|
|
770
812
|
return
|
|
771
813
|
}
|
|
772
814
|
|
|
815
|
+
if (action === "capability") {
|
|
816
|
+
const model = args[2]
|
|
817
|
+
if (!validateModelID(model)) {
|
|
818
|
+
console.error("Usage: ocskill models capability <provider/model> [capability flags]")
|
|
819
|
+
process.exitCode = 2
|
|
820
|
+
return
|
|
821
|
+
}
|
|
822
|
+
const current = policy.capabilities?.[model] || {}
|
|
823
|
+
const boolFlag = (name, prior) => {
|
|
824
|
+
const value = optionValue(args, name)
|
|
825
|
+
if (value == null) return prior
|
|
826
|
+
if (!["on", "off", "true", "false"].includes(String(value).toLowerCase())) throw new Error(name + " must be on/off")
|
|
827
|
+
return ["on", "true"].includes(String(value).toLowerCase())
|
|
828
|
+
}
|
|
829
|
+
const qualityRaw = optionValue(args, "--quality")
|
|
830
|
+
const quality = qualityRaw == null ? current.quality : Number(qualityRaw)
|
|
831
|
+
if (qualityRaw != null && (!Number.isFinite(quality) || quality < 0 || quality > 1)) throw new Error("--quality must be from 0 to 1")
|
|
832
|
+
const cost = optionValue(args, "--cost") || current.costClass
|
|
833
|
+
const latency = optionValue(args, "--latency") || current.latencyClass
|
|
834
|
+
if (cost && !["low", "medium", "high"].includes(cost)) throw new Error("--cost must be low|medium|high")
|
|
835
|
+
if (latency && !["fast", "medium", "slow"].includes(latency)) throw new Error("--latency must be fast|medium|slow")
|
|
836
|
+
policy = await writeModelPolicy(getConfigDir(), {
|
|
837
|
+
capabilities: {
|
|
838
|
+
[model]: {
|
|
839
|
+
...current,
|
|
840
|
+
coding: boolFlag("--coding", current.coding),
|
|
841
|
+
reasoning: boolFlag("--reasoning", current.reasoning),
|
|
842
|
+
toolCalling: boolFlag("--tool-calling", current.toolCalling),
|
|
843
|
+
vision: boolFlag("--vision", current.vision),
|
|
844
|
+
browser: boolFlag("--browser", current.browser),
|
|
845
|
+
filesystem: boolFlag("--filesystem", current.filesystem),
|
|
846
|
+
longContext: boolFlag("--long-context", current.longContext),
|
|
847
|
+
...(cost ? { costClass: cost } : {}),
|
|
848
|
+
...(latency ? { latencyClass: latency } : {}),
|
|
849
|
+
...(quality != null ? { quality } : {}),
|
|
850
|
+
},
|
|
851
|
+
},
|
|
852
|
+
})
|
|
853
|
+
printJson(policy)
|
|
854
|
+
return
|
|
855
|
+
}
|
|
856
|
+
|
|
773
857
|
console.error("Usage: ocskill models <status|on|off|set|role> ...")
|
|
774
858
|
process.exitCode = 2
|
|
775
859
|
}
|
|
@@ -932,6 +1016,40 @@ async function hermesControl() {
|
|
|
932
1016
|
printJson(hermesStatus())
|
|
933
1017
|
return
|
|
934
1018
|
}
|
|
1019
|
+
if (action === "workflow" || action === "exec-workflow") {
|
|
1020
|
+
const slug = args[2]
|
|
1021
|
+
const root = positionalArg(args, 3) || process.cwd()
|
|
1022
|
+
if (!slug) {
|
|
1023
|
+
console.error("Usage: ocskill hermes <workflow|exec-workflow> <slug> [dir] [--max-concurrent N]")
|
|
1024
|
+
process.exitCode = 2
|
|
1025
|
+
return
|
|
1026
|
+
}
|
|
1027
|
+
const planFile = path.join(path.resolve(root), ".ues-work", slug, "PLAN.json")
|
|
1028
|
+
const plan = readJsonFile(planFile)
|
|
1029
|
+
const schedule = planDynamicWorkflow(plan.tasks || [], {
|
|
1030
|
+
maxConcurrent: optionInt(args, "--max-concurrent", 4),
|
|
1031
|
+
})
|
|
1032
|
+
const sidecar = hermesSidecarPlan({ mode: "dynamic-workflow", maxConcurrent: schedule.maxConcurrent })
|
|
1033
|
+
const prompt = buildHermesWorkflowPrompt({ slug, goal: plan.goal || null, plan }, schedule)
|
|
1034
|
+
if (action === "workflow") {
|
|
1035
|
+
printJson({ sidecar, schedule, prompt })
|
|
1036
|
+
return
|
|
1037
|
+
}
|
|
1038
|
+
const status = hermesStatus()
|
|
1039
|
+
if (!status.available) {
|
|
1040
|
+
console.error(status.error || "Hermes CLI is unavailable")
|
|
1041
|
+
process.exitCode = 1
|
|
1042
|
+
return
|
|
1043
|
+
}
|
|
1044
|
+
const result = runCapture("hermes", hermesOneShotArgs(prompt), {
|
|
1045
|
+
cwd: path.resolve(root),
|
|
1046
|
+
maxBuffer: 8 * 1024 * 1024,
|
|
1047
|
+
})
|
|
1048
|
+
if (result.stdout) process.stdout.write(result.stdout)
|
|
1049
|
+
if (result.stderr) process.stderr.write(result.stderr)
|
|
1050
|
+
if ((result.status ?? 1) !== 0) process.exitCode = result.status ?? 1
|
|
1051
|
+
return
|
|
1052
|
+
}
|
|
935
1053
|
if (action === "prompt" || action === "exec") {
|
|
936
1054
|
const slug = args[2]
|
|
937
1055
|
const taskID = args[3]
|
|
@@ -962,10 +1080,220 @@ async function hermesControl() {
|
|
|
962
1080
|
if ((result.status ?? 1) !== 0) process.exitCode = result.status ?? 1
|
|
963
1081
|
return
|
|
964
1082
|
}
|
|
965
|
-
console.error("Usage: ocskill hermes <status|prompt|exec> ...")
|
|
1083
|
+
console.error("Usage: ocskill hermes <status|prompt|exec|workflow|exec-workflow> ...")
|
|
966
1084
|
process.exitCode = 2
|
|
967
1085
|
}
|
|
968
1086
|
|
|
1087
|
+
async function evidenceStoreControl() {
|
|
1088
|
+
const action = args[1] || "status"
|
|
1089
|
+
try {
|
|
1090
|
+
if (action === "status") {
|
|
1091
|
+
printJson(await evidenceStoreStatus(positionalArg(args, 2) || process.cwd()))
|
|
1092
|
+
return
|
|
1093
|
+
}
|
|
1094
|
+
if (action === "put") {
|
|
1095
|
+
const file = args[2]
|
|
1096
|
+
const root = positionalArg(args, 3) || process.cwd()
|
|
1097
|
+
if (!file) throw new Error("Usage: ocskill store put <file> [dir] [--kind <kind>] [--summary <text>]")
|
|
1098
|
+
const content = readTextFile(file)
|
|
1099
|
+
printJson(await putEvidence(root, content, {
|
|
1100
|
+
kind: optionValue(args, "--kind") || "file",
|
|
1101
|
+
source: path.resolve(file),
|
|
1102
|
+
summary: optionValue(args, "--summary"),
|
|
1103
|
+
}))
|
|
1104
|
+
return
|
|
1105
|
+
}
|
|
1106
|
+
if (action === "get") {
|
|
1107
|
+
const ref = args[2]
|
|
1108
|
+
const root = positionalArg(args, 3) || process.cwd()
|
|
1109
|
+
if (!ref) throw new Error("Usage: ocskill store get <evidence-ref> [dir] [--max N] [--start N]")
|
|
1110
|
+
printJson(await getEvidence(root, ref, {
|
|
1111
|
+
maxChars: optionInt(args, "--max", 24_000),
|
|
1112
|
+
start: optionInt(args, "--start", 0),
|
|
1113
|
+
}))
|
|
1114
|
+
return
|
|
1115
|
+
}
|
|
1116
|
+
if (action === "gc") {
|
|
1117
|
+
const root = positionalArg(args, 2) || process.cwd()
|
|
1118
|
+
printJson(await gcEvidenceStore(root, {
|
|
1119
|
+
maxEntries: optionInt(args, "--max-entries", 2000),
|
|
1120
|
+
maxAgeDays: optionInt(args, "--max-age-days", 30),
|
|
1121
|
+
}))
|
|
1122
|
+
return
|
|
1123
|
+
}
|
|
1124
|
+
throw new Error("Usage: ocskill store <status|put|get|gc> ...")
|
|
1125
|
+
} catch (error) {
|
|
1126
|
+
console.error(errorMessage(error))
|
|
1127
|
+
process.exitCode = 1
|
|
1128
|
+
}
|
|
1129
|
+
}
|
|
1130
|
+
|
|
1131
|
+
async function capabilityControl() {
|
|
1132
|
+
const text = args.slice(1).join(" ").trim()
|
|
1133
|
+
if (!text) {
|
|
1134
|
+
console.error("Usage: ocskill capabilities <task text>")
|
|
1135
|
+
process.exitCode = 2
|
|
1136
|
+
return
|
|
1137
|
+
}
|
|
1138
|
+
printJson(inferTaskCapabilities(text))
|
|
1139
|
+
}
|
|
1140
|
+
|
|
1141
|
+
async function visualControl() {
|
|
1142
|
+
const action = args[1]
|
|
1143
|
+
try {
|
|
1144
|
+
if (action === "spec") {
|
|
1145
|
+
const file = args[2]
|
|
1146
|
+
if (!file) throw new Error("Usage: ocskill visual spec <VISUAL_SPEC.json>")
|
|
1147
|
+
printJson(validateVisualSpec(readJsonFile(file)))
|
|
1148
|
+
return
|
|
1149
|
+
}
|
|
1150
|
+
if (action === "geometry") {
|
|
1151
|
+
const specFile = args[2]
|
|
1152
|
+
const actualFile = args[3]
|
|
1153
|
+
if (!specFile || !actualFile) throw new Error("Usage: ocskill visual geometry <VISUAL_SPEC.json> <actual-boxes.json>")
|
|
1154
|
+
printJson(createGeometryReceipt(readJsonFile(specFile), readJsonFile(actualFile)))
|
|
1155
|
+
return
|
|
1156
|
+
}
|
|
1157
|
+
if (action === "compare") {
|
|
1158
|
+
const expected = args[2]
|
|
1159
|
+
const actual = args[3]
|
|
1160
|
+
if (!expected || !actual) throw new Error("Usage: ocskill visual compare <expected.png> <actual.png> [--threshold N] [--max-diff-ratio N]")
|
|
1161
|
+
const threshold = Number(optionValue(args, "--threshold") ?? 16)
|
|
1162
|
+
const maxDiffRatio = Number(optionValue(args, "--max-diff-ratio") ?? 0)
|
|
1163
|
+
printJson(await comparePngFiles(expected, actual, { threshold, maxDiffRatio }))
|
|
1164
|
+
return
|
|
1165
|
+
}
|
|
1166
|
+
if (action === "crop") {
|
|
1167
|
+
const input = args[2]
|
|
1168
|
+
const output = args[3]
|
|
1169
|
+
if (!input || !output) throw new Error("Usage: ocskill visual crop <input.png> <output.png> --x N --y N --width N --height N")
|
|
1170
|
+
printJson(await cropPngFile(input, output, {
|
|
1171
|
+
x: optionInt(args, "--x", 0),
|
|
1172
|
+
y: optionInt(args, "--y", 0),
|
|
1173
|
+
width: optionInt(args, "--width", 1),
|
|
1174
|
+
height: optionInt(args, "--height", 1),
|
|
1175
|
+
}))
|
|
1176
|
+
return
|
|
1177
|
+
}
|
|
1178
|
+
if (action === "viewports") {
|
|
1179
|
+
printJson(responsiveViewportMatrix())
|
|
1180
|
+
return
|
|
1181
|
+
}
|
|
1182
|
+
throw new Error("Usage: ocskill visual <spec|geometry|compare|crop|viewports> ...")
|
|
1183
|
+
} catch (error) {
|
|
1184
|
+
console.error(errorMessage(error))
|
|
1185
|
+
process.exitCode = 1
|
|
1186
|
+
}
|
|
1187
|
+
}
|
|
1188
|
+
|
|
1189
|
+
async function browserControl() {
|
|
1190
|
+
const action = args[1] || "capability"
|
|
1191
|
+
try {
|
|
1192
|
+
if (action === "capability") {
|
|
1193
|
+
printJson(await browserCapability(positionalArg(args, 2) || process.cwd()))
|
|
1194
|
+
return
|
|
1195
|
+
}
|
|
1196
|
+
if (action === "plan") {
|
|
1197
|
+
const url = args[2] || null
|
|
1198
|
+
printJson(buildBrowserVerificationPlan({
|
|
1199
|
+
url,
|
|
1200
|
+
target: optionValue(args, "--target"),
|
|
1201
|
+
}))
|
|
1202
|
+
return
|
|
1203
|
+
}
|
|
1204
|
+
if (action === "inspect") {
|
|
1205
|
+
const url = args[2]
|
|
1206
|
+
if (!url) throw new Error("Usage: ocskill browser inspect <url> [dir] [--selector <css>] [--screenshot <path>] [--width N] [--height N] [--max-elements N]")
|
|
1207
|
+
const root = positionalArg(args, 3) || process.cwd()
|
|
1208
|
+
const report = await inspectBrowserPage(root, url, {
|
|
1209
|
+
selector: optionValue(args, "--selector"),
|
|
1210
|
+
screenshot: optionValue(args, "--screenshot"),
|
|
1211
|
+
width: optionInt(args, "--width", 1440),
|
|
1212
|
+
height: optionInt(args, "--height", 900),
|
|
1213
|
+
maxElements: optionInt(args, "--max-elements", 80),
|
|
1214
|
+
timeoutMs: optionInt(args, "--timeout-ms", 30000),
|
|
1215
|
+
waitMs: optionInt(args, "--wait-ms", 0),
|
|
1216
|
+
fullPage: !args.includes("--viewport-only"),
|
|
1217
|
+
})
|
|
1218
|
+
printJson(args.includes("--full") ? report : summarizeBrowserInspection(report, { limit: optionInt(args, "--limit", 20) }))
|
|
1219
|
+
return
|
|
1220
|
+
}
|
|
1221
|
+
throw new Error("Usage: ocskill browser <capability|plan|inspect> ...")
|
|
1222
|
+
} catch (error) {
|
|
1223
|
+
console.error(errorMessage(error))
|
|
1224
|
+
process.exitCode = 1
|
|
1225
|
+
}
|
|
1226
|
+
}
|
|
1227
|
+
|
|
1228
|
+
async function uiControl() {
|
|
1229
|
+
const action = args[1]
|
|
1230
|
+
try {
|
|
1231
|
+
if (action === "tokens") {
|
|
1232
|
+
const file = args[2]
|
|
1233
|
+
if (!file) throw new Error("Usage: ocskill ui tokens <styles.css>")
|
|
1234
|
+
const tokens = extractDesignTokens(readTextFile(file))
|
|
1235
|
+
printJson({ tokens, evidence: designTokenEvidence(tokens) })
|
|
1236
|
+
return
|
|
1237
|
+
}
|
|
1238
|
+
if (action === "layout") {
|
|
1239
|
+
const file = args[2]
|
|
1240
|
+
if (!file) throw new Error("Usage: ocskill ui layout <boxes.json> --width N --height N [--min-touch N] [--overlap-ratio N]")
|
|
1241
|
+
const payload = readJsonFile(file)
|
|
1242
|
+
const items = Array.isArray(payload) ? payload : payload.elements || payload.boxes || []
|
|
1243
|
+
printJson(inspectResponsiveLayout(items, {
|
|
1244
|
+
width: optionInt(args, "--width", Number(payload.viewport?.width || 0)),
|
|
1245
|
+
height: optionInt(args, "--height", Number(payload.viewport?.height || 0)),
|
|
1246
|
+
}, {
|
|
1247
|
+
minTouchTarget: optionInt(args, "--min-touch", 44),
|
|
1248
|
+
overlapRatio: Number(optionValue(args, "--overlap-ratio") ?? 0.15),
|
|
1249
|
+
}))
|
|
1250
|
+
return
|
|
1251
|
+
}
|
|
1252
|
+
throw new Error("Usage: ocskill ui <tokens|layout> ...")
|
|
1253
|
+
} catch (error) {
|
|
1254
|
+
console.error(errorMessage(error))
|
|
1255
|
+
process.exitCode = 1
|
|
1256
|
+
}
|
|
1257
|
+
}
|
|
1258
|
+
|
|
1259
|
+
async function workflowPlanControl() {
|
|
1260
|
+
const file = args[1]
|
|
1261
|
+
if (!file) {
|
|
1262
|
+
console.error("Usage: ocskill workflow-plan <PLAN.json> [--max-concurrent N]")
|
|
1263
|
+
process.exitCode = 2
|
|
1264
|
+
return
|
|
1265
|
+
}
|
|
1266
|
+
try {
|
|
1267
|
+
const plan = readJsonFile(file)
|
|
1268
|
+
printJson(planDynamicWorkflow(plan.tasks || [], {
|
|
1269
|
+
maxConcurrent: optionInt(args, "--max-concurrent", 4),
|
|
1270
|
+
maxLLMConcurrent: optionInt(args, "--max-llm-concurrent", optionInt(args, "--max-concurrent", 4)),
|
|
1271
|
+
maxVisionConcurrent: optionInt(args, "--max-vision-concurrent", 2),
|
|
1272
|
+
maxWaveCost: optionInt(args, "--max-wave-cost", 24),
|
|
1273
|
+
minAgentCost: optionInt(args, "--min-agent-cost", 5),
|
|
1274
|
+
minVisionAgentCost: optionInt(args, "--min-vision-agent-cost", 4),
|
|
1275
|
+
}))
|
|
1276
|
+
} catch (error) {
|
|
1277
|
+
console.error(errorMessage(error))
|
|
1278
|
+
process.exitCode = 1
|
|
1279
|
+
}
|
|
1280
|
+
}
|
|
1281
|
+
|
|
1282
|
+
async function skillsControl() {
|
|
1283
|
+
const action = args[1] || "lint"
|
|
1284
|
+
if (action !== "lint") {
|
|
1285
|
+
console.error("Usage: ocskill skills lint [dir]")
|
|
1286
|
+
process.exitCode = 2
|
|
1287
|
+
return
|
|
1288
|
+
}
|
|
1289
|
+
try {
|
|
1290
|
+
printJson(await lintSkillCatalog(positionalArg(args, 2) || packageRoot))
|
|
1291
|
+
} catch (error) {
|
|
1292
|
+
console.error(errorMessage(error))
|
|
1293
|
+
process.exitCode = 1
|
|
1294
|
+
}
|
|
1295
|
+
}
|
|
1296
|
+
|
|
969
1297
|
async function dashboardControl() {
|
|
970
1298
|
const forwarded = args.slice(1)
|
|
971
1299
|
const code = run(process.execPath, [path.join(packageRoot, "scripts", "control-center.mjs"), ...forwarded])
|
|
@@ -1138,6 +1466,27 @@ switch (command) {
|
|
|
1138
1466
|
case "hermes":
|
|
1139
1467
|
await hermesControl()
|
|
1140
1468
|
break
|
|
1469
|
+
case "store":
|
|
1470
|
+
await evidenceStoreControl()
|
|
1471
|
+
break
|
|
1472
|
+
case "capabilities":
|
|
1473
|
+
await capabilityControl()
|
|
1474
|
+
break
|
|
1475
|
+
case "visual":
|
|
1476
|
+
await visualControl()
|
|
1477
|
+
break
|
|
1478
|
+
case "browser":
|
|
1479
|
+
await browserControl()
|
|
1480
|
+
break
|
|
1481
|
+
case "workflow-plan":
|
|
1482
|
+
await workflowPlanControl()
|
|
1483
|
+
break
|
|
1484
|
+
case "ui":
|
|
1485
|
+
await uiControl()
|
|
1486
|
+
break
|
|
1487
|
+
case "skills":
|
|
1488
|
+
await skillsControl()
|
|
1489
|
+
break
|
|
1141
1490
|
case "dashboard":
|
|
1142
1491
|
await dashboardControl()
|
|
1143
1492
|
break
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
# V11 Perception & Adaptive Execution
|
|
2
|
+
|
|
3
|
+
## Goal
|
|
4
|
+
|
|
5
|
+
V11 minimizes context and model cost without deleting evidence needed for correctness. It adds perception-aware UI/browser verification so the system can reason about **what an element is, where it is and how it looks** using different evidence channels.
|
|
6
|
+
|
|
7
|
+
## Runtime layers
|
|
8
|
+
|
|
9
|
+
```text
|
|
10
|
+
Task
|
|
11
|
+
-> intent/risk
|
|
12
|
+
-> capability requirements
|
|
13
|
+
-> adaptive evidence budget
|
|
14
|
+
-> semantic/index evidence
|
|
15
|
+
-> content-addressed evidence pointers
|
|
16
|
+
-> focused skills
|
|
17
|
+
-> capability-aware model
|
|
18
|
+
-> fresh executor
|
|
19
|
+
-> deterministic verification
|
|
20
|
+
-> visual/browser verifier when required
|
|
21
|
+
-> diagnosis + evidence expansion only on failure
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
## Evidence Store
|
|
25
|
+
|
|
26
|
+
Large raw tool output, durable specs and dependency reports are stored under:
|
|
27
|
+
|
|
28
|
+
```text
|
|
29
|
+
.ues-cache/evidence-v1/<hash-prefix>/<sha256>.blob
|
|
30
|
+
.ues-cache/evidence-v1/<hash-prefix>/<sha256>.json
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
A prompt receives a bounded excerpt and a reference such as `evidence:sha256:<hash>`. Use `ocskill store get` to retrieve only the needed slice. `.ues-cache` is runtime state and is excluded from workspace verification fingerprints.
|
|
34
|
+
|
|
35
|
+
## Adaptive evidence budget
|
|
36
|
+
|
|
37
|
+
FAST/STANDARD/DEEP remain outer safety ceilings. Inside that ceiling V11 allocates characters by evidence role instead of treating all context as equally valuable. Debugging shifts budget toward tests/history; high-risk work shifts toward tests/references; browser/visual work shifts toward deterministic tool evidence.
|
|
38
|
+
|
|
39
|
+
Failure expands evidence through the existing initial -> diagnose -> deep-recovery stages instead of loading maximum context on the first attempt.
|
|
40
|
+
|
|
41
|
+
## Prompt cache shape
|
|
42
|
+
|
|
43
|
+
Stable material is separated conceptually from dynamic task/evidence. The runtime records stable/dynamic hashes and cacheable ratio. This is telemetry, not a promise that every provider supports prompt caching.
|
|
44
|
+
|
|
45
|
+
## Capability-aware routing
|
|
46
|
+
|
|
47
|
+
Model profiles may declare coding, reasoning, toolCalling, vision, browser, filesystem, longContext, cost/latency class and a quality hint. UES never invents an unavailable capability. If no configured candidate satisfies a requirement it exposes capability fallback instead of silently claiming that a text-only model can see screenshots.
|
|
48
|
+
|
|
49
|
+
## Visual fidelity
|
|
50
|
+
|
|
51
|
+
Visual verification uses three complementary layers: semantic DOM/accessibility identity, geometry/bounding boxes, and screenshot pixels. VISUAL_SPEC describes important anchors and tolerances. Geometry receipts prove position/size claims. PNG diff finds changed pixels and their bounding region. A failed region can be cropped so vision only sees the area that needs judgment.
|
|
52
|
+
|
|
53
|
+
Screenshot equality does not prove accessibility or interaction; DOM equality does not prove appearance.
|
|
54
|
+
|
|
55
|
+
## Browser QA and security
|
|
56
|
+
|
|
57
|
+
Browser workflows prefer bounded deterministic scripts/CLI for ordinary verification. Remote webpage content is untrusted and cannot change UES/tool permissions, request secrets, expand the approved task or authorize external side effects.
|
|
58
|
+
|
|
59
|
+
## Dynamic workflows
|
|
60
|
+
|
|
61
|
+
The scheduler classifies units as deterministic, LLM judgment or vision judgment. Deterministic work does not spawn agents. Independent tasks may share a wave only when dependencies are ready and file ownership does not conflict. Each wave is integrated and verified before later waves rely on it.
|
|
62
|
+
|
|
63
|
+
## Skill system
|
|
64
|
+
|
|
65
|
+
V11 keeps progressive disclosure: description metadata for routing, short SKILL.md entrypoint, references only when the selected mode needs them, and deterministic logic in runtime/scripts rather than repeated prompt text.
|
|
66
|
+
|
|
67
|
+
`ocskill skills lint` flags oversized entrypoints and highly overlapping descriptions.
|
|
68
|
+
|
|
69
|
+
## Hermes sidecar
|
|
70
|
+
|
|
71
|
+
Hermes remains optional. UES owns durable `.ues-work` state, evidence references, task leases/runId fencing, verification receipts and safety/permission boundaries. Hermes may execute a bounded task/workflow when explicitly available but does not become the source of truth.
|
|
72
|
+
|
|
73
|
+
## Release evidence
|
|
74
|
+
|
|
75
|
+
V11 must pass syntax/resource/router/V11 contract tests, all Node tests, package/install smokes, no-regression live suites, capability-routing tests, visual geometry/pixel fixtures, browser security/targeted-evidence fixtures and token/cache/evidence telemetry benchmarks before stable promotion.
|