opencode-agent-skill 10.0.0 → 12.0.0-beta.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +85 -0
- package/README.md +60 -8
- package/bin/ocskill.mjs +354 -6
- package/docs/DETERMINISTIC-TOOLS.md +1 -1
- package/docs/ENGINEERING-DESIGN.md +4 -4
- package/docs/EVALS.md +3 -3
- package/docs/GITHUB-RULESET.md +50 -0
- package/docs/NPM-PUBLISH.md +4 -4
- package/docs/OPENCODE-COMPAT.md +3 -3
- package/docs/TRACE-SCHEMA.md +1 -1
- package/docs/V11-PERCEPTION-ADAPTIVE-EXECUTION.md +75 -0
- package/docs/V11-PERCEPTION-ADAPTIVE.md +220 -0
- package/docs/V12-WEAK-MODEL-INTELLIGENCE.md +27 -0
- package/evals/repo-scale/tasks.json +62 -0
- package/evals/router-triggers.json +82 -0
- package/evals/routing.json +76 -0
- package/evals/v11/tasks.json +122 -0
- package/global-config/agents/merge-arbiter.md +12 -0
- package/global-config/agents/visual-verifier.md +12 -0
- package/global-config/plugins/ues-router/index.js +272 -2
- package/global-config/plugins/ues-router/router.js +27 -3
- package/global-config/skills/browser-qa/SKILL.md +14 -0
- package/global-config/skills/browser-qa/references/workflow.md +11 -0
- package/global-config/skills/browser-security/SKILL.md +12 -0
- package/global-config/skills/component-visual-testing/SKILL.md +10 -0
- package/global-config/skills/design-source/SKILL.md +10 -0
- package/global-config/skills/design-source/references/workflow.md +12 -0
- package/global-config/skills/dynamic-workflow/SKILL.md +18 -0
- package/global-config/skills/dynamic-workflow/references/workflow.md +19 -0
- package/global-config/skills/responsive-verification/SKILL.md +10 -0
- package/global-config/skills/skill-authoring/SKILL.md +12 -0
- package/global-config/skills/skill-evaluation/SKILL.md +17 -0
- package/global-config/skills/visual-fidelity/SKILL.md +14 -0
- package/global-config/skills/visual-fidelity/references/workflow.md +14 -0
- package/lib/browser-adapter.mjs +82 -0
- package/lib/browser-runtime.mjs +193 -0
- package/lib/capability-registry.mjs +109 -0
- package/lib/context-engine-v11.mjs +150 -0
- package/lib/context-manifest.mjs +16 -3
- package/lib/context-quality.mjs +59 -0
- package/lib/control-center.mjs +12 -2
- package/lib/decision-policy.mjs +23 -0
- package/lib/dynamic-workflow.mjs +179 -0
- package/lib/eval-ablation.mjs +43 -1
- package/lib/eval-report.mjs +72 -0
- package/lib/eval-telemetry.mjs +61 -0
- package/lib/evidence-budget.mjs +84 -0
- package/lib/evidence-store.mjs +178 -0
- package/lib/hermes-bridge.mjs +45 -1
- package/lib/model-config.mjs +21 -1
- package/lib/model-performance.mjs +113 -0
- package/lib/model-policy.mjs +59 -1
- package/lib/orchestrator-policy.mjs +1 -1
- package/lib/png-diff.mjs +229 -0
- package/lib/prompt-cache.mjs +60 -0
- package/lib/repo-scale-fixture.mjs +45 -0
- package/lib/skill-quality.mjs +72 -0
- package/lib/task-engine.mjs +95 -7
- package/lib/ui-inspector.mjs +152 -0
- package/lib/v11-metrics.mjs +64 -0
- package/lib/visual-spec.mjs +159 -0
- package/lib/work-plan-scope.mjs +49 -0
- package/package.json +13 -5
- package/scripts/check-release-consistency.mjs +228 -0
- package/scripts/eval-ablation.mjs +4 -1
- package/scripts/validate-repo-scale-suite.mjs +27 -0
- package/scripts/validate-v11-suite.mjs +58 -0
- package/scripts/validate-v12-foundation.mjs +24 -0
- package/scripts/validate.mjs +16 -4
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# UES engineering design
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
V11 evolves the project from an engineering workflow harness into a **perception-aware adaptive execution engine** designed to reduce context pressure on coding models while keeping evidence needed for correctness.
|
|
4
4
|
|
|
5
5
|
The selected model remains the selected model. UES improves orchestration, evidence, task boundaries, state persistence and verification; it does not claim model equivalence.
|
|
6
6
|
|
|
@@ -44,11 +44,11 @@ Models still reason about semantics and read affected code.
|
|
|
44
44
|
|
|
45
45
|
### Progressive disclosure
|
|
46
46
|
|
|
47
|
-
The catalog
|
|
47
|
+
The catalog spans 48 skills. UES prefers a small active skill set and loads deeper references only when needed.
|
|
48
48
|
|
|
49
49
|
### Hard gates, not reminders
|
|
50
50
|
|
|
51
|
-
|
|
51
|
+
V11 machine-enforces the important boundaries:
|
|
52
52
|
|
|
53
53
|
1. long/high-risk plans are not executable until a structured plan-verification receipt matches the current plan hash;
|
|
54
54
|
2. long/high-risk task completion requires a successful verification receipt for the active run and the current workspace fingerprint;
|
|
@@ -133,7 +133,7 @@ Only UES-managed resources are rewritten/removed.
|
|
|
133
133
|
|
|
134
134
|
UES separates:
|
|
135
135
|
|
|
136
|
-
1. **static skill contract** —
|
|
136
|
+
1. **static skill contract** — 43 scenarios covering the 48-skill catalog;
|
|
137
137
|
2. **V2 router precision matrix** — 120 required-route/negative-guard cases;
|
|
138
138
|
3. **standard live benchmark** — 20 executable hidden-graded tasks;
|
|
139
139
|
4. **long-horizon benchmark** — 5 tasks, including one 15-source-file integration workload;
|
package/docs/EVALS.md
CHANGED
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
# UES evaluations
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
V11 separates catalog correctness, routing precision, benchmark integrity, final behavior, long-horizon orchestration and cross-stack coverage.
|
|
4
4
|
|
|
5
5
|
## 1. Static skill-routing contract
|
|
6
6
|
|
|
7
|
-
`evals/routing.json` keeps
|
|
7
|
+
`evals/routing.json` keeps 43 representative scenarios and covers all installed skills.
|
|
8
8
|
|
|
9
9
|
```bash
|
|
10
10
|
npm run evals
|
|
@@ -141,7 +141,7 @@ Compare the same model, variant, prompt, fixture, grader and environment. Report
|
|
|
141
141
|
A benchmark result is evidence only for the measured workload. UES does not claim to turn one base model into another.
|
|
142
142
|
|
|
143
143
|
|
|
144
|
-
##
|
|
144
|
+
## V11 live-run observability and evidence gate
|
|
145
145
|
|
|
146
146
|
Live runs accept:
|
|
147
147
|
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
# GitHub Ruleset Readiness
|
|
2
|
+
|
|
3
|
+
This document records the recommended GitHub repository ruleset configuration for the `main` branch, to be activated only after CI Gate and Security Gate have each achieved at least one successful run.
|
|
4
|
+
|
|
5
|
+
## Recommended ruleset
|
|
6
|
+
|
|
7
|
+
- **Ruleset name:** `main-protection`
|
|
8
|
+
- **Target:** default branch / `main`
|
|
9
|
+
- **Enforcement:** Active
|
|
10
|
+
|
|
11
|
+
## Rules
|
|
12
|
+
|
|
13
|
+
1. **Restrict deletions** — block deleting the default branch or force-pushing over it.
|
|
14
|
+
2. **Block force pushes** — no `git push --force` or `git push -f` to `main`.
|
|
15
|
+
3. **Require pull request before merge** — all changes must go through a PR.
|
|
16
|
+
4. **Require conversation resolution** — PR reviewers must resolve all inline comments before merge.
|
|
17
|
+
5. **Require status checks** — PRs must have passing CI Gate and Security Gate checks before merge.
|
|
18
|
+
6. **Require branch up to date** — PR branches must be up to date with `main` before merge (applies when there are multiple contributors; can be relaxed for single-maintainer setups).
|
|
19
|
+
7. **Required check: CI Gate** — the aggregate CI check from `.github/workflows/ci.yml`.
|
|
20
|
+
8. **Required check: Security Gate** — the aggregate security check from `.github/workflows/security.yml`.
|
|
21
|
+
|
|
22
|
+
## Approval policy for single-maintainer repo
|
|
23
|
+
|
|
24
|
+
This repository currently has one maintainer. Setting `required approval = 1` would prevent the owner from merging their own PRs (they cannot approve their own PR). Therefore:
|
|
25
|
+
|
|
26
|
+
- **Required approval = 0** for now.
|
|
27
|
+
- When a collaborator or external reviewer is added, raise to `required approval = 1` to restore oversight.
|
|
28
|
+
|
|
29
|
+
## Classic branch protection vs ruleset
|
|
30
|
+
|
|
31
|
+
GitHub supports both classic branch protection rules and repository rulesets simultaneously. When both are configured:
|
|
32
|
+
|
|
33
|
+
- They can apply independently to different aspects of branch protection.
|
|
34
|
+
- Be cautious of duplicate or conflicting protections (e.g., two rules both requiring status checks but with different required lists).
|
|
35
|
+
- If migrating from classic protection to ruleset, remove the classic rule after confirming the ruleset is active and working.
|
|
36
|
+
|
|
37
|
+
## Activation prerequisite
|
|
38
|
+
|
|
39
|
+
Do not enable the ruleset until:
|
|
40
|
+
|
|
41
|
+
1. CI Gate has at least one successful run on `main`.
|
|
42
|
+
2. Security Gate has at least one successful run on `main` or a PR.
|
|
43
|
+
|
|
44
|
+
Without these prerequisites, the ruleset would block all merges immediately after activation, effectively locking the repository.
|
|
45
|
+
|
|
46
|
+
## Current status
|
|
47
|
+
|
|
48
|
+
- **CI Gate:** Added in `.github/workflows/ci.yml`. Awaiting first successful run.
|
|
49
|
+
- **Security Gate:** Added in `.github/workflows/security.yml`. Awaiting first successful run.
|
|
50
|
+
- **Ruleset:** NOT YET ACTIVATED. Will be configured via GitHub admin settings after both gates have successful runs.
|
package/docs/NPM-PUBLISH.md
CHANGED
|
@@ -17,7 +17,7 @@ opencode-agent-skill
|
|
|
17
17
|
npm run ci
|
|
18
18
|
```
|
|
19
19
|
|
|
20
|
-
|
|
20
|
+
CI includes syntax validation, resource validation, static skill routing, the 129-case V2 router matrix, 13 V11 contract tasks, standard/long/polyglot hidden-grader integrity checks, unit/integration tests, package dry-run, packed global-install smoke, and a plain one-command install/resource sync smoke.
|
|
21
21
|
|
|
22
22
|
## Manual release-like test
|
|
23
23
|
|
|
@@ -27,7 +27,7 @@ Use:
|
|
|
27
27
|
|
|
28
28
|
```cmd
|
|
29
29
|
npm pack
|
|
30
|
-
npm install -g .\opencode-agent-skill-
|
|
30
|
+
npm install -g .\opencode-agent-skill-11.0.0.tgz --allow-scripts=opencode-agent-skill
|
|
31
31
|
ocskill status
|
|
32
32
|
ocskill doctor
|
|
33
33
|
```
|
|
@@ -47,14 +47,14 @@ After publication verify:
|
|
|
47
47
|
|
|
48
48
|
```cmd
|
|
49
49
|
npm view opencode-agent-skill versions --json
|
|
50
|
-
npm view opencode-agent-skill@
|
|
50
|
+
npm view opencode-agent-skill@11.0.0 version
|
|
51
51
|
npm dist-tag ls opencode-agent-skill
|
|
52
52
|
```
|
|
53
53
|
|
|
54
54
|
The expected release tag is:
|
|
55
55
|
|
|
56
56
|
```text
|
|
57
|
-
latest:
|
|
57
|
+
latest: 11.0.0
|
|
58
58
|
```
|
|
59
59
|
|
|
60
60
|
## GitHub Actions publishing
|
package/docs/OPENCODE-COMPAT.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# OpenCode compatibility
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
V11 ships one npm package for OpenCode 1.x and 2.x, while only enabling V2-native runtime features when V2 is detected.
|
|
4
4
|
|
|
5
5
|
## Detection
|
|
6
6
|
|
|
@@ -23,9 +23,9 @@ The detected major is recorded in the managed state.
|
|
|
23
23
|
|
|
24
24
|
UES installs:
|
|
25
25
|
|
|
26
|
-
-
|
|
26
|
+
- 48 namespaced skills
|
|
27
27
|
- 11 namespaced commands
|
|
28
|
-
-
|
|
28
|
+
- 12 namespaced subagents using compatible V1 `permission` frontmatter
|
|
29
29
|
- managed global `AGENTS.md` block
|
|
30
30
|
|
|
31
31
|
The V2 runtime plugin is not installed.
|
package/docs/TRACE-SCHEMA.md
CHANGED
|
@@ -75,7 +75,7 @@ Keep constant:
|
|
|
75
75
|
Compare observable success, regressions, elapsed time, tool behavior and cost rather than narrative confidence.
|
|
76
76
|
|
|
77
77
|
|
|
78
|
-
##
|
|
78
|
+
## V11 runtime and evidence fields
|
|
79
79
|
|
|
80
80
|
Each live result may additionally contain:
|
|
81
81
|
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
# V11 Perception & Adaptive Execution
|
|
2
|
+
|
|
3
|
+
## Goal
|
|
4
|
+
|
|
5
|
+
V11 minimizes context and model cost without deleting evidence needed for correctness. It adds perception-aware UI/browser verification so the system can reason about **what an element is, where it is and how it looks** using different evidence channels.
|
|
6
|
+
|
|
7
|
+
## Runtime layers
|
|
8
|
+
|
|
9
|
+
```text
|
|
10
|
+
Task
|
|
11
|
+
-> intent/risk
|
|
12
|
+
-> capability requirements
|
|
13
|
+
-> adaptive evidence budget
|
|
14
|
+
-> semantic/index evidence
|
|
15
|
+
-> content-addressed evidence pointers
|
|
16
|
+
-> focused skills
|
|
17
|
+
-> capability-aware model
|
|
18
|
+
-> fresh executor
|
|
19
|
+
-> deterministic verification
|
|
20
|
+
-> visual/browser verifier when required
|
|
21
|
+
-> diagnosis + evidence expansion only on failure
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
## Evidence Store
|
|
25
|
+
|
|
26
|
+
Large raw tool output, durable specs and dependency reports are stored under:
|
|
27
|
+
|
|
28
|
+
```text
|
|
29
|
+
.ues-cache/evidence-v1/<hash-prefix>/<sha256>.blob
|
|
30
|
+
.ues-cache/evidence-v1/<hash-prefix>/<sha256>.json
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
A prompt receives a bounded excerpt and a reference such as `evidence:sha256:<hash>`. Use `ocskill store get` to retrieve only the needed slice. `.ues-cache` is runtime state and is excluded from workspace verification fingerprints.
|
|
34
|
+
|
|
35
|
+
## Adaptive evidence budget
|
|
36
|
+
|
|
37
|
+
FAST/STANDARD/DEEP remain outer safety ceilings. Inside that ceiling V11 allocates characters by evidence role instead of treating all context as equally valuable. Debugging shifts budget toward tests/history; high-risk work shifts toward tests/references; browser/visual work shifts toward deterministic tool evidence.
|
|
38
|
+
|
|
39
|
+
Failure expands evidence through the existing initial -> diagnose -> deep-recovery stages instead of loading maximum context on the first attempt.
|
|
40
|
+
|
|
41
|
+
## Prompt cache shape
|
|
42
|
+
|
|
43
|
+
Stable material is separated conceptually from dynamic task/evidence. The runtime records stable/dynamic hashes and cacheable ratio. This is telemetry, not a promise that every provider supports prompt caching.
|
|
44
|
+
|
|
45
|
+
## Capability-aware routing
|
|
46
|
+
|
|
47
|
+
Model profiles may declare coding, reasoning, toolCalling, vision, browser, filesystem, longContext, cost/latency class and a quality hint. UES never invents an unavailable capability. If no configured candidate satisfies a requirement it exposes capability fallback instead of silently claiming that a text-only model can see screenshots.
|
|
48
|
+
|
|
49
|
+
## Visual fidelity
|
|
50
|
+
|
|
51
|
+
Visual verification uses three complementary layers: semantic DOM/accessibility identity, geometry/bounding boxes, and screenshot pixels. VISUAL_SPEC describes important anchors and tolerances. Geometry receipts prove position/size claims. PNG diff finds changed pixels and their bounding region. A failed region can be cropped so vision only sees the area that needs judgment.
|
|
52
|
+
|
|
53
|
+
Screenshot equality does not prove accessibility or interaction; DOM equality does not prove appearance.
|
|
54
|
+
|
|
55
|
+
## Browser QA and security
|
|
56
|
+
|
|
57
|
+
Browser workflows prefer bounded deterministic scripts/CLI for ordinary verification. Remote webpage content is untrusted and cannot change UES/tool permissions, request secrets, expand the approved task or authorize external side effects.
|
|
58
|
+
|
|
59
|
+
## Dynamic workflows
|
|
60
|
+
|
|
61
|
+
The scheduler classifies units as deterministic, LLM judgment or vision judgment. Deterministic work does not spawn agents. Independent tasks may share a wave only when dependencies are ready and file ownership does not conflict. Each wave is integrated and verified before later waves rely on it.
|
|
62
|
+
|
|
63
|
+
## Skill system
|
|
64
|
+
|
|
65
|
+
V11 keeps progressive disclosure: description metadata for routing, short SKILL.md entrypoint, references only when the selected mode needs them, and deterministic logic in runtime/scripts rather than repeated prompt text.
|
|
66
|
+
|
|
67
|
+
`ocskill skills lint` flags oversized entrypoints and highly overlapping descriptions.
|
|
68
|
+
|
|
69
|
+
## Hermes sidecar
|
|
70
|
+
|
|
71
|
+
Hermes remains optional. UES owns durable `.ues-work` state, evidence references, task leases/runId fencing, verification receipts and safety/permission boundaries. Hermes may execute a bounded task/workflow when explicitly available but does not become the source of truth.
|
|
72
|
+
|
|
73
|
+
## Release evidence
|
|
74
|
+
|
|
75
|
+
V11 must pass syntax/resource/router/V11 contract tests, all Node tests, package/install smokes, no-regression live suites, capability-routing tests, visual geometry/pixel fixtures, browser security/targeted-evidence fixtures and token/cache/evidence telemetry benchmarks before stable promotion.
|
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
# UES V11 — Perception & Adaptive Execution
|
|
2
|
+
|
|
3
|
+
Status: stable (`11.0.0`). V11 is npm `latest`.
|
|
4
|
+
|
|
5
|
+
## Goal
|
|
6
|
+
|
|
7
|
+
V11 extends UES from a reliability-focused coding harness into a perception-aware adaptive execution system. The core rule remains:
|
|
8
|
+
|
|
9
|
+
> minimum context necessary for maximum verified task success
|
|
10
|
+
|
|
11
|
+
V11 must improve perception, routing and context efficiency without weakening V10 durability, verification or safety.
|
|
12
|
+
|
|
13
|
+
## Runtime layers
|
|
14
|
+
|
|
15
|
+
```text
|
|
16
|
+
user task
|
|
17
|
+
-> intent/risk classification
|
|
18
|
+
-> capability requirements
|
|
19
|
+
-> evidence budget
|
|
20
|
+
-> semantic/repository evidence
|
|
21
|
+
-> content-addressed evidence pointers
|
|
22
|
+
-> selective skill/model routing
|
|
23
|
+
-> fresh executor
|
|
24
|
+
-> deterministic verification
|
|
25
|
+
-> visual/browser verifier when required
|
|
26
|
+
-> recovery/escalation only from observed failure
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
## Content-addressed Evidence Store
|
|
30
|
+
|
|
31
|
+
Large evidence is stored below:
|
|
32
|
+
|
|
33
|
+
```text
|
|
34
|
+
.ues-cache/evidence-v1/
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
Records are addressed by SHA-256. Executor context receives bounded slices and evidence references instead of repeatedly embedding large tool output.
|
|
38
|
+
|
|
39
|
+
Commands:
|
|
40
|
+
|
|
41
|
+
```cmd
|
|
42
|
+
ocskill store status .
|
|
43
|
+
ocskill store put report.txt . --kind tool-output
|
|
44
|
+
ocskill store get evidence:sha256:<hash> . --max 12000
|
|
45
|
+
ocskill store gc . --max-entries 2000 --max-age-days 30
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
Evidence pointers are not proof by themselves. Any claim still needs the relevant content or a verification receipt.
|
|
49
|
+
|
|
50
|
+
## Adaptive Evidence Budget
|
|
51
|
+
|
|
52
|
+
V11 partitions context among instructions, task, declared code, tests, references, history and tools. Debugging, high-risk, visual and browser work can reallocate budget without automatically consuming the full 48k ceiling.
|
|
53
|
+
|
|
54
|
+
Failure expands evidence in stages; it does not justify replaying already-proven exploration.
|
|
55
|
+
|
|
56
|
+
## Prompt cache layout
|
|
57
|
+
|
|
58
|
+
The prompt envelope is split into a stable prefix and dynamic tail.
|
|
59
|
+
|
|
60
|
+
Stable:
|
|
61
|
+
- role and invariants
|
|
62
|
+
- loaded skill IDs
|
|
63
|
+
- stable project facts
|
|
64
|
+
- tool policy
|
|
65
|
+
|
|
66
|
+
Dynamic:
|
|
67
|
+
- current task
|
|
68
|
+
- evidence pointers
|
|
69
|
+
- recent failure
|
|
70
|
+
- next action
|
|
71
|
+
- recent messages
|
|
72
|
+
|
|
73
|
+
Telemetry records stable/dynamic hashes, sizes and cacheable ratio. This is diagnostic; provider-side cache behavior is provider-dependent.
|
|
74
|
+
|
|
75
|
+
## Capability-aware model routing
|
|
76
|
+
|
|
77
|
+
Models may declare:
|
|
78
|
+
- coding
|
|
79
|
+
- reasoning
|
|
80
|
+
- tool calling
|
|
81
|
+
- vision
|
|
82
|
+
- browser
|
|
83
|
+
- filesystem
|
|
84
|
+
- long context
|
|
85
|
+
- cost class
|
|
86
|
+
- latency class
|
|
87
|
+
- quality hint
|
|
88
|
+
|
|
89
|
+
Example:
|
|
90
|
+
|
|
91
|
+
```cmd
|
|
92
|
+
ocskill models capability provider/model --vision on --browser on --reasoning on --quality 0.9
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
Task text is converted to capability requirements. The existing light/standard/heavy tiers remain compatible, but V11 can select an eligible configured model instead of assuming every model has the same modalities.
|
|
96
|
+
|
|
97
|
+
## Visual fidelity
|
|
98
|
+
|
|
99
|
+
V11 verifies UI through three independent evidence layers:
|
|
100
|
+
|
|
101
|
+
1. semantic/accessibility identity
|
|
102
|
+
2. geometry/bounding boxes
|
|
103
|
+
3. rendered pixels
|
|
104
|
+
|
|
105
|
+
A visual task can use `VISUAL_SPEC.json` to define acceptance elements and tolerances.
|
|
106
|
+
|
|
107
|
+
```cmd
|
|
108
|
+
ocskill visual spec VISUAL_SPEC.json
|
|
109
|
+
ocskill visual geometry VISUAL_SPEC.json actual-boxes.json
|
|
110
|
+
ocskill visual compare expected.png actual.png --threshold 16 --max-diff-ratio 0.01
|
|
111
|
+
ocskill visual crop actual.png failed-region.png --x 100 --y 200 --width 300 --height 120
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
PNG comparison is deterministic and dependency-free for supported 8-bit non-interlaced PNGs. Vision models are used for appearance judgment or cropped failure regions, not for measurements that DOM/geometry can prove exactly.
|
|
115
|
+
|
|
116
|
+
## Browser QA and trust boundary
|
|
117
|
+
|
|
118
|
+
Browser workflows are CLI/script-first. Rich browser tooling is used only when persistent exploration or richer introspection is necessary.
|
|
119
|
+
|
|
120
|
+
```cmd
|
|
121
|
+
ocskill browser capability .
|
|
122
|
+
ocskill browser plan http://localhost:3000 --target Checkout
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
Remote webpage text, DOM, ARIA labels and downloaded content are untrusted evidence. They cannot:
|
|
126
|
+
- change UES/system policy
|
|
127
|
+
- expand permissions or filesystem scope
|
|
128
|
+
- request secrets
|
|
129
|
+
- authorize publish/deploy/purchases
|
|
130
|
+
- weaken verification requirements
|
|
131
|
+
|
|
132
|
+
## Dynamic workflow
|
|
133
|
+
|
|
134
|
+
```cmd
|
|
135
|
+
ocskill workflow-plan .ues-work/<slug>/PLAN.json --max-concurrent 4
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
The scheduler separates deterministic work from LLM/vision work, respects dependencies and serializes declared write conflicts. Deterministic commands should not consume agent slots. A schedule is planning evidence, not authorization for external side effects.
|
|
139
|
+
|
|
140
|
+
## Hermes sidecar
|
|
141
|
+
|
|
142
|
+
Hermes remains optional. UES owns durable state, safety and verification. Hermes can consume one-task or bounded workflow prompts and evidence pointers but does not own merge/push/publish/deploy.
|
|
143
|
+
|
|
144
|
+
```cmd
|
|
145
|
+
ocskill hermes status
|
|
146
|
+
ocskill hermes workflow <slug> .
|
|
147
|
+
ocskill hermes exec-workflow <slug> .
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
If Hermes is absent, core UES remains functional.
|
|
151
|
+
|
|
152
|
+
## New skills and agents
|
|
153
|
+
|
|
154
|
+
V11 adds nine focused skills:
|
|
155
|
+
- visual-fidelity
|
|
156
|
+
- browser-qa
|
|
157
|
+
- design-source
|
|
158
|
+
- responsive-verification
|
|
159
|
+
- component-visual-testing
|
|
160
|
+
- browser-security
|
|
161
|
+
- skill-authoring
|
|
162
|
+
- skill-evaluation
|
|
163
|
+
- dynamic-workflow
|
|
164
|
+
|
|
165
|
+
V11 adds two subagents:
|
|
166
|
+
- `ues-visual-verifier`: read-only independent rendered-evidence verification
|
|
167
|
+
- `ues-merge-arbiter`: conflict resolution across already-verified task changes
|
|
168
|
+
|
|
169
|
+
More agents are not automatically better. New agents require a distinct capability/verification boundary and benchmark evidence.
|
|
170
|
+
|
|
171
|
+
## Skill quality
|
|
172
|
+
|
|
173
|
+
```cmd
|
|
174
|
+
ocskill skills lint .
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
The linter checks metadata, entrypoint size and likely routing-description collisions. Skill instructions should use progressive disclosure and deterministic scripts for repeatable mechanics.
|
|
178
|
+
|
|
179
|
+
## Evaluation metrics
|
|
180
|
+
|
|
181
|
+
V11 keeps correctness first and additionally measures, when available:
|
|
182
|
+
- initial and total tokens
|
|
183
|
+
- cacheable prompt ratio
|
|
184
|
+
- repeated stable input
|
|
185
|
+
- evidence-reference reuse
|
|
186
|
+
- visual repair attempts
|
|
187
|
+
- context expansion count
|
|
188
|
+
- model escalation count
|
|
189
|
+
- latency/cost/tool calls
|
|
190
|
+
|
|
191
|
+
Missing telemetry is `null`, not zero.
|
|
192
|
+
|
|
193
|
+
Reference-vs-candidate gates can optionally require V11 telemetry:
|
|
194
|
+
|
|
195
|
+
```cmd
|
|
196
|
+
npm run evals:ablation -- v10-summary.json v11-summary.json --require-gate --min-cacheable-ratio 0.70 --min-evidence-reuse-ratio 0.20 --max-repeated-stable-ratio 0.20
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
An explicitly requested metric gate fails closed when its telemetry is unavailable.
|
|
200
|
+
|
|
201
|
+
## Release gates
|
|
202
|
+
|
|
203
|
+
Do not promote V11 to stable until all are satisfied:
|
|
204
|
+
|
|
205
|
+
1. `npm run ci` passes on the V11 tree.
|
|
206
|
+
2. No correctness/pass-rate regression versus the accepted V10 reference.
|
|
207
|
+
3. Required initial-input/token efficiency gate passes.
|
|
208
|
+
4. Visual geometry and PNG fixtures pass.
|
|
209
|
+
5. Browser trust-boundary and routing tests pass.
|
|
210
|
+
6. Compaction, timeout, provider recovery, lease recovery and loop guards remain green.
|
|
211
|
+
7. Packed and plain npm-install smoke tests pass.
|
|
212
|
+
8. Real weak-model evaluation shows no suite regression.
|
|
213
|
+
9. Any configured cache/evidence target has sufficient telemetry and passes.
|
|
214
|
+
10. npm `latest` is V11 stable; no release gate prevents promotion.
|
|
215
|
+
|
|
216
|
+
## Compatibility
|
|
217
|
+
|
|
218
|
+
OpenCode 1.x continues to receive resource/CLI behavior that does not require V2 runtime hooks.
|
|
219
|
+
|
|
220
|
+
OpenCode 2.x receives the managed router plugin and fresh-session runtime, including V11 deterministic helper tools for capability inference, evidence retrieval, visual geometry/diff planning, browser planning and dynamic workflow scheduling.
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
# V12 Weak-Model Intelligence Foundation
|
|
2
|
+
|
|
3
|
+
Status: beta prerelease (`12.0.0-beta.0`, npm dist-tag `next`). V11 (`11.0.0`) remains the stable `latest` release until V12 earns stable-release evidence.
|
|
4
|
+
|
|
5
|
+
## Goal
|
|
6
|
+
|
|
7
|
+
V12 focuses on making weaker coding models more reliable on large repositories by improving measured context quality, empirical model routing, plan identity, bounded autonomous decisions, and repo-scale evaluation. It does not claim that orchestration makes one base model equivalent to a stronger model.
|
|
8
|
+
|
|
9
|
+
## Foundations
|
|
10
|
+
|
|
11
|
+
- Empirical model performance: observed pass rate, retries, token use and latency can rerank capability-eligible models by task class.
|
|
12
|
+
- Context quality receipts: adaptive context reports required-file recall and irrelevant-context ratio.
|
|
13
|
+
- Plan-scoped snapshots: every imported plan gets a SHA-256 keyed snapshot and active-plan fence before execution.
|
|
14
|
+
- Decision policy: reversible local engineering choices can be auto-resolvable; publish/deploy/destructive/product decisions remain human-gated.
|
|
15
|
+
- Repo-scale contract suite: deterministic generation of a 300-module monorepo fixture for larger-repository validation.
|
|
16
|
+
|
|
17
|
+
## Release policy
|
|
18
|
+
|
|
19
|
+
V12 is not stable merely because unit tests pass. Promotion requires healthy GitHub CI/Security gates, repo-scale validation, real weak-model baseline-vs-UES trials, no regression in existing suites, measured context recall, sufficient empirical routing samples, and Windows/Linux package/install smoke evidence.
|
|
20
|
+
|
|
21
|
+
## Beta install
|
|
22
|
+
|
|
23
|
+
```cmd
|
|
24
|
+
npm install -g opencode-agent-skill@next
|
|
25
|
+
ocskill install
|
|
26
|
+
ocskill doctor
|
|
27
|
+
```
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
{
|
|
2
|
+
"version": 1,
|
|
3
|
+
"kind": "repo-scale-contract-suite",
|
|
4
|
+
"description": "Deterministic large-repository contract tasks used to validate V12 weak-model context/routing foundations before live model runs.",
|
|
5
|
+
"minimumGeneratedModules": 300,
|
|
6
|
+
"tasks": [
|
|
7
|
+
{
|
|
8
|
+
"id": "cross-package-regression",
|
|
9
|
+
"category": "repo-scale",
|
|
10
|
+
"objective": "Trace a regression across package boundaries without loading the entire generated monorepo into context.",
|
|
11
|
+
"requiredFiles": [
|
|
12
|
+
"packages/pkg-0/src/mod049.mjs",
|
|
13
|
+
"packages/pkg-1/src/mod049.mjs"
|
|
14
|
+
],
|
|
15
|
+
"acceptance": [
|
|
16
|
+
"Required-file context recall is measurable",
|
|
17
|
+
"Unrelated packages stay bounded"
|
|
18
|
+
]
|
|
19
|
+
},
|
|
20
|
+
{
|
|
21
|
+
"id": "public-contract-ripple",
|
|
22
|
+
"category": "contract",
|
|
23
|
+
"objective": "Change a public API field and identify the API producer and web consumer that must move together.",
|
|
24
|
+
"requiredFiles": [
|
|
25
|
+
"contracts/public-api.json",
|
|
26
|
+
"apps/api/src/service.mjs",
|
|
27
|
+
"apps/web/src/consumer.mjs"
|
|
28
|
+
],
|
|
29
|
+
"acceptance": [
|
|
30
|
+
"Contract producer and consumer are both discovered",
|
|
31
|
+
"Change-impact evidence is explicit"
|
|
32
|
+
]
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
"id": "deep-chain-debug",
|
|
36
|
+
"category": "debugging",
|
|
37
|
+
"objective": "Diagnose a defect near the end of a fifty-module import chain using bounded evidence expansion.",
|
|
38
|
+
"requiredFiles": [
|
|
39
|
+
"packages/pkg-2/src/mod048.mjs",
|
|
40
|
+
"packages/pkg-2/src/mod049.mjs"
|
|
41
|
+
],
|
|
42
|
+
"acceptance": [
|
|
43
|
+
"Initial context remains bounded",
|
|
44
|
+
"Recovery can expand around the failing chain"
|
|
45
|
+
]
|
|
46
|
+
},
|
|
47
|
+
{
|
|
48
|
+
"id": "multi-package-refactor",
|
|
49
|
+
"category": "refactor",
|
|
50
|
+
"objective": "Coordinate a refactor touching several package endpoints while serializing shared contract writes.",
|
|
51
|
+
"requiredFiles": [
|
|
52
|
+
"packages/pkg-3/src/index.mjs",
|
|
53
|
+
"packages/pkg-4/src/index.mjs",
|
|
54
|
+
"contracts/public-api.json"
|
|
55
|
+
],
|
|
56
|
+
"acceptance": [
|
|
57
|
+
"Cross-package scope is explicit",
|
|
58
|
+
"Shared contract changes are treated as a write-conflict surface"
|
|
59
|
+
]
|
|
60
|
+
}
|
|
61
|
+
]
|
|
62
|
+
}
|
|
@@ -1233,6 +1233,88 @@
|
|
|
1233
1233
|
"ues-task-planner"
|
|
1234
1234
|
],
|
|
1235
1235
|
"exclude": []
|
|
1236
|
+
},
|
|
1237
|
+
{
|
|
1238
|
+
"id": "v11-visual-fidelity",
|
|
1239
|
+
"prompt": "Match this reference screenshot pixel-perfect and verify visual fidelity.",
|
|
1240
|
+
"include": [
|
|
1241
|
+
"ues-visual-fidelity"
|
|
1242
|
+
],
|
|
1243
|
+
"exclude": [
|
|
1244
|
+
"ues-payment-engineering"
|
|
1245
|
+
]
|
|
1246
|
+
},
|
|
1247
|
+
{
|
|
1248
|
+
"id": "v11-browser-qa",
|
|
1249
|
+
"prompt": "Use Playwright browser QA to verify the checkout flow.",
|
|
1250
|
+
"include": [
|
|
1251
|
+
"ues-browser-qa"
|
|
1252
|
+
],
|
|
1253
|
+
"exclude": [
|
|
1254
|
+
"ues-react-native-engineering"
|
|
1255
|
+
]
|
|
1256
|
+
},
|
|
1257
|
+
{
|
|
1258
|
+
"id": "v11-design-source",
|
|
1259
|
+
"prompt": "Read the Figma design source and extract design tokens.",
|
|
1260
|
+
"include": [
|
|
1261
|
+
"ues-design-source"
|
|
1262
|
+
],
|
|
1263
|
+
"exclude": [
|
|
1264
|
+
"ues-database-engineering"
|
|
1265
|
+
]
|
|
1266
|
+
},
|
|
1267
|
+
{
|
|
1268
|
+
"id": "v11-responsive",
|
|
1269
|
+
"prompt": "Verify responsive breakpoint behavior across mobile and tablet layout.",
|
|
1270
|
+
"include": [
|
|
1271
|
+
"ues-responsive-verification"
|
|
1272
|
+
],
|
|
1273
|
+
"exclude": []
|
|
1274
|
+
},
|
|
1275
|
+
{
|
|
1276
|
+
"id": "v11-storybook",
|
|
1277
|
+
"prompt": "Add Storybook visual regression coverage for this component.",
|
|
1278
|
+
"include": [
|
|
1279
|
+
"ues-engineering-orchestrator",
|
|
1280
|
+
"ues-component-visual-testing"
|
|
1281
|
+
],
|
|
1282
|
+
"exclude": []
|
|
1283
|
+
},
|
|
1284
|
+
{
|
|
1285
|
+
"id": "v11-skill-authoring",
|
|
1286
|
+
"prompt": "Create an agent skill with a precise trigger description.",
|
|
1287
|
+
"include": [
|
|
1288
|
+
"ues-engineering-orchestrator",
|
|
1289
|
+
"ues-skill-authoring"
|
|
1290
|
+
],
|
|
1291
|
+
"exclude": []
|
|
1292
|
+
},
|
|
1293
|
+
{
|
|
1294
|
+
"id": "v11-skill-eval",
|
|
1295
|
+
"prompt": "Evaluate skill routing precision and recall with a benchmark.",
|
|
1296
|
+
"include": [
|
|
1297
|
+
"ues-skill-evaluation"
|
|
1298
|
+
],
|
|
1299
|
+
"exclude": []
|
|
1300
|
+
},
|
|
1301
|
+
{
|
|
1302
|
+
"id": "v11-dynamic-workflow",
|
|
1303
|
+
"prompt": "Plan a dynamic workflow fan-out in bounded waves for many independent tasks.",
|
|
1304
|
+
"include": [
|
|
1305
|
+
"ues-engineering-orchestrator",
|
|
1306
|
+
"ues-dynamic-workflow"
|
|
1307
|
+
],
|
|
1308
|
+
"exclude": []
|
|
1309
|
+
},
|
|
1310
|
+
{
|
|
1311
|
+
"id": "v11-browser-security",
|
|
1312
|
+
"prompt": "Audit browser security against prompt injection from an untrusted webpage.",
|
|
1313
|
+
"include": [
|
|
1314
|
+
"ues-engineering-orchestrator",
|
|
1315
|
+
"ues-browser-security"
|
|
1316
|
+
],
|
|
1317
|
+
"exclude": []
|
|
1236
1318
|
}
|
|
1237
1319
|
]
|
|
1238
1320
|
}
|