opencode-agent-skill 9.0.0 → 10.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +49 -0
- package/README.md +696 -675
- package/bin/ocskill.mjs +24 -0
- package/global-config/AGENTS.md +78 -163
- package/global-config/plugins/ues-router/index.js +413 -59
- package/global-config/plugins/ues-router/router.js +35 -0
- package/global-config/plugins/ues-router/runtime-guard.js +265 -0
- package/lib/benchmark-confidence.mjs +49 -11
- package/lib/eval-ablation.mjs +104 -0
- package/lib/eval-report.mjs +11 -0
- package/lib/eval-telemetry.mjs +3 -0
- package/lib/model-policy.mjs +6 -0
- package/lib/orchestrator-policy.mjs +99 -6
- package/lib/task-engine.mjs +146 -9
- package/package.json +4 -3
- package/scripts/eval-ablation.mjs +44 -0
- package/scripts/eval-matrix.mjs +13 -2
package/bin/ocskill.mjs
CHANGED
|
@@ -54,6 +54,8 @@ import {
|
|
|
54
54
|
createPlanVerificationReceipt,
|
|
55
55
|
createIntegrationVerificationReceipt,
|
|
56
56
|
runtimeEvents,
|
|
57
|
+
checkpointWork,
|
|
58
|
+
markCheckpointResumed,
|
|
57
59
|
} from "../lib/task-engine.mjs"
|
|
58
60
|
import { reviewScope } from "../lib/review-scope.mjs"
|
|
59
61
|
import { buildVerificationPlan } from "../lib/verification-plan.mjs"
|
|
@@ -589,6 +591,28 @@ async function workControl() {
|
|
|
589
591
|
}))
|
|
590
592
|
return
|
|
591
593
|
}
|
|
594
|
+
if (action === "checkpoint") {
|
|
595
|
+
const taskID = args[3]
|
|
596
|
+
const root = positionalArg(args, 4) || process.cwd()
|
|
597
|
+
if (!taskID) throw new Error("Usage: ocskill work checkpoint <slug> <task-id> [dir] --run-id <id> [--reason <text>]")
|
|
598
|
+
printJson(await checkpointWork(root, slug, {
|
|
599
|
+
taskID,
|
|
600
|
+
runId: optionValue(args, "--run-id"),
|
|
601
|
+
reason: optionValue(args, "--reason") || "pre-compaction",
|
|
602
|
+
}))
|
|
603
|
+
return
|
|
604
|
+
}
|
|
605
|
+
if (action === "checkpoint-resumed") {
|
|
606
|
+
const taskID = args[3]
|
|
607
|
+
const root = positionalArg(args, 4) || process.cwd()
|
|
608
|
+
if (!taskID) throw new Error("Usage: ocskill work checkpoint-resumed <slug> <task-id> [dir] --run-id <id> [--reason <text>]")
|
|
609
|
+
printJson(await markCheckpointResumed(root, slug, {
|
|
610
|
+
taskID,
|
|
611
|
+
runId: optionValue(args, "--run-id"),
|
|
612
|
+
reason: optionValue(args, "--reason"),
|
|
613
|
+
}))
|
|
614
|
+
return
|
|
615
|
+
}
|
|
592
616
|
if (action === "events") {
|
|
593
617
|
const root = positionalArg(args, 3) || process.cwd()
|
|
594
618
|
const limit = optionInt(args, "--limit", 200)
|
package/global-config/AGENTS.md
CHANGED
|
@@ -1,215 +1,130 @@
|
|
|
1
1
|
# Universal Engineering System
|
|
2
2
|
|
|
3
|
-
These instructions apply to software-engineering work in OpenCode when
|
|
3
|
+
These instructions apply to software-engineering work in OpenCode when UES is installed.
|
|
4
4
|
|
|
5
|
-
##
|
|
5
|
+
## Core rule
|
|
6
6
|
|
|
7
|
-
|
|
7
|
+
Use the minimum context and orchestration that preserve correctness. Do not trade acceptance criteria, repository evidence, verification, or safety for lower token use.
|
|
8
8
|
|
|
9
|
-
|
|
10
|
-
understand -> route -> plan when needed -> implement -> verify -> review -> finish
|
|
11
|
-
^ |
|
|
12
|
-
+---- diagnose <-----+
|
|
13
|
-
```
|
|
9
|
+
The model remains the model. UES improves task routing, evidence selection, verification and recovery; it does not replace model capability.
|
|
14
10
|
|
|
15
|
-
|
|
11
|
+
## Start by classifying the task
|
|
16
12
|
|
|
17
|
-
|
|
13
|
+
When `ocskill` is available, use `ocskill task-policy <text>` as the deterministic starting point.
|
|
18
14
|
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
4. For non-trivial work, load `ues-engineering-orchestrator` first. Then load only the process and domain skills that materially help.
|
|
23
|
-
5. Prefer process skills before framework skills: exploration/planning/debugging/verification determine how to work; domain skills determine what framework-specific details to apply.
|
|
24
|
-
6. Keep the active skill set focused. Usually 2-4 skills are enough; do not load the entire catalog.
|
|
15
|
+
- **FAST** — focused, low-risk work with a clear target. Read the target, nearest relevant test/analogue and only direct dependencies needed to prove the change. Prefer at most two directly useful skills. Do not load the engineering orchestrator, planner, critic, repo-wide graph or broad framework context unless concrete uncertainty or failure requires escalation.
|
|
16
|
+
- **STANDARD** — moderate uncertainty, several related files or a behavior change. Make a short file-aware plan, inspect affected callers/tests, and load only the process/domain skills that materially help.
|
|
17
|
+
- **DEEP** — high-risk, public-contract, auth/security/payment/schema/migration, cross-module or long-horizon work. Use impact analysis, planning, durable state and independent verification as required by policy.
|
|
25
18
|
|
|
26
|
-
|
|
19
|
+
Risk overrides convenience. A short prompt can still require DEEP handling when the blast radius is high.
|
|
27
20
|
|
|
28
|
-
|
|
21
|
+
## Evidence-first work
|
|
29
22
|
|
|
30
|
-
|
|
31
|
-
- `ocskill impact <symbol-or-term> [dir]` — bounded path/content impact search
|
|
32
|
-
- `ocskill evidence [dir]` — stack + verification + Git evidence snapshot
|
|
33
|
-
- `ocskill working-tree [dir]` — branch, HEAD and uncommitted-change state
|
|
34
|
-
- `ocskill repo-graph [dir]` — bounded source import graph and coupling hotspots
|
|
35
|
-
- `ocskill review-scope [base] [dir]` — deterministic changed-file coverage and risk hints
|
|
36
|
-
- `ocskill verification-plan [dir]` — project-native verification recommendations
|
|
37
|
-
- `ocskill task-graph <PLAN.json>` — validate dependencies and compute safe execution waves
|
|
38
|
-
- `ocskill context-pack <slug> <task> [dir]` — bounded durable handoff enriched with declared files, import neighbors, likely tests, instruction/manifests and accepted lessons
|
|
39
|
-
- `ocskill work verify-command ... -- <command>` — structured verification receipt (exit code, hashes, timing, workspace fingerprints)
|
|
40
|
-
- `ocskill sandbox create|integrate|list|remove ...` — isolated Git worktree primitives with conflict-aware integration
|
|
41
|
-
- `ocskill work events <slug> [dir]` — append-only runtime event journal for start/heartbeat/receipt/failure/recovery/completion/integration/finalization
|
|
42
|
-
- `ocskill learn status|analyze|accept|promote ...` — evidence-gated learning loop; shadow-required lessons are retrieved only after measured improvement
|
|
43
|
-
- `ocskill dashboard [dir] --serve` — local Control Center for work state, evidence, learning and eval summaries
|
|
23
|
+
Never invent repository structure, files, functions, APIs, schemas, package versions, runtime behavior or test results when they can be checked.
|
|
44
24
|
|
|
45
|
-
|
|
25
|
+
Read narrowly in this order when practical:
|
|
46
26
|
|
|
47
|
-
|
|
27
|
+
1. applicable repository instructions/manifests;
|
|
28
|
+
2. the named target or failure location;
|
|
29
|
+
3. nearest working analogue and direct callers/dependencies;
|
|
30
|
+
4. tests that encode the requested behavior;
|
|
31
|
+
5. broader graph/repository evidence only if uncertainty remains.
|
|
48
32
|
|
|
49
|
-
|
|
33
|
+
Useful deterministic helpers include:
|
|
50
34
|
|
|
51
|
-
-
|
|
52
|
-
-
|
|
53
|
-
-
|
|
54
|
-
-
|
|
35
|
+
- `ocskill inspect [dir]`
|
|
36
|
+
- `ocskill impact <symbol-or-term> [dir]`
|
|
37
|
+
- `ocskill aci search|refs|view|text ...`
|
|
38
|
+
- `ocskill working-tree [dir]`
|
|
39
|
+
- `ocskill verification-plan [dir]`
|
|
40
|
+
- `ocskill context-pack <slug> <task> [dir]`
|
|
41
|
+
- `ocskill work verify-command ... -- <command>`
|
|
55
42
|
|
|
56
|
-
|
|
43
|
+
Treat search/routing results as evidence hints, not semantic proof.
|
|
57
44
|
|
|
58
|
-
##
|
|
45
|
+
## Exact-contract discipline
|
|
59
46
|
|
|
60
|
-
|
|
47
|
+
For every edit:
|
|
61
48
|
|
|
62
|
-
-
|
|
63
|
-
-
|
|
64
|
-
-
|
|
65
|
-
-
|
|
66
|
-
-
|
|
67
|
-
-
|
|
68
|
-
- bug, crash, failed build/test, regression -> `ues-bug-diagnosis`
|
|
69
|
-
- meaningful edits -> `ues-test-verification`
|
|
70
|
-
- completed substantial change -> `ues-code-review`
|
|
71
|
-
- long task that must survive interruption -> `ues-long-task-state`
|
|
49
|
+
- preserve the user's observable acceptance criteria literally;
|
|
50
|
+
- preserve requested exception classes, type/range distinctions, return shapes, field names/order, mutation rules, idempotency and boundary behavior;
|
|
51
|
+
- preserve unrelated user changes;
|
|
52
|
+
- follow the repository's package manager, formatter, test/build conventions and generated-file policy;
|
|
53
|
+
- make the smallest coherent change; avoid opportunistic refactors and unrelated dependency upgrades;
|
|
54
|
+
- for public contracts, persistence, auth, payments, migrations or deployment, inspect downstream compatibility and rollback impact.
|
|
72
55
|
|
|
73
|
-
|
|
56
|
+
When tests are absent or hidden, use focused runtime probes for each stated criterion, especially boundary and mutation cases.
|
|
74
57
|
|
|
58
|
+
## Selective skill loading
|
|
59
|
+
|
|
60
|
+
Skills are on-demand context, not a checklist.
|
|
61
|
+
|
|
62
|
+
FAST should prefer the direct debugging/domain/verification skill and avoid generic orchestration unless needed. STANDARD may add `ues-engineering-orchestrator` plus a small number of directly relevant skills. DEEP may use planner, change-impact, long-task and critic/reviewer roles.
|
|
63
|
+
|
|
64
|
+
Typical direct routing:
|
|
65
|
+
|
|
66
|
+
- bug/crash/test failure -> `ues-bug-diagnosis`
|
|
67
|
+
- current external API/version/package -> `ues-research-verification`
|
|
75
68
|
- API contract -> `ues-api-contract`
|
|
76
69
|
- database/schema -> `ues-database-engineering`
|
|
77
70
|
- auth/permissions -> `ues-auth-security`
|
|
78
|
-
-
|
|
79
|
-
- Next.js -> `ues-nextjs-engineering`
|
|
71
|
+
- payment/webhook -> `ues-payment-engineering`
|
|
80
72
|
- React Native -> `ues-react-native-engineering`
|
|
73
|
+
- Next.js -> `ues-nextjs-engineering`
|
|
74
|
+
- React -> `ues-react-engineering`
|
|
81
75
|
- Node/Nest -> `ues-nodejs-engineering` / `ues-nestjs-engineering`
|
|
76
|
+
- Python/Django/FastAPI -> corresponding UES domain skill
|
|
82
77
|
- .NET -> `ues-dotnet-engineering`
|
|
83
78
|
- Java/Spring -> `ues-java-spring-engineering`
|
|
84
|
-
- Python/Django/FastAPI -> `ues-python-engineering` / `ues-django-engineering` / `ues-fastapi-engineering`
|
|
85
79
|
- Flutter -> `ues-flutter-engineering`
|
|
86
|
-
- UI/UX -> `ues-ui-ux-engineering`
|
|
87
|
-
- ecommerce/marketplace -> `ues-ecommerce-engineering`
|
|
88
|
-
- payment -> `ues-payment-engineering`
|
|
89
80
|
- Docker/CI/deploy -> `ues-devops-engineering`
|
|
90
|
-
- Git -> `ues-git-safety`
|
|
91
|
-
|
|
92
|
-
## Evidence and research
|
|
93
|
-
|
|
94
|
-
- Never invent files, functions, endpoints, schemas, commands, package names, package versions, framework behavior, or project structure when they can be checked.
|
|
95
|
-
- Prefer repository evidence for repository facts.
|
|
96
|
-
- For external APIs, libraries, versions, security guidance, or behavior that may have changed, use `ues-research-verification` and prefer primary/current sources.
|
|
97
|
-
- Distinguish observed facts, sourced facts, hypotheses, and recommendations.
|
|
98
|
-
- If a tool/source is unavailable, say what could not be verified instead of filling the gap with confidence.
|
|
99
|
-
|
|
100
|
-
## Debugging and retry discipline
|
|
101
|
-
|
|
102
|
-
- Reproduce or capture the exact failure before proposing a fix.
|
|
103
|
-
- Trace the bad value/state backward to the earliest supported cause.
|
|
104
|
-
- Change one causal variable at a time.
|
|
105
|
-
- If two attempted fixes fail, stop stacking patches and re-investigate from fresh evidence.
|
|
106
|
-
- If three distinct root-cause hypotheses fail or fixes expose widening coupling, question the architecture and surface that to the user before another broad change.
|
|
107
|
-
- Do not clear caches, delete lockfiles, disable checks, or upgrade dependencies as generic debugging rituals.
|
|
108
|
-
|
|
109
|
-
## Context discipline
|
|
110
81
|
|
|
111
|
-
|
|
112
|
-
- Prefer exact symbol/error searches over broad directory dumps.
|
|
113
|
-
- Summarize what is known before expanding the search.
|
|
114
|
-
- Use supporting files inside skills only when their section is needed.
|
|
115
|
-
- Do not repeatedly reread unchanged large files unless new evidence requires it.
|
|
82
|
+
Do not load the full catalog.
|
|
116
83
|
|
|
117
|
-
##
|
|
84
|
+
## Failure and weak-model recovery
|
|
118
85
|
|
|
119
|
-
|
|
86
|
+
A failed attempt is a signal to improve evidence, not to repeat the same prompt with more prose.
|
|
120
87
|
|
|
121
|
-
-
|
|
122
|
-
-
|
|
123
|
-
-
|
|
124
|
-
- architecture/implementation decisions and material alternatives
|
|
125
|
-
- acceptance-criteria status, changed files, fresh verification, unresolved risks, and one next action
|
|
88
|
+
- **Attempt 1:** use the normal FAST/STANDARD/DEEP context budget.
|
|
89
|
+
- **Attempt 2:** capture the exact failure, load failure-adjacent caller/test evidence, add debugging context when useful, and allow the configured model tier to escalate. Do not stack a speculative patch.
|
|
90
|
+
- **Attempt 3+:** re-investigate from fresh evidence, expand to callers/dependencies/contracts and repository graph, challenge architecture/coupling, and use critic/reviewer verification before accepting another repair.
|
|
126
91
|
|
|
127
|
-
|
|
92
|
+
If two fixes fail, stop patch stacking and re-diagnose. If three distinct root-cause hypotheses fail or coupling keeps widening, surface the architectural issue before another broad change.
|
|
128
93
|
|
|
129
|
-
|
|
94
|
+
A stronger model is not a substitute for missing evidence.
|
|
130
95
|
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
1. Map the relevant repository surface with deterministic evidence and `ues-codebase-mapper` when useful.
|
|
134
|
-
2. Persist observable requirements in `.ues-work/<slug>/SPEC.md`.
|
|
135
|
-
3. Create a machine-checkable `PLAN.json` and validate it with `ocskill task-graph`.
|
|
136
|
-
4. Ask `ues-plan-checker` to challenge the plan before edits begin. For long/high-risk work, create a structured plan-verification receipt bound to the current plan hash, then record approval with `ocskill work approve-plan --receipt-file ...`; `work start` is blocked until this happens.
|
|
137
|
-
5. Execute each approved task in a fresh `ues-executor` context. Active V7 tasks carry a runId, heartbeat and lease expiry so interrupted work can be recovered deterministically. On OpenCode V2 prefer `ues.dispatch_task`, which creates the fresh session and applies configured attempt-based model escalation.
|
|
138
|
-
6. Inspect each child diff and use `ocskill work verify-command` before marking completion. Long/high-risk tasks require at least one successful receipt for the active run; narrative-only completion is rejected.
|
|
139
|
-
7. Parallelize only dependency-safe tasks with no write/read conflict. V8 may isolate concurrent writers automatically; manual sandboxes use `ocskill sandbox create` and `ocskill sandbox integrate`, which refuses overlap with dirty root files.
|
|
140
|
-
8. On resume, trust durable state plus current Git evidence over conversational memory. Recover expired executor leases before retrying; preserve runId fences for active attempts.
|
|
141
|
-
9. After all tasks complete, run `ues-integration-verifier`. Long/high-risk PASS must be bound to the current workspace fingerprint with a structured integration-verification receipt before `ocskill work verify-integration` records it.
|
|
142
|
-
10. `work finalize` requires a recorded integration PASS and rejects completion if the Git workspace changed after that PASS.
|
|
143
|
-
11. Merge/push/publish/deploy remain external side effects and require explicit user intent.
|
|
144
|
-
|
|
145
|
-
Use `ocskill model-policy <role> --attempt N` when configured model tiers exist. Escalate only after diagnosis/fresh context; never use a stronger model as a substitute for missing evidence.
|
|
146
|
-
|
|
147
|
-
## Critic and repair discipline
|
|
148
|
-
|
|
149
|
-
For substantial or high-risk behavior changes, verification is followed by an independent falsification pass:
|
|
150
|
-
|
|
151
|
-
1. self-check the diff against observable acceptance criteria
|
|
152
|
-
2. run fresh behavior-matched verification
|
|
153
|
-
3. ask `ues-critic` or `ues-reviewer` to challenge assumptions and search for concrete counterexamples
|
|
154
|
-
4. repair only evidence-backed blocking findings
|
|
155
|
-
5. rerun affected verification
|
|
156
|
-
6. repeat the critic pass only when the repair materially changed risky behavior
|
|
157
|
-
|
|
158
|
-
Bound this loop to at most two repair cycles before returning to root-cause/architecture analysis. Do not churn code to satisfy speculative feedback. Unresolved blocking findings must be fixed or surfaced explicitly.
|
|
159
|
-
|
|
160
|
-
## Subagent discipline
|
|
96
|
+
## Verification gate
|
|
161
97
|
|
|
162
|
-
|
|
98
|
+
Before claiming completion:
|
|
163
99
|
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
- `ues-critic` — read-only adversarial falsification
|
|
172
|
-
- `ues-verifier` — read-only task/acceptance verification
|
|
173
|
-
- `ues-integration-verifier` — read-only cross-task/end-to-end verification
|
|
100
|
+
1. identify what observable evidence proves the requested behavior;
|
|
101
|
+
2. run the narrowest relevant check, then expand based on risk and repository conventions;
|
|
102
|
+
3. read the actual output and exit status;
|
|
103
|
+
4. re-test the original failure/acceptance criterion;
|
|
104
|
+
5. inspect the final diff for accidental changes;
|
|
105
|
+
6. for substantial/high-risk work, run independent review/critic verification and resolve evidence-backed blockers;
|
|
106
|
+
7. report exactly what passed, failed or was not run.
|
|
174
107
|
|
|
175
|
-
|
|
108
|
+
Never claim a test, build, migration, deployment, push or release succeeded unless it actually did.
|
|
176
109
|
|
|
177
|
-
##
|
|
110
|
+
## Long-horizon work
|
|
178
111
|
|
|
179
|
-
For
|
|
112
|
+
For interruption-prone or dependent multi-task work, use durable `.ues-work/<slug>/` state instead of relying on conversation memory.
|
|
180
113
|
|
|
181
|
-
|
|
182
|
-
- Follow the repository's package manager, formatter, linter, tests, build scripts, architecture, and generated-file policy.
|
|
183
|
-
- Preserve unrelated user changes.
|
|
184
|
-
- Avoid opportunistic refactors and broad dependency upgrades during unrelated fixes.
|
|
185
|
-
- For behavior changes where a practical test harness exists, prefer a failing regression/behavior test before implementation.
|
|
186
|
-
- For public contracts, persistence, auth, payments, migrations, and deployment, explicitly inspect downstream consumers and rollback/compatibility impact.
|
|
114
|
+
The required sequence is:
|
|
187
115
|
|
|
188
|
-
|
|
116
|
+
`SPEC -> PLAN -> plan check/receipt -> approved tasks -> fresh executor per task -> task verification receipts -> integration verification/receipt -> finalize`
|
|
189
117
|
|
|
190
|
-
|
|
118
|
+
Use dependency-safe waves and isolated worktrees only when their write/read scopes are safe. On resume, trust durable state plus current Git evidence over conversational memory. Long/high-risk completion must remain bound to the active run and current workspace fingerprint.
|
|
191
119
|
|
|
192
|
-
|
|
193
|
-
2. Run the narrowest relevant checks, then expand based on risk and project conventions.
|
|
194
|
-
3. Read the actual output and exit status.
|
|
195
|
-
4. Re-test the original failure/acceptance criterion, not only compilation.
|
|
196
|
-
5. Inspect the final diff for accidental changes and regressions.
|
|
197
|
-
6. Run or request an independent review/critic pass for substantial or high-risk work and resolve evidence-backed blocking findings.
|
|
198
|
-
7. Report exactly what passed, failed, was repaired, or was not run.
|
|
120
|
+
Do not store hidden chain-of-thought. Persist observable facts, decisions, acceptance status, evidence and next actions only.
|
|
199
121
|
|
|
200
|
-
|
|
122
|
+
## Safety
|
|
201
123
|
|
|
202
|
-
|
|
124
|
+
Ask before destructive or irreversible actions such as force-pushing, destructive reset/clean, deleting important data, dropping database objects, broad production migrations, production deployment or credential rotation. Never print secrets.
|
|
203
125
|
|
|
204
|
-
|
|
126
|
+
Merge, push, publish and deploy are external side effects and require explicit user intent.
|
|
205
127
|
|
|
206
128
|
## Completion standard
|
|
207
129
|
|
|
208
|
-
A task is complete only when requested behavior is implemented, acceptance criteria are addressed, relevant verification
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
## V7 learning and optional external executors
|
|
212
|
-
|
|
213
|
-
UES may analyze its own `.ues-evals` traces with `ocskill learn analyze`. V8 clusters recurring failure signatures and emits candidate rules. Acceptance stages a proposal; shadow-required lessons appear in future context only after `ocskill learn promote` records a measured benchmark improvement.
|
|
214
|
-
|
|
215
|
-
Hermes support is optional and adapter-style. `ocskill hermes status` checks availability and `ocskill hermes prompt <slug> <task> .` emits a bounded delegation prompt. Hermes is not embedded into the UES runtime and may not mutate UES durable state on its own.
|
|
130
|
+
A task is complete only when the requested behavior is implemented, the acceptance criteria are addressed, fresh relevant verification supports the result, the final diff is reviewed, and remaining limitations are stated accurately.
|