opencode-agent-skill 7.7.0 → 10.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +112 -3
- package/README.md +396 -281
- package/bin/ocskill.mjs +382 -156
- package/docs/DETERMINISTIC-TOOLS.md +25 -8
- package/docs/ENGINEERING-DESIGN.md +31 -13
- package/docs/EVALS.md +34 -12
- package/docs/NPM-PUBLISH.md +6 -6
- package/docs/OPENCODE-COMPAT.md +11 -8
- package/docs/TRACE-SCHEMA.md +15 -2
- package/docs/V8-INTELLIGENCE-RELIABILITY.md +206 -0
- package/docs/V9-SPEED-INTELLIGENCE.md +102 -0
- package/evals/live/tasks.json +6 -6
- package/evals/polyglot/fixtures/polyglot-bench/api/generated/client.ts +2 -0
- package/evals/polyglot/fixtures/polyglot-bench/api/openapi.json +25 -0
- package/evals/polyglot/fixtures/polyglot-bench/db/migrations/20260920_add_order_key.sql +1 -0
- package/evals/polyglot/fixtures/polyglot-bench/dotnet/OrderService.cs +8 -0
- package/evals/polyglot/fixtures/polyglot-bench/java/PriceService.java +5 -0
- package/evals/polyglot/fixtures/polyglot-bench/monorepo/package.json +6 -0
- package/evals/polyglot/fixtures/polyglot-bench/monorepo/packages/api/package.json +4 -0
- package/evals/polyglot/fixtures/polyglot-bench/monorepo/packages/web/package.json +7 -0
- package/evals/polyglot/fixtures/polyglot-bench/monorepo/pnpm-lock.yaml +5 -0
- package/evals/polyglot/fixtures/polyglot-bench/next/app/api/products/route.ts +7 -0
- package/evals/polyglot/fixtures/polyglot-bench/python/tenant_auth.py +4 -0
- package/evals/polyglot/fixtures/polyglot-bench/react-native/keyboard.ts +3 -0
- package/evals/polyglot/graders/polyglot-bench.mjs +101 -0
- package/evals/polyglot/tasks.json +54 -0
- package/global-config/AGENTS.md +78 -160
- package/global-config/agents/integration-verifier.md +1 -1
- package/global-config/agents/plan-checker.md +1 -1
- package/global-config/commands/run.md +9 -5
- package/global-config/plugins/ues-router/capabilities.js +4 -0
- package/global-config/plugins/ues-router/index.js +784 -37
- package/global-config/plugins/ues-router/router.js +175 -23
- package/global-config/plugins/ues-router/runtime-guard.js +265 -0
- package/global-config/skills/engineering-orchestrator/references/long-horizon.md +6 -4
- package/lib/aci.mjs +128 -0
- package/lib/benchmark-confidence.mjs +173 -0
- package/lib/cli-utils.mjs +41 -0
- package/lib/container-sandbox.mjs +102 -0
- package/lib/context-manifest.mjs +300 -22
- package/lib/control-center.mjs +36 -3
- package/lib/eval-ablation.mjs +104 -0
- package/lib/eval-order.mjs +9 -0
- package/lib/eval-report.mjs +11 -0
- package/lib/eval-telemetry.mjs +8 -2
- package/lib/gate-receipt.mjs +52 -0
- package/lib/installer.mjs +39 -25
- package/lib/learning-engine.mjs +236 -38
- package/lib/model-policy.mjs +6 -0
- package/lib/opencode-compat.mjs +25 -10
- package/lib/orchestrator-policy.mjs +195 -21
- package/lib/process-runner.mjs +30 -9
- package/lib/runtime-events.mjs +31 -0
- package/lib/semantic-index.mjs +318 -0
- package/lib/task-engine.mjs +557 -28
- package/lib/trajectory.mjs +89 -0
- package/lib/windows-shim.mjs +227 -0
- package/lib/worktree-sandbox.mjs +85 -3
- package/package.json +11 -4
- package/scripts/check-release-tag.mjs +22 -0
- package/scripts/control-center.mjs +25 -0
- package/scripts/eval-ablation.mjs +44 -0
- package/scripts/eval-live.mjs +39 -58
- package/scripts/eval-matrix.mjs +166 -0
- package/scripts/smoke-packed-install.mjs +138 -4
- package/scripts/smoke-plain-install.mjs +91 -0
- package/scripts/validate-live-suite.mjs +3 -3
- package/scripts/validate.mjs +27 -5
package/global-config/AGENTS.md
CHANGED
|
@@ -1,212 +1,130 @@
|
|
|
1
1
|
# Universal Engineering System
|
|
2
2
|
|
|
3
|
-
These instructions apply to software-engineering work in OpenCode when
|
|
3
|
+
These instructions apply to software-engineering work in OpenCode when UES is installed.
|
|
4
4
|
|
|
5
|
-
##
|
|
5
|
+
## Core rule
|
|
6
6
|
|
|
7
|
-
|
|
7
|
+
Use the minimum context and orchestration that preserve correctness. Do not trade acceptance criteria, repository evidence, verification, or safety for lower token use.
|
|
8
8
|
|
|
9
|
-
|
|
10
|
-
understand -> route -> plan when needed -> implement -> verify -> review -> finish
|
|
11
|
-
^ |
|
|
12
|
-
+---- diagnose <-----+
|
|
13
|
-
```
|
|
9
|
+
The model remains the model. UES improves task routing, evidence selection, verification and recovery; it does not replace model capability.
|
|
14
10
|
|
|
15
|
-
|
|
11
|
+
## Start by classifying the task
|
|
16
12
|
|
|
17
|
-
|
|
13
|
+
When `ocskill` is available, use `ocskill task-policy <text>` as the deterministic starting point.
|
|
18
14
|
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
4. For non-trivial work, load `ues-engineering-orchestrator` first. Then load only the process and domain skills that materially help.
|
|
23
|
-
5. Prefer process skills before framework skills: exploration/planning/debugging/verification determine how to work; domain skills determine what framework-specific details to apply.
|
|
24
|
-
6. Keep the active skill set focused. Usually 2-4 skills are enough; do not load the entire catalog.
|
|
15
|
+
- **FAST** — focused, low-risk work with a clear target. Read the target, nearest relevant test/analogue and only direct dependencies needed to prove the change. Prefer at most two directly useful skills. Do not load the engineering orchestrator, planner, critic, repo-wide graph or broad framework context unless concrete uncertainty or failure requires escalation.
|
|
16
|
+
- **STANDARD** — moderate uncertainty, several related files or a behavior change. Make a short file-aware plan, inspect affected callers/tests, and load only the process/domain skills that materially help.
|
|
17
|
+
- **DEEP** — high-risk, public-contract, auth/security/payment/schema/migration, cross-module or long-horizon work. Use impact analysis, planning, durable state and independent verification as required by policy.
|
|
25
18
|
|
|
26
|
-
|
|
19
|
+
Risk overrides convenience. A short prompt can still require DEEP handling when the blast radius is high.
|
|
27
20
|
|
|
28
|
-
|
|
21
|
+
## Evidence-first work
|
|
29
22
|
|
|
30
|
-
|
|
31
|
-
- `ocskill impact <symbol-or-term> [dir]` — bounded path/content impact search
|
|
32
|
-
- `ocskill evidence [dir]` — stack + verification + Git evidence snapshot
|
|
33
|
-
- `ocskill working-tree [dir]` — branch, HEAD and uncommitted-change state
|
|
34
|
-
- `ocskill repo-graph [dir]` — bounded source import graph and coupling hotspots
|
|
35
|
-
- `ocskill review-scope [base] [dir]` — deterministic changed-file coverage and risk hints
|
|
36
|
-
- `ocskill verification-plan [dir]` — project-native verification recommendations
|
|
37
|
-
- `ocskill task-graph <PLAN.json>` — validate dependencies and compute safe execution waves
|
|
38
|
-
- `ocskill context-pack <slug> <task> [dir]` — bounded durable handoff enriched with declared files, import neighbors, likely tests, instruction/manifests and accepted lessons
|
|
39
|
-
- `ocskill work verify-command ... -- <command>` — structured verification receipt (exit code, hashes, timing, workspace fingerprints)
|
|
40
|
-
- `ocskill sandbox create|list|remove ...` — isolated Git worktree primitives for parallel write tasks
|
|
41
|
-
- `ocskill learn status|analyze|accept ...` — evidence-gated learning loop; proposals never auto-edit skills
|
|
42
|
-
- `ocskill dashboard [dir] --serve` — local Control Center for work state, evidence, learning and eval summaries
|
|
23
|
+
Never invent repository structure, files, functions, APIs, schemas, package versions, runtime behavior or test results when they can be checked.
|
|
43
24
|
|
|
44
|
-
|
|
25
|
+
Read narrowly in this order when practical:
|
|
45
26
|
|
|
46
|
-
|
|
27
|
+
1. applicable repository instructions/manifests;
|
|
28
|
+
2. the named target or failure location;
|
|
29
|
+
3. nearest working analogue and direct callers/dependencies;
|
|
30
|
+
4. tests that encode the requested behavior;
|
|
31
|
+
5. broader graph/repository evidence only if uncertainty remains.
|
|
47
32
|
|
|
48
|
-
|
|
33
|
+
Useful deterministic helpers include:
|
|
49
34
|
|
|
50
|
-
-
|
|
51
|
-
-
|
|
52
|
-
-
|
|
53
|
-
-
|
|
35
|
+
- `ocskill inspect [dir]`
|
|
36
|
+
- `ocskill impact <symbol-or-term> [dir]`
|
|
37
|
+
- `ocskill aci search|refs|view|text ...`
|
|
38
|
+
- `ocskill working-tree [dir]`
|
|
39
|
+
- `ocskill verification-plan [dir]`
|
|
40
|
+
- `ocskill context-pack <slug> <task> [dir]`
|
|
41
|
+
- `ocskill work verify-command ... -- <command>`
|
|
54
42
|
|
|
55
|
-
|
|
43
|
+
Treat search/routing results as evidence hints, not semantic proof.
|
|
56
44
|
|
|
57
|
-
##
|
|
45
|
+
## Exact-contract discipline
|
|
58
46
|
|
|
59
|
-
|
|
47
|
+
For every edit:
|
|
60
48
|
|
|
61
|
-
-
|
|
62
|
-
-
|
|
63
|
-
-
|
|
64
|
-
-
|
|
65
|
-
-
|
|
66
|
-
-
|
|
67
|
-
- bug, crash, failed build/test, regression -> `ues-bug-diagnosis`
|
|
68
|
-
- meaningful edits -> `ues-test-verification`
|
|
69
|
-
- completed substantial change -> `ues-code-review`
|
|
70
|
-
- long task that must survive interruption -> `ues-long-task-state`
|
|
49
|
+
- preserve the user's observable acceptance criteria literally;
|
|
50
|
+
- preserve requested exception classes, type/range distinctions, return shapes, field names/order, mutation rules, idempotency and boundary behavior;
|
|
51
|
+
- preserve unrelated user changes;
|
|
52
|
+
- follow the repository's package manager, formatter, test/build conventions and generated-file policy;
|
|
53
|
+
- make the smallest coherent change; avoid opportunistic refactors and unrelated dependency upgrades;
|
|
54
|
+
- for public contracts, persistence, auth, payments, migrations or deployment, inspect downstream compatibility and rollback impact.
|
|
71
55
|
|
|
72
|
-
|
|
56
|
+
When tests are absent or hidden, use focused runtime probes for each stated criterion, especially boundary and mutation cases.
|
|
73
57
|
|
|
58
|
+
## Selective skill loading
|
|
59
|
+
|
|
60
|
+
Skills are on-demand context, not a checklist.
|
|
61
|
+
|
|
62
|
+
FAST should prefer the direct debugging/domain/verification skill and avoid generic orchestration unless needed. STANDARD may add `ues-engineering-orchestrator` plus a small number of directly relevant skills. DEEP may use planner, change-impact, long-task and critic/reviewer roles.
|
|
63
|
+
|
|
64
|
+
Typical direct routing:
|
|
65
|
+
|
|
66
|
+
- bug/crash/test failure -> `ues-bug-diagnosis`
|
|
67
|
+
- current external API/version/package -> `ues-research-verification`
|
|
74
68
|
- API contract -> `ues-api-contract`
|
|
75
69
|
- database/schema -> `ues-database-engineering`
|
|
76
70
|
- auth/permissions -> `ues-auth-security`
|
|
77
|
-
-
|
|
78
|
-
- Next.js -> `ues-nextjs-engineering`
|
|
71
|
+
- payment/webhook -> `ues-payment-engineering`
|
|
79
72
|
- React Native -> `ues-react-native-engineering`
|
|
73
|
+
- Next.js -> `ues-nextjs-engineering`
|
|
74
|
+
- React -> `ues-react-engineering`
|
|
80
75
|
- Node/Nest -> `ues-nodejs-engineering` / `ues-nestjs-engineering`
|
|
76
|
+
- Python/Django/FastAPI -> corresponding UES domain skill
|
|
81
77
|
- .NET -> `ues-dotnet-engineering`
|
|
82
78
|
- Java/Spring -> `ues-java-spring-engineering`
|
|
83
|
-
- Python/Django/FastAPI -> `ues-python-engineering` / `ues-django-engineering` / `ues-fastapi-engineering`
|
|
84
79
|
- Flutter -> `ues-flutter-engineering`
|
|
85
|
-
- UI/UX -> `ues-ui-ux-engineering`
|
|
86
|
-
- ecommerce/marketplace -> `ues-ecommerce-engineering`
|
|
87
|
-
- payment -> `ues-payment-engineering`
|
|
88
80
|
- Docker/CI/deploy -> `ues-devops-engineering`
|
|
89
|
-
- Git -> `ues-git-safety`
|
|
90
|
-
|
|
91
|
-
## Evidence and research
|
|
92
|
-
|
|
93
|
-
- Never invent files, functions, endpoints, schemas, commands, package names, package versions, framework behavior, or project structure when they can be checked.
|
|
94
|
-
- Prefer repository evidence for repository facts.
|
|
95
|
-
- For external APIs, libraries, versions, security guidance, or behavior that may have changed, use `ues-research-verification` and prefer primary/current sources.
|
|
96
|
-
- Distinguish observed facts, sourced facts, hypotheses, and recommendations.
|
|
97
|
-
- If a tool/source is unavailable, say what could not be verified instead of filling the gap with confidence.
|
|
98
|
-
|
|
99
|
-
## Debugging and retry discipline
|
|
100
|
-
|
|
101
|
-
- Reproduce or capture the exact failure before proposing a fix.
|
|
102
|
-
- Trace the bad value/state backward to the earliest supported cause.
|
|
103
|
-
- Change one causal variable at a time.
|
|
104
|
-
- If two attempted fixes fail, stop stacking patches and re-investigate from fresh evidence.
|
|
105
|
-
- If three distinct root-cause hypotheses fail or fixes expose widening coupling, question the architecture and surface that to the user before another broad change.
|
|
106
|
-
- Do not clear caches, delete lockfiles, disable checks, or upgrade dependencies as generic debugging rituals.
|
|
107
|
-
|
|
108
|
-
## Context discipline
|
|
109
81
|
|
|
110
|
-
|
|
111
|
-
- Prefer exact symbol/error searches over broad directory dumps.
|
|
112
|
-
- Summarize what is known before expanding the search.
|
|
113
|
-
- Use supporting files inside skills only when their section is needed.
|
|
114
|
-
- Do not repeatedly reread unchanged large files unless new evidence requires it.
|
|
82
|
+
Do not load the full catalog.
|
|
115
83
|
|
|
116
|
-
##
|
|
84
|
+
## Failure and weak-model recovery
|
|
117
85
|
|
|
118
|
-
|
|
86
|
+
A failed attempt is a signal to improve evidence, not to repeat the same prompt with more prose.
|
|
119
87
|
|
|
120
|
-
-
|
|
121
|
-
-
|
|
122
|
-
-
|
|
123
|
-
- architecture/implementation decisions and material alternatives
|
|
124
|
-
- acceptance-criteria status, changed files, fresh verification, unresolved risks, and one next action
|
|
88
|
+
- **Attempt 1:** use the normal FAST/STANDARD/DEEP context budget.
|
|
89
|
+
- **Attempt 2:** capture the exact failure, load failure-adjacent caller/test evidence, add debugging context when useful, and allow the configured model tier to escalate. Do not stack a speculative patch.
|
|
90
|
+
- **Attempt 3+:** re-investigate from fresh evidence, expand to callers/dependencies/contracts and repository graph, challenge architecture/coupling, and use critic/reviewer verification before accepting another repair.
|
|
125
91
|
|
|
126
|
-
|
|
92
|
+
If two fixes fail, stop patch stacking and re-diagnose. If three distinct root-cause hypotheses fail or coupling keeps widening, surface the architectural issue before another broad change.
|
|
127
93
|
|
|
128
|
-
|
|
94
|
+
A stronger model is not a substitute for missing evidence.
|
|
129
95
|
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
1. Map the relevant repository surface with deterministic evidence and `ues-codebase-mapper` when useful.
|
|
133
|
-
2. Persist observable requirements in `.ues-work/<slug>/SPEC.md`.
|
|
134
|
-
3. Create a machine-checkable `PLAN.json` and validate it with `ocskill task-graph`.
|
|
135
|
-
4. Ask `ues-plan-checker` to challenge the plan before edits begin. Record PASS with `ocskill work approve-plan`; `work start` is blocked until this happens.
|
|
136
|
-
5. Execute each approved task in a fresh `ues-executor` context. Active V7 tasks carry a runId, heartbeat and lease expiry so interrupted work can be recovered deterministically. On OpenCode V2 prefer `ues.dispatch_task`, which creates the fresh session and applies configured attempt-based model escalation.
|
|
137
|
-
6. Inspect each child diff and prefer receipt-backed verification using `ocskill work verify-command` before marking completion with `ocskill work complete`; record failures with `ocskill work fail`.
|
|
138
|
-
7. Parallelize only dependency-safe tasks with no write/read conflict. For concurrent writers, use isolated worktrees/sandboxes and an explicit integration step instead of sharing one working tree. UES serializes durable state writes but cannot make conflicting source edits safe.
|
|
139
|
-
8. On resume, trust durable state plus current Git evidence over conversational memory. Recover expired executor leases before retrying; preserve runId fences for active attempts.
|
|
140
|
-
9. After all tasks complete, run `ues-integration-verifier` against cross-task contracts and end-to-end acceptance criteria, then persist its actual verdict with `ocskill work verify-integration`.
|
|
141
|
-
10. `work finalize` requires a recorded integration PASS and rejects completion if the Git workspace changed after that PASS.
|
|
142
|
-
11. Merge/push/publish/deploy remain external side effects and require explicit user intent.
|
|
143
|
-
|
|
144
|
-
Use `ocskill model-policy <role> --attempt N` when configured model tiers exist. Escalate only after diagnosis/fresh context; never use a stronger model as a substitute for missing evidence.
|
|
145
|
-
|
|
146
|
-
## Critic and repair discipline
|
|
147
|
-
|
|
148
|
-
For substantial or high-risk behavior changes, verification is followed by an independent falsification pass:
|
|
149
|
-
|
|
150
|
-
1. self-check the diff against observable acceptance criteria
|
|
151
|
-
2. run fresh behavior-matched verification
|
|
152
|
-
3. ask `ues-critic` or `ues-reviewer` to challenge assumptions and search for concrete counterexamples
|
|
153
|
-
4. repair only evidence-backed blocking findings
|
|
154
|
-
5. rerun affected verification
|
|
155
|
-
6. repeat the critic pass only when the repair materially changed risky behavior
|
|
156
|
-
|
|
157
|
-
Bound this loop to at most two repair cycles before returning to root-cause/architecture analysis. Do not churn code to satisfy speculative feedback. Unresolved blocking findings must be fixed or surfaced explicitly.
|
|
96
|
+
## Verification gate
|
|
158
97
|
|
|
159
|
-
|
|
98
|
+
Before claiming completion:
|
|
160
99
|
|
|
161
|
-
|
|
100
|
+
1. identify what observable evidence proves the requested behavior;
|
|
101
|
+
2. run the narrowest relevant check, then expand based on risk and repository conventions;
|
|
102
|
+
3. read the actual output and exit status;
|
|
103
|
+
4. re-test the original failure/acceptance criterion;
|
|
104
|
+
5. inspect the final diff for accidental changes;
|
|
105
|
+
6. for substantial/high-risk work, run independent review/critic verification and resolve evidence-backed blockers;
|
|
106
|
+
7. report exactly what passed, failed or was not run.
|
|
162
107
|
|
|
163
|
-
|
|
164
|
-
- `ues-architect` — read-only architecture/change-impact analysis
|
|
165
|
-
- `ues-plan-checker` — read-only independent plan gate
|
|
166
|
-
- `ues-executor` — fresh-context implementation of exactly one approved task
|
|
167
|
-
- `ues-debugger` — read-only root-cause analysis
|
|
168
|
-
- `ues-researcher` — read-only current-source research
|
|
169
|
-
- `ues-reviewer` — read-only final/diff review
|
|
170
|
-
- `ues-critic` — read-only adversarial falsification
|
|
171
|
-
- `ues-verifier` — read-only task/acceptance verification
|
|
172
|
-
- `ues-integration-verifier` — read-only cross-task/end-to-end verification
|
|
108
|
+
Never claim a test, build, migration, deployment, push or release succeeded unless it actually did.
|
|
173
109
|
|
|
174
|
-
|
|
110
|
+
## Long-horizon work
|
|
175
111
|
|
|
176
|
-
|
|
112
|
+
For interruption-prone or dependent multi-task work, use durable `.ues-work/<slug>/` state instead of relying on conversation memory.
|
|
177
113
|
|
|
178
|
-
|
|
179
|
-
- Follow the repository's package manager, formatter, linter, tests, build scripts, architecture, and generated-file policy.
|
|
180
|
-
- Preserve unrelated user changes.
|
|
181
|
-
- Avoid opportunistic refactors and broad dependency upgrades during unrelated fixes.
|
|
182
|
-
- For behavior changes where a practical test harness exists, prefer a failing regression/behavior test before implementation.
|
|
183
|
-
- For public contracts, persistence, auth, payments, migrations, and deployment, explicitly inspect downstream consumers and rollback/compatibility impact.
|
|
114
|
+
The required sequence is:
|
|
184
115
|
|
|
185
|
-
|
|
116
|
+
`SPEC -> PLAN -> plan check/receipt -> approved tasks -> fresh executor per task -> task verification receipts -> integration verification/receipt -> finalize`
|
|
186
117
|
|
|
187
|
-
|
|
118
|
+
Use dependency-safe waves and isolated worktrees only when their write/read scopes are safe. On resume, trust durable state plus current Git evidence over conversational memory. Long/high-risk completion must remain bound to the active run and current workspace fingerprint.
|
|
188
119
|
|
|
189
|
-
|
|
190
|
-
2. Run the narrowest relevant checks, then expand based on risk and project conventions.
|
|
191
|
-
3. Read the actual output and exit status.
|
|
192
|
-
4. Re-test the original failure/acceptance criterion, not only compilation.
|
|
193
|
-
5. Inspect the final diff for accidental changes and regressions.
|
|
194
|
-
6. Run or request an independent review/critic pass for substantial or high-risk work and resolve evidence-backed blocking findings.
|
|
195
|
-
7. Report exactly what passed, failed, was repaired, or was not run.
|
|
120
|
+
Do not store hidden chain-of-thought. Persist observable facts, decisions, acceptance status, evidence and next actions only.
|
|
196
121
|
|
|
197
|
-
|
|
122
|
+
## Safety
|
|
198
123
|
|
|
199
|
-
|
|
124
|
+
Ask before destructive or irreversible actions such as force-pushing, destructive reset/clean, deleting important data, dropping database objects, broad production migrations, production deployment or credential rotation. Never print secrets.
|
|
200
125
|
|
|
201
|
-
|
|
126
|
+
Merge, push, publish and deploy are external side effects and require explicit user intent.
|
|
202
127
|
|
|
203
128
|
## Completion standard
|
|
204
129
|
|
|
205
|
-
A task is complete only when requested behavior is implemented, acceptance criteria are addressed, relevant verification
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
## V7 learning and optional external executors
|
|
209
|
-
|
|
210
|
-
UES may analyze its own `.ues-evals` traces with `ocskill learn analyze`. The output is a proposal set, not an automatic self-modification. A human/parent explicitly accepts a proposal before it can appear in future task context. This keeps learning evidence-gated and reversible.
|
|
211
|
-
|
|
212
|
-
Hermes support is optional and adapter-style. `ocskill hermes status` checks availability and `ocskill hermes prompt <slug> <task> .` emits a bounded delegation prompt. Hermes is not embedded into the UES runtime and may not mutate UES durable state on its own.
|
|
130
|
+
A task is complete only when the requested behavior is implemented, the acceptance criteria are addressed, fresh relevant verification supports the result, the final diff is reviewed, and remaining limitations are stated accurately.
|
|
@@ -40,4 +40,4 @@ Only concrete issues that prevent completion.
|
|
|
40
40
|
## Completion evidence
|
|
41
41
|
What the parent may truthfully claim after this verification.
|
|
42
42
|
|
|
43
|
-
|
|
43
|
+
For a long/high-risk PASS, the parent must create an `integration-verification` receipt bound to the current workspace fingerprint and pass it to `ocskill work verify-integration --receipt-file <file>`. Finalization remains blocked without PASS and is invalidated by later workspace changes.
|
|
@@ -40,4 +40,4 @@ Persistence, auth, payment, public API, deployment, destructive or migration con
|
|
|
40
40
|
## Required revisions
|
|
41
41
|
Only blocking changes required before execution.
|
|
42
42
|
|
|
43
|
-
A PASS means the plan is executable, not that implementation is correct. The parent must
|
|
43
|
+
A PASS means the plan is executable, not that implementation is correct. The parent must bind a genuine PASS to the exact plan hash with `ocskill work gate-receipt <slug> plan ...`, then call `ocskill work approve-plan ... --receipt-file <file>`. Do not approve a plan that you returned as REVISE.
|
|
@@ -13,17 +13,21 @@ Required workflow:
|
|
|
13
13
|
3. Create a concise SPEC with observable acceptance criteria.
|
|
14
14
|
4. Initialize persistent state with `ocskill work init`.
|
|
15
15
|
5. Produce a file-aware `PLAN.json` using the UES plan schema, then import it with `ocskill work plan`.
|
|
16
|
-
6. Dispatch `ues-plan-checker` in fresh context.
|
|
16
|
+
6. Dispatch `ues-plan-checker` in fresh context. For long/high-risk work, bind the PASS to the current plan with a structured receipt, then approve it:
|
|
17
|
+
`ocskill work gate-receipt <slug> plan . --verifier ues-plan-checker --evidence "<summary>" --out .ues-work/<slug>/reports/plan-receipt.json`
|
|
18
|
+
followed by `ocskill work approve-plan <slug> . --evidence "<summary>" --receipt-file .ues-work/<slug>/reports/plan-receipt.json`.
|
|
17
19
|
7. Use `ocskill task-graph` and execute only ready dependency-safe tasks. On OpenCode V2 prefer `ues.dispatch_task`: it starts the task, creates a fresh `ues-executor` session, applies configured attempt-based model escalation, waits for that executor, and returns its report. Inspect the diff and evidence, then record `ocskill work complete` or `ocskill work fail`.
|
|
18
|
-
8. Independent tasks may run concurrently only when safe-wave analysis reports no write/read conflict.
|
|
20
|
+
8. Independent tasks may run concurrently only when safe-wave analysis reports no write/read conflict. V8 can isolate concurrent writers in Git worktrees and integrate them with conflict detection; manual fallback is `ocskill sandbox create ...` followed by `ocskill sandbox integrate <worktree> .`. Never integrate over overlapping dirty root files.
|
|
19
21
|
9. During long execution keep the task lease alive with `ocskill work heartbeat` (the V2 dispatcher does this automatically). On resume, `ocskill work recover` or `ocskill work resume` recovers expired leases instead of leaving tasks stuck in `running`.
|
|
20
|
-
10. Run declared checks through `ocskill work verify-command <slug> <task> . -- <command> [args...]
|
|
22
|
+
10. Run declared checks through `ocskill work verify-command <slug> <task> . --run-id <run-id> -- <command> [args...]`. For long/high-risk plans, a successful receipt for the active run is mandatory before `work complete`; narrative-only completion is rejected.
|
|
21
23
|
11. On executor failure, diagnose from fresh evidence and retry in a fresh executor. Adaptive model policy may raise the model tier based on risk/complexity plus attempt count; do not escalate blindly.
|
|
22
|
-
12. After all tasks complete, dispatch `ues-integration-verifier`.
|
|
24
|
+
12. After all tasks complete, dispatch `ues-integration-verifier`. For PASS on long/high-risk work, create an integration receipt bound to the current workspace fingerprint:
|
|
25
|
+
`ocskill work gate-receipt <slug> integration . --verifier ues-integration-verifier --verdict PASS --evidence "<summary>" --out .ues-work/<slug>/reports/integration-receipt.json`
|
|
26
|
+
then record it with `ocskill work verify-integration <slug> . --verdict PASS --evidence "<summary>" --receipt-file .ues-work/<slug>/reports/integration-receipt.json`.
|
|
23
27
|
13. `ocskill work finalize` is allowed only after a recorded PASS and only if the workspace fingerprint has not changed since that PASS. Re-run verification if it changed.
|
|
24
28
|
14. Inspect final diff/status and report only evidence-backed completion.
|
|
25
29
|
|
|
26
30
|
Do not merge, push, publish, deploy or perform destructive operations without explicit user approval.
|
|
27
31
|
|
|
28
32
|
|
|
29
|
-
After a meaningful eval run, `ocskill learn analyze . --eval-dir .ues-evals`
|
|
33
|
+
After a meaningful eval run, `ocskill learn analyze . --eval-dir .ues-evals` clusters recurring failures into candidate lessons. `ocskill learn accept <id> .` stages a proposal, but shadow-required lessons enter future context only after `ocskill learn promote <id> . --baseline <rate> --candidate <rate> --samples N` proves an improvement.
|
|
@@ -1,18 +1,22 @@
|
|
|
1
1
|
export function runtimeCapabilities(ctx) {
|
|
2
2
|
const session = ctx?.session || {}
|
|
3
|
+
const permission = ctx?.permission || {}
|
|
3
4
|
const capabilities = {
|
|
4
5
|
sessionCreate: typeof session.create === "function",
|
|
5
6
|
sessionPrompt: typeof session.prompt === "function",
|
|
6
7
|
sessionWait: typeof session.wait === "function",
|
|
8
|
+
sessionInterrupt: typeof session.interrupt === "function",
|
|
7
9
|
sessionContext: typeof session.context === "function",
|
|
8
10
|
sessionSwitchAgent: typeof session.switchAgent === "function",
|
|
9
11
|
sessionSwitchModel: typeof session.switchModel === "function",
|
|
10
12
|
sessionHook: typeof session.hook === "function",
|
|
13
|
+
permissionHook: typeof permission.hook === "function",
|
|
11
14
|
}
|
|
12
15
|
capabilities.freshDispatch =
|
|
13
16
|
capabilities.sessionCreate &&
|
|
14
17
|
capabilities.sessionPrompt &&
|
|
15
18
|
capabilities.sessionWait &&
|
|
19
|
+
capabilities.sessionInterrupt &&
|
|
16
20
|
capabilities.sessionContext &&
|
|
17
21
|
capabilities.sessionSwitchAgent
|
|
18
22
|
capabilities.modelSwitch = capabilities.sessionSwitchModel
|