opencode-agent-skill 7.7.0 → 10.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/CHANGELOG.md +112 -3
  2. package/README.md +396 -281
  3. package/bin/ocskill.mjs +382 -156
  4. package/docs/DETERMINISTIC-TOOLS.md +25 -8
  5. package/docs/ENGINEERING-DESIGN.md +31 -13
  6. package/docs/EVALS.md +34 -12
  7. package/docs/NPM-PUBLISH.md +6 -6
  8. package/docs/OPENCODE-COMPAT.md +11 -8
  9. package/docs/TRACE-SCHEMA.md +15 -2
  10. package/docs/V8-INTELLIGENCE-RELIABILITY.md +206 -0
  11. package/docs/V9-SPEED-INTELLIGENCE.md +102 -0
  12. package/evals/live/tasks.json +6 -6
  13. package/evals/polyglot/fixtures/polyglot-bench/api/generated/client.ts +2 -0
  14. package/evals/polyglot/fixtures/polyglot-bench/api/openapi.json +25 -0
  15. package/evals/polyglot/fixtures/polyglot-bench/db/migrations/20260920_add_order_key.sql +1 -0
  16. package/evals/polyglot/fixtures/polyglot-bench/dotnet/OrderService.cs +8 -0
  17. package/evals/polyglot/fixtures/polyglot-bench/java/PriceService.java +5 -0
  18. package/evals/polyglot/fixtures/polyglot-bench/monorepo/package.json +6 -0
  19. package/evals/polyglot/fixtures/polyglot-bench/monorepo/packages/api/package.json +4 -0
  20. package/evals/polyglot/fixtures/polyglot-bench/monorepo/packages/web/package.json +7 -0
  21. package/evals/polyglot/fixtures/polyglot-bench/monorepo/pnpm-lock.yaml +5 -0
  22. package/evals/polyglot/fixtures/polyglot-bench/next/app/api/products/route.ts +7 -0
  23. package/evals/polyglot/fixtures/polyglot-bench/python/tenant_auth.py +4 -0
  24. package/evals/polyglot/fixtures/polyglot-bench/react-native/keyboard.ts +3 -0
  25. package/evals/polyglot/graders/polyglot-bench.mjs +101 -0
  26. package/evals/polyglot/tasks.json +54 -0
  27. package/global-config/AGENTS.md +78 -160
  28. package/global-config/agents/integration-verifier.md +1 -1
  29. package/global-config/agents/plan-checker.md +1 -1
  30. package/global-config/commands/run.md +9 -5
  31. package/global-config/plugins/ues-router/capabilities.js +4 -0
  32. package/global-config/plugins/ues-router/index.js +784 -37
  33. package/global-config/plugins/ues-router/router.js +175 -23
  34. package/global-config/plugins/ues-router/runtime-guard.js +265 -0
  35. package/global-config/skills/engineering-orchestrator/references/long-horizon.md +6 -4
  36. package/lib/aci.mjs +128 -0
  37. package/lib/benchmark-confidence.mjs +173 -0
  38. package/lib/cli-utils.mjs +41 -0
  39. package/lib/container-sandbox.mjs +102 -0
  40. package/lib/context-manifest.mjs +300 -22
  41. package/lib/control-center.mjs +36 -3
  42. package/lib/eval-ablation.mjs +104 -0
  43. package/lib/eval-order.mjs +9 -0
  44. package/lib/eval-report.mjs +11 -0
  45. package/lib/eval-telemetry.mjs +8 -2
  46. package/lib/gate-receipt.mjs +52 -0
  47. package/lib/installer.mjs +39 -25
  48. package/lib/learning-engine.mjs +236 -38
  49. package/lib/model-policy.mjs +6 -0
  50. package/lib/opencode-compat.mjs +25 -10
  51. package/lib/orchestrator-policy.mjs +195 -21
  52. package/lib/process-runner.mjs +30 -9
  53. package/lib/runtime-events.mjs +31 -0
  54. package/lib/semantic-index.mjs +318 -0
  55. package/lib/task-engine.mjs +557 -28
  56. package/lib/trajectory.mjs +89 -0
  57. package/lib/windows-shim.mjs +227 -0
  58. package/lib/worktree-sandbox.mjs +85 -3
  59. package/package.json +11 -4
  60. package/scripts/check-release-tag.mjs +22 -0
  61. package/scripts/control-center.mjs +25 -0
  62. package/scripts/eval-ablation.mjs +44 -0
  63. package/scripts/eval-live.mjs +39 -58
  64. package/scripts/eval-matrix.mjs +166 -0
  65. package/scripts/smoke-packed-install.mjs +138 -4
  66. package/scripts/smoke-plain-install.mjs +91 -0
  67. package/scripts/validate-live-suite.mjs +3 -3
  68. package/scripts/validate.mjs +27 -5
@@ -1,212 +1,130 @@
1
1
  # Universal Engineering System
2
2
 
3
- These instructions apply to software-engineering work in OpenCode when this package is installed.
3
+ These instructions apply to software-engineering work in OpenCode when UES is installed.
4
4
 
5
- ## Operating model
5
+ ## Core rule
6
6
 
7
- Treat engineering as an evidence-driven loop:
7
+ Use the minimum context and orchestration that preserve correctness. Do not trade acceptance criteria, repository evidence, verification, or safety for lower token use.
8
8
 
9
- ```text
10
- understand -> route -> plan when needed -> implement -> verify -> review -> finish
11
- ^ |
12
- +---- diagnose <-----+
13
- ```
9
+ The model remains the model. UES improves task routing, evidence selection, verification and recovery; it does not replace model capability.
14
10
 
15
- The selected model remains the model. These instructions improve process, context selection, verification, and recovery; they do not replace model capability.
11
+ ## Start by classifying the task
16
12
 
17
- ## First actions
13
+ When `ocskill` is available, use `ocskill task-policy <text>` as the deterministic starting point.
18
14
 
19
- 1. When `ocskill` is available for non-trivial work, classify the request with `ocskill task-policy <text>` and use its mode/risk/model/context guidance as a deterministic starting point.
20
- 2. Read the repository's applicable `AGENTS.md`, manifests, package-manager files, and nearby conventions before editing.
21
- 3. Establish the actual request, acceptance criteria, constraints, and current behavior from evidence.
22
- 4. For non-trivial work, load `ues-engineering-orchestrator` first. Then load only the process and domain skills that materially help.
23
- 5. Prefer process skills before framework skills: exploration/planning/debugging/verification determine how to work; domain skills determine what framework-specific details to apply.
24
- 6. Keep the active skill set focused. Usually 2-4 skills are enough; do not load the entire catalog.
15
+ - **FAST** — focused, low-risk work with a clear target. Read the target, nearest relevant test/analogue and only direct dependencies needed to prove the change. Prefer at most two directly useful skills. Do not load the engineering orchestrator, planner, critic, repo-wide graph or broad framework context unless concrete uncertainty or failure requires escalation.
16
+ - **STANDARD** — moderate uncertainty, several related files or a behavior change. Make a short file-aware plan, inspect affected callers/tests, and load only the process/domain skills that materially help.
17
+ - **DEEP** — high-risk, public-contract, auth/security/payment/schema/migration, cross-module or long-horizon work. Use impact analysis, planning, durable state and independent verification as required by policy.
25
18
 
26
- ## Deterministic evidence helpers
19
+ Risk overrides convenience. A short prompt can still require DEEP handling when the blast radius is high.
27
20
 
28
- When the `ocskill` CLI is available, prefer deterministic repository evidence before spending model context on broad exploration:
21
+ ## Evidence-first work
29
22
 
30
- - `ocskill inspect [dir]` — stack, package manager, top-level map and project-native verification commands
31
- - `ocskill impact <symbol-or-term> [dir]` — bounded path/content impact search
32
- - `ocskill evidence [dir]` — stack + verification + Git evidence snapshot
33
- - `ocskill working-tree [dir]` — branch, HEAD and uncommitted-change state
34
- - `ocskill repo-graph [dir]` — bounded source import graph and coupling hotspots
35
- - `ocskill review-scope [base] [dir]` — deterministic changed-file coverage and risk hints
36
- - `ocskill verification-plan [dir]` — project-native verification recommendations
37
- - `ocskill task-graph <PLAN.json>` — validate dependencies and compute safe execution waves
38
- - `ocskill context-pack <slug> <task> [dir]` — bounded durable handoff enriched with declared files, import neighbors, likely tests, instruction/manifests and accepted lessons
39
- - `ocskill work verify-command ... -- <command>` — structured verification receipt (exit code, hashes, timing, workspace fingerprints)
40
- - `ocskill sandbox create|list|remove ...` — isolated Git worktree primitives for parallel write tasks
41
- - `ocskill learn status|analyze|accept ...` — evidence-gated learning loop; proposals never auto-edit skills
42
- - `ocskill dashboard [dir] --serve` — local Control Center for work state, evidence, learning and eval summaries
23
+ Never invent repository structure, files, functions, APIs, schemas, package versions, runtime behavior or test results when they can be checked.
43
24
 
44
- These helpers are evidence accelerators, not substitutes for reading the exact affected code. Use repository-native search/tools when they provide more precise symbol/call-graph information.
25
+ Read narrowly in this order when practical:
45
26
 
46
- On OpenCode v2, UES may install a managed runtime router that preselects at most a small focused set of relevant skills from the incoming prompt. The V2 plugin also upgrades destructive/high-impact shell actions such as forceful Git history operations, publishing, infrastructure destruction, or destructive SQL to an explicit permission prompt. Treat router selections as hints: keep useful skills, load deeper references only when needed, and do not assume a routed skill proves anything about the repository.
27
+ 1. applicable repository instructions/manifests;
28
+ 2. the named target or failure location;
29
+ 3. nearest working analogue and direct callers/dependencies;
30
+ 4. tests that encode the requested behavior;
31
+ 5. broader graph/repository evidence only if uncertainty remains.
47
32
 
48
- ## Scope classification
33
+ Useful deterministic helpers include:
49
34
 
50
- - **Small:** one local area, low risk, obvious verification. Work inline; no ceremonial plan.
51
- - **Standard:** behavior change, 2-5 related files, or moderate uncertainty. Make a short file-aware plan and identify verification before editing.
52
- - **Complex:** cross-module/public API/schema/auth/security/migration/dependency-major changes, more than about five files, or high rollback risk. Use `ues-task-planner`, `ues-change-impact-analysis`, and architecture/research skills as appropriate before implementation.
53
- - **Long-horizon:** many dependent work units, interruption/compaction risk, or work expected to outlive one context. When the user explicitly selects the long workflow (for example `/ues-run`), use durable `.ues-work/<slug>/` state, an independent plan gate, fresh task executors, dependency-safe waves, and final integration verification.
35
+ - `ocskill inspect [dir]`
36
+ - `ocskill impact <symbol-or-term> [dir]`
37
+ - `ocskill aci search|refs|view|text ...`
38
+ - `ocskill working-tree [dir]`
39
+ - `ocskill verification-plan [dir]`
40
+ - `ocskill context-pack <slug> <task> [dir]`
41
+ - `ocskill work verify-command ... -- <command>`
54
42
 
55
- These are routing heuristics, not quotas. Risk matters more than file count.
43
+ Treat search/routing results as evidence hints, not semantic proof.
56
44
 
57
- ## Automatic skill routing
45
+ ## Exact-contract discipline
58
46
 
59
- Common process routing:
47
+ For every edit:
60
48
 
61
- - unfamiliar or large repository -> `ues-repo-explorer` + optionally `ues-context-engineering`
62
- - non-trivial multi-step work -> `ues-engineering-orchestrator`
63
- - multi-file/risky change -> `ues-task-planner`
64
- - cross-boundary contract or blast-radius question -> `ues-change-impact-analysis`
65
- - current or uncertain external API/version/package -> `ues-research-verification`
66
- - feature/bugfix with a practical test harness -> `ues-test-driven-development`
67
- - bug, crash, failed build/test, regression -> `ues-bug-diagnosis`
68
- - meaningful edits -> `ues-test-verification`
69
- - completed substantial change -> `ues-code-review`
70
- - long task that must survive interruption -> `ues-long-task-state`
49
+ - preserve the user's observable acceptance criteria literally;
50
+ - preserve requested exception classes, type/range distinctions, return shapes, field names/order, mutation rules, idempotency and boundary behavior;
51
+ - preserve unrelated user changes;
52
+ - follow the repository's package manager, formatter, test/build conventions and generated-file policy;
53
+ - make the smallest coherent change; avoid opportunistic refactors and unrelated dependency upgrades;
54
+ - for public contracts, persistence, auth, payments, migrations or deployment, inspect downstream compatibility and rollback impact.
71
55
 
72
- Domain routing remains specific:
56
+ When tests are absent or hidden, use focused runtime probes for each stated criterion, especially boundary and mutation cases.
73
57
 
58
+ ## Selective skill loading
59
+
60
+ Skills are on-demand context, not a checklist.
61
+
62
+ FAST should prefer the direct debugging/domain/verification skill and avoid generic orchestration unless needed. STANDARD may add `ues-engineering-orchestrator` plus a small number of directly relevant skills. DEEP may use planner, change-impact, long-task and critic/reviewer roles.
63
+
64
+ Typical direct routing:
65
+
66
+ - bug/crash/test failure -> `ues-bug-diagnosis`
67
+ - current external API/version/package -> `ues-research-verification`
74
68
  - API contract -> `ues-api-contract`
75
69
  - database/schema -> `ues-database-engineering`
76
70
  - auth/permissions -> `ues-auth-security`
77
- - React -> `ues-react-engineering`
78
- - Next.js -> `ues-nextjs-engineering`
71
+ - payment/webhook -> `ues-payment-engineering`
79
72
  - React Native -> `ues-react-native-engineering`
73
+ - Next.js -> `ues-nextjs-engineering`
74
+ - React -> `ues-react-engineering`
80
75
  - Node/Nest -> `ues-nodejs-engineering` / `ues-nestjs-engineering`
76
+ - Python/Django/FastAPI -> corresponding UES domain skill
81
77
  - .NET -> `ues-dotnet-engineering`
82
78
  - Java/Spring -> `ues-java-spring-engineering`
83
- - Python/Django/FastAPI -> `ues-python-engineering` / `ues-django-engineering` / `ues-fastapi-engineering`
84
79
  - Flutter -> `ues-flutter-engineering`
85
- - UI/UX -> `ues-ui-ux-engineering`
86
- - ecommerce/marketplace -> `ues-ecommerce-engineering`
87
- - payment -> `ues-payment-engineering`
88
80
  - Docker/CI/deploy -> `ues-devops-engineering`
89
- - Git -> `ues-git-safety`
90
-
91
- ## Evidence and research
92
-
93
- - Never invent files, functions, endpoints, schemas, commands, package names, package versions, framework behavior, or project structure when they can be checked.
94
- - Prefer repository evidence for repository facts.
95
- - For external APIs, libraries, versions, security guidance, or behavior that may have changed, use `ues-research-verification` and prefer primary/current sources.
96
- - Distinguish observed facts, sourced facts, hypotheses, and recommendations.
97
- - If a tool/source is unavailable, say what could not be verified instead of filling the gap with confidence.
98
-
99
- ## Debugging and retry discipline
100
-
101
- - Reproduce or capture the exact failure before proposing a fix.
102
- - Trace the bad value/state backward to the earliest supported cause.
103
- - Change one causal variable at a time.
104
- - If two attempted fixes fail, stop stacking patches and re-investigate from fresh evidence.
105
- - If three distinct root-cause hypotheses fail or fixes expose widening coupling, question the architecture and surface that to the user before another broad change.
106
- - Do not clear caches, delete lockfiles, disable checks, or upgrade dependencies as generic debugging rituals.
107
-
108
- ## Context discipline
109
81
 
110
- - Read narrowly: instructions/manifests -> relevant entry point -> nearest working analogue -> direct dependencies/callers -> tests.
111
- - Prefer exact symbol/error searches over broad directory dumps.
112
- - Summarize what is known before expanding the search.
113
- - Use supporting files inside skills only when their section is needed.
114
- - Do not repeatedly reread unchanged large files unless new evidence requires it.
82
+ Do not load the full catalog.
115
83
 
116
- ## Reasoning-state discipline
84
+ ## Failure and weak-model recovery
117
85
 
118
- For complex, ambiguous, or interruption-prone work, maintain a compact reasoning ledger rather than relying on conversational memory:
86
+ A failed attempt is a signal to improve evidence, not to repeat the same prompt with more prose.
119
87
 
120
- - confirmed facts with repository/runtime evidence
121
- - assumptions with confidence and a concrete way to verify them
122
- - rejected hypotheses with the evidence that disproved them
123
- - architecture/implementation decisions and material alternatives
124
- - acceptance-criteria status, changed files, fresh verification, unresolved risks, and one next action
88
+ - **Attempt 1:** use the normal FAST/STANDARD/DEEP context budget.
89
+ - **Attempt 2:** capture the exact failure, load failure-adjacent caller/test evidence, add debugging context when useful, and allow the configured model tier to escalate. Do not stack a speculative patch.
90
+ - **Attempt 3+:** re-investigate from fresh evidence, expand to callers/dependencies/contracts and repository graph, challenge architecture/coupling, and use critic/reviewer verification before accepting another repair.
125
91
 
126
- Do not store hidden chain-of-thought. Preserve actionable evidence and decisions. Use `ues-long-task-state` when this state must survive context compaction or another session.
92
+ If two fixes fail, stop patch stacking and re-diagnose. If three distinct root-cause hypotheses fail or coupling keeps widening, surface the architectural issue before another broad change.
127
93
 
128
- ## Long-horizon execution discipline
94
+ A stronger model is not a substitute for missing evidence.
129
95
 
130
- For explicit long-running/autonomous work, do not ask one context to remember the whole implementation.
131
-
132
- 1. Map the relevant repository surface with deterministic evidence and `ues-codebase-mapper` when useful.
133
- 2. Persist observable requirements in `.ues-work/<slug>/SPEC.md`.
134
- 3. Create a machine-checkable `PLAN.json` and validate it with `ocskill task-graph`.
135
- 4. Ask `ues-plan-checker` to challenge the plan before edits begin. Record PASS with `ocskill work approve-plan`; `work start` is blocked until this happens.
136
- 5. Execute each approved task in a fresh `ues-executor` context. Active V7 tasks carry a runId, heartbeat and lease expiry so interrupted work can be recovered deterministically. On OpenCode V2 prefer `ues.dispatch_task`, which creates the fresh session and applies configured attempt-based model escalation.
137
- 6. Inspect each child diff and prefer receipt-backed verification using `ocskill work verify-command` before marking completion with `ocskill work complete`; record failures with `ocskill work fail`.
138
- 7. Parallelize only dependency-safe tasks with no write/read conflict. For concurrent writers, use isolated worktrees/sandboxes and an explicit integration step instead of sharing one working tree. UES serializes durable state writes but cannot make conflicting source edits safe.
139
- 8. On resume, trust durable state plus current Git evidence over conversational memory. Recover expired executor leases before retrying; preserve runId fences for active attempts.
140
- 9. After all tasks complete, run `ues-integration-verifier` against cross-task contracts and end-to-end acceptance criteria, then persist its actual verdict with `ocskill work verify-integration`.
141
- 10. `work finalize` requires a recorded integration PASS and rejects completion if the Git workspace changed after that PASS.
142
- 11. Merge/push/publish/deploy remain external side effects and require explicit user intent.
143
-
144
- Use `ocskill model-policy <role> --attempt N` when configured model tiers exist. Escalate only after diagnosis/fresh context; never use a stronger model as a substitute for missing evidence.
145
-
146
- ## Critic and repair discipline
147
-
148
- For substantial or high-risk behavior changes, verification is followed by an independent falsification pass:
149
-
150
- 1. self-check the diff against observable acceptance criteria
151
- 2. run fresh behavior-matched verification
152
- 3. ask `ues-critic` or `ues-reviewer` to challenge assumptions and search for concrete counterexamples
153
- 4. repair only evidence-backed blocking findings
154
- 5. rerun affected verification
155
- 6. repeat the critic pass only when the repair materially changed risky behavior
156
-
157
- Bound this loop to at most two repair cycles before returning to root-cause/architecture analysis. Do not churn code to satisfy speculative feedback. Unresolved blocking findings must be fixed or surfaced explicitly.
96
+ ## Verification gate
158
97
 
159
- ## Subagent discipline
98
+ Before claiming completion:
160
99
 
161
- OpenCode may expose these installed subagents:
100
+ 1. identify what observable evidence proves the requested behavior;
101
+ 2. run the narrowest relevant check, then expand based on risk and repository conventions;
102
+ 3. read the actual output and exit status;
103
+ 4. re-test the original failure/acceptance criterion;
104
+ 5. inspect the final diff for accidental changes;
105
+ 6. for substantial/high-risk work, run independent review/critic verification and resolve evidence-backed blockers;
106
+ 7. report exactly what passed, failed or was not run.
162
107
 
163
- - `ues-codebase-mapper` — read-only mapping for large/unfamiliar repositories
164
- - `ues-architect` — read-only architecture/change-impact analysis
165
- - `ues-plan-checker` — read-only independent plan gate
166
- - `ues-executor` — fresh-context implementation of exactly one approved task
167
- - `ues-debugger` — read-only root-cause analysis
168
- - `ues-researcher` — read-only current-source research
169
- - `ues-reviewer` — read-only final/diff review
170
- - `ues-critic` — read-only adversarial falsification
171
- - `ues-verifier` — read-only task/acceptance verification
172
- - `ues-integration-verifier` — read-only cross-task/end-to-end verification
108
+ Never claim a test, build, migration, deployment, push or release succeeded unless it actually did.
173
109
 
174
- Use them selectively. Keep trivial work inline. The editable `ues-executor` must not launch child agents or broaden its task silently. Never allow concurrent executors to edit overlapping files in one working tree. Treat every subagent report as evidence to inspect, not authority. The parent remains responsible for orchestration, integration and final claims.
110
+ ## Long-horizon work
175
111
 
176
- ## Implementation discipline
112
+ For interruption-prone or dependent multi-task work, use durable `.ues-work/<slug>/` state instead of relying on conversation memory.
177
113
 
178
- - Make the smallest coherent change that satisfies the request.
179
- - Follow the repository's package manager, formatter, linter, tests, build scripts, architecture, and generated-file policy.
180
- - Preserve unrelated user changes.
181
- - Avoid opportunistic refactors and broad dependency upgrades during unrelated fixes.
182
- - For behavior changes where a practical test harness exists, prefer a failing regression/behavior test before implementation.
183
- - For public contracts, persistence, auth, payments, migrations, and deployment, explicitly inspect downstream consumers and rollback/compatibility impact.
114
+ The required sequence is:
184
115
 
185
- ## Verification gate
116
+ `SPEC -> PLAN -> plan check/receipt -> approved tasks -> fresh executor per task -> task verification receipts -> integration verification/receipt -> finalize`
186
117
 
187
- Before saying a task is complete:
118
+ Use dependency-safe waves and isolated worktrees only when their write/read scopes are safe. On resume, trust durable state plus current Git evidence over conversational memory. Long/high-risk completion must remain bound to the active run and current workspace fingerprint.
188
119
 
189
- 1. Identify what evidence would prove the requested behavior.
190
- 2. Run the narrowest relevant checks, then expand based on risk and project conventions.
191
- 3. Read the actual output and exit status.
192
- 4. Re-test the original failure/acceptance criterion, not only compilation.
193
- 5. Inspect the final diff for accidental changes and regressions.
194
- 6. Run or request an independent review/critic pass for substantial or high-risk work and resolve evidence-backed blocking findings.
195
- 7. Report exactly what passed, failed, was repaired, or was not run.
120
+ Do not store hidden chain-of-thought. Persist observable facts, decisions, acceptance status, evidence and next actions only.
196
121
 
197
- Never claim a command, test, build, deployment, migration, push, or release succeeded unless it actually did.
122
+ ## Safety
198
123
 
199
- ## Destructive operations
124
+ Ask before destructive or irreversible actions such as force-pushing, destructive reset/clean, deleting important data, dropping database objects, broad production migrations, production deployment or credential rotation. Never print secrets.
200
125
 
201
- Ask before destructive or irreversible actions such as deleting important data, dropping database objects, force pushing, resetting/cleaning uncommitted work, rewriting history, production deployment, credential rotation, or broad migration execution. Never print secrets.
126
+ Merge, push, publish and deploy are external side effects and require explicit user intent.
202
127
 
203
128
  ## Completion standard
204
129
 
205
- A task is complete only when requested behavior is implemented, acceptance criteria are addressed, relevant verification has fresh evidence, the final diff has been reviewed, no known blocking critic finding is being hidden, and any remaining limitations are stated accurately.
206
-
207
-
208
- ## V7 learning and optional external executors
209
-
210
- UES may analyze its own `.ues-evals` traces with `ocskill learn analyze`. The output is a proposal set, not an automatic self-modification. A human/parent explicitly accepts a proposal before it can appear in future task context. This keeps learning evidence-gated and reversible.
211
-
212
- Hermes support is optional and adapter-style. `ocskill hermes status` checks availability and `ocskill hermes prompt <slug> <task> .` emits a bounded delegation prompt. Hermes is not embedded into the UES runtime and may not mutate UES durable state on its own.
130
+ A task is complete only when the requested behavior is implemented, the acceptance criteria are addressed, fresh relevant verification supports the result, the final diff is reviewed, and remaining limitations are stated accurately.
@@ -40,4 +40,4 @@ Only concrete issues that prevent completion.
40
40
  ## Completion evidence
41
41
  What the parent may truthfully claim after this verification.
42
42
 
43
- The parent must record your actual verdict with `ocskill work verify-integration <slug> . --verdict PASS|FAIL|PARTIAL --evidence <summary>`. Finalization is intentionally blocked without a recorded PASS and will be invalidated if the workspace changes afterward.
43
+ For a long/high-risk PASS, the parent must create an `integration-verification` receipt bound to the current workspace fingerprint and pass it to `ocskill work verify-integration --receipt-file <file>`. Finalization remains blocked without PASS and is invalidated by later workspace changes.
@@ -40,4 +40,4 @@ Persistence, auth, payment, public API, deployment, destructive or migration con
40
40
  ## Required revisions
41
41
  Only blocking changes required before execution.
42
42
 
43
- A PASS means the plan is executable, not that implementation is correct. The parent must persist a genuine PASS with `ocskill work approve-plan <slug> . --evidence <summary>`; do not approve a plan that you returned as REVISE.
43
+ A PASS means the plan is executable, not that implementation is correct. The parent must bind a genuine PASS to the exact plan hash with `ocskill work gate-receipt <slug> plan ...`, then call `ocskill work approve-plan ... --receipt-file <file>`. Do not approve a plan that you returned as REVISE.
@@ -13,17 +13,21 @@ Required workflow:
13
13
  3. Create a concise SPEC with observable acceptance criteria.
14
14
  4. Initialize persistent state with `ocskill work init`.
15
15
  5. Produce a file-aware `PLAN.json` using the UES plan schema, then import it with `ocskill work plan`.
16
- 6. Dispatch `ues-plan-checker` in fresh context. A task cannot start until the checker returns PASS and the parent records it with `ocskill work approve-plan <slug> . --evidence <summary>`.
16
+ 6. Dispatch `ues-plan-checker` in fresh context. For long/high-risk work, bind the PASS to the current plan with a structured receipt, then approve it:
17
+ `ocskill work gate-receipt <slug> plan . --verifier ues-plan-checker --evidence "<summary>" --out .ues-work/<slug>/reports/plan-receipt.json`
18
+ followed by `ocskill work approve-plan <slug> . --evidence "<summary>" --receipt-file .ues-work/<slug>/reports/plan-receipt.json`.
17
19
  7. Use `ocskill task-graph` and execute only ready dependency-safe tasks. On OpenCode V2 prefer `ues.dispatch_task`: it starts the task, creates a fresh `ues-executor` session, applies configured attempt-based model escalation, waits for that executor, and returns its report. Inspect the diff and evidence, then record `ocskill work complete` or `ocskill work fail`.
18
- 8. Independent tasks may run concurrently only when safe-wave analysis reports no write/read conflict. For parallel write tasks, prefer isolated Git worktrees with `ocskill sandbox create <slug> <task> .` (created beside the main checkout) and integrate deliberately; do not point two executors at overlapping write surfaces in one working tree.
20
+ 8. Independent tasks may run concurrently only when safe-wave analysis reports no write/read conflict. V8 can isolate concurrent writers in Git worktrees and integrate them with conflict detection; manual fallback is `ocskill sandbox create ...` followed by `ocskill sandbox integrate <worktree> .`. Never integrate over overlapping dirty root files.
19
21
  9. During long execution keep the task lease alive with `ocskill work heartbeat` (the V2 dispatcher does this automatically). On resume, `ocskill work recover` or `ocskill work resume` recovers expired leases instead of leaving tasks stuck in `running`.
20
- 10. Run declared checks through `ocskill work verify-command <slug> <task> . -- <command> [args...]` when practical so EVIDENCE.json contains structured exit-code/output-hash/workspace receipts. Then record completion or failure with the current runId.
22
+ 10. Run declared checks through `ocskill work verify-command <slug> <task> . --run-id <run-id> -- <command> [args...]`. For long/high-risk plans, a successful receipt for the active run is mandatory before `work complete`; narrative-only completion is rejected.
21
23
  11. On executor failure, diagnose from fresh evidence and retry in a fresh executor. Adaptive model policy may raise the model tier based on risk/complexity plus attempt count; do not escalate blindly.
22
- 12. After all tasks complete, dispatch `ues-integration-verifier`. Record its actual verdict with `ocskill work verify-integration <slug> . --verdict PASS|FAIL|PARTIAL --evidence <summary>`.
24
+ 12. After all tasks complete, dispatch `ues-integration-verifier`. For PASS on long/high-risk work, create an integration receipt bound to the current workspace fingerprint:
25
+ `ocskill work gate-receipt <slug> integration . --verifier ues-integration-verifier --verdict PASS --evidence "<summary>" --out .ues-work/<slug>/reports/integration-receipt.json`
26
+ then record it with `ocskill work verify-integration <slug> . --verdict PASS --evidence "<summary>" --receipt-file .ues-work/<slug>/reports/integration-receipt.json`.
23
27
  13. `ocskill work finalize` is allowed only after a recorded PASS and only if the workspace fingerprint has not changed since that PASS. Re-run verification if it changed.
24
28
  14. Inspect final diff/status and report only evidence-backed completion.
25
29
 
26
30
  Do not merge, push, publish, deploy or perform destructive operations without explicit user approval.
27
31
 
28
32
 
29
- After a meaningful eval run, `ocskill learn analyze . --eval-dir .ues-evals` may produce deterministic learning proposals. Accepted lessons are explicit (`ocskill learn accept <id> .`) and can be surfaced in future context packs; UES never silently rewrites skills from one run.
33
+ After a meaningful eval run, `ocskill learn analyze . --eval-dir .ues-evals` clusters recurring failures into candidate lessons. `ocskill learn accept <id> .` stages a proposal, but shadow-required lessons enter future context only after `ocskill learn promote <id> . --baseline <rate> --candidate <rate> --samples N` proves an improvement.
@@ -1,18 +1,22 @@
1
1
  export function runtimeCapabilities(ctx) {
2
2
  const session = ctx?.session || {}
3
+ const permission = ctx?.permission || {}
3
4
  const capabilities = {
4
5
  sessionCreate: typeof session.create === "function",
5
6
  sessionPrompt: typeof session.prompt === "function",
6
7
  sessionWait: typeof session.wait === "function",
8
+ sessionInterrupt: typeof session.interrupt === "function",
7
9
  sessionContext: typeof session.context === "function",
8
10
  sessionSwitchAgent: typeof session.switchAgent === "function",
9
11
  sessionSwitchModel: typeof session.switchModel === "function",
10
12
  sessionHook: typeof session.hook === "function",
13
+ permissionHook: typeof permission.hook === "function",
11
14
  }
12
15
  capabilities.freshDispatch =
13
16
  capabilities.sessionCreate &&
14
17
  capabilities.sessionPrompt &&
15
18
  capabilities.sessionWait &&
19
+ capabilities.sessionInterrupt &&
16
20
  capabilities.sessionContext &&
17
21
  capabilities.sessionSwitchAgent
18
22
  capabilities.modelSwitch = capabilities.sessionSwitchModel