opencode-agent-skill 7.7.0 → 9.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +63 -3
- package/README.md +675 -581
- package/bin/ocskill.mjs +358 -156
- package/docs/DETERMINISTIC-TOOLS.md +25 -8
- package/docs/ENGINEERING-DESIGN.md +31 -13
- package/docs/EVALS.md +34 -12
- package/docs/NPM-PUBLISH.md +6 -6
- package/docs/OPENCODE-COMPAT.md +11 -8
- package/docs/TRACE-SCHEMA.md +15 -2
- package/docs/V8-INTELLIGENCE-RELIABILITY.md +206 -0
- package/docs/V9-SPEED-INTELLIGENCE.md +102 -0
- package/evals/live/tasks.json +6 -6
- package/evals/polyglot/fixtures/polyglot-bench/api/generated/client.ts +2 -0
- package/evals/polyglot/fixtures/polyglot-bench/api/openapi.json +25 -0
- package/evals/polyglot/fixtures/polyglot-bench/db/migrations/20260920_add_order_key.sql +1 -0
- package/evals/polyglot/fixtures/polyglot-bench/dotnet/OrderService.cs +8 -0
- package/evals/polyglot/fixtures/polyglot-bench/java/PriceService.java +5 -0
- package/evals/polyglot/fixtures/polyglot-bench/monorepo/package.json +6 -0
- package/evals/polyglot/fixtures/polyglot-bench/monorepo/packages/api/package.json +4 -0
- package/evals/polyglot/fixtures/polyglot-bench/monorepo/packages/web/package.json +7 -0
- package/evals/polyglot/fixtures/polyglot-bench/monorepo/pnpm-lock.yaml +5 -0
- package/evals/polyglot/fixtures/polyglot-bench/next/app/api/products/route.ts +7 -0
- package/evals/polyglot/fixtures/polyglot-bench/python/tenant_auth.py +4 -0
- package/evals/polyglot/fixtures/polyglot-bench/react-native/keyboard.ts +3 -0
- package/evals/polyglot/graders/polyglot-bench.mjs +101 -0
- package/evals/polyglot/tasks.json +54 -0
- package/global-config/AGENTS.md +10 -7
- package/global-config/agents/integration-verifier.md +1 -1
- package/global-config/agents/plan-checker.md +1 -1
- package/global-config/commands/run.md +9 -5
- package/global-config/plugins/ues-router/capabilities.js +4 -0
- package/global-config/plugins/ues-router/index.js +414 -21
- package/global-config/plugins/ues-router/router.js +140 -23
- package/global-config/skills/engineering-orchestrator/references/long-horizon.md +6 -4
- package/lib/aci.mjs +128 -0
- package/lib/benchmark-confidence.mjs +135 -0
- package/lib/cli-utils.mjs +41 -0
- package/lib/container-sandbox.mjs +102 -0
- package/lib/context-manifest.mjs +300 -22
- package/lib/control-center.mjs +36 -3
- package/lib/eval-order.mjs +9 -0
- package/lib/eval-telemetry.mjs +5 -2
- package/lib/gate-receipt.mjs +52 -0
- package/lib/installer.mjs +39 -25
- package/lib/learning-engine.mjs +236 -38
- package/lib/opencode-compat.mjs +25 -10
- package/lib/orchestrator-policy.mjs +101 -20
- package/lib/process-runner.mjs +30 -9
- package/lib/runtime-events.mjs +31 -0
- package/lib/semantic-index.mjs +318 -0
- package/lib/task-engine.mjs +416 -24
- package/lib/trajectory.mjs +89 -0
- package/lib/windows-shim.mjs +227 -0
- package/lib/worktree-sandbox.mjs +85 -3
- package/package.json +10 -4
- package/scripts/check-release-tag.mjs +22 -0
- package/scripts/control-center.mjs +25 -0
- package/scripts/eval-live.mjs +39 -58
- package/scripts/eval-matrix.mjs +155 -0
- package/scripts/smoke-packed-install.mjs +138 -4
- package/scripts/smoke-plain-install.mjs +91 -0
- package/scripts/validate-live-suite.mjs +3 -3
- package/scripts/validate.mjs +27 -5
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
# UES 9.0 Speed & Intelligence
|
|
2
|
+
|
|
3
|
+
UES 9.0 builds on the V8 evidence and reliability gates. The goal is not to make the underlying model generate tokens faster; it is to reduce time-to-correct-result by finding the right code sooner, shrinking unnecessary context, avoiding unsafe retries and requiring evidence before completion.
|
|
4
|
+
|
|
5
|
+
## Execution profiles
|
|
6
|
+
|
|
7
|
+
UES classifies work into three execution profiles:
|
|
8
|
+
|
|
9
|
+
- **FAST**: small/low-risk work, up to 2 skills, 12k context budget, targeted verification, no durable worktree/critic overhead by default.
|
|
10
|
+
- **STANDARD**: medium work, up to 4 skills, 24k context budget, semantic+Git context and targeted/affected verification.
|
|
11
|
+
- **DEEP**: long-horizon or high-risk work, up to 5 skills, 48k context budget, durable state, worktree isolation, critic/integration verification and full CI.
|
|
12
|
+
|
|
13
|
+
Risk still overrides speed. Authentication, security, payment, schema/migration, production/deploy and public-contract work remains fail-closed.
|
|
14
|
+
|
|
15
|
+
## Incremental evidence index
|
|
16
|
+
|
|
17
|
+
The V9 index caches bounded source metadata under `.ues-cache/semantic-index-v1.json`. Unchanged files reuse cached evidence while changed files are reparsed. It records:
|
|
18
|
+
|
|
19
|
+
- concrete source paths;
|
|
20
|
+
- bounded symbol definitions with line numbers;
|
|
21
|
+
- lexical identifier counts;
|
|
22
|
+
- cache reuse/reparse statistics.
|
|
23
|
+
|
|
24
|
+
This is intentionally labelled **syntax-aware lexical evidence**. It must not be presented as proof of program semantics or as a full AST/LSP call graph.
|
|
25
|
+
|
|
26
|
+
Commands:
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
ocskill index status .
|
|
30
|
+
ocskill index build .
|
|
31
|
+
ocskill index rebuild .
|
|
32
|
+
ocskill aci search "checkout total" .
|
|
33
|
+
ocskill aci refs calculateTotal .
|
|
34
|
+
ocskill aci view src/checkout.ts . --line 120 --lines 80
|
|
35
|
+
ocskill aci text "exact marker" .
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
## Weak-model ACI
|
|
39
|
+
|
|
40
|
+
The ACI keeps search/view output bounded and evidence-first. Path traversal outside the repository is rejected. Large/binary files are refused by the bounded viewer. Reference results distinguish lexical references from concrete definition lines so an agent cannot honestly claim deeper semantic certainty than the tool measured.
|
|
41
|
+
|
|
42
|
+
## Runtime traces
|
|
43
|
+
|
|
44
|
+
Operational trace events are appended under `.ues-traces/*.jsonl`. Common credentials/tokens are redacted before persistence, oversized payloads are hashed/truncated, and traces explicitly exclude hidden chain-of-thought.
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
ocskill trace show <trace-id> .
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
## Verification sandbox
|
|
51
|
+
|
|
52
|
+
When Docker or Podman is available, deterministic verification commands can run with:
|
|
53
|
+
|
|
54
|
+
- network disabled by default;
|
|
55
|
+
- Linux capabilities dropped;
|
|
56
|
+
- no-new-privileges;
|
|
57
|
+
- read-only container root;
|
|
58
|
+
- bounded PIDs/memory/CPU;
|
|
59
|
+
- only the requested workspace bind-mounted;
|
|
60
|
+
- no host secrets forwarded by default.
|
|
61
|
+
|
|
62
|
+
This protects verification commands; it does **not** claim to sandbox the OpenCode model process itself.
|
|
63
|
+
|
|
64
|
+
## Benchmark confidence gate
|
|
65
|
+
|
|
66
|
+
The matrix now embeds paired baseline/UES evidence and computes:
|
|
67
|
+
|
|
68
|
+
- paired wins/losses/ties;
|
|
69
|
+
- pass-rate delta;
|
|
70
|
+
- exact two-sided sign-test p-value;
|
|
71
|
+
- per-suite regression checks;
|
|
72
|
+
- mean duration ratio;
|
|
73
|
+
- optional cost ratio.
|
|
74
|
+
|
|
75
|
+
Normal reporting:
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
npm run evals:matrix -- --model provider/model --trials 3
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
Fail-closed release gate:
|
|
82
|
+
|
|
83
|
+
```bash
|
|
84
|
+
npm run evals:matrix:gate -- --model provider/model --trials 3
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
The gate requires enough paired samples, positive uplift, more UES wins than losses, statistical support, no suite regression and acceptable speed. This is evidence for the measured benchmark only; it is not evidence that UES turns a weaker model into a different model.
|
|
88
|
+
|
|
89
|
+
## Lock and Windows hardening
|
|
90
|
+
|
|
91
|
+
State locking now uses owner tokens and heartbeats. Stale takeover renames the old lock before removal, and release only removes a lock whose token still belongs to the caller. Windows CLI/router invocation resolves Node-backed shims and refuses unrecognized batch shims rather than falling back to shell execution.
|
|
92
|
+
|
|
93
|
+
## Release validation
|
|
94
|
+
|
|
95
|
+
Before publishing 9.0.0:
|
|
96
|
+
|
|
97
|
+
```bash
|
|
98
|
+
npm run ci
|
|
99
|
+
npm run evals:matrix:gate -- --model provider/model --trials 3
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
If GitHub-hosted Actions still fails before the first workflow step, treat that as an infrastructure/repository-action issue rather than test evidence; local CI and the paired benchmark remain required release gates.
|
package/evals/live/tasks.json
CHANGED
|
@@ -11,13 +11,13 @@
|
|
|
11
11
|
{
|
|
12
12
|
"id": "auth-resource-ownership",
|
|
13
13
|
"fixture": "fixtures/engineering-bench",
|
|
14
|
-
"prompt": "Fix canEditResource(user, resource) in src/authz.mjs. Missing inputs deny.
|
|
14
|
+
"prompt": "Fix canEditResource(user, resource) in src/authz.mjs. Missing inputs deny. Treat user.role === 'admin' as admin. Treat user.suspended === true as suspended; suspended users always deny, including admins. Admin users may edit any non-missing resource. Non-admin users may edit only when user.id equals resource.ownerId AND user.tenantId equals resource.tenantId. Return a boolean and do not mutate inputs.",
|
|
15
15
|
"grader": "graders/engineering-bench.mjs"
|
|
16
16
|
},
|
|
17
17
|
{
|
|
18
18
|
"id": "api-pagination-contract",
|
|
19
19
|
"fixture": "fixtures/engineering-bench",
|
|
20
|
-
"prompt": "Fix paginate(items, options) in src/pagination.mjs.
|
|
20
|
+
"prompt": "Fix paginate(items, options) in src/pagination.mjs. Pages are 1-indexed; defaults are page=1 and pageSize=10. items must be an array. page/pageSize must be finite integers: non-number, non-finite, or non-integer values throw TypeError; integer values <= 0 throw RangeError. Return {items,total,page,pageSize,totalPages}; out-of-range pages return an empty items array.",
|
|
21
21
|
"grader": "graders/engineering-bench.mjs"
|
|
22
22
|
},
|
|
23
23
|
{
|
|
@@ -29,13 +29,13 @@
|
|
|
29
29
|
{
|
|
30
30
|
"id": "payment-idempotency",
|
|
31
31
|
"fixture": "fixtures/engineering-bench",
|
|
32
|
-
"prompt": "Fix applyPaymentEvent(order, event) in src/payment.mjs. Do not mutate order. Duplicate event.id values already in order.processedEvents must be idempotent. A new event must
|
|
32
|
+
"prompt": "Fix applyPaymentEvent(order, event) in src/payment.mjs. Do not mutate order. Duplicate event.id values already in order.processedEvents must be idempotent. A new event.id must be a non-empty string; malformed, missing, or empty ids throw TypeError. For status='succeeded', amountCents and currency must exactly match order.totalCents/order.currency or throw RangeError, then mark status paid. Failed events must not mark paid. Record each new event id exactly once.",
|
|
33
33
|
"grader": "graders/engineering-bench.mjs"
|
|
34
34
|
},
|
|
35
35
|
{
|
|
36
36
|
"id": "webhook-ordering",
|
|
37
37
|
"fixture": "fixtures/engineering-bench",
|
|
38
|
-
"prompt": "Fix advancePaymentState(state, event) in src/webhook.mjs. Ignore duplicate/stale events whose integer sequence <= state.lastSequence. Newer events must follow legal transitions pending->authorized|failed, authorized->paid|failed, paid->refunded; same-status newer events are allowed. Illegal transitions throw RangeError.
|
|
38
|
+
"prompt": "Fix advancePaymentState(state, event) in src/webhook.mjs. Ignore duplicate/stale events whose integer sequence <= state.lastSequence by returning the original state object unchanged (the exact same reference). Newer events must follow legal transitions pending->authorized|failed, authorized->paid|failed, paid->refunded; same-status newer events are allowed. Illegal transitions throw RangeError. For accepted newer events, return a new state object and never mutate inputs.",
|
|
39
39
|
"grader": "graders/engineering-bench.mjs"
|
|
40
40
|
},
|
|
41
41
|
{
|
|
@@ -77,7 +77,7 @@
|
|
|
77
77
|
{
|
|
78
78
|
"id": "money-integer-invariants",
|
|
79
79
|
"fixture": "fixtures/engineering-bench",
|
|
80
|
-
"prompt": "Fix orderTotal(lines) in src/money.mjs. Each line uses integer unitPriceCents>=0, integer quantity>0, optional integer discountCents>=0.
|
|
80
|
+
"prompt": "Fix orderTotal(lines) in src/money.mjs. Each line uses integer unitPriceCents>=0, integer quantity>0, optional integer discountCents>=0. Non-number, non-finite, or non-integer numeric fields throw TypeError. Range violations throw RangeError. discountCents may not exceed unitPriceCents*quantity; treat that as a RangeError rather than clamping. Return the sum as an integer cent total.",
|
|
81
81
|
"grader": "graders/engineering-bench.mjs"
|
|
82
82
|
},
|
|
83
83
|
{
|
|
@@ -113,7 +113,7 @@
|
|
|
113
113
|
{
|
|
114
114
|
"id": "cache-invalidation-tags",
|
|
115
115
|
"fixture": "fixtures/engineering-bench",
|
|
116
|
-
"prompt": "Fix tagsForProductMutation(product) in src/cache-tags.mjs. product must contain a non-empty id. Return unique tags in this order: 'products', 'product:<id>', then optional 'category:<categoryId>' and 'seller:<sellerId>' when those values are non-null/non-empty. Do not emit undefined/empty tags and do not mutate product.",
|
|
116
|
+
"prompt": "Fix tagsForProductMutation(product) in src/cache-tags.mjs. product must contain a non-empty string id; malformed product/id or an empty id throws TypeError. Return unique tags in this order: 'products', 'product:<id>', then optional 'category:<categoryId>' and 'seller:<sellerId>' when those values are non-null/non-empty. Do not emit undefined/empty tags and do not mutate product.",
|
|
117
117
|
"grader": "graders/engineering-bench.mjs"
|
|
118
118
|
},
|
|
119
119
|
{
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
{
|
|
2
|
+
"openapi": "3.1.0",
|
|
3
|
+
"info": {
|
|
4
|
+
"title": "Demo",
|
|
5
|
+
"version": "1.0.0"
|
|
6
|
+
},
|
|
7
|
+
"components": {
|
|
8
|
+
"schemas": {
|
|
9
|
+
"User": {
|
|
10
|
+
"type": "object",
|
|
11
|
+
"required": [
|
|
12
|
+
"id"
|
|
13
|
+
],
|
|
14
|
+
"properties": {
|
|
15
|
+
"id": {
|
|
16
|
+
"type": "string"
|
|
17
|
+
},
|
|
18
|
+
"name": {
|
|
19
|
+
"type": "string"
|
|
20
|
+
}
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
ALTER TABLE orders ADD COLUMN order_key TEXT NOT NULL;
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
import assert from "node:assert/strict"
|
|
2
|
+
import { createHash } from "node:crypto"
|
|
3
|
+
import { readFile } from "node:fs/promises"
|
|
4
|
+
import path from "node:path"
|
|
5
|
+
|
|
6
|
+
const root = process.env.UES_EVAL_WORKSPACE || process.cwd()
|
|
7
|
+
const task = process.env.UES_EVAL_TASK
|
|
8
|
+
|
|
9
|
+
const text = async (file) => readFile(path.join(root, file), "utf8")
|
|
10
|
+
const sha = (value) => createHash("sha256").update(value).digest("hex")
|
|
11
|
+
const noDangerousSql = (value) => assert.doesNotMatch(value, /\b(?:DROP|TRUNCATE|DELETE\s+FROM)\b/i)
|
|
12
|
+
|
|
13
|
+
switch (task) {
|
|
14
|
+
case "polyglot-python-tenant-auth": {
|
|
15
|
+
const source = await text("python/tenant_auth.py")
|
|
16
|
+
assert.match(source, /def\s+can_edit\s*\(/)
|
|
17
|
+
assert.match(source, /suspended/i)
|
|
18
|
+
assert.match(source, /tenant/i)
|
|
19
|
+
assert.match(source, /owner_id/)
|
|
20
|
+
assert.match(source, /user(?:\.|\[).*id|user\.id|user\.get\(["']id["']\)/)
|
|
21
|
+
assert.match(source, /admin/i)
|
|
22
|
+
assert.doesNotMatch(source, /return\s+True\s*$/m)
|
|
23
|
+
break
|
|
24
|
+
}
|
|
25
|
+
case "polyglot-java-money-validation": {
|
|
26
|
+
const source = await text("java/PriceService.java")
|
|
27
|
+
assert.match(source, /long\s+totalCents\s*\(/)
|
|
28
|
+
assert.match(source, /IllegalArgumentException/)
|
|
29
|
+
assert.match(source, /unitPriceCents\s*\*\s*quantity/)
|
|
30
|
+
assert.match(source, /discountCents/)
|
|
31
|
+
assert.match(source, /discountCents\s*>\s*(?:gross|unitPriceCents\s*\*\s*quantity)/)
|
|
32
|
+
assert.doesNotMatch(source, /\b(?:double|float)\b/)
|
|
33
|
+
break
|
|
34
|
+
}
|
|
35
|
+
case "polyglot-dotnet-order-authorization": {
|
|
36
|
+
const source = await text("dotnet/OrderService.cs")
|
|
37
|
+
assert.match(source, /bool\s+CanUpdate\s*\(/)
|
|
38
|
+
assert.match(source, /Suspended/)
|
|
39
|
+
assert.match(source, /TenantId/)
|
|
40
|
+
assert.match(source, /Admin/)
|
|
41
|
+
assert.match(source, /OwnerId\s*==\s*user\.Id|order\.OwnerId\s*==\s*user\.Id/)
|
|
42
|
+
assert.doesNotMatch(source, /return\s+true\s*;/i)
|
|
43
|
+
break
|
|
44
|
+
}
|
|
45
|
+
case "polyglot-nextjs-api-errors": {
|
|
46
|
+
const source = await text("next/app/api/products/route.ts")
|
|
47
|
+
assert.match(source, /export\s+async\s+function\s+GET/)
|
|
48
|
+
assert.match(source, /NOT_FOUND/)
|
|
49
|
+
assert.match(source, /status\s*:\s*404/)
|
|
50
|
+
assert.match(source, /INTERNAL/)
|
|
51
|
+
assert.match(source, /status\s*:\s*500/)
|
|
52
|
+
assert.doesNotMatch(source, /error\.stack|stack\s*:/)
|
|
53
|
+
break
|
|
54
|
+
}
|
|
55
|
+
case "polyglot-react-native-keyboard": {
|
|
56
|
+
const source = await text("react-native/keyboard.ts")
|
|
57
|
+
assert.match(source, /platform\s*===\s*["']ios["']/)
|
|
58
|
+
assert.match(source, /safeAreaTop\s*\+\s*headerHeight/)
|
|
59
|
+
assert.match(source, /return\s+headerHeight/)
|
|
60
|
+
assert.match(source, /Number\.isFinite/)
|
|
61
|
+
assert.match(source, /android/)
|
|
62
|
+
break
|
|
63
|
+
}
|
|
64
|
+
case "polyglot-safe-sql-migration": {
|
|
65
|
+
const source = await text("db/migrations/20260920_add_order_key.sql")
|
|
66
|
+
noDangerousSql(source)
|
|
67
|
+
const add = source.search(/ADD\s+COLUMN\s+order_key/i)
|
|
68
|
+
const backfill = source.search(/UPDATE\s+orders[\s\S]*order_key/i)
|
|
69
|
+
const notNull = source.search(/order_key[\s\S]*SET\s+NOT\s+NULL/i)
|
|
70
|
+
const unique = source.search(/CREATE\s+UNIQUE\s+INDEX/i)
|
|
71
|
+
assert.ok(add >= 0 && backfill > add && notNull > backfill && unique > notNull)
|
|
72
|
+
assert.match(source, /order_key\s*=\s*[^;]*\bid\b/i)
|
|
73
|
+
break
|
|
74
|
+
}
|
|
75
|
+
case "polyglot-monorepo-workspace-boundary": {
|
|
76
|
+
const rootPkg = JSON.parse(await text("monorepo/package.json"))
|
|
77
|
+
const webPkg = JSON.parse(await text("monorepo/packages/web/package.json"))
|
|
78
|
+
assert.ok(Array.isArray(rootPkg.workspaces))
|
|
79
|
+
assert.ok(rootPkg.workspaces.includes("packages/*"))
|
|
80
|
+
assert.equal(webPkg.dependencies?.["@demo/api"], "workspace:*")
|
|
81
|
+
const lock = await text("monorepo/pnpm-lock.yaml")
|
|
82
|
+
assert.equal(sha(lock), "130256c4e0c0b4db7ec66736ba9afbadb60e908556b993e92fd5c70244bba8d8")
|
|
83
|
+
break
|
|
84
|
+
}
|
|
85
|
+
case "polyglot-generated-contract-discipline": {
|
|
86
|
+
const api = JSON.parse(await text("api/openapi.json"))
|
|
87
|
+
const user = api.components?.schemas?.User
|
|
88
|
+
assert.ok(user?.properties?.id)
|
|
89
|
+
assert.ok(user?.properties?.name)
|
|
90
|
+
assert.ok(user?.properties?.displayName)
|
|
91
|
+
assert.ok(user.required?.includes("id"))
|
|
92
|
+
assert.ok(user.required?.includes("displayName"))
|
|
93
|
+
const generated = await text("api/generated/client.ts")
|
|
94
|
+
assert.equal(sha(generated), "3b4a1e4902a2d678c751f9673a74a5efd352e434142195747268516a30b0ce88")
|
|
95
|
+
break
|
|
96
|
+
}
|
|
97
|
+
default:
|
|
98
|
+
throw new Error("Unknown polyglot task: " + task)
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
console.log("polyglot grader PASS: " + task)
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
{
|
|
2
|
+
"version": 1,
|
|
3
|
+
"description": "Polyglot engineering-surface benchmark covering Python, Java/Spring, .NET, Next.js, React Native, SQL migration, monorepo dependency boundaries and generated-code discipline.",
|
|
4
|
+
"tasks": [
|
|
5
|
+
{
|
|
6
|
+
"id": "polyglot-python-tenant-auth",
|
|
7
|
+
"fixture": "fixtures/polyglot-bench",
|
|
8
|
+
"prompt": "Repair python/tenant_auth.py. can_edit(user, resource) must deny missing inputs and suspended users, allow admins only inside the same tenant, and allow members only when both tenant IDs match and resource.owner_id equals user.id. Keep the function pure and return a boolean.",
|
|
9
|
+
"grader": "graders/polyglot-bench.mjs"
|
|
10
|
+
},
|
|
11
|
+
{
|
|
12
|
+
"id": "polyglot-java-money-validation",
|
|
13
|
+
"fixture": "fixtures/polyglot-bench",
|
|
14
|
+
"prompt": "Repair java/PriceService.java. totalCents(unitPriceCents, quantity, discountCents) must reject negative unit price/discount and non-positive quantity with IllegalArgumentException, calculate integer-cent total exactly, and reject discounts larger than the gross line total. Do not switch to floating point.",
|
|
15
|
+
"grader": "graders/polyglot-bench.mjs"
|
|
16
|
+
},
|
|
17
|
+
{
|
|
18
|
+
"id": "polyglot-dotnet-order-authorization",
|
|
19
|
+
"fixture": "fixtures/polyglot-bench",
|
|
20
|
+
"prompt": "Repair dotnet/OrderService.cs. CanUpdate(User user, Order order) must deny null inputs and suspended users, require matching TenantId for every role, allow Admin within the same tenant, and otherwise require order.OwnerId == user.Id. Return bool without mutating inputs.",
|
|
21
|
+
"grader": "graders/polyglot-bench.mjs"
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"id": "polyglot-nextjs-api-errors",
|
|
25
|
+
"fixture": "fixtures/polyglot-bench",
|
|
26
|
+
"prompt": "Repair next/app/api/products/route.ts. GET must return JSON with status 200 on success, map a NOT_FOUND error to a sanitized 404 body {error:{code:'NOT_FOUND',message:'Not found'}}, and map unknown errors to 500 INTERNAL without exposing stack/message details. Preserve the route export.",
|
|
27
|
+
"grader": "graders/polyglot-bench.mjs"
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"id": "polyglot-react-native-keyboard",
|
|
31
|
+
"fixture": "fixtures/polyglot-bench",
|
|
32
|
+
"prompt": "Repair react-native/keyboard.ts. keyboardOffset(platform, safeAreaTop, headerHeight) must validate non-negative finite offsets, accept only ios/android, return safeAreaTop+headerHeight for iOS and headerHeight for Android, and throw on invalid input.",
|
|
33
|
+
"grader": "graders/polyglot-bench.mjs"
|
|
34
|
+
},
|
|
35
|
+
{
|
|
36
|
+
"id": "polyglot-safe-sql-migration",
|
|
37
|
+
"fixture": "fixtures/polyglot-bench",
|
|
38
|
+
"prompt": "Repair db/migrations/20260920_add_order_key.sql as a forward-compatible migration: add order_key as nullable first, backfill existing rows deterministically from id, then make it NOT NULL and add a UNIQUE index. Do not drop/truncate tables or delete data.",
|
|
39
|
+
"grader": "graders/polyglot-bench.mjs"
|
|
40
|
+
},
|
|
41
|
+
{
|
|
42
|
+
"id": "polyglot-monorepo-workspace-boundary",
|
|
43
|
+
"fixture": "fixtures/polyglot-bench",
|
|
44
|
+
"prompt": "Repair the monorepo dependency boundary in monorepo/package.json and monorepo/packages/web/package.json. Root workspaces must include packages/* and web must depend on @demo/api using workspace:*. Do not edit the generated monorepo/pnpm-lock.yaml.",
|
|
45
|
+
"grader": "graders/polyglot-bench.mjs"
|
|
46
|
+
},
|
|
47
|
+
{
|
|
48
|
+
"id": "polyglot-generated-contract-discipline",
|
|
49
|
+
"fixture": "fixtures/polyglot-bench",
|
|
50
|
+
"prompt": "Repair api/openapi.json so the User schema preserves the legacy name field and also exposes displayName, with id and displayName required. Do not edit api/generated/client.ts because it is generated output; only the source contract should change.",
|
|
51
|
+
"grader": "graders/polyglot-bench.mjs"
|
|
52
|
+
}
|
|
53
|
+
]
|
|
54
|
+
}
|
package/global-config/AGENTS.md
CHANGED
|
@@ -37,8 +37,9 @@ When the `ocskill` CLI is available, prefer deterministic repository evidence be
|
|
|
37
37
|
- `ocskill task-graph <PLAN.json>` — validate dependencies and compute safe execution waves
|
|
38
38
|
- `ocskill context-pack <slug> <task> [dir]` — bounded durable handoff enriched with declared files, import neighbors, likely tests, instruction/manifests and accepted lessons
|
|
39
39
|
- `ocskill work verify-command ... -- <command>` — structured verification receipt (exit code, hashes, timing, workspace fingerprints)
|
|
40
|
-
- `ocskill sandbox create|list|remove ...` — isolated Git worktree primitives
|
|
41
|
-
- `ocskill
|
|
40
|
+
- `ocskill sandbox create|integrate|list|remove ...` — isolated Git worktree primitives with conflict-aware integration
|
|
41
|
+
- `ocskill work events <slug> [dir]` — append-only runtime event journal for start/heartbeat/receipt/failure/recovery/completion/integration/finalization
|
|
42
|
+
- `ocskill learn status|analyze|accept|promote ...` — evidence-gated learning loop; shadow-required lessons are retrieved only after measured improvement
|
|
42
43
|
- `ocskill dashboard [dir] --serve` — local Control Center for work state, evidence, learning and eval summaries
|
|
43
44
|
|
|
44
45
|
These helpers are evidence accelerators, not substitutes for reading the exact affected code. Use repository-native search/tools when they provide more precise symbol/call-graph information.
|
|
@@ -132,12 +133,12 @@ For explicit long-running/autonomous work, do not ask one context to remember th
|
|
|
132
133
|
1. Map the relevant repository surface with deterministic evidence and `ues-codebase-mapper` when useful.
|
|
133
134
|
2. Persist observable requirements in `.ues-work/<slug>/SPEC.md`.
|
|
134
135
|
3. Create a machine-checkable `PLAN.json` and validate it with `ocskill task-graph`.
|
|
135
|
-
4. Ask `ues-plan-checker` to challenge the plan before edits begin.
|
|
136
|
+
4. Ask `ues-plan-checker` to challenge the plan before edits begin. For long/high-risk work, create a structured plan-verification receipt bound to the current plan hash, then record approval with `ocskill work approve-plan --receipt-file ...`; `work start` is blocked until this happens.
|
|
136
137
|
5. Execute each approved task in a fresh `ues-executor` context. Active V7 tasks carry a runId, heartbeat and lease expiry so interrupted work can be recovered deterministically. On OpenCode V2 prefer `ues.dispatch_task`, which creates the fresh session and applies configured attempt-based model escalation.
|
|
137
|
-
6. Inspect each child diff and
|
|
138
|
-
7. Parallelize only dependency-safe tasks with no write/read conflict.
|
|
138
|
+
6. Inspect each child diff and use `ocskill work verify-command` before marking completion. Long/high-risk tasks require at least one successful receipt for the active run; narrative-only completion is rejected.
|
|
139
|
+
7. Parallelize only dependency-safe tasks with no write/read conflict. V8 may isolate concurrent writers automatically; manual sandboxes use `ocskill sandbox create` and `ocskill sandbox integrate`, which refuses overlap with dirty root files.
|
|
139
140
|
8. On resume, trust durable state plus current Git evidence over conversational memory. Recover expired executor leases before retrying; preserve runId fences for active attempts.
|
|
140
|
-
9. After all tasks complete, run `ues-integration-verifier
|
|
141
|
+
9. After all tasks complete, run `ues-integration-verifier`. Long/high-risk PASS must be bound to the current workspace fingerprint with a structured integration-verification receipt before `ocskill work verify-integration` records it.
|
|
141
142
|
10. `work finalize` requires a recorded integration PASS and rejects completion if the Git workspace changed after that PASS.
|
|
142
143
|
11. Merge/push/publish/deploy remain external side effects and require explicit user intent.
|
|
143
144
|
|
|
@@ -175,6 +176,8 @@ Use them selectively. Keep trivial work inline. The editable `ues-executor` must
|
|
|
175
176
|
|
|
176
177
|
## Implementation discipline
|
|
177
178
|
|
|
179
|
+
For focused small/FAST fixes where the request names the exact file/function and observable contract, keep the workflow literal and bounded: read the target first, maintain a compact acceptance checklist, and avoid repo-wide discovery unless a concrete uncertainty or dependency requires it. Preserve explicitly requested exception classes, type-vs-range distinctions, return shapes, field names/order, mutation rules, idempotency, and boundary behavior exactly. Do not silently strengthen, weaken, or substitute those semantics. When tests are absent or hidden, use focused runtime probes that cover each stated criterion, especially type/range boundaries, before declaring success.
|
|
180
|
+
|
|
178
181
|
- Make the smallest coherent change that satisfies the request.
|
|
179
182
|
- Follow the repository's package manager, formatter, linter, tests, build scripts, architecture, and generated-file policy.
|
|
180
183
|
- Preserve unrelated user changes.
|
|
@@ -207,6 +210,6 @@ A task is complete only when requested behavior is implemented, acceptance crite
|
|
|
207
210
|
|
|
208
211
|
## V7 learning and optional external executors
|
|
209
212
|
|
|
210
|
-
UES may analyze its own `.ues-evals` traces with `ocskill learn analyze`.
|
|
213
|
+
UES may analyze its own `.ues-evals` traces with `ocskill learn analyze`. V8 clusters recurring failure signatures and emits candidate rules. Acceptance stages a proposal; shadow-required lessons appear in future context only after `ocskill learn promote` records a measured benchmark improvement.
|
|
211
214
|
|
|
212
215
|
Hermes support is optional and adapter-style. `ocskill hermes status` checks availability and `ocskill hermes prompt <slug> <task> .` emits a bounded delegation prompt. Hermes is not embedded into the UES runtime and may not mutate UES durable state on its own.
|
|
@@ -40,4 +40,4 @@ Only concrete issues that prevent completion.
|
|
|
40
40
|
## Completion evidence
|
|
41
41
|
What the parent may truthfully claim after this verification.
|
|
42
42
|
|
|
43
|
-
|
|
43
|
+
For a long/high-risk PASS, the parent must create an `integration-verification` receipt bound to the current workspace fingerprint and pass it to `ocskill work verify-integration --receipt-file <file>`. Finalization remains blocked without PASS and is invalidated by later workspace changes.
|
|
@@ -40,4 +40,4 @@ Persistence, auth, payment, public API, deployment, destructive or migration con
|
|
|
40
40
|
## Required revisions
|
|
41
41
|
Only blocking changes required before execution.
|
|
42
42
|
|
|
43
|
-
A PASS means the plan is executable, not that implementation is correct. The parent must
|
|
43
|
+
A PASS means the plan is executable, not that implementation is correct. The parent must bind a genuine PASS to the exact plan hash with `ocskill work gate-receipt <slug> plan ...`, then call `ocskill work approve-plan ... --receipt-file <file>`. Do not approve a plan that you returned as REVISE.
|
|
@@ -13,17 +13,21 @@ Required workflow:
|
|
|
13
13
|
3. Create a concise SPEC with observable acceptance criteria.
|
|
14
14
|
4. Initialize persistent state with `ocskill work init`.
|
|
15
15
|
5. Produce a file-aware `PLAN.json` using the UES plan schema, then import it with `ocskill work plan`.
|
|
16
|
-
6. Dispatch `ues-plan-checker` in fresh context.
|
|
16
|
+
6. Dispatch `ues-plan-checker` in fresh context. For long/high-risk work, bind the PASS to the current plan with a structured receipt, then approve it:
|
|
17
|
+
`ocskill work gate-receipt <slug> plan . --verifier ues-plan-checker --evidence "<summary>" --out .ues-work/<slug>/reports/plan-receipt.json`
|
|
18
|
+
followed by `ocskill work approve-plan <slug> . --evidence "<summary>" --receipt-file .ues-work/<slug>/reports/plan-receipt.json`.
|
|
17
19
|
7. Use `ocskill task-graph` and execute only ready dependency-safe tasks. On OpenCode V2 prefer `ues.dispatch_task`: it starts the task, creates a fresh `ues-executor` session, applies configured attempt-based model escalation, waits for that executor, and returns its report. Inspect the diff and evidence, then record `ocskill work complete` or `ocskill work fail`.
|
|
18
|
-
8. Independent tasks may run concurrently only when safe-wave analysis reports no write/read conflict.
|
|
20
|
+
8. Independent tasks may run concurrently only when safe-wave analysis reports no write/read conflict. V8 can isolate concurrent writers in Git worktrees and integrate them with conflict detection; manual fallback is `ocskill sandbox create ...` followed by `ocskill sandbox integrate <worktree> .`. Never integrate over overlapping dirty root files.
|
|
19
21
|
9. During long execution keep the task lease alive with `ocskill work heartbeat` (the V2 dispatcher does this automatically). On resume, `ocskill work recover` or `ocskill work resume` recovers expired leases instead of leaving tasks stuck in `running`.
|
|
20
|
-
10. Run declared checks through `ocskill work verify-command <slug> <task> . -- <command> [args...]
|
|
22
|
+
10. Run declared checks through `ocskill work verify-command <slug> <task> . --run-id <run-id> -- <command> [args...]`. For long/high-risk plans, a successful receipt for the active run is mandatory before `work complete`; narrative-only completion is rejected.
|
|
21
23
|
11. On executor failure, diagnose from fresh evidence and retry in a fresh executor. Adaptive model policy may raise the model tier based on risk/complexity plus attempt count; do not escalate blindly.
|
|
22
|
-
12. After all tasks complete, dispatch `ues-integration-verifier`.
|
|
24
|
+
12. After all tasks complete, dispatch `ues-integration-verifier`. For PASS on long/high-risk work, create an integration receipt bound to the current workspace fingerprint:
|
|
25
|
+
`ocskill work gate-receipt <slug> integration . --verifier ues-integration-verifier --verdict PASS --evidence "<summary>" --out .ues-work/<slug>/reports/integration-receipt.json`
|
|
26
|
+
then record it with `ocskill work verify-integration <slug> . --verdict PASS --evidence "<summary>" --receipt-file .ues-work/<slug>/reports/integration-receipt.json`.
|
|
23
27
|
13. `ocskill work finalize` is allowed only after a recorded PASS and only if the workspace fingerprint has not changed since that PASS. Re-run verification if it changed.
|
|
24
28
|
14. Inspect final diff/status and report only evidence-backed completion.
|
|
25
29
|
|
|
26
30
|
Do not merge, push, publish, deploy or perform destructive operations without explicit user approval.
|
|
27
31
|
|
|
28
32
|
|
|
29
|
-
After a meaningful eval run, `ocskill learn analyze . --eval-dir .ues-evals`
|
|
33
|
+
After a meaningful eval run, `ocskill learn analyze . --eval-dir .ues-evals` clusters recurring failures into candidate lessons. `ocskill learn accept <id> .` stages a proposal, but shadow-required lessons enter future context only after `ocskill learn promote <id> . --baseline <rate> --candidate <rate> --samples N` proves an improvement.
|
|
@@ -1,18 +1,22 @@
|
|
|
1
1
|
export function runtimeCapabilities(ctx) {
|
|
2
2
|
const session = ctx?.session || {}
|
|
3
|
+
const permission = ctx?.permission || {}
|
|
3
4
|
const capabilities = {
|
|
4
5
|
sessionCreate: typeof session.create === "function",
|
|
5
6
|
sessionPrompt: typeof session.prompt === "function",
|
|
6
7
|
sessionWait: typeof session.wait === "function",
|
|
8
|
+
sessionInterrupt: typeof session.interrupt === "function",
|
|
7
9
|
sessionContext: typeof session.context === "function",
|
|
8
10
|
sessionSwitchAgent: typeof session.switchAgent === "function",
|
|
9
11
|
sessionSwitchModel: typeof session.switchModel === "function",
|
|
10
12
|
sessionHook: typeof session.hook === "function",
|
|
13
|
+
permissionHook: typeof permission.hook === "function",
|
|
11
14
|
}
|
|
12
15
|
capabilities.freshDispatch =
|
|
13
16
|
capabilities.sessionCreate &&
|
|
14
17
|
capabilities.sessionPrompt &&
|
|
15
18
|
capabilities.sessionWait &&
|
|
19
|
+
capabilities.sessionInterrupt &&
|
|
16
20
|
capabilities.sessionContext &&
|
|
17
21
|
capabilities.sessionSwitchAgent
|
|
18
22
|
capabilities.modelSwitch = capabilities.sessionSwitchModel
|