opencode-agent-skill 7.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +163 -0
- package/LICENSE +9 -0
- package/README.md +581 -0
- package/bin/ocskill.mjs +975 -0
- package/docs/DETERMINISTIC-TOOLS.md +88 -0
- package/docs/ENGINEERING-DESIGN.md +176 -0
- package/docs/EVALS.md +136 -0
- package/docs/NPM-PUBLISH.md +102 -0
- package/docs/OPENCODE-COMPAT.md +109 -0
- package/docs/RESEARCH-SOURCES.md +37 -0
- package/docs/TRACE-SCHEMA.md +109 -0
- package/docs/V7-INTELLIGENCE-RUNTIME.md +166 -0
- package/evals/live/fixtures/engineering-bench/package.json +1 -0
- package/evals/live/fixtures/engineering-bench/src/api-errors.mjs +3 -0
- package/evals/live/fixtures/engineering-bench/src/authz.mjs +3 -0
- package/evals/live/fixtures/engineering-bench/src/cache-tags.mjs +3 -0
- package/evals/live/fixtures/engineering-bench/src/config.mjs +3 -0
- package/evals/live/fixtures/engineering-bench/src/contract-consumer.mjs +6 -0
- package/evals/live/fixtures/engineering-bench/src/contract-producer.mjs +3 -0
- package/evals/live/fixtures/engineering-bench/src/dedupe.mjs +3 -0
- package/evals/live/fixtures/engineering-bench/src/dependency.mjs +3 -0
- package/evals/live/fixtures/engineering-bench/src/discount.mjs +3 -0
- package/evals/live/fixtures/engineering-bench/src/inventory.mjs +3 -0
- package/evals/live/fixtures/engineering-bench/src/migration.mjs +3 -0
- package/evals/live/fixtures/engineering-bench/src/money.mjs +3 -0
- package/evals/live/fixtures/engineering-bench/src/pagination.mjs +5 -0
- package/evals/live/fixtures/engineering-bench/src/path-safe.mjs +5 -0
- package/evals/live/fixtures/engineering-bench/src/payment.mjs +5 -0
- package/evals/live/fixtures/engineering-bench/src/query-sort.mjs +3 -0
- package/evals/live/fixtures/engineering-bench/src/react-state.mjs +8 -0
- package/evals/live/fixtures/engineering-bench/src/retry.mjs +3 -0
- package/evals/live/fixtures/engineering-bench/src/rn-platform.mjs +3 -0
- package/evals/live/fixtures/engineering-bench/src/upload.mjs +3 -0
- package/evals/live/fixtures/engineering-bench/src/webhook.mjs +3 -0
- package/evals/live/graders/engineering-bench.mjs +239 -0
- package/evals/live/tasks.json +126 -0
- package/evals/long/fixtures/long-horizon/package.json +1 -0
- package/evals/long/fixtures/long-horizon/src/auth.mjs +5 -0
- package/evals/long/fixtures/long-horizon/src/checkout.mjs +9 -0
- package/evals/long/fixtures/long-horizon/src/inventory.mjs +5 -0
- package/evals/long/fixtures/long-horizon/src/money.mjs +3 -0
- package/evals/long/fixtures/long-horizon/src/payment.mjs +7 -0
- package/evals/long/fixtures/long-horizon/src/product-api.mjs +10 -0
- package/evals/long/fixtures/long-horizon/src/product-cache.mjs +3 -0
- package/evals/long/fixtures/long-horizon/src/product-service.mjs +6 -0
- package/evals/long/fixtures/long-horizon/src/product-state.mjs +8 -0
- package/evals/long/fixtures/long-horizon/src/project-api.mjs +9 -0
- package/evals/long/fixtures/long-horizon/src/project-service.mjs +7 -0
- package/evals/long/fixtures/long-horizon/src/user-consumer.mjs +3 -0
- package/evals/long/fixtures/long-horizon/src/user-migration.mjs +7 -0
- package/evals/long/fixtures/long-horizon/src/user-serializer.mjs +3 -0
- package/evals/long/fixtures/long-horizon/src/user-validation.mjs +3 -0
- package/evals/long/graders/long-horizon.mjs +209 -0
- package/evals/long/tasks.json +36 -0
- package/evals/router-triggers.json +1238 -0
- package/evals/routing.json +321 -0
- package/global-config/AGENTS.md +212 -0
- package/global-config/agents/architect.md +38 -0
- package/global-config/agents/codebase-mapper.md +41 -0
- package/global-config/agents/critic.md +37 -0
- package/global-config/agents/debugger.md +36 -0
- package/global-config/agents/executor.md +48 -0
- package/global-config/agents/integration-verifier.md +43 -0
- package/global-config/agents/plan-checker.md +43 -0
- package/global-config/agents/researcher.md +30 -0
- package/global-config/agents/reviewer.md +30 -0
- package/global-config/agents/verifier.md +33 -0
- package/global-config/commands/audit.md +8 -0
- package/global-config/commands/critique.md +8 -0
- package/global-config/commands/debug.md +8 -0
- package/global-config/commands/feature.md +8 -0
- package/global-config/commands/fix.md +8 -0
- package/global-config/commands/plan.md +8 -0
- package/global-config/commands/research.md +8 -0
- package/global-config/commands/resume.md +22 -0
- package/global-config/commands/review.md +8 -0
- package/global-config/commands/run.md +29 -0
- package/global-config/commands/verify.md +8 -0
- package/global-config/plugins/ues-router/capabilities.js +20 -0
- package/global-config/plugins/ues-router/index.js +277 -0
- package/global-config/plugins/ues-router/router.js +74 -0
- package/global-config/plugins/ues-router/safety.js +16 -0
- package/global-config/skills/accessibility/SKILL.md +10 -0
- package/global-config/skills/accessibility/references/workflow.md +17 -0
- package/global-config/skills/api-contract/SKILL.md +12 -0
- package/global-config/skills/api-contract/references/workflow.md +17 -0
- package/global-config/skills/auth-security/SKILL.md +12 -0
- package/global-config/skills/auth-security/references/workflow.md +15 -0
- package/global-config/skills/bug-diagnosis/SKILL.md +28 -0
- package/global-config/skills/change-impact-analysis/SKILL.md +23 -0
- package/global-config/skills/code-review/SKILL.md +19 -0
- package/global-config/skills/context-engineering/SKILL.md +18 -0
- package/global-config/skills/context-engineering/references/large-repo.md +16 -0
- package/global-config/skills/database-engineering/SKILL.md +12 -0
- package/global-config/skills/database-engineering/references/workflow.md +15 -0
- package/global-config/skills/dependency-management/SKILL.md +12 -0
- package/global-config/skills/dependency-management/references/workflow.md +14 -0
- package/global-config/skills/devops-engineering/SKILL.md +10 -0
- package/global-config/skills/devops-engineering/references/workflow.md +11 -0
- package/global-config/skills/django-engineering/SKILL.md +10 -0
- package/global-config/skills/django-engineering/references/workflow.md +11 -0
- package/global-config/skills/documentation-engineering/SKILL.md +10 -0
- package/global-config/skills/documentation-engineering/references/workflow.md +15 -0
- package/global-config/skills/dotnet-engineering/SKILL.md +10 -0
- package/global-config/skills/dotnet-engineering/references/workflow.md +11 -0
- package/global-config/skills/ecommerce-engineering/SKILL.md +10 -0
- package/global-config/skills/ecommerce-engineering/references/workflow.md +17 -0
- package/global-config/skills/engineering-orchestrator/SKILL.md +29 -0
- package/global-config/skills/engineering-orchestrator/references/delegation.md +22 -0
- package/global-config/skills/engineering-orchestrator/references/evaluator-loop.md +18 -0
- package/global-config/skills/engineering-orchestrator/references/long-horizon.md +57 -0
- package/global-config/skills/engineering-orchestrator/references/model-escalation.md +19 -0
- package/global-config/skills/engineering-orchestrator/references/retry-policy.md +12 -0
- package/global-config/skills/engineering-orchestrator/references/routing.md +39 -0
- package/global-config/skills/engineering-orchestrator/references/verification-matrix.md +18 -0
- package/global-config/skills/fastapi-engineering/SKILL.md +10 -0
- package/global-config/skills/fastapi-engineering/references/workflow.md +13 -0
- package/global-config/skills/file-upload-engineering/SKILL.md +10 -0
- package/global-config/skills/file-upload-engineering/references/workflow.md +17 -0
- package/global-config/skills/flutter-engineering/SKILL.md +10 -0
- package/global-config/skills/flutter-engineering/references/workflow.md +13 -0
- package/global-config/skills/git-safety/SKILL.md +10 -0
- package/global-config/skills/git-safety/references/workflow.md +15 -0
- package/global-config/skills/implementation-engineer/SKILL.md +10 -0
- package/global-config/skills/implementation-engineer/references/workflow.md +14 -0
- package/global-config/skills/java-spring-engineering/SKILL.md +10 -0
- package/global-config/skills/java-spring-engineering/references/workflow.md +13 -0
- package/global-config/skills/long-task-state/SKILL.md +28 -0
- package/global-config/skills/long-task-state/references/context-ledger.md +27 -0
- package/global-config/skills/long-task-state/templates/STATE.md +48 -0
- package/global-config/skills/nestjs-engineering/SKILL.md +10 -0
- package/global-config/skills/nestjs-engineering/references/workflow.md +11 -0
- package/global-config/skills/nextjs-engineering/SKILL.md +12 -0
- package/global-config/skills/nextjs-engineering/references/workflow.md +15 -0
- package/global-config/skills/nodejs-engineering/SKILL.md +12 -0
- package/global-config/skills/nodejs-engineering/references/workflow.md +11 -0
- package/global-config/skills/payment-engineering/SKILL.md +12 -0
- package/global-config/skills/payment-engineering/references/workflow.md +19 -0
- package/global-config/skills/performance-engineering/SKILL.md +10 -0
- package/global-config/skills/performance-engineering/references/workflow.md +17 -0
- package/global-config/skills/python-engineering/SKILL.md +10 -0
- package/global-config/skills/python-engineering/references/workflow.md +11 -0
- package/global-config/skills/react-engineering/SKILL.md +14 -0
- package/global-config/skills/react-engineering/references/workflow.md +16 -0
- package/global-config/skills/react-native-engineering/SKILL.md +14 -0
- package/global-config/skills/react-native-engineering/references/workflow.md +16 -0
- package/global-config/skills/repo-explorer/SKILL.md +17 -0
- package/global-config/skills/research-verification/SKILL.md +18 -0
- package/global-config/skills/research-verification/references/source-hierarchy.md +12 -0
- package/global-config/skills/rest-api-design/SKILL.md +10 -0
- package/global-config/skills/rest-api-design/references/workflow.md +18 -0
- package/global-config/skills/software-architect/SKILL.md +10 -0
- package/global-config/skills/software-architect/references/workflow.md +18 -0
- package/global-config/skills/task-planner/SKILL.md +21 -0
- package/global-config/skills/task-planner/references/plan-schema.md +56 -0
- package/global-config/skills/test-driven-development/SKILL.md +22 -0
- package/global-config/skills/test-driven-development/references/writing-good-tests.md +21 -0
- package/global-config/skills/test-verification/SKILL.md +20 -0
- package/global-config/skills/ui-ux-engineering/SKILL.md +10 -0
- package/global-config/skills/ui-ux-engineering/references/workflow.md +17 -0
- package/global-config/skills/web-security-review/SKILL.md +10 -0
- package/global-config/skills/web-security-review/references/workflow.md +22 -0
- package/lib/context-manifest.mjs +101 -0
- package/lib/control-center.mjs +148 -0
- package/lib/eval-auth.mjs +21 -0
- package/lib/eval-report.mjs +72 -0
- package/lib/eval-telemetry.mjs +155 -0
- package/lib/evidence-receipt.mjs +50 -0
- package/lib/hermes-bridge.mjs +28 -0
- package/lib/ids.mjs +21 -0
- package/lib/installer.mjs +636 -0
- package/lib/learning-engine.mjs +161 -0
- package/lib/model-config.mjs +88 -0
- package/lib/model-policy.mjs +71 -0
- package/lib/opencode-compat.mjs +86 -0
- package/lib/orchestrator-policy.mjs +35 -0
- package/lib/process-runner.mjs +117 -0
- package/lib/repo-graph.mjs +135 -0
- package/lib/repo-inspect.mjs +264 -0
- package/lib/review-scope.mjs +98 -0
- package/lib/router-config.mjs +40 -0
- package/lib/task-engine.mjs +777 -0
- package/lib/task-graph.mjs +262 -0
- package/lib/update-resolver.mjs +43 -0
- package/lib/verification-plan.mjs +52 -0
- package/lib/version.mjs +47 -0
- package/lib/workspace-snapshot.mjs +45 -0
- package/lib/worktree-sandbox.mjs +59 -0
- package/package.json +69 -0
- package/scripts/check-working-tree.mjs +2 -0
- package/scripts/collect-evidence.mjs +2 -0
- package/scripts/control-center.mjs +37 -0
- package/scripts/detect-stack.mjs +2 -0
- package/scripts/detect-test-commands.mjs +2 -0
- package/scripts/eval-live.mjs +437 -0
- package/scripts/eval-report.mjs +49 -0
- package/scripts/eval-router.mjs +45 -0
- package/scripts/eval-skills.mjs +71 -0
- package/scripts/impact-map.mjs +4 -0
- package/scripts/install.mjs +61 -0
- package/scripts/repo-map.mjs +2 -0
- package/scripts/smoke-packed-install.mjs +326 -0
- package/scripts/syntax-check.mjs +35 -0
- package/scripts/uninstall.mjs +21 -0
- package/scripts/validate-live-suite.mjs +65 -0
- package/scripts/validate.mjs +133 -0
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
# UES evaluation trace schema
|
|
2
|
+
|
|
3
|
+
Live evaluations write machine-readable JSON under `.ues-evals/`.
|
|
4
|
+
|
|
5
|
+
Each result item contains:
|
|
6
|
+
|
|
7
|
+
- `task`
|
|
8
|
+
- `mode`: `baseline` or `ues`
|
|
9
|
+
- `model` and optional `variant`
|
|
10
|
+
- `trial`
|
|
11
|
+
- `passed`
|
|
12
|
+
- `agentExit` and `graderExit`
|
|
13
|
+
- `durationMs`
|
|
14
|
+
- `authMode`: `env-only` or `current`
|
|
15
|
+
- `changedFiles`: added/removed/modified workspace paths
|
|
16
|
+
- bounded agent/grader stdout and stderr
|
|
17
|
+
- optional kept workspace path when `--keep` is used
|
|
18
|
+
- timestamp
|
|
19
|
+
|
|
20
|
+
## Telemetry
|
|
21
|
+
|
|
22
|
+
`telemetry` currently has schema version 1:
|
|
23
|
+
|
|
24
|
+
```json
|
|
25
|
+
{
|
|
26
|
+
"schemaVersion": 1,
|
|
27
|
+
"format": "best-effort-opencode-jsonl",
|
|
28
|
+
"jsonLines": 42,
|
|
29
|
+
"parseErrors": 0,
|
|
30
|
+
"toolCalls": 12,
|
|
31
|
+
"tools": {
|
|
32
|
+
"bash": 4,
|
|
33
|
+
"read": 5,
|
|
34
|
+
"skill": 2,
|
|
35
|
+
"subagent": 1
|
|
36
|
+
},
|
|
37
|
+
"skillsLoaded": ["ues-bug-diagnosis"],
|
|
38
|
+
"subagents": ["ues-verifier"],
|
|
39
|
+
"tokens": {
|
|
40
|
+
"input": 12000,
|
|
41
|
+
"output": 2200,
|
|
42
|
+
"total": 14200
|
|
43
|
+
},
|
|
44
|
+
"cost": 0.18
|
|
45
|
+
}
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
OpenCode JSON event shapes can evolve, so telemetry extraction is best-effort. Hidden-grader correctness and process exit status remain the primary benchmark evidence.
|
|
49
|
+
|
|
50
|
+
The trace intentionally does not collect or score hidden chain-of-thought.
|
|
51
|
+
|
|
52
|
+
## Run summary
|
|
53
|
+
|
|
54
|
+
Each result file also contains:
|
|
55
|
+
- suite version
|
|
56
|
+
- auth mode
|
|
57
|
+
- selected model/variant
|
|
58
|
+
- trial count
|
|
59
|
+
- optional task filter
|
|
60
|
+
- modes executed
|
|
61
|
+
- pass counts and pass rates per mode
|
|
62
|
+
|
|
63
|
+
Use `ocskill eval-report` or `npm run evals:report -- <paths>` to aggregate multiple result files.
|
|
64
|
+
|
|
65
|
+
## Fair comparisons
|
|
66
|
+
|
|
67
|
+
Keep constant:
|
|
68
|
+
- model and variant
|
|
69
|
+
- task fixture
|
|
70
|
+
- prompt
|
|
71
|
+
- grader
|
|
72
|
+
- runtime/provider environment
|
|
73
|
+
- trial count when possible
|
|
74
|
+
|
|
75
|
+
Compare observable success, regressions, elapsed time, tool behavior and cost rather than narrative confidence.
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
## V7 runtime fields
|
|
79
|
+
|
|
80
|
+
Each live result may additionally contain:
|
|
81
|
+
|
|
82
|
+
- `timedOut`
|
|
83
|
+
- `idleTimedOut`
|
|
84
|
+
- `aborted`
|
|
85
|
+
- detected `opencodeVersion` / `opencodeMajor`
|
|
86
|
+
- configured heartbeat/hard/idle timeout values
|
|
87
|
+
- long-suite receipt coverage inside orchestration inspection
|
|
88
|
+
|
|
89
|
+
Long-task `EVIDENCE.json` schema 3 may contain a `receipts` array. Receipt fields include:
|
|
90
|
+
|
|
91
|
+
```json
|
|
92
|
+
{
|
|
93
|
+
"schemaVersion": 1,
|
|
94
|
+
"id": "uuid",
|
|
95
|
+
"task": "T1",
|
|
96
|
+
"runId": "attempt-uuid",
|
|
97
|
+
"command": "npm",
|
|
98
|
+
"args": ["test"],
|
|
99
|
+
"exitCode": 0,
|
|
100
|
+
"passed": true,
|
|
101
|
+
"durationMs": 1234,
|
|
102
|
+
"stdoutSha256": "...",
|
|
103
|
+
"stderrSha256": "...",
|
|
104
|
+
"workspaceBefore": "...",
|
|
105
|
+
"workspaceAfter": "..."
|
|
106
|
+
}
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
Full stdout/stderr are not stored in receipts; hashes provide binding without persisting potentially sensitive logs.
|
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
# UES 7.7 Intelligence Runtime
|
|
2
|
+
|
|
3
|
+
UES 7.7 upgrades the V6 long-horizon harness into a more observable, crash-safe and adaptive engineering runtime. The model is still the model; UES improves how work is decomposed, bounded, verified, recovered and learned from.
|
|
4
|
+
|
|
5
|
+
## V7.0 — Runtime reliability
|
|
6
|
+
|
|
7
|
+
- live evals use an async process runner with periodic heartbeats
|
|
8
|
+
- hard timeout and idle timeout are separate
|
|
9
|
+
- Ctrl+C aborts the active OpenCode process tree instead of leaving an orphan
|
|
10
|
+
- durable tasks carry `runId`, owner metadata, heartbeat time and lease expiry
|
|
11
|
+
- stale running tasks can be recovered with `ocskill work recover`
|
|
12
|
+
- `ocskill work resume` performs stale-lease recovery before reporting ready work
|
|
13
|
+
- the V2 plugin probes actual session/runtime capabilities before fresh dispatch
|
|
14
|
+
|
|
15
|
+
Example:
|
|
16
|
+
|
|
17
|
+
```bash
|
|
18
|
+
ocskill work status checkout .
|
|
19
|
+
ocskill work recover checkout .
|
|
20
|
+
ocskill work heartbeat checkout T1 . --run-id <run-id>
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
## V7.1 — Structured evidence
|
|
24
|
+
|
|
25
|
+
`ocskill work verify-command` executes a concrete verification command and records a receipt in `EVIDENCE.json`.
|
|
26
|
+
|
|
27
|
+
A receipt contains:
|
|
28
|
+
|
|
29
|
+
- command and arguments
|
|
30
|
+
- exit code and pass/fail
|
|
31
|
+
- start/end/duration
|
|
32
|
+
- SHA-256 of stdout/stderr rather than full output
|
|
33
|
+
- workspace fingerprint before and after
|
|
34
|
+
- task/runId binding
|
|
35
|
+
|
|
36
|
+
This complements narrative evidence. Task state reports whether completed work is `receipt-backed` or `narrative`.
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
ocskill work verify-command checkout T1 . --run-id <run-id> -- npm test
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
## V7.2 — Context intelligence
|
|
43
|
+
|
|
44
|
+
Fresh executor context packs now include a bounded context manifest:
|
|
45
|
+
|
|
46
|
+
- declared task files
|
|
47
|
+
- local import neighbors and reverse importers
|
|
48
|
+
- likely related tests
|
|
49
|
+
- top-level repository instruction/manifests
|
|
50
|
+
- bounded source excerpts
|
|
51
|
+
- repository graph hotspots
|
|
52
|
+
- accepted learning items relevant to the task
|
|
53
|
+
|
|
54
|
+
The goal is to reduce rediscovery cost without dumping the whole repository into one prompt.
|
|
55
|
+
|
|
56
|
+
## V7.3 — Adaptive orchestration/model policy
|
|
57
|
+
|
|
58
|
+
`ocskill task-policy <text>` deterministically classifies work by complexity/risk and recommends:
|
|
59
|
+
|
|
60
|
+
- `inline`, `standard` or `long-horizon`
|
|
61
|
+
- light/standard/heavy model tier
|
|
62
|
+
- context budget
|
|
63
|
+
- retry budget
|
|
64
|
+
- whether plan/integration gates are required
|
|
65
|
+
|
|
66
|
+
`ocskill model-policy <role> --attempt N --text <task>` combines role defaults, task risk and retry escalation.
|
|
67
|
+
|
|
68
|
+
## V7.4 — Isolated parallel execution primitives
|
|
69
|
+
|
|
70
|
+
The task graph distinguishes reads from writes. By default linked worktrees are created in a sibling `.REPO.ues-sandboxes/` directory on the same drive, avoiding nested worktrees inside the main checkout:
|
|
71
|
+
|
|
72
|
+
- read/read overlap can share a safe wave
|
|
73
|
+
- write/read and write/write conflicts serialize
|
|
74
|
+
|
|
75
|
+
For parallel writers UES provides Git worktree sandboxes:
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
ocskill sandbox create <slug> <task-id> .
|
|
79
|
+
ocskill sandbox list .
|
|
80
|
+
ocskill sandbox remove <worktree-path> . --force
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
Sandbox creation is deterministic infrastructure. Integration/merging remains an explicit parent responsibility; UES does not silently merge branches.
|
|
84
|
+
|
|
85
|
+
## V7.5 — Evidence-gated learning loop
|
|
86
|
+
|
|
87
|
+
UES can mine its own eval traces for recurring failure patterns:
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
ocskill learn analyze . --eval-dir .ues-evals
|
|
91
|
+
ocskill learn status .
|
|
92
|
+
ocskill learn accept <proposal-id> .
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
Accepted lessons can appear in later context packs when relevant. UES never auto-edits skills or promotes a lesson from a single run without an explicit accept step.
|
|
96
|
+
|
|
97
|
+
Current deterministic proposal classes include:
|
|
98
|
+
|
|
99
|
+
- hard timeout
|
|
100
|
+
- idle timeout
|
|
101
|
+
- agent process failure
|
|
102
|
+
- hidden grader failure
|
|
103
|
+
- orchestration failure
|
|
104
|
+
- telemetry parse drift
|
|
105
|
+
|
|
106
|
+
## V7.6 — Optional Hermes bridge
|
|
107
|
+
|
|
108
|
+
Hermes is treated as an optional external executor, not embedded as another runtime layer.
|
|
109
|
+
|
|
110
|
+
```bash
|
|
111
|
+
ocskill hermes status
|
|
112
|
+
ocskill hermes prompt <slug> <task-id> .
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
The bridge emits a bounded UES delegation prompt. Hermes must not independently mutate durable UES state, merge, push, publish or deploy.
|
|
116
|
+
|
|
117
|
+
## V7.7 — Control Center
|
|
118
|
+
|
|
119
|
+
Generate a zero-dependency local dashboard:
|
|
120
|
+
|
|
121
|
+
```bash
|
|
122
|
+
ocskill dashboard .
|
|
123
|
+
ocskill dashboard . --serve --port 4177
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
When served with `--serve`, the Control Center refreshes its data every few seconds without requiring a frontend build.
|
|
127
|
+
|
|
128
|
+
The Control Center summarizes:
|
|
129
|
+
|
|
130
|
+
- long-horizon work items and task status
|
|
131
|
+
- attempts, blockers and receipt count
|
|
132
|
+
- learning proposals/accepted lessons
|
|
133
|
+
- recent baseline/UES evaluation summaries
|
|
134
|
+
|
|
135
|
+
Generated state lives under `.ues-dashboard/` and is git-ignored.
|
|
136
|
+
|
|
137
|
+
## Live benchmark observability
|
|
138
|
+
|
|
139
|
+
The live harness accepts:
|
|
140
|
+
|
|
141
|
+
```bash
|
|
142
|
+
node scripts/eval-live.mjs \
|
|
143
|
+
--suite long \
|
|
144
|
+
--model provider/model \
|
|
145
|
+
--mode both \
|
|
146
|
+
--trials 3 \
|
|
147
|
+
--heartbeat-ms 30000 \
|
|
148
|
+
--idle-timeout-ms 300000 \
|
|
149
|
+
--timeout-ms 900000
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
Every active trial prints a start line and periodic heartbeat rather than appearing frozen.
|
|
153
|
+
|
|
154
|
+
## Safety boundaries
|
|
155
|
+
|
|
156
|
+
V7.7 deliberately keeps these actions explicit:
|
|
157
|
+
|
|
158
|
+
- merging sandbox branches
|
|
159
|
+
- forceful Git/history actions
|
|
160
|
+
- publishing/releases
|
|
161
|
+
- deployment
|
|
162
|
+
- destructive database/filesystem operations
|
|
163
|
+
- automatic promotion of learned rules
|
|
164
|
+
- automatic execution through Hermes
|
|
165
|
+
|
|
166
|
+
The goal is a stronger runtime without removing human control over irreversible side effects.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"name":"ues-engineering-bench","private":true,"type":"module"}
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
export const initialState = { requestId: 0, loading: false, data: null, error: null }
|
|
2
|
+
|
|
3
|
+
export function reducer(state, action) {
|
|
4
|
+
if (action.type === "start") return { ...state, loading: true, requestId: action.requestId }
|
|
5
|
+
if (action.type === "success") return { ...state, loading: false, data: action.data }
|
|
6
|
+
if (action.type === "error") return { ...state, loading: false, error: action.error }
|
|
7
|
+
return state
|
|
8
|
+
}
|
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
import assert from "node:assert/strict"
|
|
2
|
+
import path from "node:path"
|
|
3
|
+
import { pathToFileURL } from "node:url"
|
|
4
|
+
|
|
5
|
+
const workspace = process.env.UES_EVAL_WORKSPACE
|
|
6
|
+
const task = process.env.UES_EVAL_TASK
|
|
7
|
+
if (!workspace || !task) {
|
|
8
|
+
console.error("UES_EVAL_WORKSPACE and UES_EVAL_TASK are required")
|
|
9
|
+
process.exit(2)
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
const load = async (file) => import(pathToFileURL(path.join(workspace, "src", file)).href + `?eval=${Date.now()}-${Math.random()}`)
|
|
13
|
+
const throws = (fn, ErrorType) => assert.throws(fn, ErrorType)
|
|
14
|
+
const close = (actual, expected) => assert.ok(Math.abs(actual - expected) < 1e-9, `expected ${actual} ~= ${expected}`)
|
|
15
|
+
|
|
16
|
+
switch (task) {
|
|
17
|
+
case "js-discount-regression": {
|
|
18
|
+
const { calculateDiscount } = await load("discount.mjs")
|
|
19
|
+
assert.equal(calculateDiscount(100, 20), 80)
|
|
20
|
+
close(calculateDiscount(199.99, 15), 169.9915)
|
|
21
|
+
assert.equal(calculateDiscount(42, 0), 42)
|
|
22
|
+
assert.equal(calculateDiscount(42, 100), 0)
|
|
23
|
+
for (const args of [[NaN, 10], [100, Infinity], ["100", 10]]) throws(() => calculateDiscount(...args), TypeError)
|
|
24
|
+
for (const percent of [-1, 101]) throws(() => calculateDiscount(100, percent), RangeError)
|
|
25
|
+
break
|
|
26
|
+
}
|
|
27
|
+
case "auth-resource-ownership": {
|
|
28
|
+
const { canEditResource } = await load("authz.mjs")
|
|
29
|
+
const own = { id: "u1", tenantId: "t1", role: "member" }
|
|
30
|
+
const resource = { ownerId: "u1", tenantId: "t1" }
|
|
31
|
+
assert.equal(canEditResource(own, resource), true)
|
|
32
|
+
assert.equal(canEditResource({ ...own, id: "u2" }, resource), false)
|
|
33
|
+
assert.equal(canEditResource({ ...own, tenantId: "t2" }, resource), false)
|
|
34
|
+
assert.equal(canEditResource({ ...own, role: "admin", id: "x", tenantId: "x" }, resource), true)
|
|
35
|
+
assert.equal(canEditResource({ ...own, role: "admin", suspended: true }, resource), false)
|
|
36
|
+
assert.equal(canEditResource(null, resource), false)
|
|
37
|
+
assert.equal(canEditResource(own, null), false)
|
|
38
|
+
break
|
|
39
|
+
}
|
|
40
|
+
case "api-pagination-contract": {
|
|
41
|
+
const { paginate } = await load("pagination.mjs")
|
|
42
|
+
const items = Array.from({ length: 23 }, (_, i) => i + 1)
|
|
43
|
+
assert.deepEqual(paginate(items, { page: 2, pageSize: 5 }), { items: [6,7,8,9,10], total: 23, page: 2, pageSize: 5, totalPages: 5 })
|
|
44
|
+
assert.deepEqual(paginate(items, {}), { items: items.slice(0,10), total: 23, page: 1, pageSize: 10, totalPages: 3 })
|
|
45
|
+
assert.deepEqual(paginate(items, { page: 99, pageSize: 10 }).items, [])
|
|
46
|
+
throws(() => paginate("x", {}), TypeError)
|
|
47
|
+
throws(() => paginate(items, { page: 0 }), RangeError)
|
|
48
|
+
throws(() => paginate(items, { pageSize: 1.5 }), TypeError)
|
|
49
|
+
break
|
|
50
|
+
}
|
|
51
|
+
case "inventory-reservation": {
|
|
52
|
+
const { reserveStock } = await load("inventory.mjs")
|
|
53
|
+
assert.equal(reserveStock(10, 3), 7)
|
|
54
|
+
assert.equal(reserveStock(1, 1), 0)
|
|
55
|
+
throws(() => reserveStock(2, 3), RangeError)
|
|
56
|
+
throws(() => reserveStock(-1, 1), RangeError)
|
|
57
|
+
throws(() => reserveStock(2, 0), RangeError)
|
|
58
|
+
throws(() => reserveStock(2.5, 1), TypeError)
|
|
59
|
+
throws(() => reserveStock("2", 1), TypeError)
|
|
60
|
+
break
|
|
61
|
+
}
|
|
62
|
+
case "payment-idempotency": {
|
|
63
|
+
const { applyPaymentEvent } = await load("payment.mjs")
|
|
64
|
+
const order = { status: "pending", totalCents: 1500, currency: "USD", processedEvents: [] }
|
|
65
|
+
const paid = applyPaymentEvent(order, { id: "evt1", status: "succeeded", amountCents: 1500, currency: "USD" })
|
|
66
|
+
assert.equal(order.status, "pending")
|
|
67
|
+
assert.deepEqual(order.processedEvents, [])
|
|
68
|
+
assert.equal(paid.status, "paid")
|
|
69
|
+
assert.deepEqual(paid.processedEvents, ["evt1"])
|
|
70
|
+
const duplicate = applyPaymentEvent(paid, { id: "evt1", status: "succeeded", amountCents: 1500, currency: "USD" })
|
|
71
|
+
assert.deepEqual(duplicate, paid)
|
|
72
|
+
const failed = applyPaymentEvent(order, { id: "evt2", status: "failed", amountCents: 1500, currency: "USD" })
|
|
73
|
+
assert.equal(failed.status, "pending")
|
|
74
|
+
assert.deepEqual(failed.processedEvents, ["evt2"])
|
|
75
|
+
throws(() => applyPaymentEvent(order, { id: "evt3", status: "succeeded", amountCents: 1499, currency: "USD" }), RangeError)
|
|
76
|
+
throws(() => applyPaymentEvent(order, { id: "", status: "failed" }), TypeError)
|
|
77
|
+
break
|
|
78
|
+
}
|
|
79
|
+
case "webhook-ordering": {
|
|
80
|
+
const { advancePaymentState } = await load("webhook.mjs")
|
|
81
|
+
const pending = { status: "pending", lastSequence: 1 }
|
|
82
|
+
assert.deepEqual(advancePaymentState(pending, { status: "authorized", sequence: 2 }), { status: "authorized", lastSequence: 2 })
|
|
83
|
+
assert.equal(advancePaymentState(pending, { status: "paid", sequence: 1 }), pending)
|
|
84
|
+
const paid = advancePaymentState({ status: "authorized", lastSequence: 2 }, { status: "paid", sequence: 3 })
|
|
85
|
+
assert.deepEqual(paid, { status: "paid", lastSequence: 3 })
|
|
86
|
+
assert.deepEqual(advancePaymentState(paid, { status: "refunded", sequence: 4 }), { status: "refunded", lastSequence: 4 })
|
|
87
|
+
throws(() => advancePaymentState(pending, { status: "refunded", sequence: 2 }), RangeError)
|
|
88
|
+
break
|
|
89
|
+
}
|
|
90
|
+
case "secure-file-upload": {
|
|
91
|
+
const { validateUpload } = await load("upload.mjs")
|
|
92
|
+
const opts = { maxBytes: 1024, allowedTypes: ["image/png", "image/jpeg"] }
|
|
93
|
+
assert.equal(validateUpload({ name: "photo.png", type: "image/png", size: 100 }, opts), true)
|
|
94
|
+
throws(() => validateUpload({ name: "../x.png", type: "image/png", size: 100 }, opts), TypeError)
|
|
95
|
+
throws(() => validateUpload({ name: "a/b.png", type: "image/png", size: 100 }, opts), TypeError)
|
|
96
|
+
throws(() => validateUpload({ name: "x.exe", type: "application/octet-stream", size: 100 }, opts), TypeError)
|
|
97
|
+
throws(() => validateUpload({ name: "x.png", type: "image/png", size: 2048 }, opts), RangeError)
|
|
98
|
+
throws(() => validateUpload({ name: "x.png", type: "image/png", size: 1.2 }, opts), TypeError)
|
|
99
|
+
break
|
|
100
|
+
}
|
|
101
|
+
case "path-traversal-defense": {
|
|
102
|
+
const { safeJoin } = await load("path-safe.mjs")
|
|
103
|
+
const root = path.resolve(workspace, "uploads")
|
|
104
|
+
assert.equal(safeJoin(root, "user/file.txt"), path.resolve(root, "user/file.txt"))
|
|
105
|
+
throws(() => safeJoin(root, "../secret.txt"), RangeError)
|
|
106
|
+
throws(() => safeJoin(root, path.resolve(root, "..", "secret.txt")), RangeError)
|
|
107
|
+
throws(() => safeJoin(root, "x\0y"), RangeError)
|
|
108
|
+
throws(() => safeJoin(root, 42), TypeError)
|
|
109
|
+
break
|
|
110
|
+
}
|
|
111
|
+
case "stable-api-errors": {
|
|
112
|
+
const { toHttpError } = await load("api-errors.mjs")
|
|
113
|
+
assert.deepEqual(toHttpError({ code: "NOT_FOUND", message: "Missing", stack: "secret" }), { status: 404, body: { error: { code: "NOT_FOUND", message: "Missing" } } })
|
|
114
|
+
assert.equal(toHttpError({ code: "VALIDATION", message: "Bad" }).status, 400)
|
|
115
|
+
assert.equal(toHttpError({ code: "FORBIDDEN", message: "No" }).status, 403)
|
|
116
|
+
assert.equal(toHttpError({ code: "CONFLICT", message: "Dup" }).status, 409)
|
|
117
|
+
assert.deepEqual(toHttpError(new Error("db secret")), { status: 500, body: { error: { code: "INTERNAL", message: "Internal server error" } } })
|
|
118
|
+
break
|
|
119
|
+
}
|
|
120
|
+
case "http-retry-policy": {
|
|
121
|
+
const { shouldRetry } = await load("retry.mjs")
|
|
122
|
+
assert.equal(shouldRetry({ status: 429, attempt: 1, maxAttempts: 3 }), true)
|
|
123
|
+
assert.equal(shouldRetry({ status: 503, attempt: 2, maxAttempts: 3 }), true)
|
|
124
|
+
assert.equal(shouldRetry({ status: 500, attempt: 3, maxAttempts: 3 }), false)
|
|
125
|
+
assert.equal(shouldRetry({ status: 400, attempt: 1, maxAttempts: 3 }), false)
|
|
126
|
+
throws(() => shouldRetry({ status: 500, attempt: 0, maxAttempts: 3 }), RangeError)
|
|
127
|
+
throws(() => shouldRetry({ status: "500", attempt: 1, maxAttempts: 3 }), TypeError)
|
|
128
|
+
break
|
|
129
|
+
}
|
|
130
|
+
case "dedupe-stable-order": {
|
|
131
|
+
const { dedupeById } = await load("dedupe.mjs")
|
|
132
|
+
const input = [{id:1,v:"a"},{id:1,v:"b"},{id:"1",v:"c"},{id:2,v:"d"}]
|
|
133
|
+
const out = dedupeById(input)
|
|
134
|
+
assert.deepEqual(out, [{id:1,v:"a"},{id:"1",v:"c"},{id:2,v:"d"}])
|
|
135
|
+
assert.equal(input.length, 4)
|
|
136
|
+
throws(() => dedupeById([{id:null}]), TypeError)
|
|
137
|
+
throws(() => dedupeById("x"), TypeError)
|
|
138
|
+
break
|
|
139
|
+
}
|
|
140
|
+
case "legacy-user-migration": {
|
|
141
|
+
const { migrateUsers } = await load("migration.mjs")
|
|
142
|
+
const rows = [
|
|
143
|
+
{ id: 1, fullName: " Ada Lovelace ", extra: true },
|
|
144
|
+
{ id: 2, fullName: "Prince" },
|
|
145
|
+
{ id: 3, fullName: "", x: 1 },
|
|
146
|
+
{ id: 4, fullName: "Ignored", firstName: "Existing", lastName: "Name" },
|
|
147
|
+
]
|
|
148
|
+
const out = migrateUsers(rows)
|
|
149
|
+
assert.deepEqual(out, [
|
|
150
|
+
{ id: 1, firstName: "Ada", lastName: "Lovelace", extra: true },
|
|
151
|
+
{ id: 2, firstName: "Prince", lastName: "" },
|
|
152
|
+
{ id: 3, firstName: "", lastName: "", x: 1 },
|
|
153
|
+
{ id: 4, firstName: "Existing", lastName: "Name" },
|
|
154
|
+
])
|
|
155
|
+
assert.equal(rows[0].fullName.trim().startsWith("Ada"), true)
|
|
156
|
+
break
|
|
157
|
+
}
|
|
158
|
+
case "money-integer-invariants": {
|
|
159
|
+
const { orderTotal } = await load("money.mjs")
|
|
160
|
+
assert.equal(orderTotal([{ unitPriceCents: 250, quantity: 2 }, { unitPriceCents: 100, quantity: 3, discountCents: 50 }]), 750)
|
|
161
|
+
assert.equal(orderTotal([]), 0)
|
|
162
|
+
throws(() => orderTotal([{ unitPriceCents: 100, quantity: 1, discountCents: 101 }]), RangeError)
|
|
163
|
+
throws(() => orderTotal([{ unitPriceCents: 1.5, quantity: 1 }]), TypeError)
|
|
164
|
+
throws(() => orderTotal([{ unitPriceCents: 100, quantity: 0 }]), RangeError)
|
|
165
|
+
break
|
|
166
|
+
}
|
|
167
|
+
case "sql-sort-allowlist": {
|
|
168
|
+
const { buildSort } = await load("query-sort.mjs")
|
|
169
|
+
assert.equal(buildSort({ field: "createdAt", direction: "desc" }), "ORDER BY created_at DESC")
|
|
170
|
+
assert.equal(buildSort({ field: "name", direction: "ASC" }), "ORDER BY name ASC")
|
|
171
|
+
assert.equal(buildSort({ field: "price", direction: "asc" }), "ORDER BY price_cents ASC")
|
|
172
|
+
throws(() => buildSort({ field: "name; DROP TABLE users", direction: "asc" }), RangeError)
|
|
173
|
+
throws(() => buildSort({ field: "name", direction: "sideways" }), RangeError)
|
|
174
|
+
throws(() => buildSort("name"), TypeError)
|
|
175
|
+
break
|
|
176
|
+
}
|
|
177
|
+
case "react-stale-request": {
|
|
178
|
+
const { reducer, initialState } = await load("react-state.mjs")
|
|
179
|
+
const started = reducer(initialState, { type: "start", requestId: 2 })
|
|
180
|
+
assert.deepEqual(started, { requestId: 2, loading: true, data: null, error: null })
|
|
181
|
+
const stale = reducer(started, { type: "success", requestId: 1, data: "old" })
|
|
182
|
+
assert.equal(stale, started)
|
|
183
|
+
assert.deepEqual(reducer(started, { type: "success", requestId: 2, data: "new" }), { requestId: 2, loading: false, data: "new", error: null })
|
|
184
|
+
assert.deepEqual(reducer(started, { type: "error", requestId: 2, error: "boom" }), { requestId: 2, loading: false, data: null, error: "boom" })
|
|
185
|
+
break
|
|
186
|
+
}
|
|
187
|
+
case "react-native-keyboard-offset": {
|
|
188
|
+
const { keyboardOffset } = await load("rn-platform.mjs")
|
|
189
|
+
assert.equal(keyboardOffset("ios", 44, 56), 100)
|
|
190
|
+
assert.equal(keyboardOffset("android", 24, 56), 56)
|
|
191
|
+
throws(() => keyboardOffset("windows", 0, 0), RangeError)
|
|
192
|
+
throws(() => keyboardOffset("ios", -1, 10), RangeError)
|
|
193
|
+
throws(() => keyboardOffset("ios", "1", 10), TypeError)
|
|
194
|
+
break
|
|
195
|
+
}
|
|
196
|
+
case "dependency-caret-range": {
|
|
197
|
+
const { satisfiesCaret } = await load("dependency.mjs")
|
|
198
|
+
assert.equal(satisfiesCaret("1.2.3", "^1.2.3"), true)
|
|
199
|
+
assert.equal(satisfiesCaret("1.9.9", "^1.2.3"), true)
|
|
200
|
+
assert.equal(satisfiesCaret("2.0.0", "^1.2.3"), false)
|
|
201
|
+
assert.equal(satisfiesCaret("0.2.9", "^0.2.3"), true)
|
|
202
|
+
assert.equal(satisfiesCaret("0.3.0", "^0.2.3"), false)
|
|
203
|
+
assert.equal(satisfiesCaret("0.0.3", "^0.0.3"), true)
|
|
204
|
+
assert.equal(satisfiesCaret("0.0.4", "^0.0.3"), false)
|
|
205
|
+
assert.equal(satisfiesCaret("bad", "^1.2.3"), false)
|
|
206
|
+
break
|
|
207
|
+
}
|
|
208
|
+
case "strict-env-boolean": {
|
|
209
|
+
const { readBooleanEnv } = await load("config.mjs")
|
|
210
|
+
for (const value of ["true"," TRUE ","1","yes","On"]) assert.equal(readBooleanEnv(value), true)
|
|
211
|
+
for (const value of ["false"," FALSE ","0","no","off"]) assert.equal(readBooleanEnv(value, true), false)
|
|
212
|
+
assert.equal(readBooleanEnv(undefined, true), true)
|
|
213
|
+
assert.equal(readBooleanEnv("", false), false)
|
|
214
|
+
assert.equal(readBooleanEnv(true), true)
|
|
215
|
+
throws(() => readBooleanEnv("maybe"), TypeError)
|
|
216
|
+
throws(() => readBooleanEnv(1), TypeError)
|
|
217
|
+
break
|
|
218
|
+
}
|
|
219
|
+
case "cache-invalidation-tags": {
|
|
220
|
+
const { tagsForProductMutation } = await load("cache-tags.mjs")
|
|
221
|
+
assert.deepEqual(tagsForProductMutation({ id: "p1", categoryId: "c1", sellerId: "s1" }), ["products","product:p1","category:c1","seller:s1"])
|
|
222
|
+
assert.deepEqual(tagsForProductMutation({ id: "p1", categoryId: "", sellerId: null }), ["products","product:p1"])
|
|
223
|
+
throws(() => tagsForProductMutation({ id: "" }), TypeError)
|
|
224
|
+
break
|
|
225
|
+
}
|
|
226
|
+
case "multi-file-contract-compatibility": {
|
|
227
|
+
const { serializeUser } = await load("contract-producer.mjs")
|
|
228
|
+
const { userLabel } = await load("contract-consumer.mjs")
|
|
229
|
+
assert.deepEqual(serializeUser({ id: 1, name: "Legacy" }), { id: 1, name: "Legacy", displayName: "Legacy" })
|
|
230
|
+
assert.deepEqual(serializeUser({ id: 2, name: "Legacy", displayName: "Preferred" }), { id: 2, name: "Legacy", displayName: "Preferred" })
|
|
231
|
+
assert.equal(userLabel({ id: 3, name: "Legacy" }), "Legacy")
|
|
232
|
+
assert.equal(userLabel({ id: 4, name: "Legacy", displayName: "Preferred" }), "Preferred")
|
|
233
|
+
break
|
|
234
|
+
}
|
|
235
|
+
default:
|
|
236
|
+
throw new Error("Unknown live eval task: " + task)
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
console.log("hidden grader passed:", task)
|