opencode-agent-skill 11.0.0 → 12.0.0-beta.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +18 -0
- package/README.md +13 -7
- package/bin/ocskill.mjs +24 -1
- package/docs/DETERMINISTIC-TOOLS.md +1 -1
- package/docs/ENGINEERING-DESIGN.md +4 -4
- package/docs/EVALS.md +3 -3
- package/docs/GITHUB-RULESET.md +50 -0
- package/docs/NPM-PUBLISH.md +4 -4
- package/docs/OPENCODE-COMPAT.md +3 -3
- package/docs/TRACE-SCHEMA.md +1 -1
- package/docs/V11-PERCEPTION-ADAPTIVE.md +2 -2
- package/docs/V12-WEAK-MODEL-INTELLIGENCE.md +27 -0
- package/evals/repo-scale/tasks.json +62 -0
- package/lib/context-engine-v11.mjs +4 -0
- package/lib/context-quality.mjs +59 -0
- package/lib/decision-policy.mjs +23 -0
- package/lib/model-config.mjs +13 -1
- package/lib/model-performance.mjs +113 -0
- package/lib/model-policy.mjs +9 -2
- package/lib/repo-scale-fixture.mjs +45 -0
- package/lib/task-engine.mjs +17 -3
- package/lib/work-plan-scope.mjs +49 -0
- package/package.json +6 -3
- package/scripts/check-release-consistency.mjs +228 -0
- package/scripts/validate-repo-scale-suite.mjs +27 -0
- package/scripts/validate-v12-foundation.mjs +24 -0
package/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,24 @@ The project follows Semantic Versioning.
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [12.0.0-beta.0] - 2026-09-22
|
|
10
|
+
|
|
11
|
+
### Beta
|
|
12
|
+
- Added empirical per-task-class model performance history and capability-preserving reranking with a minimum evidence threshold before reranking.
|
|
13
|
+
- Added context quality receipts for required-file recall and irrelevant-context ratio.
|
|
14
|
+
- Added hash-keyed plan snapshots with active-plan execution fencing.
|
|
15
|
+
- Added bounded decision policy for reversible local rulings versus human-gated destructive/external actions.
|
|
16
|
+
- Added deterministic repo-scale benchmark fixture generation and V12/repo-scale validation gates.
|
|
17
|
+
- Hardened release consistency checks to derive eval counts and validate aggregate workflow structure.
|
|
18
|
+
- Fixed missing empirical-history handling so models without benchmark history safely fall back to static capability routing.
|
|
19
|
+
|
|
20
|
+
### Verified locally on Windows
|
|
21
|
+
- 255 tests total: 253 passed, 0 failed, 2 platform-specific skips.
|
|
22
|
+
- Syntax, catalog validation, docs consistency, routing, V11/V12/repo-scale/live/long/polyglot validation: PASS.
|
|
23
|
+
- npm pack, packed-install smoke and plain one-command install/resource auto-sync smoke: PASS.
|
|
24
|
+
- Published prerelease intent: npm dist-tag `next`; V11 remains `latest` until V12 stable release gates are satisfied.
|
|
25
|
+
|
|
26
|
+
|
|
9
27
|
## [11.0.0] - 2026-09-22
|
|
10
28
|
|
|
11
29
|
### Released
|
package/README.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# OpenCode Universal Engineering System (UES)
|
|
2
2
|
|
|
3
|
-
> **
|
|
4
|
-
>
|
|
3
|
+
> **V12 beta: 12.0.0-beta.0** — weak-model intelligence foundation với empirical routing, context-quality metrics, plan-scoped execution và repo-scale validation.
|
|
4
|
+
> **Stable latest vẫn là V11: 11.0.0.** Dùng `npm install -g opencode-agent-skill@next` để thử V12 beta.
|
|
5
5
|
|
|
6
6
|
[](https://www.npmjs.com/package/opencode-agent-skill)
|
|
7
7
|
[](LICENSE)
|
|
@@ -431,7 +431,7 @@ ocskill work status checkout .
|
|
|
431
431
|
|
|
432
432
|
## Skills
|
|
433
433
|
|
|
434
|
-
V11
|
|
434
|
+
V11 stable hiện có **48 skills**; V10 stable có 39. Router vẫn chỉ chọn tập skill phù hợp thay vì nạp toàn bộ catalog vào mỗi task.
|
|
435
435
|
|
|
436
436
|
Một số process skill quan trọng:
|
|
437
437
|
|
|
@@ -592,10 +592,11 @@ Integration sẽ từ chối ghi đè lên file đang dirty ở root. Safe-wave
|
|
|
592
592
|
|
|
593
593
|
## Evaluation
|
|
594
594
|
|
|
595
|
-
|
|
595
|
+
V11 hiện có:
|
|
596
596
|
|
|
597
|
-
- **
|
|
598
|
-
- **
|
|
597
|
+
- **43 static skill-routing scenarios** phủ 48 skills;
|
|
598
|
+
- **129 V2 router cases** với required routes và negative guards;
|
|
599
|
+
- **13 V11 contract tasks** across 6 categories;
|
|
599
600
|
- **20 standard live tasks**;
|
|
600
601
|
- **5 long-horizon tasks**;
|
|
601
602
|
- **8 polyglot tasks** cho Python, Java, .NET, Next.js, React Native, SQL migration, monorepo và generated contract;
|
|
@@ -667,11 +668,13 @@ Pipeline hiện kiểm tra:
|
|
|
667
668
|
```text
|
|
668
669
|
syntax
|
|
669
670
|
→ resource validation
|
|
671
|
+
→ docs:check
|
|
670
672
|
→ static skill routing
|
|
671
673
|
→ V2 router matrix
|
|
672
674
|
→ standard hidden-grader integrity
|
|
673
675
|
→ long hidden-grader integrity
|
|
674
676
|
→ polyglot hidden-grader integrity
|
|
677
|
+
→ V11 contract validation
|
|
675
678
|
→ Node tests
|
|
676
679
|
→ npm pack --dry-run
|
|
677
680
|
→ packed global-install smoke
|
|
@@ -712,6 +715,7 @@ UES chỉ quản lý resource có namespace/marker của chính nó và cố g
|
|
|
712
715
|
- [V7 Intelligence Runtime](docs/V7-INTELLIGENCE-RUNTIME.md)
|
|
713
716
|
- [V8 Intelligence & Reliability](docs/V8-INTELLIGENCE-RELIABILITY.md)
|
|
714
717
|
- [V9 Speed & Intelligence](docs/V9-SPEED-INTELLIGENCE.md)
|
|
718
|
+
- [V12 Weak-Model Intelligence (beta)](docs/V12-WEAK-MODEL-INTELLIGENCE.md)
|
|
715
719
|
|
|
716
720
|
---
|
|
717
721
|
|
|
@@ -732,9 +736,11 @@ npm install -g opencode-agent-skill
|
|
|
732
736
|
Phiên bản hiện tại:
|
|
733
737
|
|
|
734
738
|
```text
|
|
735
|
-
|
|
739
|
+
12.0.0-beta.0
|
|
736
740
|
```
|
|
737
741
|
|
|
742
|
+
V12 beta dùng npm dist-tag `next`; `latest` tiếp tục trỏ tới V11 stable cho đến khi các release gate V12 hoàn tất.
|
|
743
|
+
|
|
738
744
|
---
|
|
739
745
|
|
|
740
746
|
## License
|
package/bin/ocskill.mjs
CHANGED
|
@@ -65,9 +65,10 @@ import { classifyEngineeringTask } from "../lib/orchestrator-policy.mjs"
|
|
|
65
65
|
import { createTaskSandbox, integrateTaskSandbox, listTaskSandboxes, removeTaskSandbox } from "../lib/worktree-sandbox.mjs"
|
|
66
66
|
import { analyzeEvalTraces, saveLearningAnalysis, readLearningState, acceptLearning, promoteLearning } from "../lib/learning-engine.mjs"
|
|
67
67
|
import { hermesStatus, buildHermesDelegationPrompt, buildHermesWorkflowPrompt, hermesOneShotArgs, hermesSidecarPlan } from "../lib/hermes-bridge.mjs"
|
|
68
|
-
import { readModelPolicy, validateModelID, writeModelPolicy } from "../lib/model-config.mjs"
|
|
68
|
+
import { readModelPolicy, recordModelPerformance, validateModelID, writeModelPolicy } from "../lib/model-config.mjs"
|
|
69
69
|
import { evidenceStoreStatus, gcEvidenceStore, getEvidence, putEvidence } from "../lib/evidence-store.mjs"
|
|
70
70
|
import { inferTaskCapabilities } from "../lib/capability-registry.mjs"
|
|
71
|
+
import { MODEL_TASK_CLASSES } from "../lib/model-performance.mjs"
|
|
71
72
|
import { browserCapability, buildBrowserVerificationPlan } from "../lib/browser-adapter.mjs"
|
|
72
73
|
import { inspectBrowserPage, summarizeBrowserInspection } from "../lib/browser-runtime.mjs"
|
|
73
74
|
import { comparePngFiles, cropPngFile } from "../lib/png-diff.mjs"
|
|
@@ -171,6 +172,7 @@ Usage:
|
|
|
171
172
|
ocskill models set <light|standard|heavy> <provider/model[#variant]>
|
|
172
173
|
ocskill models role <role> <light|standard|heavy>
|
|
173
174
|
ocskill models capability <provider/model> [--vision on|off] [--browser on|off] [--reasoning on|off] [--long-context on|off] [--cost low|medium|high] [--latency fast|medium|slow] [--quality 0..1]
|
|
175
|
+
ocskill models observe <provider/model> --task-class <class> --passed on|off [--retries N] [--tokens N] [--latency-ms N]
|
|
174
176
|
|
|
175
177
|
--force backs up and replaces/removes state owned by another package.
|
|
176
178
|
`)
|
|
@@ -812,6 +814,27 @@ async function modelsControl() {
|
|
|
812
814
|
return
|
|
813
815
|
}
|
|
814
816
|
|
|
817
|
+
if (action === "observe") {
|
|
818
|
+
const model = args[2]
|
|
819
|
+
if (!validateModelID(model)) {
|
|
820
|
+
console.error("Usage: ocskill models observe <provider/model> --task-class <class> --passed on|off")
|
|
821
|
+
process.exitCode = 2
|
|
822
|
+
return
|
|
823
|
+
}
|
|
824
|
+
const taskClass = optionValue(args, "--task-class") || "general"
|
|
825
|
+
if (!MODEL_TASK_CLASSES.includes(taskClass)) throw new Error("--task-class must be one of: " + MODEL_TASK_CLASSES.join(", "))
|
|
826
|
+
const passedRaw = String(optionValue(args, "--passed") || "").toLowerCase()
|
|
827
|
+
if (!["on","off","true","false","pass","fail"].includes(passedRaw)) throw new Error("--passed must be on/off, true/false, or pass/fail")
|
|
828
|
+
policy = await recordModelPerformance(getConfigDir(), {
|
|
829
|
+
model, taskClass, passed: ["on","true","pass"].includes(passedRaw),
|
|
830
|
+
retries: optionInt(args, "--retries", 0) || 0,
|
|
831
|
+
tokens: optionInt(args, "--tokens", 0) || 0,
|
|
832
|
+
latencyMs: optionInt(args, "--latency-ms", 0) || 0,
|
|
833
|
+
})
|
|
834
|
+
printJson({ model, taskClass, recorded: true, performance: policy.performance?.[model]?.[taskClass] || null })
|
|
835
|
+
return
|
|
836
|
+
}
|
|
837
|
+
|
|
815
838
|
if (action === "capability") {
|
|
816
839
|
const model = args[2]
|
|
817
840
|
if (!validateModelID(model)) {
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Deterministic evidence and execution tools
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
V11 uses dependency-light Node helpers for work that should not rely on a model guessing or remembering it.
|
|
4
4
|
|
|
5
5
|
## Repository evidence
|
|
6
6
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# UES engineering design
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
V11 evolves the project from an engineering workflow harness into a **perception-aware adaptive execution engine** designed to reduce context pressure on coding models while keeping evidence needed for correctness.
|
|
4
4
|
|
|
5
5
|
The selected model remains the selected model. UES improves orchestration, evidence, task boundaries, state persistence and verification; it does not claim model equivalence.
|
|
6
6
|
|
|
@@ -44,11 +44,11 @@ Models still reason about semantics and read affected code.
|
|
|
44
44
|
|
|
45
45
|
### Progressive disclosure
|
|
46
46
|
|
|
47
|
-
The catalog
|
|
47
|
+
The catalog spans 48 skills. UES prefers a small active skill set and loads deeper references only when needed.
|
|
48
48
|
|
|
49
49
|
### Hard gates, not reminders
|
|
50
50
|
|
|
51
|
-
|
|
51
|
+
V11 machine-enforces the important boundaries:
|
|
52
52
|
|
|
53
53
|
1. long/high-risk plans are not executable until a structured plan-verification receipt matches the current plan hash;
|
|
54
54
|
2. long/high-risk task completion requires a successful verification receipt for the active run and the current workspace fingerprint;
|
|
@@ -133,7 +133,7 @@ Only UES-managed resources are rewritten/removed.
|
|
|
133
133
|
|
|
134
134
|
UES separates:
|
|
135
135
|
|
|
136
|
-
1. **static skill contract** —
|
|
136
|
+
1. **static skill contract** — 43 scenarios covering the 48-skill catalog;
|
|
137
137
|
2. **V2 router precision matrix** — 120 required-route/negative-guard cases;
|
|
138
138
|
3. **standard live benchmark** — 20 executable hidden-graded tasks;
|
|
139
139
|
4. **long-horizon benchmark** — 5 tasks, including one 15-source-file integration workload;
|
package/docs/EVALS.md
CHANGED
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
# UES evaluations
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
V11 separates catalog correctness, routing precision, benchmark integrity, final behavior, long-horizon orchestration and cross-stack coverage.
|
|
4
4
|
|
|
5
5
|
## 1. Static skill-routing contract
|
|
6
6
|
|
|
7
|
-
`evals/routing.json` keeps
|
|
7
|
+
`evals/routing.json` keeps 43 representative scenarios and covers all installed skills.
|
|
8
8
|
|
|
9
9
|
```bash
|
|
10
10
|
npm run evals
|
|
@@ -141,7 +141,7 @@ Compare the same model, variant, prompt, fixture, grader and environment. Report
|
|
|
141
141
|
A benchmark result is evidence only for the measured workload. UES does not claim to turn one base model into another.
|
|
142
142
|
|
|
143
143
|
|
|
144
|
-
##
|
|
144
|
+
## V11 live-run observability and evidence gate
|
|
145
145
|
|
|
146
146
|
Live runs accept:
|
|
147
147
|
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
# GitHub Ruleset Readiness
|
|
2
|
+
|
|
3
|
+
This document records the recommended GitHub repository ruleset configuration for the `main` branch, to be activated only after CI Gate and Security Gate have each achieved at least one successful run.
|
|
4
|
+
|
|
5
|
+
## Recommended ruleset
|
|
6
|
+
|
|
7
|
+
- **Ruleset name:** `main-protection`
|
|
8
|
+
- **Target:** default branch / `main`
|
|
9
|
+
- **Enforcement:** Active
|
|
10
|
+
|
|
11
|
+
## Rules
|
|
12
|
+
|
|
13
|
+
1. **Restrict deletions** — block deleting the default branch or force-pushing over it.
|
|
14
|
+
2. **Block force pushes** — no `git push --force` or `git push -f` to `main`.
|
|
15
|
+
3. **Require pull request before merge** — all changes must go through a PR.
|
|
16
|
+
4. **Require conversation resolution** — PR reviewers must resolve all inline comments before merge.
|
|
17
|
+
5. **Require status checks** — PRs must have passing CI Gate and Security Gate checks before merge.
|
|
18
|
+
6. **Require branch up to date** — PR branches must be up to date with `main` before merge (applies when there are multiple contributors; can be relaxed for single-maintainer setups).
|
|
19
|
+
7. **Required check: CI Gate** — the aggregate CI check from `.github/workflows/ci.yml`.
|
|
20
|
+
8. **Required check: Security Gate** — the aggregate security check from `.github/workflows/security.yml`.
|
|
21
|
+
|
|
22
|
+
## Approval policy for single-maintainer repo
|
|
23
|
+
|
|
24
|
+
This repository currently has one maintainer. Setting `required approval = 1` would prevent the owner from merging their own PRs (they cannot approve their own PR). Therefore:
|
|
25
|
+
|
|
26
|
+
- **Required approval = 0** for now.
|
|
27
|
+
- When a collaborator or external reviewer is added, raise to `required approval = 1` to restore oversight.
|
|
28
|
+
|
|
29
|
+
## Classic branch protection vs ruleset
|
|
30
|
+
|
|
31
|
+
GitHub supports both classic branch protection rules and repository rulesets simultaneously. When both are configured:
|
|
32
|
+
|
|
33
|
+
- They can apply independently to different aspects of branch protection.
|
|
34
|
+
- Be cautious of duplicate or conflicting protections (e.g., two rules both requiring status checks but with different required lists).
|
|
35
|
+
- If migrating from classic protection to ruleset, remove the classic rule after confirming the ruleset is active and working.
|
|
36
|
+
|
|
37
|
+
## Activation prerequisite
|
|
38
|
+
|
|
39
|
+
Do not enable the ruleset until:
|
|
40
|
+
|
|
41
|
+
1. CI Gate has at least one successful run on `main`.
|
|
42
|
+
2. Security Gate has at least one successful run on `main` or a PR.
|
|
43
|
+
|
|
44
|
+
Without these prerequisites, the ruleset would block all merges immediately after activation, effectively locking the repository.
|
|
45
|
+
|
|
46
|
+
## Current status
|
|
47
|
+
|
|
48
|
+
- **CI Gate:** Added in `.github/workflows/ci.yml`. Awaiting first successful run.
|
|
49
|
+
- **Security Gate:** Added in `.github/workflows/security.yml`. Awaiting first successful run.
|
|
50
|
+
- **Ruleset:** NOT YET ACTIVATED. Will be configured via GitHub admin settings after both gates have successful runs.
|
package/docs/NPM-PUBLISH.md
CHANGED
|
@@ -17,7 +17,7 @@ opencode-agent-skill
|
|
|
17
17
|
npm run ci
|
|
18
18
|
```
|
|
19
19
|
|
|
20
|
-
|
|
20
|
+
CI includes syntax validation, resource validation, static skill routing, the 129-case V2 router matrix, 13 V11 contract tasks, standard/long/polyglot hidden-grader integrity checks, unit/integration tests, package dry-run, packed global-install smoke, and a plain one-command install/resource sync smoke.
|
|
21
21
|
|
|
22
22
|
## Manual release-like test
|
|
23
23
|
|
|
@@ -27,7 +27,7 @@ Use:
|
|
|
27
27
|
|
|
28
28
|
```cmd
|
|
29
29
|
npm pack
|
|
30
|
-
npm install -g .\opencode-agent-skill-
|
|
30
|
+
npm install -g .\opencode-agent-skill-11.0.0.tgz --allow-scripts=opencode-agent-skill
|
|
31
31
|
ocskill status
|
|
32
32
|
ocskill doctor
|
|
33
33
|
```
|
|
@@ -47,14 +47,14 @@ After publication verify:
|
|
|
47
47
|
|
|
48
48
|
```cmd
|
|
49
49
|
npm view opencode-agent-skill versions --json
|
|
50
|
-
npm view opencode-agent-skill@
|
|
50
|
+
npm view opencode-agent-skill@11.0.0 version
|
|
51
51
|
npm dist-tag ls opencode-agent-skill
|
|
52
52
|
```
|
|
53
53
|
|
|
54
54
|
The expected release tag is:
|
|
55
55
|
|
|
56
56
|
```text
|
|
57
|
-
latest:
|
|
57
|
+
latest: 11.0.0
|
|
58
58
|
```
|
|
59
59
|
|
|
60
60
|
## GitHub Actions publishing
|
package/docs/OPENCODE-COMPAT.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# OpenCode compatibility
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
V11 ships one npm package for OpenCode 1.x and 2.x, while only enabling V2-native runtime features when V2 is detected.
|
|
4
4
|
|
|
5
5
|
## Detection
|
|
6
6
|
|
|
@@ -23,9 +23,9 @@ The detected major is recorded in the managed state.
|
|
|
23
23
|
|
|
24
24
|
UES installs:
|
|
25
25
|
|
|
26
|
-
-
|
|
26
|
+
- 48 namespaced skills
|
|
27
27
|
- 11 namespaced commands
|
|
28
|
-
-
|
|
28
|
+
- 12 namespaced subagents using compatible V1 `permission` frontmatter
|
|
29
29
|
- managed global `AGENTS.md` block
|
|
30
30
|
|
|
31
31
|
The V2 runtime plugin is not installed.
|
package/docs/TRACE-SCHEMA.md
CHANGED
|
@@ -75,7 +75,7 @@ Keep constant:
|
|
|
75
75
|
Compare observable success, regressions, elapsed time, tool behavior and cost rather than narrative confidence.
|
|
76
76
|
|
|
77
77
|
|
|
78
|
-
##
|
|
78
|
+
## V11 runtime and evidence fields
|
|
79
79
|
|
|
80
80
|
Each live result may additionally contain:
|
|
81
81
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# UES V11 — Perception & Adaptive Execution
|
|
2
2
|
|
|
3
|
-
Status:
|
|
3
|
+
Status: stable (`11.0.0`). V11 is npm `latest`.
|
|
4
4
|
|
|
5
5
|
## Goal
|
|
6
6
|
|
|
@@ -211,7 +211,7 @@ Do not promote V11 to stable until all are satisfied:
|
|
|
211
211
|
7. Packed and plain npm-install smoke tests pass.
|
|
212
212
|
8. Real weak-model evaluation shows no suite regression.
|
|
213
213
|
9. Any configured cache/evidence target has sufficient telemetry and passes.
|
|
214
|
-
10. npm `latest`
|
|
214
|
+
10. npm `latest` is V11 stable; no release gate prevents promotion.
|
|
215
215
|
|
|
216
216
|
## Compatibility
|
|
217
217
|
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
# V12 Weak-Model Intelligence Foundation
|
|
2
|
+
|
|
3
|
+
Status: beta prerelease (`12.0.0-beta.0`, npm dist-tag `next`). V11 (`11.0.0`) remains the stable `latest` release until V12 earns stable-release evidence.
|
|
4
|
+
|
|
5
|
+
## Goal
|
|
6
|
+
|
|
7
|
+
V12 focuses on making weaker coding models more reliable on large repositories by improving measured context quality, empirical model routing, plan identity, bounded autonomous decisions, and repo-scale evaluation. It does not claim that orchestration makes one base model equivalent to a stronger model.
|
|
8
|
+
|
|
9
|
+
## Foundations
|
|
10
|
+
|
|
11
|
+
- Empirical model performance: observed pass rate, retries, token use and latency can rerank capability-eligible models by task class.
|
|
12
|
+
- Context quality receipts: adaptive context reports required-file recall and irrelevant-context ratio.
|
|
13
|
+
- Plan-scoped snapshots: every imported plan gets a SHA-256 keyed snapshot and active-plan fence before execution.
|
|
14
|
+
- Decision policy: reversible local engineering choices can be auto-resolvable; publish/deploy/destructive/product decisions remain human-gated.
|
|
15
|
+
- Repo-scale contract suite: deterministic generation of a 300-module monorepo fixture for larger-repository validation.
|
|
16
|
+
|
|
17
|
+
## Release policy
|
|
18
|
+
|
|
19
|
+
V12 is not stable merely because unit tests pass. Promotion requires healthy GitHub CI/Security gates, repo-scale validation, real weak-model baseline-vs-UES trials, no regression in existing suites, measured context recall, sufficient empirical routing samples, and Windows/Linux package/install smoke evidence.
|
|
20
|
+
|
|
21
|
+
## Beta install
|
|
22
|
+
|
|
23
|
+
```cmd
|
|
24
|
+
npm install -g opencode-agent-skill@next
|
|
25
|
+
ocskill install
|
|
26
|
+
ocskill doctor
|
|
27
|
+
```
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
{
|
|
2
|
+
"version": 1,
|
|
3
|
+
"kind": "repo-scale-contract-suite",
|
|
4
|
+
"description": "Deterministic large-repository contract tasks used to validate V12 weak-model context/routing foundations before live model runs.",
|
|
5
|
+
"minimumGeneratedModules": 300,
|
|
6
|
+
"tasks": [
|
|
7
|
+
{
|
|
8
|
+
"id": "cross-package-regression",
|
|
9
|
+
"category": "repo-scale",
|
|
10
|
+
"objective": "Trace a regression across package boundaries without loading the entire generated monorepo into context.",
|
|
11
|
+
"requiredFiles": [
|
|
12
|
+
"packages/pkg-0/src/mod049.mjs",
|
|
13
|
+
"packages/pkg-1/src/mod049.mjs"
|
|
14
|
+
],
|
|
15
|
+
"acceptance": [
|
|
16
|
+
"Required-file context recall is measurable",
|
|
17
|
+
"Unrelated packages stay bounded"
|
|
18
|
+
]
|
|
19
|
+
},
|
|
20
|
+
{
|
|
21
|
+
"id": "public-contract-ripple",
|
|
22
|
+
"category": "contract",
|
|
23
|
+
"objective": "Change a public API field and identify the API producer and web consumer that must move together.",
|
|
24
|
+
"requiredFiles": [
|
|
25
|
+
"contracts/public-api.json",
|
|
26
|
+
"apps/api/src/service.mjs",
|
|
27
|
+
"apps/web/src/consumer.mjs"
|
|
28
|
+
],
|
|
29
|
+
"acceptance": [
|
|
30
|
+
"Contract producer and consumer are both discovered",
|
|
31
|
+
"Change-impact evidence is explicit"
|
|
32
|
+
]
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
"id": "deep-chain-debug",
|
|
36
|
+
"category": "debugging",
|
|
37
|
+
"objective": "Diagnose a defect near the end of a fifty-module import chain using bounded evidence expansion.",
|
|
38
|
+
"requiredFiles": [
|
|
39
|
+
"packages/pkg-2/src/mod048.mjs",
|
|
40
|
+
"packages/pkg-2/src/mod049.mjs"
|
|
41
|
+
],
|
|
42
|
+
"acceptance": [
|
|
43
|
+
"Initial context remains bounded",
|
|
44
|
+
"Recovery can expand around the failing chain"
|
|
45
|
+
]
|
|
46
|
+
},
|
|
47
|
+
{
|
|
48
|
+
"id": "multi-package-refactor",
|
|
49
|
+
"category": "refactor",
|
|
50
|
+
"objective": "Coordinate a refactor touching several package endpoints while serializing shared contract writes.",
|
|
51
|
+
"requiredFiles": [
|
|
52
|
+
"packages/pkg-3/src/index.mjs",
|
|
53
|
+
"packages/pkg-4/src/index.mjs",
|
|
54
|
+
"contracts/public-api.json"
|
|
55
|
+
],
|
|
56
|
+
"acceptance": [
|
|
57
|
+
"Cross-package scope is explicit",
|
|
58
|
+
"Shared contract changes are treated as a write-conflict surface"
|
|
59
|
+
]
|
|
60
|
+
}
|
|
61
|
+
]
|
|
62
|
+
}
|
|
@@ -3,6 +3,7 @@ import { planEvidenceBudget, evidenceValueScore } from "./evidence-budget.mjs"
|
|
|
3
3
|
import { putEvidence } from "./evidence-store.mjs"
|
|
4
4
|
import { inferTaskCapabilities } from "./capability-registry.mjs"
|
|
5
5
|
import { buildPromptEnvelope, comparePromptEnvelopes } from "./prompt-cache.mjs"
|
|
6
|
+
import { measureContextQuality } from "./context-quality.mjs"
|
|
6
7
|
|
|
7
8
|
function taskText(task = {}) {
|
|
8
9
|
return [
|
|
@@ -118,6 +119,8 @@ export async function buildAdaptiveTaskContext(root, task, options = {}) {
|
|
|
118
119
|
recentMessages: options.recentMessages || [],
|
|
119
120
|
})
|
|
120
121
|
|
|
122
|
+
const contextQuality = measureContextQuality(task, externalized.manifest, { minRequiredRecall: options.minRequiredRecall })
|
|
123
|
+
|
|
121
124
|
const cache = options.previousPromptEnvelope
|
|
122
125
|
? comparePromptEnvelopes(options.previousPromptEnvelope, promptEnvelope)
|
|
123
126
|
: null
|
|
@@ -128,6 +131,7 @@ export async function buildAdaptiveTaskContext(root, task, options = {}) {
|
|
|
128
131
|
capabilities,
|
|
129
132
|
evidenceBudget,
|
|
130
133
|
contextManifest: externalized.manifest,
|
|
134
|
+
contextQuality,
|
|
131
135
|
evidenceStore: {
|
|
132
136
|
refs: externalized.externalized.length,
|
|
133
137
|
externalizedBytes: externalized.externalizedBytes,
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
function normalizePath(value) {
|
|
2
|
+
return String(value || "").replaceAll("\\", "/").replace(/^\.\//, "")
|
|
3
|
+
}
|
|
4
|
+
|
|
5
|
+
function declaredTaskFiles(task = {}) {
|
|
6
|
+
const files = task.files
|
|
7
|
+
const values = []
|
|
8
|
+
if (Array.isArray(files)) values.push(...files)
|
|
9
|
+
else if (files && typeof files === "object") {
|
|
10
|
+
for (const [kind, list] of Object.entries(files)) {
|
|
11
|
+
if (kind === "create") continue
|
|
12
|
+
if (Array.isArray(list)) values.push(...list)
|
|
13
|
+
else if (typeof list === "string") values.push(list)
|
|
14
|
+
}
|
|
15
|
+
}
|
|
16
|
+
if (Array.isArray(task.requiredFiles)) values.push(...task.requiredFiles)
|
|
17
|
+
return [...new Set(values.map(normalizePath).filter(Boolean))]
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
function manifestPaths(manifest = {}) {
|
|
21
|
+
const values = []
|
|
22
|
+
for (const item of manifest.excerpts || []) if (item?.path) values.push(item.path)
|
|
23
|
+
for (const item of manifest.rankedReferences || []) if (item?.path) values.push(item.path)
|
|
24
|
+
for (const item of manifest.instructions || []) {
|
|
25
|
+
if (typeof item === "string") values.push(item)
|
|
26
|
+
else if (item?.path) values.push(item.path)
|
|
27
|
+
}
|
|
28
|
+
return [...new Set(values.map(normalizePath).filter(Boolean))]
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
export function measureContextQuality(task = {}, manifest = {}, options = {}) {
|
|
32
|
+
const required = declaredTaskFiles(task)
|
|
33
|
+
const included = manifestPaths(manifest)
|
|
34
|
+
const includedSet = new Set(included)
|
|
35
|
+
const hits = required.filter((file) => includedSet.has(file))
|
|
36
|
+
const requiredFileRecall = required.length ? hits.length / required.length : null
|
|
37
|
+
const relevant = new Set(required)
|
|
38
|
+
for (const item of manifest.instructions || []) {
|
|
39
|
+
if (typeof item === "string") relevant.add(normalizePath(item))
|
|
40
|
+
else if (item?.path) relevant.add(normalizePath(item.path))
|
|
41
|
+
}
|
|
42
|
+
for (const item of manifest.excerpts || []) {
|
|
43
|
+
if (["declared", "test", "instruction"].includes(item?.role) && item?.path) relevant.add(normalizePath(item.path))
|
|
44
|
+
}
|
|
45
|
+
const irrelevant = included.filter((file) => !relevant.has(file))
|
|
46
|
+
const irrelevantRatio = included.length ? irrelevant.length / included.length : 0
|
|
47
|
+
const minRequiredRecall = Number.isFinite(Number(options.minRequiredRecall))
|
|
48
|
+
? Math.max(0, Math.min(1, Number(options.minRequiredRecall))) : 1
|
|
49
|
+
return {
|
|
50
|
+
schemaVersion: 1,
|
|
51
|
+
requiredFiles: required,
|
|
52
|
+
includedFiles: included,
|
|
53
|
+
requiredFileHits: hits,
|
|
54
|
+
requiredFileRecall,
|
|
55
|
+
irrelevantFiles: irrelevant,
|
|
56
|
+
irrelevantRatio,
|
|
57
|
+
checks: { requiredRecallAcceptable: requiredFileRecall == null ? true : requiredFileRecall >= minRequiredRecall },
|
|
58
|
+
}
|
|
59
|
+
}
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
const HIGH_RISK = /(npm publish|publish package|deploy|production|git push|force push|reset --hard|git clean|delete branch|drop table|truncate|rotate secret|credential|api key|purchase|payment|irreversible|public api break)/i
|
|
2
|
+
const MEDIUM_RISK = /(migration|schema|dependency upgrade|lockfile|generated code|shared config|auth|permission|security)/i
|
|
3
|
+
const REVERSIBLE = /(local|test|rename|format|refactor|temporary|fixture|internal|reversible)/i
|
|
4
|
+
|
|
5
|
+
export function classifyDecisionPolicy(text = "", facts = {}) {
|
|
6
|
+
const value = String(text || "").trim()
|
|
7
|
+
const explicitlyIrreversible = facts.irreversible === true
|
|
8
|
+
const externalSideEffect = facts.externalSideEffect === true
|
|
9
|
+
const destructive = facts.destructive === true || HIGH_RISK.test(value)
|
|
10
|
+
const medium = MEDIUM_RISK.test(value)
|
|
11
|
+
const reversible = facts.reversible === true || (!destructive && REVERSIBLE.test(value))
|
|
12
|
+
const risk = destructive || explicitlyIrreversible || externalSideEffect ? "high" : medium ? "medium" : "low"
|
|
13
|
+
const requiresUser = risk === "high" || facts.productDecision === true
|
|
14
|
+
return {
|
|
15
|
+
schemaVersion: 1, risk, reversible, destructive, externalSideEffect, requiresUser,
|
|
16
|
+
autoResolvable: !requiresUser && (reversible || risk === "low"),
|
|
17
|
+
reason: requiresUser ? "human-approval-required" : reversible ? "reversible-local-decision" : "bounded-engineering-decision",
|
|
18
|
+
}
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
export function canAutoResolveDecision(text = "", facts = {}) {
|
|
22
|
+
return classifyDecisionPolicy(text, facts).autoResolvable
|
|
23
|
+
}
|
package/lib/model-config.mjs
CHANGED
|
@@ -3,6 +3,7 @@ import { mkdir, readFile, writeFile } from "node:fs/promises"
|
|
|
3
3
|
import path from "node:path"
|
|
4
4
|
import { defaultModelPolicy, resolveModel } from "./model-policy.mjs"
|
|
5
5
|
import { normalizeCapabilityProfile } from "./capability-registry.mjs"
|
|
6
|
+
import { normalizePerformanceHistory, recordPerformanceOutcome } from "./model-performance.mjs"
|
|
6
7
|
|
|
7
8
|
const TIERS = new Set(["light", "standard", "heavy"])
|
|
8
9
|
|
|
@@ -29,8 +30,10 @@ function normalize(policy) {
|
|
|
29
30
|
if (validModelID(model)) capabilities[model] = normalizeCapabilityProfile(profile)
|
|
30
31
|
}
|
|
31
32
|
|
|
33
|
+
const performance = normalizePerformanceHistory(input.performance || {})
|
|
34
|
+
|
|
32
35
|
return {
|
|
33
|
-
schemaVersion:
|
|
36
|
+
schemaVersion: 3,
|
|
34
37
|
enabled: input.enabled === true,
|
|
35
38
|
maxEscalations: Number.isInteger(input.maxEscalations)
|
|
36
39
|
? Math.max(0, Math.min(input.maxEscalations, 2))
|
|
@@ -38,6 +41,8 @@ function normalize(policy) {
|
|
|
38
41
|
tiers,
|
|
39
42
|
roleTiers,
|
|
40
43
|
capabilities,
|
|
44
|
+
performance,
|
|
45
|
+
performanceMinSamples: Number.isInteger(input.performanceMinSamples) ? Math.max(1, Math.min(input.performanceMinSamples, 20)) : base.performanceMinSamples,
|
|
41
46
|
}
|
|
42
47
|
}
|
|
43
48
|
|
|
@@ -64,6 +69,7 @@ export async function writeModelPolicy(configDir, patch = {}) {
|
|
|
64
69
|
tiers: { ...current.tiers, ...(patch.tiers || {}) },
|
|
65
70
|
roleTiers: { ...current.roleTiers, ...(patch.roleTiers || {}) },
|
|
66
71
|
capabilities: { ...(current.capabilities || {}), ...(patch.capabilities || {}) },
|
|
72
|
+
performance: patch.performance || current.performance || {},
|
|
67
73
|
})
|
|
68
74
|
const file = modelPolicyFile(configDir)
|
|
69
75
|
await mkdir(path.dirname(file), { recursive: true })
|
|
@@ -94,3 +100,9 @@ export function applyConfiguredModel(source, role, policy) {
|
|
|
94
100
|
}
|
|
95
101
|
return lines.join("\n")
|
|
96
102
|
}
|
|
103
|
+
|
|
104
|
+
export async function recordModelPerformance(configDir, outcome = {}) {
|
|
105
|
+
const current = await readModelPolicy(configDir)
|
|
106
|
+
const performance = recordPerformanceOutcome(current.performance || {}, outcome)
|
|
107
|
+
return writeModelPolicy(configDir, { performance })
|
|
108
|
+
}
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
export const MODEL_TASK_CLASSES = Object.freeze([
|
|
2
|
+
"general", "repo-scale", "debugging", "architecture", "security",
|
|
3
|
+
"migration", "frontend", "backend", "visual", "browser",
|
|
4
|
+
])
|
|
5
|
+
const KNOWN_TASK_CLASSES = new Set(MODEL_TASK_CLASSES)
|
|
6
|
+
|
|
7
|
+
function boundedNumber(value, fallback = 0, min = 0, max = Number.MAX_SAFE_INTEGER) {
|
|
8
|
+
const number = Number(value)
|
|
9
|
+
if (!Number.isFinite(number)) return fallback
|
|
10
|
+
return Math.max(min, Math.min(max, number))
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
export function inferTaskClass(text = "", facts = {}) {
|
|
14
|
+
const explicit = String(facts.taskClass || "").trim().toLowerCase()
|
|
15
|
+
if (KNOWN_TASK_CLASSES.has(explicit)) return explicit
|
|
16
|
+
const value = String(text || "").toLowerCase()
|
|
17
|
+
if (/(whole repo|entire project|large monorepo|repo[- ]scale|cross[- ]module|toàn bộ dự án|nhiều module)/.test(value)) return "repo-scale"
|
|
18
|
+
if (/(prompt injection|security|auth|authorization|permission|secret|credential|bảo mật|phân quyền)/.test(value)) return "security"
|
|
19
|
+
if (/(migration|schema|database|sql|backfill|migrate)/.test(value)) return "migration"
|
|
20
|
+
if (/(screenshot|visual|figma|pixel|responsive|storybook)/.test(value)) return "visual"
|
|
21
|
+
if (/(browser|playwright|e2e|web page|click flow)/.test(value)) return "browser"
|
|
22
|
+
if (/(root cause|debug|regression|crash|failing|bug|lỗi)/.test(value)) return "debugging"
|
|
23
|
+
if (/(architecture|architect|design decision|system design|kiến trúc)/.test(value)) return "architecture"
|
|
24
|
+
if (/(react|next\.js|vue|svelte|css|frontend|ui\b)/.test(value)) return "frontend"
|
|
25
|
+
if (/(api|service|node|python|java|dotnet|backend|server)/.test(value)) return "backend"
|
|
26
|
+
return "general"
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
export function normalizePerformanceRecord(record = {}) {
|
|
30
|
+
record = record && typeof record === "object" ? record : {}
|
|
31
|
+
const samples = Math.floor(boundedNumber(record.samples, 0, 0))
|
|
32
|
+
const successes = Math.floor(boundedNumber(record.successes, Math.round(samples * boundedNumber(record.passRate, 0, 0, 1)), 0, samples))
|
|
33
|
+
return {
|
|
34
|
+
samples,
|
|
35
|
+
successes,
|
|
36
|
+
passRate: samples ? successes / samples : 0,
|
|
37
|
+
avgRetries: boundedNumber(record.avgRetries, 0, 0, 100),
|
|
38
|
+
avgTokens: boundedNumber(record.avgTokens, 0, 0),
|
|
39
|
+
avgLatencyMs: boundedNumber(record.avgLatencyMs, 0, 0),
|
|
40
|
+
updatedAt: typeof record.updatedAt === "string" ? record.updatedAt : null,
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
export function normalizePerformanceHistory(history = {}) {
|
|
45
|
+
const output = {}
|
|
46
|
+
for (const [model, classes] of Object.entries(history || {})) {
|
|
47
|
+
if (!model || !classes || typeof classes !== "object") continue
|
|
48
|
+
const normalizedClasses = {}
|
|
49
|
+
for (const [taskClass, record] of Object.entries(classes)) {
|
|
50
|
+
if (!KNOWN_TASK_CLASSES.has(taskClass) && taskClass !== "overall") continue
|
|
51
|
+
normalizedClasses[taskClass] = normalizePerformanceRecord(record)
|
|
52
|
+
}
|
|
53
|
+
if (Object.keys(normalizedClasses).length) output[model] = normalizedClasses
|
|
54
|
+
}
|
|
55
|
+
return output
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
function mergeAverage(previousAverage, previousSamples, value) {
|
|
59
|
+
return previousSamples <= 0 ? value : ((previousAverage * previousSamples) + value) / (previousSamples + 1)
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export function recordPerformanceOutcome(history = {}, outcome = {}) {
|
|
63
|
+
const model = String(outcome.model || "").trim()
|
|
64
|
+
if (!model) throw new Error("model performance outcome requires model")
|
|
65
|
+
const taskClass = inferTaskClass(outcome.text || "", { taskClass: outcome.taskClass })
|
|
66
|
+
const normalized = normalizePerformanceHistory(history)
|
|
67
|
+
const current = normalizePerformanceRecord(normalized[model]?.[taskClass] || {})
|
|
68
|
+
const samples = current.samples
|
|
69
|
+
const passed = outcome.passed === true
|
|
70
|
+
const next = {
|
|
71
|
+
samples: samples + 1,
|
|
72
|
+
successes: current.successes + (passed ? 1 : 0),
|
|
73
|
+
passRate: 0,
|
|
74
|
+
avgRetries: mergeAverage(current.avgRetries, samples, boundedNumber(outcome.retries, 0, 0, 100)),
|
|
75
|
+
avgTokens: mergeAverage(current.avgTokens, samples, boundedNumber(outcome.tokens, 0, 0)),
|
|
76
|
+
avgLatencyMs: mergeAverage(current.avgLatencyMs, samples, boundedNumber(outcome.latencyMs, 0, 0)),
|
|
77
|
+
updatedAt: new Date().toISOString(),
|
|
78
|
+
}
|
|
79
|
+
next.passRate = next.successes / next.samples
|
|
80
|
+
return { ...normalized, [model]: { ...(normalized[model] || {}), [taskClass]: next } }
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
function performanceAdjustment(record, minSamples) {
|
|
84
|
+
const normalized = normalizePerformanceRecord(record)
|
|
85
|
+
if (normalized.samples <= 0) return { adjustment: 0, confidence: 0, record: normalized }
|
|
86
|
+
const confidence = Math.min(1, normalized.samples / Math.max(1, minSamples))
|
|
87
|
+
if (normalized.samples < minSamples) return { adjustment: 0, confidence, record: normalized }
|
|
88
|
+
const correctness = (normalized.passRate - 0.5) * 80
|
|
89
|
+
const retryPenalty = Math.min(20, normalized.avgRetries * 5)
|
|
90
|
+
const latencyPenalty = normalized.avgLatencyMs > 0 ? Math.min(10, Math.max(0, Math.log10(Math.max(1, normalized.avgLatencyMs / 1000)) * 3)) : 0
|
|
91
|
+
return { adjustment: (correctness - retryPenalty - latencyPenalty) * confidence, confidence, record: normalized }
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
export function rerankCapabilitySelection(selection = {}, history = {}, options = {}) {
|
|
95
|
+
const taskClass = inferTaskClass(options.text || "", { taskClass: options.taskClass })
|
|
96
|
+
const minSamples = Math.max(1, Number(options.minSamples || 3))
|
|
97
|
+
const normalized = normalizePerformanceHistory(history)
|
|
98
|
+
const candidates = (selection.candidates || []).map((candidate) => {
|
|
99
|
+
const record = normalized[candidate.id]?.[taskClass] || normalized[candidate.id]?.overall || null
|
|
100
|
+
const evidence = performanceAdjustment(record, minSamples)
|
|
101
|
+
return {
|
|
102
|
+
...candidate,
|
|
103
|
+
baseScore: Number(candidate.score || 0),
|
|
104
|
+
empiricalTaskClass: taskClass,
|
|
105
|
+
empiricalEvidence: evidence.record,
|
|
106
|
+
empiricalConfidence: Number(evidence.confidence.toFixed(4)),
|
|
107
|
+
adjustedScore: Number((Number(candidate.score || 0) + evidence.adjustment).toFixed(6)),
|
|
108
|
+
}
|
|
109
|
+
})
|
|
110
|
+
const eligible = candidates.filter((candidate) => candidate.eligible)
|
|
111
|
+
.sort((a, b) => b.adjustedScore - a.adjustedScore || b.baseScore - a.baseScore)
|
|
112
|
+
return { ...selection, selected: eligible[0] || null, candidates, taskClass, empirical: true }
|
|
113
|
+
}
|
package/lib/model-policy.mjs
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { inferTaskCapabilities, modelCandidatesFromPolicy, selectCapabilityCandidate } from "./capability-registry.mjs"
|
|
2
|
+
import { inferTaskClass, rerankCapabilitySelection } from "./model-performance.mjs"
|
|
2
3
|
|
|
3
4
|
const ROLE_TIERS = {
|
|
4
5
|
"codebase-mapper": "standard",
|
|
@@ -45,12 +46,14 @@ export function resolveModel(role, attempt, config = {}) {
|
|
|
45
46
|
|
|
46
47
|
export function defaultModelPolicy() {
|
|
47
48
|
return {
|
|
48
|
-
schemaVersion:
|
|
49
|
+
schemaVersion: 3,
|
|
49
50
|
enabled: false,
|
|
50
51
|
maxEscalations: 2,
|
|
51
52
|
tiers: { light: null, standard: null, heavy: null },
|
|
52
53
|
roleTiers: { ...ROLE_TIERS },
|
|
53
54
|
capabilities: {},
|
|
55
|
+
performance: {},
|
|
56
|
+
performanceMinSamples: 3,
|
|
54
57
|
}
|
|
55
58
|
}
|
|
56
59
|
|
|
@@ -90,10 +93,14 @@ export function resolveCapabilityModel(role, attempt, taskText = "", taskPolicy
|
|
|
90
93
|
})
|
|
91
94
|
const candidates = modelCandidatesFromPolicy(config)
|
|
92
95
|
.filter((candidate) => tierIndex(candidate.tier) >= tierIndex(base.tier))
|
|
93
|
-
const
|
|
96
|
+
const staticSelection = selectCapabilityCandidate(requirements, candidates, {
|
|
94
97
|
role,
|
|
95
98
|
preferredTier: base.tier,
|
|
96
99
|
})
|
|
100
|
+
const taskClass = inferTaskClass(taskText, facts)
|
|
101
|
+
const selection = rerankCapabilitySelection(staticSelection, config.performance || {}, {
|
|
102
|
+
taskClass, text: taskText, minSamples: config.performanceMinSamples || 3,
|
|
103
|
+
})
|
|
97
104
|
const capabilityEnforced = config.enabled === true && candidates.length > 0
|
|
98
105
|
if (capabilityEnforced && !selection.selected) {
|
|
99
106
|
return {
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
import { mkdir, writeFile } from "node:fs/promises"
|
|
2
|
+
import path from "node:path"
|
|
3
|
+
|
|
4
|
+
function moduleSource(packageIndex, moduleIndex) {
|
|
5
|
+
const previous = moduleIndex > 0
|
|
6
|
+
? 'import { value as previous } from "./mod' + String(moduleIndex - 1).padStart(3, "0") + '.mjs"\n'
|
|
7
|
+
: ""
|
|
8
|
+
return previous + "export const value = " + (moduleIndex > 0 ? "previous + 1" : packageIndex * 1000) +
|
|
9
|
+
"\nexport function compute(input) { return value + Number(input || 0) }\n"
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
export async function generateRepoScaleFixture(root, options = {}) {
|
|
13
|
+
const packageCount = Math.max(2, Number(options.packageCount || 6))
|
|
14
|
+
const modulesPerPackage = Math.max(5, Number(options.modulesPerPackage || 50))
|
|
15
|
+
await mkdir(root, { recursive: true })
|
|
16
|
+
await writeFile(path.join(root, "package.json"), JSON.stringify({
|
|
17
|
+
name: "ues-repo-scale-fixture", private: true, type: "module", workspaces: ["packages/*", "apps/*"],
|
|
18
|
+
}, null, 2) + "\n")
|
|
19
|
+
|
|
20
|
+
let generatedModules = 0
|
|
21
|
+
for (let p = 0; p < packageCount; p += 1) {
|
|
22
|
+
const pkgRoot = path.join(root, "packages", "pkg-" + p)
|
|
23
|
+
const src = path.join(pkgRoot, "src")
|
|
24
|
+
await mkdir(src, { recursive: true })
|
|
25
|
+
await writeFile(path.join(pkgRoot, "package.json"), JSON.stringify({
|
|
26
|
+
name: "@fixture/pkg-" + p, private: true, type: "module", exports: "./src/index.mjs",
|
|
27
|
+
}, null, 2) + "\n")
|
|
28
|
+
for (let m = 0; m < modulesPerPackage; m += 1) {
|
|
29
|
+
await writeFile(path.join(src, "mod" + String(m).padStart(3, "0") + ".mjs"), moduleSource(p, m))
|
|
30
|
+
generatedModules += 1
|
|
31
|
+
}
|
|
32
|
+
await writeFile(path.join(src, "index.mjs"),
|
|
33
|
+
'export { value, compute } from "./mod' + String(modulesPerPackage - 1).padStart(3, "0") + '.mjs"\n')
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
const apiRoot = path.join(root, "apps", "api", "src")
|
|
37
|
+
const webRoot = path.join(root, "apps", "web", "src")
|
|
38
|
+
const contracts = path.join(root, "contracts")
|
|
39
|
+
await mkdir(apiRoot, { recursive: true }); await mkdir(webRoot, { recursive: true }); await mkdir(contracts, { recursive: true })
|
|
40
|
+
await writeFile(path.join(contracts, "public-api.json"), JSON.stringify({ version: 1, fields: ["id", "name", "status"] }, null, 2) + "\n")
|
|
41
|
+
await writeFile(path.join(apiRoot, "service.mjs"), 'export function serialize(row) { return { id: row.id, name: row.name, status: row.status } }\n')
|
|
42
|
+
await writeFile(path.join(webRoot, "consumer.mjs"), 'export function render(item) { return item.name + ":" + item.status }\n')
|
|
43
|
+
await writeFile(path.join(root, "AGENTS.md"), "# Fixture instructions\n\nPreserve public contracts and update focused tests.\n")
|
|
44
|
+
return { root, packageCount, modulesPerPackage, generatedModules, totalPrimaryFiles: generatedModules + packageCount + 5 }
|
|
45
|
+
}
|
package/lib/task-engine.mjs
CHANGED
|
@@ -15,6 +15,8 @@ import { putEvidence } from "./evidence-store.mjs"
|
|
|
15
15
|
import { inferTaskCapabilities } from "./capability-registry.mjs"
|
|
16
16
|
import { buildPromptEnvelope } from "./prompt-cache.mjs"
|
|
17
17
|
import { externalizeContextExcerpts } from "./context-engine-v11.mjs"
|
|
18
|
+
import { assertActivePlanScope, persistPlanScope } from "./work-plan-scope.mjs"
|
|
19
|
+
import { classifyDecisionPolicy } from "./decision-policy.mjs"
|
|
18
20
|
|
|
19
21
|
const WORK_DIR = ".ues-work"
|
|
20
22
|
const STATE_SCHEMA = 4
|
|
@@ -49,6 +51,8 @@ export function workPaths(root, slug) {
|
|
|
49
51
|
events: path.join(dir, "EVENTS.jsonl"),
|
|
50
52
|
tasks: path.join(dir, "tasks"),
|
|
51
53
|
reports: path.join(dir, "reports"),
|
|
54
|
+
plans: path.join(dir, "plans"),
|
|
55
|
+
activePlan: path.join(dir, "ACTIVE_PLAN.json"),
|
|
52
56
|
lock: path.join(dir, ".state-lock"),
|
|
53
57
|
}
|
|
54
58
|
}
|
|
@@ -268,6 +272,7 @@ export async function initWork(root, slug, goal) {
|
|
|
268
272
|
|
|
269
273
|
await mkdir(paths.tasks, { recursive: true })
|
|
270
274
|
await mkdir(paths.reports, { recursive: true })
|
|
275
|
+
await mkdir(paths.plans, { recursive: true })
|
|
271
276
|
const createdAt = now()
|
|
272
277
|
const cleanGoal = String(goal || "").trim() || "Define the requested engineering outcome."
|
|
273
278
|
|
|
@@ -285,6 +290,7 @@ export async function initWork(root, slug, goal) {
|
|
|
285
290
|
planImportedAt: null,
|
|
286
291
|
planHash: null,
|
|
287
292
|
planApproval: null,
|
|
293
|
+
planScope: null,
|
|
288
294
|
integrationVerification: null,
|
|
289
295
|
tasks: {},
|
|
290
296
|
decisions: [],
|
|
@@ -383,6 +389,7 @@ export async function importPlan(root, slug, planInput) {
|
|
|
383
389
|
}
|
|
384
390
|
|
|
385
391
|
const updatedAt = now()
|
|
392
|
+
const planScope = await persistPlanScope(loaded.paths, hash, plan, { importedAt: updatedAt, previousPlanHash: loaded.state.planHash || null })
|
|
386
393
|
const state = {
|
|
387
394
|
...loaded.state,
|
|
388
395
|
schemaVersion: STATE_SCHEMA,
|
|
@@ -392,6 +399,7 @@ export async function importPlan(root, slug, planInput) {
|
|
|
392
399
|
planImportedAt: updatedAt,
|
|
393
400
|
planHash: hash,
|
|
394
401
|
planApproval: { status: "pending", planHash: hash, at: updatedAt, evidence: null },
|
|
402
|
+
planScope,
|
|
395
403
|
integrationVerification: null,
|
|
396
404
|
checkpoint: null,
|
|
397
405
|
evidencePolicy: evidencePolicyForPlan(plan),
|
|
@@ -414,6 +422,7 @@ export async function approvePlan(root, slug, evidence, options = {}) {
|
|
|
414
422
|
const loaded = await loadWork(root, slug)
|
|
415
423
|
if (!loaded.plan) throw new Error("PLAN.json is missing")
|
|
416
424
|
const hash = planHash(loaded.plan)
|
|
425
|
+
await assertActivePlanScope(loaded.paths, loaded.state.planHash || hash)
|
|
417
426
|
if (loaded.state.planHash && loaded.state.planHash !== hash) {
|
|
418
427
|
throw new Error("PLAN.json changed after import; re-import it before approval")
|
|
419
428
|
}
|
|
@@ -540,6 +549,7 @@ export async function workStatus(root, slug) {
|
|
|
540
549
|
blockers: loaded.state.blockers || [],
|
|
541
550
|
planApproval: loaded.state.planApproval || null,
|
|
542
551
|
integrationVerification: loaded.state.integrationVerification || null,
|
|
552
|
+
planScope: loaded.state.planScope || null,
|
|
543
553
|
evidence: {
|
|
544
554
|
receiptBacked,
|
|
545
555
|
completed,
|
|
@@ -557,6 +567,7 @@ export async function startTask(root, slug, taskID, options = {}) {
|
|
|
557
567
|
const result = await withWorkLock(root, slug, async () => {
|
|
558
568
|
const loaded = await loadWork(root, slug)
|
|
559
569
|
if (!loaded.plan) throw new Error("PLAN.json is missing; import a valid plan first")
|
|
570
|
+
await assertActivePlanScope(loaded.paths, loaded.state.planHash || planHash(loaded.plan))
|
|
560
571
|
if (!planIsApproved(loaded.state, loaded.plan)) {
|
|
561
572
|
throw new Error("plan is not approved; run ues-plan-checker and 'ocskill work approve-plan' first")
|
|
562
573
|
}
|
|
@@ -888,16 +899,19 @@ export async function failTask(root, slug, taskID, reason, options = {}) {
|
|
|
888
899
|
}
|
|
889
900
|
|
|
890
901
|
export async function addDecision(root, slug, decision) {
|
|
891
|
-
const
|
|
902
|
+
const input = decision && typeof decision === "object" ? decision : { text: decision }
|
|
903
|
+
const text = String(input.text || "").trim()
|
|
892
904
|
if (!text) throw new Error("decision text is required")
|
|
893
|
-
|
|
905
|
+
const policy = classifyDecisionPolicy(text, input.facts || input)
|
|
906
|
+
if (input.auto === true && !policy.autoResolvable) throw new Error("decision requires human approval before automatic resolution")
|
|
894
907
|
return withWorkLock(root, slug, async () => {
|
|
895
908
|
const loaded = await loadWork(root, slug)
|
|
896
909
|
loaded.state.decisions ??= []
|
|
897
910
|
const timestamp = now()
|
|
898
|
-
loaded.state.decisions.push({ at: timestamp, text })
|
|
911
|
+
loaded.state.decisions.push({ at: timestamp, text, policy, source: input.source || (input.auto === true ? "auto-ruling" : "human-or-agent") })
|
|
899
912
|
loaded.state.updatedAt = timestamp
|
|
900
913
|
await writeJson(loaded.paths.state, loaded.state)
|
|
914
|
+
await journal(loaded.paths, "decision.recorded", { text, policy })
|
|
901
915
|
return loaded.state
|
|
902
916
|
})
|
|
903
917
|
}
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
import { existsSync } from "node:fs"
|
|
2
|
+
import { mkdir, readFile, writeFile } from "node:fs/promises"
|
|
3
|
+
import path from "node:path"
|
|
4
|
+
|
|
5
|
+
function validPlanHash(value) {
|
|
6
|
+
return /^[a-f0-9]{64}$/i.test(String(value || ""))
|
|
7
|
+
}
|
|
8
|
+
|
|
9
|
+
export function planScopePaths(paths, planHash) {
|
|
10
|
+
if (!validPlanHash(planHash)) throw new Error("plan scope requires a SHA-256 plan hash")
|
|
11
|
+
const dir = path.join(paths.plans, planHash)
|
|
12
|
+
return { dir, plan: path.join(dir, "PLAN.json"), metadata: path.join(dir, "SCOPE.json") }
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
export async function persistPlanScope(paths, planHash, plan, metadata = {}) {
|
|
16
|
+
const scoped = planScopePaths(paths, planHash)
|
|
17
|
+
await mkdir(scoped.dir, { recursive: true })
|
|
18
|
+
await writeFile(scoped.plan, JSON.stringify(plan, null, 2) + "\n", "utf8")
|
|
19
|
+
const record = {
|
|
20
|
+
schemaVersion: 1,
|
|
21
|
+
planHash,
|
|
22
|
+
importedAt: metadata.importedAt || new Date().toISOString(),
|
|
23
|
+
previousPlanHash: metadata.previousPlanHash || null,
|
|
24
|
+
relativeDir: path.relative(paths.root, scoped.dir).replaceAll("\\", "/"),
|
|
25
|
+
}
|
|
26
|
+
await writeFile(scoped.metadata, JSON.stringify(record, null, 2) + "\n", "utf8")
|
|
27
|
+
await writeFile(paths.activePlan, JSON.stringify(record, null, 2) + "\n", "utf8")
|
|
28
|
+
return record
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
export async function readActivePlanScope(paths) {
|
|
32
|
+
if (!existsSync(paths.activePlan)) return null
|
|
33
|
+
try { return JSON.parse(await readFile(paths.activePlan, "utf8")) }
|
|
34
|
+
catch { throw new Error("ACTIVE_PLAN.json is invalid; re-import the plan before execution") }
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export async function assertActivePlanScope(paths, expectedPlanHash) {
|
|
38
|
+
const active = await readActivePlanScope(paths)
|
|
39
|
+
if (!active) return { legacy: true, planHash: expectedPlanHash || null }
|
|
40
|
+
if (!validPlanHash(active.planHash)) throw new Error("ACTIVE_PLAN.json contains an invalid plan hash")
|
|
41
|
+
if (expectedPlanHash && active.planHash !== expectedPlanHash) {
|
|
42
|
+
throw new Error("active plan scope mismatch; re-import PLAN.json before executing tasks")
|
|
43
|
+
}
|
|
44
|
+
const scoped = planScopePaths(paths, active.planHash)
|
|
45
|
+
if (!existsSync(scoped.plan) || !existsSync(scoped.metadata)) {
|
|
46
|
+
throw new Error("active plan scope snapshot is incomplete; re-import PLAN.json")
|
|
47
|
+
}
|
|
48
|
+
return { ...active, legacy: false }
|
|
49
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "opencode-agent-skill",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "12.0.0-beta.0",
|
|
4
4
|
"description": "Perception-aware evidence-first engineering runtime with adaptive context, capability routing, visual/browser verification, weak-model recovery and benchmark-gated execution for OpenCode",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -23,7 +23,7 @@
|
|
|
23
23
|
"evals": "node scripts/eval-skills.mjs",
|
|
24
24
|
"evals:live:validate": "node scripts/validate-live-suite.mjs",
|
|
25
25
|
"test": "node --test test/*.test.mjs",
|
|
26
|
-
"ci": "npm run syntax && npm run validate && npm run evals && npm run evals:router && npm run evals:v11:validate && npm run evals:live:validate && npm run evals:long:validate && npm run evals:polyglot:validate && npm test && npm pack --dry-run && npm run smoke:pack && npm run smoke:plain-install",
|
|
26
|
+
"ci": "npm run syntax && npm run validate && npm run docs:check && npm run evals && npm run evals:router && npm run evals:v11:validate && npm run evals:v12:validate && npm run evals:repo-scale:validate && npm run evals:live:validate && npm run evals:long:validate && npm run evals:polyglot:validate && npm test && npm pack --dry-run && npm run smoke:pack && npm run smoke:plain-install",
|
|
27
27
|
"postinstall": "node scripts/install.mjs",
|
|
28
28
|
"preuninstall": "node scripts/uninstall.mjs",
|
|
29
29
|
"prepublishOnly": "npm run ci",
|
|
@@ -42,13 +42,16 @@
|
|
|
42
42
|
"dashboard": "node scripts/control-center.mjs",
|
|
43
43
|
"evals:matrix": "node scripts/eval-matrix.mjs",
|
|
44
44
|
"smoke:plain-install": "node scripts/smoke-plain-install.mjs",
|
|
45
|
+
"docs:check": "node scripts/check-release-consistency.mjs",
|
|
45
46
|
"release:check-tag": "node scripts/check-release-tag.mjs",
|
|
46
47
|
"evals:polyglot:validate": "node scripts/validate-live-suite.mjs --suite polyglot",
|
|
47
48
|
"evals:polyglot": "node scripts/eval-live.mjs --suite polyglot",
|
|
48
49
|
"evals:matrix:gate": "node scripts/eval-matrix.mjs --require-confidence",
|
|
49
50
|
"evals:ablation": "node scripts/eval-ablation.mjs",
|
|
50
51
|
"evals:v11": "node --test test/*-v11.test.mjs",
|
|
51
|
-
"evals:v11:validate": "node scripts/validate-v11-suite.mjs"
|
|
52
|
+
"evals:v11:validate": "node scripts/validate-v11-suite.mjs",
|
|
53
|
+
"evals:repo-scale:validate": "node scripts/validate-repo-scale-suite.mjs",
|
|
54
|
+
"evals:v12:validate": "node scripts/validate-v12-foundation.mjs"
|
|
52
55
|
},
|
|
53
56
|
"keywords": [
|
|
54
57
|
"opencode",
|
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
|
|
3
|
+
import { existsSync, readdirSync, readFileSync } from "node:fs"
|
|
4
|
+
import path from "node:path"
|
|
5
|
+
import { fileURLToPath } from "node:url"
|
|
6
|
+
|
|
7
|
+
const DEFAULT_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..")
|
|
8
|
+
|
|
9
|
+
function readJson(root, relative) {
|
|
10
|
+
const file = path.join(root, relative)
|
|
11
|
+
if (!existsSync(file)) {
|
|
12
|
+
return { ok: false, error: `${relative}: file not found`, value: null }
|
|
13
|
+
}
|
|
14
|
+
try {
|
|
15
|
+
return { ok: true, error: null, value: JSON.parse(readFileSync(file, "utf8")) }
|
|
16
|
+
} catch (e) {
|
|
17
|
+
return { ok: false, error: `${relative}: invalid JSON (${e.message})`, value: null }
|
|
18
|
+
}
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
function readText(root, relative) {
|
|
22
|
+
const file = path.join(root, relative)
|
|
23
|
+
if (!existsSync(file)) {
|
|
24
|
+
return { ok: false, error: `${relative}: file not found`, value: null }
|
|
25
|
+
}
|
|
26
|
+
return { ok: true, error: null, value: readFileSync(file, "utf8") }
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
function countSkillDirs(root) {
|
|
30
|
+
const fullPath = path.join(root, "global-config", "skills")
|
|
31
|
+
if (!existsSync(fullPath)) return { ok: false, error: "global-config/skills: directory not found", value: 0 }
|
|
32
|
+
const entries = readdirSync(fullPath, { withFileTypes: true })
|
|
33
|
+
const count = entries.filter((entry) => entry.isDirectory() && existsSync(path.join(fullPath, entry.name, "SKILL.md"))).length
|
|
34
|
+
return { ok: true, error: null, value: count }
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
function countDir(root, dirPath, filter) {
|
|
38
|
+
const fullPath = path.join(root, dirPath)
|
|
39
|
+
if (!existsSync(fullPath)) {
|
|
40
|
+
return { ok: false, error: `${dirPath}: directory not found`, value: 0 }
|
|
41
|
+
}
|
|
42
|
+
const entries = readdirSync(fullPath, { withFileTypes: true })
|
|
43
|
+
const count = filter ? entries.filter(filter).length : entries.length
|
|
44
|
+
return { ok: true, error: null, value: count }
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
export function checkReleaseConsistency(root) {
|
|
48
|
+
root = root || DEFAULT_ROOT
|
|
49
|
+
const errors = []
|
|
50
|
+
const warnings = []
|
|
51
|
+
|
|
52
|
+
function rJson(relative) {
|
|
53
|
+
return readJson(root, relative)
|
|
54
|
+
}
|
|
55
|
+
function rText(relative) {
|
|
56
|
+
return readText(root, relative)
|
|
57
|
+
}
|
|
58
|
+
function rCount(dirPath, filter) {
|
|
59
|
+
return countDir(root, dirPath, filter)
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
// 1. package.json version == package-lock.json version
|
|
63
|
+
const pkg = rJson("package.json")
|
|
64
|
+
const lock = rJson("package-lock.json")
|
|
65
|
+
if (pkg.ok && lock.ok) {
|
|
66
|
+
const pkgVersion = pkg.value.version
|
|
67
|
+
const lockVersion = lock.value.version
|
|
68
|
+
if (pkgVersion !== lockVersion) {
|
|
69
|
+
errors.push(`package.json version (${pkgVersion}) != package-lock.json version (${lockVersion})`)
|
|
70
|
+
}
|
|
71
|
+
} else {
|
|
72
|
+
if (!pkg.ok) errors.push(pkg.error)
|
|
73
|
+
if (!lock.ok) errors.push(lock.error)
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
// 2. Count skills, commands, subagents
|
|
77
|
+
const skillRes = countSkillDirs(root)
|
|
78
|
+
const commandRes = rCount("global-config/commands", (e) => e.isFile() && e.name.endsWith(".md"))
|
|
79
|
+
const agentRes = rCount("global-config/agents", (e) => e.isFile() && e.name.endsWith(".md"))
|
|
80
|
+
const skillCount = skillRes.ok ? skillRes.value : 0
|
|
81
|
+
const commandCount = commandRes.ok ? commandRes.value : 0
|
|
82
|
+
const agentCount = agentRes.ok ? agentRes.value : 0
|
|
83
|
+
if (!skillRes.ok) errors.push(skillRes.error)
|
|
84
|
+
if (!commandRes.ok) errors.push(commandRes.error)
|
|
85
|
+
if (!agentRes.ok) errors.push(agentRes.error)
|
|
86
|
+
|
|
87
|
+
const expectedVersion = pkg.ok ? pkg.value.version : "unknown"
|
|
88
|
+
const routing = rJson("evals/routing.json")
|
|
89
|
+
const routerTriggers = rJson("evals/router-triggers.json")
|
|
90
|
+
const staticScenarioCount = routing.ok && Array.isArray(routing.value.scenarios) ? routing.value.scenarios.length : 0
|
|
91
|
+
const routerCaseCount = routerTriggers.ok && Array.isArray(routerTriggers.value.cases) ? routerTriggers.value.cases.length : 0
|
|
92
|
+
if (!routing.ok) errors.push(routing.error)
|
|
93
|
+
if (!routerTriggers.ok) errors.push(routerTriggers.error)
|
|
94
|
+
|
|
95
|
+
// 3. Check README current version block
|
|
96
|
+
const readme = rText("README.md")
|
|
97
|
+
if (readme.ok) {
|
|
98
|
+
const versionMatch = readme.value.match(/Phiên bản hiện tại:\s*\n```text\n(\S+)/)
|
|
99
|
+
if (versionMatch && versionMatch[1] !== expectedVersion) {
|
|
100
|
+
errors.push(`README.md: current version says ${versionMatch[1]}, expected ${expectedVersion}`)
|
|
101
|
+
} else if (!versionMatch) {
|
|
102
|
+
errors.push("README.md: could not find current version block")
|
|
103
|
+
}
|
|
104
|
+
if (staticScenarioCount && !readme.value.includes("**" + staticScenarioCount + " static skill-routing scenarios**")) {
|
|
105
|
+
errors.push("README.md: static routing scenario count drift (actual: " + staticScenarioCount + ")")
|
|
106
|
+
}
|
|
107
|
+
if (routerCaseCount && !readme.value.includes("**" + routerCaseCount + " V2 router cases**")) {
|
|
108
|
+
errors.push("README.md: router case count drift (actual: " + routerCaseCount + ")")
|
|
109
|
+
}
|
|
110
|
+
} else {
|
|
111
|
+
errors.push(readme.error)
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
// 4. Check V11 docs for stale status markers
|
|
115
|
+
const v11Docs = ["docs/V11-PERCEPTION-ADAPTIVE.md", "docs/V11-PERCEPTION-ADAPTIVE-EXECUTION.md"]
|
|
116
|
+
for (const doc of v11Docs) {
|
|
117
|
+
const content = rText(doc)
|
|
118
|
+
if (!content.ok) {
|
|
119
|
+
errors.push(content.error)
|
|
120
|
+
continue
|
|
121
|
+
}
|
|
122
|
+
if (content.value.includes("Status: development")) {
|
|
123
|
+
errors.push(`${doc}: still marked as development`)
|
|
124
|
+
}
|
|
125
|
+
if (content.value.includes("11.0.0-dev.")) {
|
|
126
|
+
errors.push(`${doc}: still references dev version 11.0.0-dev.*`)
|
|
127
|
+
}
|
|
128
|
+
if (content.value.includes("npm `latest` remains V10")) {
|
|
129
|
+
errors.push(`${doc}: still says V10 remains npm latest`)
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
// 5. Verify docs reflect actual counts (skip historical sections)
|
|
134
|
+
const docsToCheck = ["README.md", "docs/ENGINEERING-DESIGN.md", "docs/OPENCODE-COMPAT.md"]
|
|
135
|
+
for (const doc of docsToCheck) {
|
|
136
|
+
const content = rText(doc)
|
|
137
|
+
if (!content.ok) {
|
|
138
|
+
errors.push(content.error)
|
|
139
|
+
continue
|
|
140
|
+
}
|
|
141
|
+
const sections = content.value.split(/\n(?=#{1,6}\s)/)
|
|
142
|
+
for (const section of sections) {
|
|
143
|
+
const isHistorical = /^\s*#{1,6}\s+.*(V\d+\.|UES\s+\d+\.|historical|Historical)/im.test(section)
|
|
144
|
+
if (isHistorical) continue
|
|
145
|
+
if (section.includes("39 skills") || section.includes("39 namespaced skills")) {
|
|
146
|
+
errors.push(`${doc}: current section still references 39 skills (actual: ${skillCount})`)
|
|
147
|
+
}
|
|
148
|
+
if (section.includes("10 namespaced subagents") || section.includes("10 subagents")) {
|
|
149
|
+
errors.push(`${doc}: current section still references 10 subagents (actual: ${agentCount})`)
|
|
150
|
+
}
|
|
151
|
+
if (section.includes("34 static") && section.includes("scenarios")) {
|
|
152
|
+
errors.push(`${doc}: current section still references 34 static scenarios (actual: ${staticScenarioCount})`)
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
// 6. Check CI workflow
|
|
158
|
+
const ciYaml = rText(".github/workflows/ci.yml")
|
|
159
|
+
if (ciYaml.ok) {
|
|
160
|
+
if (!ciYaml.value.includes("evals:v11:validate")) {
|
|
161
|
+
errors.push(".github/workflows/ci.yml: missing evals:v11:validate job")
|
|
162
|
+
}
|
|
163
|
+
if (!/^\s*name:\s*CI Gate\s*$/m.test(ciYaml.value)) errors.push(".github/workflows/ci.yml: missing CI Gate aggregate job name")
|
|
164
|
+
if (!/needs:\s*\[\s*static\s*,\s*unit\s*,\s*package\s*\]/m.test(ciYaml.value)) errors.push(".github/workflows/ci.yml: CI Gate must depend on static, unit and package")
|
|
165
|
+
if (!/if:\s*always\(\)/m.test(ciYaml.value)) errors.push(".github/workflows/ci.yml: aggregate gate must use if: always()")
|
|
166
|
+
if (!ciYaml.value.includes("docs:check")) errors.push(".github/workflows/ci.yml: missing docs:check job")
|
|
167
|
+
for (const check of ["evals:v12:validate","evals:repo-scale:validate"]) {
|
|
168
|
+
if (!ciYaml.value.includes(check)) errors.push(".github/workflows/ci.yml: missing " + check + " job")
|
|
169
|
+
}
|
|
170
|
+
} else {
|
|
171
|
+
errors.push(ciYaml.error)
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
// 7. Check Security workflow
|
|
175
|
+
const securityYaml = rText(".github/workflows/security.yml")
|
|
176
|
+
if (securityYaml.ok) {
|
|
177
|
+
if (!/^\s*name:\s*Security Gate\s*$/m.test(securityYaml.value)) errors.push(".github/workflows/security.yml: missing Security Gate aggregate job name")
|
|
178
|
+
if (!/needs:\s*\[\s*codeql\s*,\s*dependency-review\s*\]/m.test(securityYaml.value)) errors.push(".github/workflows/security.yml: Security Gate must depend on codeql and dependency-review")
|
|
179
|
+
} else {
|
|
180
|
+
errors.push(securityYaml.error)
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
// 8. Check publish workflow idempotency
|
|
184
|
+
const publishYaml = rText(".github/workflows/publish.yml")
|
|
185
|
+
if (publishYaml.ok) {
|
|
186
|
+
if (!publishYaml.value.includes("already published")) {
|
|
187
|
+
errors.push(".github/workflows/publish.yml: missing idempotency check for existing versions")
|
|
188
|
+
}
|
|
189
|
+
} else {
|
|
190
|
+
errors.push(publishYaml.error)
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
// 9. Check npm script docs:check exists
|
|
194
|
+
if (pkg.ok) {
|
|
195
|
+
const pkgScripts = pkg.value.scripts || {}
|
|
196
|
+
if (!pkgScripts["docs:check"]) {
|
|
197
|
+
warnings.push("package.json: missing docs:check npm script")
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
return {
|
|
202
|
+
pass: errors.length === 0,
|
|
203
|
+
errors,
|
|
204
|
+
warnings,
|
|
205
|
+
version: expectedVersion,
|
|
206
|
+
skillCount,
|
|
207
|
+
commandCount,
|
|
208
|
+
subagentCount: agentCount,
|
|
209
|
+
staticScenarioCount,
|
|
210
|
+
routerCaseCount,
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
if (process.argv[1] === fileURLToPath(import.meta.url)) {
|
|
215
|
+
const result = checkReleaseConsistency(process.env.UES_BUNDLE_ROOT || DEFAULT_ROOT)
|
|
216
|
+
if (result.warnings.length) {
|
|
217
|
+
console.warn("Warnings:")
|
|
218
|
+
for (const w of result.warnings) console.warn(` WARN: ${w}`)
|
|
219
|
+
}
|
|
220
|
+
if (result.errors.length) {
|
|
221
|
+
console.error("Release consistency check FAILED:")
|
|
222
|
+
for (const e of result.errors) console.error(` - ${e}`)
|
|
223
|
+
process.exit(1)
|
|
224
|
+
}
|
|
225
|
+
console.log(
|
|
226
|
+
`Release consistency check PASS: package=${result.version}, skills=${result.skillCount}, commands=${result.commandCount}, subagents=${result.subagentCount}`,
|
|
227
|
+
)
|
|
228
|
+
}
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { mkdtemp, readFile, rm } from "node:fs/promises"
|
|
3
|
+
import os from "node:os"
|
|
4
|
+
import path from "node:path"
|
|
5
|
+
import { fileURLToPath } from "node:url"
|
|
6
|
+
import { generateRepoScaleFixture } from "../lib/repo-scale-fixture.mjs"
|
|
7
|
+
|
|
8
|
+
const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..")
|
|
9
|
+
const suite = JSON.parse(await readFile(path.join(root, "evals", "repo-scale", "tasks.json"), "utf8"))
|
|
10
|
+
const errors = []
|
|
11
|
+
if (!Array.isArray(suite.tasks) || suite.tasks.length < 4) errors.push("repo-scale suite must define at least 4 tasks")
|
|
12
|
+
for (const task of suite.tasks || []) {
|
|
13
|
+
if (!task.id || !task.objective || task.objective.length < 40) errors.push((task.id || "<missing>") + ": objective too weak")
|
|
14
|
+
if (!Array.isArray(task.requiredFiles) || task.requiredFiles.length < 2) errors.push((task.id || "<missing>") + ": requiredFiles must contain at least 2 paths")
|
|
15
|
+
if (!Array.isArray(task.acceptance) || task.acceptance.length < 2) errors.push((task.id || "<missing>") + ": acceptance must contain at least 2 checks")
|
|
16
|
+
}
|
|
17
|
+
const tmp = await mkdtemp(path.join(os.tmpdir(), "ues-repo-scale-"))
|
|
18
|
+
try {
|
|
19
|
+
const generated = await generateRepoScaleFixture(tmp)
|
|
20
|
+
if (generated.generatedModules < Number(suite.minimumGeneratedModules || 300)) errors.push("generated repo too small: " + generated.generatedModules + " modules")
|
|
21
|
+
} finally { await rm(tmp, { recursive: true, force: true }) }
|
|
22
|
+
if (errors.length) {
|
|
23
|
+
console.error("Repo-scale suite validation failed:")
|
|
24
|
+
for (const error of errors) console.error("- " + error)
|
|
25
|
+
process.exit(1)
|
|
26
|
+
}
|
|
27
|
+
console.log("Validated repo-scale suite: " + suite.tasks.length + " tasks, >=" + suite.minimumGeneratedModules + " generated modules.")
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { existsSync } from "node:fs"
|
|
3
|
+
import { readFile } from "node:fs/promises"
|
|
4
|
+
import path from "node:path"
|
|
5
|
+
import { fileURLToPath } from "node:url"
|
|
6
|
+
|
|
7
|
+
const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..")
|
|
8
|
+
const errors = []
|
|
9
|
+
const required = [
|
|
10
|
+
"lib/model-performance.mjs","lib/context-quality.mjs","lib/work-plan-scope.mjs",
|
|
11
|
+
"lib/decision-policy.mjs","lib/repo-scale-fixture.mjs","evals/repo-scale/tasks.json",
|
|
12
|
+
"scripts/validate-repo-scale-suite.mjs",
|
|
13
|
+
]
|
|
14
|
+
for (const relative of required) if (!existsSync(path.join(root, relative))) errors.push("missing V12 foundation file: " + relative)
|
|
15
|
+
const pkg = JSON.parse(await readFile(path.join(root, "package.json"), "utf8"))
|
|
16
|
+
for (const script of ["evals:v12:validate","evals:repo-scale:validate"]) if (!pkg.scripts?.[script]) errors.push("package.json missing script " + script)
|
|
17
|
+
const tasks = JSON.parse(await readFile(path.join(root, "evals", "repo-scale", "tasks.json"), "utf8"))
|
|
18
|
+
if ((tasks.tasks || []).length < 4) errors.push("V12 repo-scale suite needs at least 4 contract tasks")
|
|
19
|
+
if (errors.length) {
|
|
20
|
+
console.error("V12 foundation validation failed:")
|
|
21
|
+
for (const error of errors) console.error("- " + error)
|
|
22
|
+
process.exit(1)
|
|
23
|
+
}
|
|
24
|
+
console.log("Validated V12 weak-model intelligence foundation files and repo-scale contracts.")
|