opencode-agent-skill 9.0.0 → 11.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +116 -0
- package/README.md +742 -675
- package/bin/ocskill.mjs +354 -5
- package/docs/V11-PERCEPTION-ADAPTIVE-EXECUTION.md +75 -0
- package/docs/V11-PERCEPTION-ADAPTIVE.md +220 -0
- package/evals/router-triggers.json +82 -0
- package/evals/routing.json +76 -0
- package/evals/v11/tasks.json +122 -0
- package/global-config/AGENTS.md +78 -163
- package/global-config/agents/merge-arbiter.md +12 -0
- package/global-config/agents/visual-verifier.md +12 -0
- package/global-config/plugins/ues-router/index.js +683 -59
- package/global-config/plugins/ues-router/router.js +62 -3
- package/global-config/plugins/ues-router/runtime-guard.js +265 -0
- package/global-config/skills/browser-qa/SKILL.md +14 -0
- package/global-config/skills/browser-qa/references/workflow.md +11 -0
- package/global-config/skills/browser-security/SKILL.md +12 -0
- package/global-config/skills/component-visual-testing/SKILL.md +10 -0
- package/global-config/skills/design-source/SKILL.md +10 -0
- package/global-config/skills/design-source/references/workflow.md +12 -0
- package/global-config/skills/dynamic-workflow/SKILL.md +18 -0
- package/global-config/skills/dynamic-workflow/references/workflow.md +19 -0
- package/global-config/skills/responsive-verification/SKILL.md +10 -0
- package/global-config/skills/skill-authoring/SKILL.md +12 -0
- package/global-config/skills/skill-evaluation/SKILL.md +17 -0
- package/global-config/skills/visual-fidelity/SKILL.md +14 -0
- package/global-config/skills/visual-fidelity/references/workflow.md +14 -0
- package/lib/benchmark-confidence.mjs +49 -11
- package/lib/browser-adapter.mjs +82 -0
- package/lib/browser-runtime.mjs +193 -0
- package/lib/capability-registry.mjs +109 -0
- package/lib/context-engine-v11.mjs +146 -0
- package/lib/context-manifest.mjs +16 -3
- package/lib/control-center.mjs +12 -2
- package/lib/dynamic-workflow.mjs +179 -0
- package/lib/eval-ablation.mjs +146 -0
- package/lib/eval-report.mjs +83 -0
- package/lib/eval-telemetry.mjs +64 -0
- package/lib/evidence-budget.mjs +84 -0
- package/lib/evidence-store.mjs +178 -0
- package/lib/hermes-bridge.mjs +45 -1
- package/lib/model-config.mjs +9 -1
- package/lib/model-policy.mjs +58 -1
- package/lib/orchestrator-policy.mjs +100 -7
- package/lib/png-diff.mjs +229 -0
- package/lib/prompt-cache.mjs +60 -0
- package/lib/skill-quality.mjs +72 -0
- package/lib/task-engine.mjs +223 -12
- package/lib/ui-inspector.mjs +152 -0
- package/lib/v11-metrics.mjs +64 -0
- package/lib/visual-spec.mjs +159 -0
- package/package.json +11 -5
- package/scripts/eval-ablation.mjs +47 -0
- package/scripts/eval-matrix.mjs +13 -2
- package/scripts/validate-v11-suite.mjs +58 -0
- package/scripts/validate.mjs +16 -4
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
import { createHash } from "node:crypto"
|
|
2
|
+
import { existsSync } from "node:fs"
|
|
3
|
+
import { mkdir, readFile, readdir, rm, stat, writeFile } from "node:fs/promises"
|
|
4
|
+
import path from "node:path"
|
|
5
|
+
|
|
6
|
+
const STORE_VERSION = 1
|
|
7
|
+
const REF_PREFIX = "evidence:sha256:"
|
|
8
|
+
|
|
9
|
+
function stableValue(value) {
|
|
10
|
+
if (Array.isArray(value)) return value.map(stableValue)
|
|
11
|
+
if (!value || typeof value !== "object" || Buffer.isBuffer(value)) return value
|
|
12
|
+
return Object.fromEntries(
|
|
13
|
+
Object.keys(value).sort().map((key) => [key, stableValue(value[key])]),
|
|
14
|
+
)
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
function toBytes(value, options = {}) {
|
|
18
|
+
if (Buffer.isBuffer(value)) return { bytes: value, encoding: "binary", mediaType: options.mediaType || "application/octet-stream" }
|
|
19
|
+
if (typeof value === "string") return { bytes: Buffer.from(value, "utf8"), encoding: "utf8", mediaType: options.mediaType || "text/plain; charset=utf-8" }
|
|
20
|
+
const text = JSON.stringify(stableValue(value), null, options.pretty === false ? 0 : 2)
|
|
21
|
+
return { bytes: Buffer.from(text, "utf8"), encoding: "utf8", mediaType: options.mediaType || "application/json" }
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
function normalizeHash(ref) {
|
|
25
|
+
const value = String(ref || "").trim()
|
|
26
|
+
const hash = value.startsWith(REF_PREFIX) ? value.slice(REF_PREFIX.length) : value.replace(/^sha256:/, "")
|
|
27
|
+
if (!/^[a-f0-9]{64}$/i.test(hash)) throw new Error("Invalid evidence reference")
|
|
28
|
+
return hash.toLowerCase()
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
export function evidenceReference(hash) {
|
|
32
|
+
return REF_PREFIX + normalizeHash(hash)
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
export function evidenceStoreRoot(root = process.cwd()) {
|
|
36
|
+
return path.join(path.resolve(root), ".ues-cache", "evidence-v1")
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
function evidencePaths(root, hash) {
|
|
40
|
+
const normalized = normalizeHash(hash)
|
|
41
|
+
const dir = path.join(evidenceStoreRoot(root), normalized.slice(0, 2))
|
|
42
|
+
return {
|
|
43
|
+
dir,
|
|
44
|
+
meta: path.join(dir, normalized + ".json"),
|
|
45
|
+
data: path.join(dir, normalized + ".blob"),
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
export async function putEvidence(root, value, options = {}) {
|
|
50
|
+
root = path.resolve(root)
|
|
51
|
+
const encoded = toBytes(value, options)
|
|
52
|
+
const hash = createHash("sha256").update(encoded.bytes).digest("hex")
|
|
53
|
+
const target = evidencePaths(root, hash)
|
|
54
|
+
await mkdir(target.dir, { recursive: true })
|
|
55
|
+
|
|
56
|
+
const now = new Date().toISOString()
|
|
57
|
+
const existing = existsSync(target.meta)
|
|
58
|
+
? JSON.parse(await readFile(target.meta, "utf8").catch(() => "{}"))
|
|
59
|
+
: null
|
|
60
|
+
|
|
61
|
+
if (!existsSync(target.data)) await writeFile(target.data, encoded.bytes)
|
|
62
|
+
|
|
63
|
+
const preview = encoded.encoding === "utf8"
|
|
64
|
+
? encoded.bytes.toString("utf8", 0, Math.min(encoded.bytes.length, 600))
|
|
65
|
+
: null
|
|
66
|
+
|
|
67
|
+
const metadata = {
|
|
68
|
+
schemaVersion: STORE_VERSION,
|
|
69
|
+
ref: evidenceReference(hash),
|
|
70
|
+
sha256: hash,
|
|
71
|
+
bytes: encoded.bytes.length,
|
|
72
|
+
encoding: encoded.encoding,
|
|
73
|
+
mediaType: encoded.mediaType,
|
|
74
|
+
kind: options.kind || existing?.kind || "tool-output",
|
|
75
|
+
source: options.source || existing?.source || null,
|
|
76
|
+
summary: options.summary || existing?.summary || null,
|
|
77
|
+
createdAt: existing?.createdAt || now,
|
|
78
|
+
lastSeenAt: now,
|
|
79
|
+
preview,
|
|
80
|
+
}
|
|
81
|
+
await writeFile(target.meta, JSON.stringify(metadata, null, 2) + "\n", "utf8")
|
|
82
|
+
return metadata
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
export async function getEvidence(root, ref, options = {}) {
|
|
86
|
+
const hash = normalizeHash(ref)
|
|
87
|
+
const target = evidencePaths(root, hash)
|
|
88
|
+
const metadata = JSON.parse(await readFile(target.meta, "utf8"))
|
|
89
|
+
const bytes = await readFile(target.data)
|
|
90
|
+
const start = Math.max(0, Number(options.start || 0))
|
|
91
|
+
const maxBytes = Math.max(1, Number(options.maxBytes || options.maxChars || 24_000))
|
|
92
|
+
const slice = bytes.subarray(start, Math.min(bytes.length, start + maxBytes))
|
|
93
|
+
return {
|
|
94
|
+
...metadata,
|
|
95
|
+
truncated: start + slice.length < bytes.length,
|
|
96
|
+
start,
|
|
97
|
+
returnedBytes: slice.length,
|
|
98
|
+
content: metadata.encoding === "utf8" ? slice.toString("utf8") : slice.toString("base64"),
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
async function metadataFiles(root) {
|
|
103
|
+
const base = evidenceStoreRoot(root)
|
|
104
|
+
const prefixes = await readdir(base, { withFileTypes: true }).catch(() => [])
|
|
105
|
+
const files = []
|
|
106
|
+
for (const prefix of prefixes) {
|
|
107
|
+
if (!prefix.isDirectory()) continue
|
|
108
|
+
const dir = path.join(base, prefix.name)
|
|
109
|
+
for (const entry of await readdir(dir, { withFileTypes: true }).catch(() => [])) {
|
|
110
|
+
if (entry.isFile() && entry.name.endsWith(".json")) files.push(path.join(dir, entry.name))
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
return files
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
export async function evidenceStoreStatus(root = process.cwd()) {
|
|
117
|
+
const files = await metadataFiles(root)
|
|
118
|
+
let bytes = 0
|
|
119
|
+
let oldest = null
|
|
120
|
+
let newest = null
|
|
121
|
+
for (const file of files) {
|
|
122
|
+
const meta = JSON.parse(await readFile(file, "utf8").catch(() => "{}"))
|
|
123
|
+
bytes += Number(meta.bytes || 0)
|
|
124
|
+
if (meta.createdAt && (!oldest || meta.createdAt < oldest)) oldest = meta.createdAt
|
|
125
|
+
if (meta.lastSeenAt && (!newest || meta.lastSeenAt > newest)) newest = meta.lastSeenAt
|
|
126
|
+
}
|
|
127
|
+
return {
|
|
128
|
+
schemaVersion: STORE_VERSION,
|
|
129
|
+
root: evidenceStoreRoot(root),
|
|
130
|
+
entries: files.length,
|
|
131
|
+
bytes,
|
|
132
|
+
oldest,
|
|
133
|
+
newest,
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
export async function gcEvidenceStore(root = process.cwd(), options = {}) {
|
|
138
|
+
const files = await metadataFiles(root)
|
|
139
|
+
const maxEntries = Math.max(10, Number(options.maxEntries || 2_000))
|
|
140
|
+
const maxAgeMs = Math.max(0, Number(options.maxAgeDays ?? 30)) * 86_400_000
|
|
141
|
+
const now = Date.now()
|
|
142
|
+
const rows = []
|
|
143
|
+
for (const metaFile of files) {
|
|
144
|
+
const meta = JSON.parse(await readFile(metaFile, "utf8").catch(() => "{}"))
|
|
145
|
+
const time = Date.parse(meta.lastSeenAt || meta.createdAt || 0) || 0
|
|
146
|
+
rows.push({ metaFile, meta, time })
|
|
147
|
+
}
|
|
148
|
+
rows.sort((a, b) => b.time - a.time)
|
|
149
|
+
|
|
150
|
+
const removed = []
|
|
151
|
+
for (let index = 0; index < rows.length; index += 1) {
|
|
152
|
+
const row = rows[index]
|
|
153
|
+
const tooMany = index >= maxEntries
|
|
154
|
+
const tooOld = maxAgeMs > 0 && row.time > 0 && now - row.time > maxAgeMs
|
|
155
|
+
if (!tooMany && !tooOld) continue
|
|
156
|
+
const hash = row.meta.sha256 || path.basename(row.metaFile, ".json")
|
|
157
|
+
const target = evidencePaths(root, hash)
|
|
158
|
+
await rm(target.meta, { force: true })
|
|
159
|
+
await rm(target.data, { force: true })
|
|
160
|
+
removed.push(evidenceReference(hash))
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
return {
|
|
164
|
+
removed,
|
|
165
|
+
removedCount: removed.length,
|
|
166
|
+
status: await evidenceStoreStatus(root),
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
export async function evidenceExists(root, ref) {
|
|
171
|
+
try {
|
|
172
|
+
const target = evidencePaths(root, normalizeHash(ref))
|
|
173
|
+
const info = await stat(target.data)
|
|
174
|
+
return info.isFile()
|
|
175
|
+
} catch {
|
|
176
|
+
return false
|
|
177
|
+
}
|
|
178
|
+
}
|
package/lib/hermes-bridge.mjs
CHANGED
|
@@ -1,11 +1,20 @@
|
|
|
1
1
|
import { spawnSync } from "node:child_process"
|
|
2
|
+
import { resolveWindowsCommand } from "./windows-shim.mjs"
|
|
3
|
+
|
|
4
|
+
function runHermesVersion() {
|
|
5
|
+
if (process.platform !== "win32") return spawnSync("hermes", ["--version"], { encoding: "utf8" })
|
|
6
|
+
const resolved = resolveWindowsCommand("hermes")
|
|
7
|
+
if (!resolved) return { status: 127, stdout: "", stderr: "Hermes CLI not found" }
|
|
8
|
+
return spawnSync(resolved.executable, [...resolved.argsPrefix, "--version"], { encoding: "utf8" })
|
|
9
|
+
}
|
|
2
10
|
|
|
3
11
|
export function hermesStatus() {
|
|
4
|
-
const result =
|
|
12
|
+
const result = runHermesVersion()
|
|
5
13
|
return {
|
|
6
14
|
available: result.status === 0,
|
|
7
15
|
version: result.status === 0 ? String(result.stdout || result.stderr || "").trim() : null,
|
|
8
16
|
error: result.status === 0 ? null : String(result.stderr || result.stdout || "Hermes CLI not found").trim(),
|
|
17
|
+
mode: "optional-sidecar",
|
|
9
18
|
}
|
|
10
19
|
}
|
|
11
20
|
|
|
@@ -15,11 +24,46 @@ export function buildHermesDelegationPrompt(contextPack) {
|
|
|
15
24
|
"Implement exactly the approved task described below.",
|
|
16
25
|
"Do not broaden scope, merge, push, publish, deploy, or alter durable UES state.",
|
|
17
26
|
"Run the declared verification and return a concise structured report.",
|
|
27
|
+
"Large evidence is referenced by evidence:sha256 pointers; fetch only the slice required for the task.",
|
|
28
|
+
"",
|
|
29
|
+
JSON.stringify(contextPack, null, 2),
|
|
30
|
+
].join("\n")
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export function buildHermesWorkflowPrompt(contextPack, schedule) {
|
|
34
|
+
return [
|
|
35
|
+
"You are the optional Hermes sidecar for a UES V11 dynamic workflow.",
|
|
36
|
+
"Follow the supplied bounded wave schedule. Deterministic tasks are not delegated to LLM children.",
|
|
37
|
+
"Never exceed declared task ownership or concurrency. Persist outputs/evidence to files instead of conversational summaries.",
|
|
38
|
+
"Do not merge, push, publish, deploy, or mutate durable UES state except through explicitly supplied UES commands.",
|
|
39
|
+
"",
|
|
40
|
+
"SCHEDULE:",
|
|
41
|
+
JSON.stringify(schedule, null, 2),
|
|
18
42
|
"",
|
|
43
|
+
"CONTEXT:",
|
|
19
44
|
JSON.stringify(contextPack, null, 2),
|
|
20
45
|
].join("\n")
|
|
21
46
|
}
|
|
22
47
|
|
|
48
|
+
export function hermesSidecarPlan(input = {}) {
|
|
49
|
+
return {
|
|
50
|
+
schemaVersion: 2,
|
|
51
|
+
adapter: "hermes",
|
|
52
|
+
optional: true,
|
|
53
|
+
mode: input.mode || "one-shot",
|
|
54
|
+
maxConcurrent: Math.max(1, Math.min(16, Number(input.maxConcurrent || 4))),
|
|
55
|
+
durableStateOwner: "ues",
|
|
56
|
+
evidenceTransport: "content-addressed-pointers",
|
|
57
|
+
allowNestedDelegation: input.allowNestedDelegation === true,
|
|
58
|
+
safety: {
|
|
59
|
+
merge: false,
|
|
60
|
+
push: false,
|
|
61
|
+
publish: false,
|
|
62
|
+
deploy: false,
|
|
63
|
+
destructiveGit: false,
|
|
64
|
+
},
|
|
65
|
+
}
|
|
66
|
+
}
|
|
23
67
|
|
|
24
68
|
export function hermesOneShotArgs(prompt) {
|
|
25
69
|
const text = String(prompt || "").trim()
|
package/lib/model-config.mjs
CHANGED
|
@@ -2,6 +2,7 @@ import { existsSync } from "node:fs"
|
|
|
2
2
|
import { mkdir, readFile, writeFile } from "node:fs/promises"
|
|
3
3
|
import path from "node:path"
|
|
4
4
|
import { defaultModelPolicy, resolveModel } from "./model-policy.mjs"
|
|
5
|
+
import { normalizeCapabilityProfile } from "./capability-registry.mjs"
|
|
5
6
|
|
|
6
7
|
const TIERS = new Set(["light", "standard", "heavy"])
|
|
7
8
|
|
|
@@ -23,14 +24,20 @@ function normalize(policy) {
|
|
|
23
24
|
if (TIERS.has(tier)) roleTiers[role] = tier
|
|
24
25
|
}
|
|
25
26
|
|
|
27
|
+
const capabilities = {}
|
|
28
|
+
for (const [model, profile] of Object.entries(input.capabilities || {})) {
|
|
29
|
+
if (validModelID(model)) capabilities[model] = normalizeCapabilityProfile(profile)
|
|
30
|
+
}
|
|
31
|
+
|
|
26
32
|
return {
|
|
27
|
-
schemaVersion:
|
|
33
|
+
schemaVersion: 2,
|
|
28
34
|
enabled: input.enabled === true,
|
|
29
35
|
maxEscalations: Number.isInteger(input.maxEscalations)
|
|
30
36
|
? Math.max(0, Math.min(input.maxEscalations, 2))
|
|
31
37
|
: base.maxEscalations,
|
|
32
38
|
tiers,
|
|
33
39
|
roleTiers,
|
|
40
|
+
capabilities,
|
|
34
41
|
}
|
|
35
42
|
}
|
|
36
43
|
|
|
@@ -56,6 +63,7 @@ export async function writeModelPolicy(configDir, patch = {}) {
|
|
|
56
63
|
...patch,
|
|
57
64
|
tiers: { ...current.tiers, ...(patch.tiers || {}) },
|
|
58
65
|
roleTiers: { ...current.roleTiers, ...(patch.roleTiers || {}) },
|
|
66
|
+
capabilities: { ...(current.capabilities || {}), ...(patch.capabilities || {}) },
|
|
59
67
|
})
|
|
60
68
|
const file = modelPolicyFile(configDir)
|
|
61
69
|
await mkdir(path.dirname(file), { recursive: true })
|
package/lib/model-policy.mjs
CHANGED
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { inferTaskCapabilities, modelCandidatesFromPolicy, selectCapabilityCandidate } from "./capability-registry.mjs"
|
|
2
|
+
|
|
1
3
|
const ROLE_TIERS = {
|
|
2
4
|
"codebase-mapper": "standard",
|
|
3
5
|
architect: "heavy",
|
|
@@ -9,6 +11,8 @@ const ROLE_TIERS = {
|
|
|
9
11
|
critic: "heavy",
|
|
10
12
|
verifier: "standard",
|
|
11
13
|
"integration-verifier": "heavy",
|
|
14
|
+
"visual-verifier": "standard",
|
|
15
|
+
"merge-arbiter": "heavy",
|
|
12
16
|
}
|
|
13
17
|
|
|
14
18
|
const ORDER = ["light", "standard", "heavy"]
|
|
@@ -41,11 +45,12 @@ export function resolveModel(role, attempt, config = {}) {
|
|
|
41
45
|
|
|
42
46
|
export function defaultModelPolicy() {
|
|
43
47
|
return {
|
|
44
|
-
schemaVersion:
|
|
48
|
+
schemaVersion: 2,
|
|
45
49
|
enabled: false,
|
|
46
50
|
maxEscalations: 2,
|
|
47
51
|
tiers: { light: null, standard: null, heavy: null },
|
|
48
52
|
roleTiers: { ...ROLE_TIERS },
|
|
53
|
+
capabilities: {},
|
|
49
54
|
}
|
|
50
55
|
}
|
|
51
56
|
|
|
@@ -58,14 +63,66 @@ export function resolveAdaptiveModel(role, attempt, taskPolicy = {}, config = {}
|
|
|
58
63
|
...config,
|
|
59
64
|
roleTiers: { ...(config.roleTiers || {}), [role]: baseTier },
|
|
60
65
|
})
|
|
66
|
+
const normalizedAttempt = Math.max(1, Number(attempt || 1))
|
|
61
67
|
return {
|
|
62
68
|
...resolved,
|
|
69
|
+
recoveryStage:
|
|
70
|
+
normalizedAttempt <= 1 ? "initial" :
|
|
71
|
+
normalizedAttempt === 2 ? "diagnose" :
|
|
72
|
+
"deep-recovery",
|
|
63
73
|
policy: {
|
|
64
74
|
mode: taskPolicy.mode || null,
|
|
65
75
|
risk: taskPolicy.risk || null,
|
|
76
|
+
executionProfile: taskPolicy.executionProfile || null,
|
|
66
77
|
score: Number(taskPolicy.score || 0),
|
|
67
78
|
maxAttempts: Number(taskPolicy.maxAttempts || 0) || null,
|
|
68
79
|
contextBudget: Number(taskPolicy.contextBudget || 0) || null,
|
|
69
80
|
},
|
|
70
81
|
}
|
|
71
82
|
}
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
export function resolveCapabilityModel(role, attempt, taskText = "", taskPolicy = {}, config = {}, facts = {}) {
|
|
86
|
+
const base = resolveAdaptiveModel(role, attempt, taskPolicy, config)
|
|
87
|
+
const requirements = inferTaskCapabilities(taskText, {
|
|
88
|
+
...facts,
|
|
89
|
+
...(role === "visual-verifier" && facts.coding === undefined ? { coding: false } : {}),
|
|
90
|
+
})
|
|
91
|
+
const candidates = modelCandidatesFromPolicy(config)
|
|
92
|
+
.filter((candidate) => tierIndex(candidate.tier) >= tierIndex(base.tier))
|
|
93
|
+
const selection = selectCapabilityCandidate(requirements, candidates, {
|
|
94
|
+
role,
|
|
95
|
+
preferredTier: base.tier,
|
|
96
|
+
})
|
|
97
|
+
const capabilityEnforced = config.enabled === true && candidates.length > 0
|
|
98
|
+
if (capabilityEnforced && !selection.selected) {
|
|
99
|
+
return {
|
|
100
|
+
...base,
|
|
101
|
+
model: null,
|
|
102
|
+
capabilityRequirements: requirements,
|
|
103
|
+
capabilitySelection: selection,
|
|
104
|
+
capabilityFallback: true,
|
|
105
|
+
capabilityBlocked: true,
|
|
106
|
+
capabilityBlockReason: "no-configured-model-satisfies-required-capabilities",
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
if (!config.enabled || !selection.selected) {
|
|
110
|
+
return {
|
|
111
|
+
...base,
|
|
112
|
+
capabilityRequirements: requirements,
|
|
113
|
+
capabilitySelection: selection,
|
|
114
|
+
capabilityFallback: selection.fallbackNeeded,
|
|
115
|
+
capabilityBlocked: false,
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
const selected = selection.selected
|
|
119
|
+
return {
|
|
120
|
+
...base,
|
|
121
|
+
tier: selected.tier || base.tier,
|
|
122
|
+
model: selected.id,
|
|
123
|
+
capabilityRequirements: requirements,
|
|
124
|
+
capabilitySelection: selection,
|
|
125
|
+
capabilityFallback: false,
|
|
126
|
+
capabilityBlocked: false,
|
|
127
|
+
}
|
|
128
|
+
}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
const HIGH_RISK = /(
|
|
1
|
+
const HIGH_RISK = /(\bauth\b|authorization|authentication|security|permission|payment|migration|schema|database|production|deploy|public api|breaking|secret|credential|phân quyền|bảo mật|thanh toán|cơ sở dữ liệu|triển khai|migrate|migration)/i
|
|
2
2
|
const LONG = /(whole repo|whole repository|whole project|entire repo|entire project|large refactor|major refactor|long[- ](?:running|horizon)|durable state|dependency graph|integration verification|resume|migration|refactor all|toàn bộ repo|toàn bộ repository|toàn bộ dự án|toàn bộ project|refactor lớn|tác vụ dài|nhiều file|nhiều module|tiếp tục công việc|refactor toàn bộ|xác minh tích hợp|kiểm tra tích hợp|chia (?:công việc|task|tác vụ).*(?:dependency|phụ thuộc))/i
|
|
3
3
|
const DEBUG = /(fix|bug|debug|crash|regression|failure|error|broken|sửa lỗi|lỗi|điều tra lỗi|không chạy)/i
|
|
4
4
|
const CONTRACT = /(public api|api contract|openapi|response schema|request schema|breaking api|hợp đồng api|api công khai)/i
|
|
@@ -13,8 +13,9 @@ function profileFor(mode, risk) {
|
|
|
13
13
|
return {
|
|
14
14
|
name: "fast",
|
|
15
15
|
maxSkills: 2,
|
|
16
|
-
contextBudget:
|
|
16
|
+
contextBudget: 8_000,
|
|
17
17
|
contextStrategy: "incremental-semantic",
|
|
18
|
+
skillLoading: "direct-only",
|
|
18
19
|
durableState: false,
|
|
19
20
|
worktree: "off",
|
|
20
21
|
critic: "off",
|
|
@@ -29,6 +30,7 @@ function profileFor(mode, risk) {
|
|
|
29
30
|
maxSkills: 5,
|
|
30
31
|
contextBudget: 48_000,
|
|
31
32
|
contextStrategy: "semantic+graph+git",
|
|
33
|
+
skillLoading: "orchestrated",
|
|
32
34
|
durableState: true,
|
|
33
35
|
worktree: "auto-writers",
|
|
34
36
|
critic: "required-for-high-risk-or-final",
|
|
@@ -40,8 +42,9 @@ function profileFor(mode, risk) {
|
|
|
40
42
|
return {
|
|
41
43
|
name: "standard",
|
|
42
44
|
maxSkills: 4,
|
|
43
|
-
contextBudget:
|
|
45
|
+
contextBudget: 20_000,
|
|
44
46
|
contextStrategy: "incremental-semantic+git",
|
|
47
|
+
skillLoading: "selective",
|
|
45
48
|
durableState: false,
|
|
46
49
|
worktree: "auto-on-conflict",
|
|
47
50
|
critic: "on-failure-or-elevated-risk",
|
|
@@ -51,12 +54,97 @@ function profileFor(mode, risk) {
|
|
|
51
54
|
}
|
|
52
55
|
}
|
|
53
56
|
|
|
57
|
+
function boundedInt(value, fallback, min, max) {
|
|
58
|
+
const parsed = Number(value)
|
|
59
|
+
if (!Number.isFinite(parsed)) return fallback
|
|
60
|
+
return Math.min(max, Math.max(min, Math.round(parsed)))
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
export function recoveryPolicyForAttempt(taskPolicy = {}, attempt = 1) {
|
|
64
|
+
const normalizedAttempt = boundedInt(attempt, 1, 1, 99)
|
|
65
|
+
const baseBudget = boundedInt(
|
|
66
|
+
taskPolicy.contextBudget ?? taskPolicy.profile?.contextBudget,
|
|
67
|
+
20_000,
|
|
68
|
+
4_000,
|
|
69
|
+
48_000,
|
|
70
|
+
)
|
|
71
|
+
const baseSkills = boundedInt(
|
|
72
|
+
taskPolicy.maxSkills ?? taskPolicy.profile?.maxSkills,
|
|
73
|
+
4,
|
|
74
|
+
1,
|
|
75
|
+
5,
|
|
76
|
+
)
|
|
77
|
+
const baseStrategy = taskPolicy.profile?.contextStrategy || "incremental-semantic+git"
|
|
78
|
+
|
|
79
|
+
if (normalizedAttempt <= 1) {
|
|
80
|
+
return {
|
|
81
|
+
schemaVersion: 1,
|
|
82
|
+
stage: "initial",
|
|
83
|
+
attempt: normalizedAttempt,
|
|
84
|
+
contextBudget: baseBudget,
|
|
85
|
+
maxSkills: baseSkills,
|
|
86
|
+
contextStrategy: baseStrategy,
|
|
87
|
+
requireDiagnosis: false,
|
|
88
|
+
requireCritic: false,
|
|
89
|
+
modelEscalation: false,
|
|
90
|
+
directives: [],
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
if (normalizedAttempt === 2) {
|
|
95
|
+
return {
|
|
96
|
+
schemaVersion: 1,
|
|
97
|
+
stage: "diagnose",
|
|
98
|
+
attempt: normalizedAttempt,
|
|
99
|
+
contextBudget: Math.min(48_000, Math.max(baseBudget, Math.round(baseBudget * 1.35))),
|
|
100
|
+
maxSkills: Math.min(5, baseSkills + 1),
|
|
101
|
+
contextStrategy:
|
|
102
|
+
taskPolicy.executionProfile === "fast"
|
|
103
|
+
? "incremental-semantic+git"
|
|
104
|
+
: baseStrategy,
|
|
105
|
+
requireDiagnosis: true,
|
|
106
|
+
requireCritic: false,
|
|
107
|
+
modelEscalation: true,
|
|
108
|
+
directives: [
|
|
109
|
+
"reproduce or capture the exact previous failure before editing",
|
|
110
|
+
"inspect the direct caller, nearest test and failure-adjacent evidence",
|
|
111
|
+
"do not stack another speculative patch on top of the failed attempt",
|
|
112
|
+
],
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
return {
|
|
117
|
+
schemaVersion: 1,
|
|
118
|
+
stage: "deep-recovery",
|
|
119
|
+
attempt: normalizedAttempt,
|
|
120
|
+
contextBudget: Math.min(48_000, Math.max(20_000, Math.round(baseBudget * 1.75))),
|
|
121
|
+
maxSkills: Math.min(5, baseSkills + 2),
|
|
122
|
+
contextStrategy: "semantic+graph+git",
|
|
123
|
+
requireDiagnosis: true,
|
|
124
|
+
requireCritic: true,
|
|
125
|
+
modelEscalation: true,
|
|
126
|
+
directives: [
|
|
127
|
+
"re-investigate from fresh evidence and explicitly reject the failed hypothesis",
|
|
128
|
+
"expand to callers, dependencies, tests and boundary contracts before editing",
|
|
129
|
+
"challenge the architecture or coupling if repeated fixes expose a wider problem",
|
|
130
|
+
"run an independent critic or review pass before accepting the recovery",
|
|
131
|
+
],
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
|
|
54
135
|
export function classifyEngineeringTask(text, facts = {}) {
|
|
55
136
|
const value = String(text || "")
|
|
56
137
|
const declaredHighRisk =
|
|
57
138
|
["high", "critical"].includes(String(facts.risk || "").toLowerCase()) ||
|
|
58
|
-
/\b(?:risk|rủi ro)\s*[
|
|
59
|
-
|
|
139
|
+
/\b(?:risk|rủi ro)\s*[:=\/-]?\s*(?:high|critical|cao|nghiêm trọng)\b/i.test(value) ||
|
|
140
|
+
/\b(?:high|critical)[-\s]?(?:risk|rủi ro)\b/i.test(value)
|
|
141
|
+
const declaredLongHorizon =
|
|
142
|
+
facts.longHorizon === true ||
|
|
143
|
+
["long", "long-horizon", "deep"].includes(String(facts.mode || "").toLowerCase())
|
|
144
|
+
const compoundLongRisk =
|
|
145
|
+
/\blong\s*[/|,+]\s*(?:high|critical)[-\s]?risk\b/i.test(value) ||
|
|
146
|
+
/\b(?:high|critical)[-\s]?risk\s*[/|,+]\s*long\b/i.test(value)
|
|
147
|
+
const explicitLongHorizon = declaredLongHorizon || compoundLongRisk || LONG.test(value)
|
|
60
148
|
const signals = [
|
|
61
149
|
signal("long-request-text", value.length > 700, 1),
|
|
62
150
|
signal("medium-request-text", value.length > 250, 1),
|
|
@@ -91,8 +179,9 @@ export function classifyEngineeringTask(text, facts = {}) {
|
|
|
91
179
|
if (/(docker|kubernetes|terraform|github actions|ci\/cd|deploy|triển khai)/i.test(value)) domains.push("devops")
|
|
92
180
|
|
|
93
181
|
return {
|
|
94
|
-
schemaVersion:
|
|
182
|
+
schemaVersion: 4,
|
|
95
183
|
score,
|
|
184
|
+
signals,
|
|
96
185
|
risk,
|
|
97
186
|
mode,
|
|
98
187
|
executionProfile: profile.name,
|
|
@@ -105,7 +194,11 @@ export function classifyEngineeringTask(text, facts = {}) {
|
|
|
105
194
|
requireFreshEvidence: true,
|
|
106
195
|
profile,
|
|
107
196
|
domains: [...new Set(domains)],
|
|
108
|
-
|
|
197
|
+
recovery: {
|
|
198
|
+
escalateAfterFailure: true,
|
|
199
|
+
diagnosisBeforePatch: true,
|
|
200
|
+
deepRecoveryFromAttempt: 3,
|
|
201
|
+
},
|
|
109
202
|
antiHallucination: {
|
|
110
203
|
evidenceFirst: true,
|
|
111
204
|
noCompletionWithoutVerification: mode !== "inline" || risk === "high",
|