opencode-agent-skill 10.0.0 → 12.0.0-beta.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +85 -0
- package/README.md +60 -8
- package/bin/ocskill.mjs +354 -6
- package/docs/DETERMINISTIC-TOOLS.md +1 -1
- package/docs/ENGINEERING-DESIGN.md +4 -4
- package/docs/EVALS.md +3 -3
- package/docs/GITHUB-RULESET.md +50 -0
- package/docs/NPM-PUBLISH.md +4 -4
- package/docs/OPENCODE-COMPAT.md +3 -3
- package/docs/TRACE-SCHEMA.md +1 -1
- package/docs/V11-PERCEPTION-ADAPTIVE-EXECUTION.md +75 -0
- package/docs/V11-PERCEPTION-ADAPTIVE.md +220 -0
- package/docs/V12-WEAK-MODEL-INTELLIGENCE.md +27 -0
- package/evals/repo-scale/tasks.json +62 -0
- package/evals/router-triggers.json +82 -0
- package/evals/routing.json +76 -0
- package/evals/v11/tasks.json +122 -0
- package/global-config/agents/merge-arbiter.md +12 -0
- package/global-config/agents/visual-verifier.md +12 -0
- package/global-config/plugins/ues-router/index.js +272 -2
- package/global-config/plugins/ues-router/router.js +27 -3
- package/global-config/skills/browser-qa/SKILL.md +14 -0
- package/global-config/skills/browser-qa/references/workflow.md +11 -0
- package/global-config/skills/browser-security/SKILL.md +12 -0
- package/global-config/skills/component-visual-testing/SKILL.md +10 -0
- package/global-config/skills/design-source/SKILL.md +10 -0
- package/global-config/skills/design-source/references/workflow.md +12 -0
- package/global-config/skills/dynamic-workflow/SKILL.md +18 -0
- package/global-config/skills/dynamic-workflow/references/workflow.md +19 -0
- package/global-config/skills/responsive-verification/SKILL.md +10 -0
- package/global-config/skills/skill-authoring/SKILL.md +12 -0
- package/global-config/skills/skill-evaluation/SKILL.md +17 -0
- package/global-config/skills/visual-fidelity/SKILL.md +14 -0
- package/global-config/skills/visual-fidelity/references/workflow.md +14 -0
- package/lib/browser-adapter.mjs +82 -0
- package/lib/browser-runtime.mjs +193 -0
- package/lib/capability-registry.mjs +109 -0
- package/lib/context-engine-v11.mjs +150 -0
- package/lib/context-manifest.mjs +16 -3
- package/lib/context-quality.mjs +59 -0
- package/lib/control-center.mjs +12 -2
- package/lib/decision-policy.mjs +23 -0
- package/lib/dynamic-workflow.mjs +179 -0
- package/lib/eval-ablation.mjs +43 -1
- package/lib/eval-report.mjs +72 -0
- package/lib/eval-telemetry.mjs +61 -0
- package/lib/evidence-budget.mjs +84 -0
- package/lib/evidence-store.mjs +178 -0
- package/lib/hermes-bridge.mjs +45 -1
- package/lib/model-config.mjs +21 -1
- package/lib/model-performance.mjs +113 -0
- package/lib/model-policy.mjs +59 -1
- package/lib/orchestrator-policy.mjs +1 -1
- package/lib/png-diff.mjs +229 -0
- package/lib/prompt-cache.mjs +60 -0
- package/lib/repo-scale-fixture.mjs +45 -0
- package/lib/skill-quality.mjs +72 -0
- package/lib/task-engine.mjs +95 -7
- package/lib/ui-inspector.mjs +152 -0
- package/lib/v11-metrics.mjs +64 -0
- package/lib/visual-spec.mjs +159 -0
- package/lib/work-plan-scope.mjs +49 -0
- package/package.json +13 -5
- package/scripts/check-release-consistency.mjs +228 -0
- package/scripts/eval-ablation.mjs +4 -1
- package/scripts/validate-repo-scale-suite.mjs +27 -0
- package/scripts/validate-v11-suite.mjs +58 -0
- package/scripts/validate-v12-foundation.mjs +24 -0
- package/scripts/validate.mjs +16 -4
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
import { buildContextManifest } from "./context-manifest.mjs"
|
|
2
|
+
import { planEvidenceBudget, evidenceValueScore } from "./evidence-budget.mjs"
|
|
3
|
+
import { putEvidence } from "./evidence-store.mjs"
|
|
4
|
+
import { inferTaskCapabilities } from "./capability-registry.mjs"
|
|
5
|
+
import { buildPromptEnvelope, comparePromptEnvelopes } from "./prompt-cache.mjs"
|
|
6
|
+
import { measureContextQuality } from "./context-quality.mjs"
|
|
7
|
+
|
|
8
|
+
function taskText(task = {}) {
|
|
9
|
+
return [
|
|
10
|
+
task.title,
|
|
11
|
+
task.summary,
|
|
12
|
+
...(task.acceptance || []),
|
|
13
|
+
...(task.verification || []),
|
|
14
|
+
].filter(Boolean).join(" ")
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
function rankExcerpt(item = {}, task = {}) {
|
|
18
|
+
const text = taskText(task).toLowerCase()
|
|
19
|
+
const pathValue = String(item.path || "").toLowerCase()
|
|
20
|
+
const declared = item.role === "declared" ? 1 : 0
|
|
21
|
+
const test = item.role === "test" ? 0.9 : 0
|
|
22
|
+
const instruction = item.role === "instruction" ? 0.85 : 0
|
|
23
|
+
const pathMatch = text && pathValue
|
|
24
|
+
? text.split(/[^a-z0-9_$.-]+/i).filter((term) => term.length >= 4 && pathValue.includes(term.toLowerCase())).length
|
|
25
|
+
: 0
|
|
26
|
+
const relevance = Math.min(1, declared + test + instruction + pathMatch * 0.15)
|
|
27
|
+
return evidenceValueScore({
|
|
28
|
+
relevance,
|
|
29
|
+
freshness: 1,
|
|
30
|
+
confidence: item.role === "reference" ? 0.7 : 1,
|
|
31
|
+
chars: String(item.text || "").length || 1,
|
|
32
|
+
})
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
export async function externalizeContextExcerpts(root, manifest, options = {}) {
|
|
36
|
+
if (!manifest) return { manifest: null, externalized: [], externalizedBytes: 0 }
|
|
37
|
+
const threshold = Math.max(512, Number(options.threshold || 2_500))
|
|
38
|
+
const keepInline = Math.max(256, Number(options.inlineChars || 1_200))
|
|
39
|
+
const externalized = []
|
|
40
|
+
let externalizedBytes = 0
|
|
41
|
+
const excerpts = []
|
|
42
|
+
|
|
43
|
+
for (const item of manifest.excerpts || []) {
|
|
44
|
+
const text = String(item.text || "")
|
|
45
|
+
if (text.length <= threshold) {
|
|
46
|
+
excerpts.push(item)
|
|
47
|
+
continue
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
const stored = await putEvidence(root, text, {
|
|
51
|
+
kind: "context-excerpt",
|
|
52
|
+
source: item.path || null,
|
|
53
|
+
summary: `Externalized ${item.role || "reference"} context excerpt for ${item.path || "unknown"}`,
|
|
54
|
+
})
|
|
55
|
+
externalized.push({
|
|
56
|
+
ref: stored.ref,
|
|
57
|
+
path: item.path || null,
|
|
58
|
+
role: item.role || null,
|
|
59
|
+
bytes: stored.bytes,
|
|
60
|
+
score: Number(rankExcerpt(item, options.task).toFixed(8)),
|
|
61
|
+
})
|
|
62
|
+
externalizedBytes += stored.bytes
|
|
63
|
+
excerpts.push({
|
|
64
|
+
...item,
|
|
65
|
+
text: text.slice(0, keepInline) + "\n...[externalized: " + stored.ref + "]",
|
|
66
|
+
evidenceRef: stored.ref,
|
|
67
|
+
originalChars: text.length,
|
|
68
|
+
externalized: true,
|
|
69
|
+
})
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
return {
|
|
73
|
+
manifest: {
|
|
74
|
+
...manifest,
|
|
75
|
+
schemaVersion: Math.max(5, Number(manifest.schemaVersion || 0)),
|
|
76
|
+
excerpts,
|
|
77
|
+
evidencePointers: externalized,
|
|
78
|
+
},
|
|
79
|
+
externalized,
|
|
80
|
+
externalizedBytes,
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
export async function buildAdaptiveTaskContext(root, task, options = {}) {
|
|
85
|
+
const policy = options.policy || {}
|
|
86
|
+
const capabilities = options.capabilities || inferTaskCapabilities(taskText(task), options.facts || {})
|
|
87
|
+
const evidenceBudget = options.evidenceBudget || planEvidenceBudget(policy, task, capabilities.required)
|
|
88
|
+
const manifest = await buildContextManifest(root, task, {
|
|
89
|
+
budget: evidenceBudget.total,
|
|
90
|
+
evidenceBudget,
|
|
91
|
+
strategy: options.strategy || policy?.profile?.contextStrategy || "incremental-semantic+git",
|
|
92
|
+
semanticMaxFiles: options.semanticMaxFiles,
|
|
93
|
+
maxFiles: options.maxFiles,
|
|
94
|
+
})
|
|
95
|
+
|
|
96
|
+
const externalized = await externalizeContextExcerpts(root, manifest, {
|
|
97
|
+
task,
|
|
98
|
+
threshold: options.externalizeThreshold,
|
|
99
|
+
inlineChars: options.inlineChars,
|
|
100
|
+
})
|
|
101
|
+
|
|
102
|
+
const promptEnvelope = buildPromptEnvelope({
|
|
103
|
+
invariants: options.invariants || "evidence-first; scoped edits; fresh verification; no unsupported completion claims",
|
|
104
|
+
role: options.role || "executor",
|
|
105
|
+
skills: options.skills || policy.domains || [],
|
|
106
|
+
projectFacts: {
|
|
107
|
+
strategy: externalized.manifest?.strategy || null,
|
|
108
|
+
instructions: externalized.manifest?.instructions || [],
|
|
109
|
+
...(options.projectFacts || {}),
|
|
110
|
+
},
|
|
111
|
+
task,
|
|
112
|
+
evidence: [
|
|
113
|
+
...(externalized.externalized || []).map((item) => item.ref),
|
|
114
|
+
...(externalized.manifest?.rankedReferences || []).slice(0, 12).map((item) => item.path),
|
|
115
|
+
...(options.evidence || []),
|
|
116
|
+
],
|
|
117
|
+
recentFailure: options.recentFailure || null,
|
|
118
|
+
nextAction: options.nextAction || null,
|
|
119
|
+
recentMessages: options.recentMessages || [],
|
|
120
|
+
})
|
|
121
|
+
|
|
122
|
+
const contextQuality = measureContextQuality(task, externalized.manifest, { minRequiredRecall: options.minRequiredRecall })
|
|
123
|
+
|
|
124
|
+
const cache = options.previousPromptEnvelope
|
|
125
|
+
? comparePromptEnvelopes(options.previousPromptEnvelope, promptEnvelope)
|
|
126
|
+
: null
|
|
127
|
+
|
|
128
|
+
return {
|
|
129
|
+
schemaVersion: 1,
|
|
130
|
+
contextSchemaVersion: 6,
|
|
131
|
+
capabilities,
|
|
132
|
+
evidenceBudget,
|
|
133
|
+
contextManifest: externalized.manifest,
|
|
134
|
+
contextQuality,
|
|
135
|
+
evidenceStore: {
|
|
136
|
+
refs: externalized.externalized.length,
|
|
137
|
+
externalizedBytes: externalized.externalizedBytes,
|
|
138
|
+
entries: externalized.externalized,
|
|
139
|
+
},
|
|
140
|
+
promptEnvelope,
|
|
141
|
+
promptCache: {
|
|
142
|
+
stablePrefixHash: promptEnvelope.stablePrefixHash,
|
|
143
|
+
dynamicHash: promptEnvelope.dynamicHash,
|
|
144
|
+
stableChars: promptEnvelope.stableChars,
|
|
145
|
+
dynamicChars: promptEnvelope.dynamicChars,
|
|
146
|
+
cacheableRatio: promptEnvelope.cacheableRatio,
|
|
147
|
+
...(cache || {}),
|
|
148
|
+
},
|
|
149
|
+
}
|
|
150
|
+
}
|
package/lib/context-manifest.mjs
CHANGED
|
@@ -243,7 +243,8 @@ async function rankedReferences(root, nodes, terms, declared, changed, limit = 2
|
|
|
243
243
|
|
|
244
244
|
export async function buildContextManifest(root, task, options = {}) {
|
|
245
245
|
root = path.resolve(root)
|
|
246
|
-
const budget = Math.max(4_000, Number(options.budget ?? 24_000))
|
|
246
|
+
const budget = Math.max(4_000, Number(options.evidenceBudget?.total ?? options.budget ?? 24_000))
|
|
247
|
+
const evidenceBudget = options.evidenceBudget || null
|
|
247
248
|
const declared = taskFiles(task)
|
|
248
249
|
const terms = taskTerms(task)
|
|
249
250
|
const changed = gitChangedFiles(root)
|
|
@@ -331,20 +332,30 @@ export async function buildContextManifest(root, task, options = {}) {
|
|
|
331
332
|
}
|
|
332
333
|
|
|
333
334
|
let remaining = budget
|
|
335
|
+
const categoryRemaining = {
|
|
336
|
+
declared: evidenceBudget?.buckets?.declared ?? Math.round(budget * 0.42),
|
|
337
|
+
test: evidenceBudget?.buckets?.tests ?? Math.round(budget * 0.20),
|
|
338
|
+
instruction: evidenceBudget?.buckets?.instructions ?? Math.round(budget * 0.12),
|
|
339
|
+
reference: evidenceBudget?.buckets?.references ?? Math.round(budget * 0.26),
|
|
340
|
+
}
|
|
334
341
|
const excerpts = []
|
|
335
342
|
for (const file of priority) {
|
|
336
343
|
if (remaining <= 0) break
|
|
337
344
|
const isDeclared = declared.includes(file)
|
|
338
345
|
const isTest = tests.includes(file)
|
|
339
346
|
const isInstruction = instructions.includes(file)
|
|
347
|
+
const role = isDeclared ? "declared" : isTest ? "test" : isInstruction ? "instruction" : "reference"
|
|
340
348
|
const desired = isDeclared ? 6_000 : isTest ? 4_000 : isInstruction ? 3_000 : 2_500
|
|
341
|
-
const
|
|
349
|
+
const category = Math.max(0, Number(categoryRemaining[role] || 0))
|
|
350
|
+
if (category <= 0) continue
|
|
351
|
+
const perFile = Math.min(desired, remaining, category)
|
|
342
352
|
const item = await excerpt(root, file, perFile, terms)
|
|
343
353
|
if (!item) continue
|
|
344
354
|
remaining -= item.text.length
|
|
355
|
+
categoryRemaining[role] = Math.max(0, categoryRemaining[role] - item.text.length)
|
|
345
356
|
excerpts.push({
|
|
346
357
|
...item,
|
|
347
|
-
role
|
|
358
|
+
role,
|
|
348
359
|
})
|
|
349
360
|
}
|
|
350
361
|
|
|
@@ -373,6 +384,8 @@ export async function buildContextManifest(root, task, options = {}) {
|
|
|
373
384
|
hotspots: graph.hotspots.slice(0, 12),
|
|
374
385
|
},
|
|
375
386
|
budget,
|
|
387
|
+
evidenceBudget,
|
|
388
|
+
categoryRemaining,
|
|
376
389
|
used: budget - remaining,
|
|
377
390
|
}
|
|
378
391
|
}
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
function normalizePath(value) {
|
|
2
|
+
return String(value || "").replaceAll("\\", "/").replace(/^\.\//, "")
|
|
3
|
+
}
|
|
4
|
+
|
|
5
|
+
function declaredTaskFiles(task = {}) {
|
|
6
|
+
const files = task.files
|
|
7
|
+
const values = []
|
|
8
|
+
if (Array.isArray(files)) values.push(...files)
|
|
9
|
+
else if (files && typeof files === "object") {
|
|
10
|
+
for (const [kind, list] of Object.entries(files)) {
|
|
11
|
+
if (kind === "create") continue
|
|
12
|
+
if (Array.isArray(list)) values.push(...list)
|
|
13
|
+
else if (typeof list === "string") values.push(list)
|
|
14
|
+
}
|
|
15
|
+
}
|
|
16
|
+
if (Array.isArray(task.requiredFiles)) values.push(...task.requiredFiles)
|
|
17
|
+
return [...new Set(values.map(normalizePath).filter(Boolean))]
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
function manifestPaths(manifest = {}) {
|
|
21
|
+
const values = []
|
|
22
|
+
for (const item of manifest.excerpts || []) if (item?.path) values.push(item.path)
|
|
23
|
+
for (const item of manifest.rankedReferences || []) if (item?.path) values.push(item.path)
|
|
24
|
+
for (const item of manifest.instructions || []) {
|
|
25
|
+
if (typeof item === "string") values.push(item)
|
|
26
|
+
else if (item?.path) values.push(item.path)
|
|
27
|
+
}
|
|
28
|
+
return [...new Set(values.map(normalizePath).filter(Boolean))]
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
export function measureContextQuality(task = {}, manifest = {}, options = {}) {
|
|
32
|
+
const required = declaredTaskFiles(task)
|
|
33
|
+
const included = manifestPaths(manifest)
|
|
34
|
+
const includedSet = new Set(included)
|
|
35
|
+
const hits = required.filter((file) => includedSet.has(file))
|
|
36
|
+
const requiredFileRecall = required.length ? hits.length / required.length : null
|
|
37
|
+
const relevant = new Set(required)
|
|
38
|
+
for (const item of manifest.instructions || []) {
|
|
39
|
+
if (typeof item === "string") relevant.add(normalizePath(item))
|
|
40
|
+
else if (item?.path) relevant.add(normalizePath(item.path))
|
|
41
|
+
}
|
|
42
|
+
for (const item of manifest.excerpts || []) {
|
|
43
|
+
if (["declared", "test", "instruction"].includes(item?.role) && item?.path) relevant.add(normalizePath(item.path))
|
|
44
|
+
}
|
|
45
|
+
const irrelevant = included.filter((file) => !relevant.has(file))
|
|
46
|
+
const irrelevantRatio = included.length ? irrelevant.length / included.length : 0
|
|
47
|
+
const minRequiredRecall = Number.isFinite(Number(options.minRequiredRecall))
|
|
48
|
+
? Math.max(0, Math.min(1, Number(options.minRequiredRecall))) : 1
|
|
49
|
+
return {
|
|
50
|
+
schemaVersion: 1,
|
|
51
|
+
requiredFiles: required,
|
|
52
|
+
includedFiles: included,
|
|
53
|
+
requiredFileHits: hits,
|
|
54
|
+
requiredFileRecall,
|
|
55
|
+
irrelevantFiles: irrelevant,
|
|
56
|
+
irrelevantRatio,
|
|
57
|
+
checks: { requiredRecallAcceptable: requiredFileRecall == null ? true : requiredFileRecall >= minRequiredRecall },
|
|
58
|
+
}
|
|
59
|
+
}
|
package/lib/control-center.mjs
CHANGED
|
@@ -3,6 +3,7 @@ import { mkdir, readFile, readdir, writeFile } from "node:fs/promises"
|
|
|
3
3
|
import path from "node:path"
|
|
4
4
|
import { readLearningState } from "./learning-engine.mjs"
|
|
5
5
|
import { readRuntimeEvents } from "./runtime-events.mjs"
|
|
6
|
+
import { evidenceStoreStatus } from "./evidence-store.mjs"
|
|
6
7
|
|
|
7
8
|
function escapeHtml(value) {
|
|
8
9
|
return String(value ?? "")
|
|
@@ -64,18 +65,23 @@ async function collectEvalSummary(root) {
|
|
|
64
65
|
|
|
65
66
|
export async function collectControlCenterData(root = process.cwd()) {
|
|
66
67
|
root = path.resolve(root)
|
|
67
|
-
const [work, learning, evals] = await Promise.all([
|
|
68
|
+
const [work, learning, evals, evidenceStore] = await Promise.all([
|
|
68
69
|
collectWork(root),
|
|
69
70
|
readLearningState(root),
|
|
70
71
|
collectEvalSummary(root),
|
|
72
|
+
evidenceStoreStatus(root),
|
|
71
73
|
])
|
|
72
74
|
return {
|
|
73
|
-
schemaVersion:
|
|
75
|
+
schemaVersion: 2,
|
|
74
76
|
generatedAt: new Date().toISOString(),
|
|
75
77
|
root,
|
|
76
78
|
work,
|
|
77
79
|
learning,
|
|
78
80
|
evals,
|
|
81
|
+
v11: {
|
|
82
|
+
evidenceStore,
|
|
83
|
+
runtime: "perception-adaptive-execution",
|
|
84
|
+
},
|
|
79
85
|
}
|
|
80
86
|
}
|
|
81
87
|
|
|
@@ -103,6 +109,7 @@ details{margin-top:10px;border-top:1px solid #243049;padding-top:8px}summary{cur
|
|
|
103
109
|
<div class="top"><div><h1>UES Control Center</h1><div class="muted" id="root"></div></div><div class="muted" id="generated"></div></div>
|
|
104
110
|
<div class="section"><h2>Long-horizon work</h2><div class="grid" id="work"></div></div>
|
|
105
111
|
<div class="section"><h2>Learning loop</h2><div class="grid" id="learning"></div></div>
|
|
112
|
+
<div class="section"><h2>V11 runtime efficiency</h2><div class="grid" id="v11"></div></div>
|
|
106
113
|
<div class="section"><h2>Recent runtime events</h2><div class="card"><table><thead><tr><th>Work</th><th>Event</th><th>Task</th><th>Time</th></tr></thead><tbody id="events"></tbody></table></div></div>
|
|
107
114
|
<div class="section"><h2>Recent evaluations</h2><div class="card"><table><thead><tr><th>Suite</th><th>Model</th><th>Baseline</th><th>UES</th></tr></thead><tbody id="evals"></tbody></table></div></div>
|
|
108
115
|
</div>
|
|
@@ -126,6 +133,9 @@ function render(next){
|
|
|
126
133
|
if(!data.work.length) work.innerHTML='<div class="card muted">No .ues-work items found.</div>';
|
|
127
134
|
const learning=document.querySelector("#learning");
|
|
128
135
|
learning.innerHTML='<div class="card"><h3>Accepted lessons</h3><strong>'+data.learning.accepted.length+'</strong></div><div class="card"><h3>Proposals</h3><strong>'+data.learning.proposals.length+'</strong></div>';
|
|
136
|
+
const v11=document.querySelector("#v11");
|
|
137
|
+
const store=data.v11?.evidenceStore||{};
|
|
138
|
+
v11.innerHTML='<div class="card"><h3>Evidence store</h3><strong>'+esc(store.entries||0)+'</strong><div class="muted">'+esc(store.bytes||0)+' bytes externalized</div></div><div class="card"><h3>Runtime</h3><strong>V11</strong><div class="muted">'+esc(data.v11?.runtime||'adaptive')+'</div></div>';
|
|
129
139
|
const eventBody=document.querySelector("#events");
|
|
130
140
|
eventBody.innerHTML="";
|
|
131
141
|
for(const item of data.work){
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
const HIGH_RISK = /(npm publish|publish package|deploy|production|git push|force push|reset --hard|git clean|delete branch|drop table|truncate|rotate secret|credential|api key|purchase|payment|irreversible|public api break)/i
|
|
2
|
+
const MEDIUM_RISK = /(migration|schema|dependency upgrade|lockfile|generated code|shared config|auth|permission|security)/i
|
|
3
|
+
const REVERSIBLE = /(local|test|rename|format|refactor|temporary|fixture|internal|reversible)/i
|
|
4
|
+
|
|
5
|
+
export function classifyDecisionPolicy(text = "", facts = {}) {
|
|
6
|
+
const value = String(text || "").trim()
|
|
7
|
+
const explicitlyIrreversible = facts.irreversible === true
|
|
8
|
+
const externalSideEffect = facts.externalSideEffect === true
|
|
9
|
+
const destructive = facts.destructive === true || HIGH_RISK.test(value)
|
|
10
|
+
const medium = MEDIUM_RISK.test(value)
|
|
11
|
+
const reversible = facts.reversible === true || (!destructive && REVERSIBLE.test(value))
|
|
12
|
+
const risk = destructive || explicitlyIrreversible || externalSideEffect ? "high" : medium ? "medium" : "low"
|
|
13
|
+
const requiresUser = risk === "high" || facts.productDecision === true
|
|
14
|
+
return {
|
|
15
|
+
schemaVersion: 1, risk, reversible, destructive, externalSideEffect, requiresUser,
|
|
16
|
+
autoResolvable: !requiresUser && (reversible || risk === "low"),
|
|
17
|
+
reason: requiresUser ? "human-approval-required" : reversible ? "reversible-local-decision" : "bounded-engineering-decision",
|
|
18
|
+
}
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
export function canAutoResolveDecision(text = "", facts = {}) {
|
|
22
|
+
return classifyDecisionPolicy(text, facts).autoResolvable
|
|
23
|
+
}
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
function taskFiles(task = {}) {
|
|
2
|
+
if (Array.isArray(task.files)) return task.files
|
|
3
|
+
const files = task.files && typeof task.files === "object" ? task.files : {}
|
|
4
|
+
return [...new Set(["create","modify","test","delete"].flatMap((key) => Array.isArray(files[key]) ? files[key] : []))]
|
|
5
|
+
}
|
|
6
|
+
|
|
7
|
+
function writes(task = {}) {
|
|
8
|
+
const files = task.files && typeof task.files === "object" && !Array.isArray(task.files) ? task.files : null
|
|
9
|
+
if (!files) return Array.isArray(task.files) && task.files.length > 0
|
|
10
|
+
return ["create","modify","delete"].some((key) => Array.isArray(files[key]) && files[key].length > 0)
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
function clampInt(value, fallback, min, max) {
|
|
14
|
+
const parsed = Number(value)
|
|
15
|
+
if (!Number.isFinite(parsed)) return fallback
|
|
16
|
+
return Math.max(min, Math.min(max, Math.round(parsed)))
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
export function classifyWorkflowTask(task = {}) {
|
|
20
|
+
const text = [task.title, task.summary, ...(task.acceptance || [])].filter(Boolean).join(" ").toLowerCase()
|
|
21
|
+
const visual = /(visual|screenshot|pixel|figma|layout|giao diện|ảnh mẫu|image reference)/.test(text)
|
|
22
|
+
const deterministic = task.deterministic === true || /(run test|typecheck|lint|format|generate manifest|build index|verify command|compile|unit test)/.test(text)
|
|
23
|
+
const kind = deterministic ? "deterministic" : visual ? "vision" : "llm"
|
|
24
|
+
const fileCount = taskFiles(task).length
|
|
25
|
+
const acceptanceCount = Array.isArray(task.acceptance) ? task.acceptance.length : 0
|
|
26
|
+
const verificationCount = Array.isArray(task.verification) ? task.verification.length : 0
|
|
27
|
+
const estimatedCost = kind === "deterministic"
|
|
28
|
+
? 1
|
|
29
|
+
: Math.max(
|
|
30
|
+
2,
|
|
31
|
+
Math.min(
|
|
32
|
+
12,
|
|
33
|
+
2 + fileCount + Math.min(3, acceptanceCount) + Math.min(2, verificationCount) + (task.risk === "high" ? 3 : 0),
|
|
34
|
+
),
|
|
35
|
+
)
|
|
36
|
+
return {
|
|
37
|
+
kind,
|
|
38
|
+
estimatedCost,
|
|
39
|
+
writes: writes(task),
|
|
40
|
+
files: taskFiles(task),
|
|
41
|
+
risk: task.risk || "medium",
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
function conflict(a, b) {
|
|
46
|
+
if (!a.writes && !b.writes) return false
|
|
47
|
+
const aa = new Set(a.files)
|
|
48
|
+
return b.files.some((file) => aa.has(file))
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
function executionMode(classification, options) {
|
|
52
|
+
if (classification.kind === "deterministic") return "deterministic"
|
|
53
|
+
if (classification.kind === "vision") {
|
|
54
|
+
return classification.estimatedCost >= options.minVisionAgentCost ? "agent" : "inline"
|
|
55
|
+
}
|
|
56
|
+
return classification.estimatedCost >= options.minAgentCost ? "agent" : "inline"
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
function kindCapacity(selected, classification, options) {
|
|
60
|
+
if (classification.kind === "vision") {
|
|
61
|
+
return selected.filter((item) => item.classification.kind === "vision" && item.execution === "agent").length < options.maxVisionConcurrent
|
|
62
|
+
}
|
|
63
|
+
if (classification.kind === "llm") {
|
|
64
|
+
return selected.filter((item) => item.classification.kind === "llm" && item.execution === "agent").length < options.maxLLMConcurrent
|
|
65
|
+
}
|
|
66
|
+
return true
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
function readyTasks(remaining, byID, complete) {
|
|
70
|
+
return [...remaining]
|
|
71
|
+
.map((id) => byID.get(id))
|
|
72
|
+
.filter((task) => (task.dependsOn || task.dependencies || []).every((dep) => complete.has(dep)))
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
export function planDynamicWorkflow(tasks = [], inputOptions = {}) {
|
|
76
|
+
const options = {
|
|
77
|
+
maxConcurrent: clampInt(inputOptions.maxConcurrent, 4, 1, 16),
|
|
78
|
+
maxLLMConcurrent: clampInt(inputOptions.maxLLMConcurrent, inputOptions.maxConcurrent || 4, 1, 16),
|
|
79
|
+
maxVisionConcurrent: clampInt(inputOptions.maxVisionConcurrent, 2, 1, 8),
|
|
80
|
+
maxWaveCost: clampInt(inputOptions.maxWaveCost, 24, 1, 128),
|
|
81
|
+
minAgentCost: clampInt(inputOptions.minAgentCost, 5, 2, 12),
|
|
82
|
+
minVisionAgentCost: clampInt(inputOptions.minVisionAgentCost, 4, 2, 12),
|
|
83
|
+
deterministicFirst: inputOptions.deterministicFirst !== false,
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
const byID = new Map(tasks.map((task) => [task.id, task]))
|
|
87
|
+
if (byID.size !== tasks.length || tasks.some((task) => !task?.id)) {
|
|
88
|
+
throw new Error("Workflow tasks require unique non-empty ids")
|
|
89
|
+
}
|
|
90
|
+
for (const task of tasks) {
|
|
91
|
+
for (const dep of task.dependsOn || task.dependencies || []) {
|
|
92
|
+
if (!byID.has(dep)) throw new Error("Workflow contains a missing dependency: " + dep)
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
const remaining = new Set(tasks.map((task) => task.id))
|
|
97
|
+
const complete = new Set()
|
|
98
|
+
const waves = []
|
|
99
|
+
let waveIndex = 0
|
|
100
|
+
|
|
101
|
+
while (remaining.size) {
|
|
102
|
+
const ready = readyTasks(remaining, byID, complete)
|
|
103
|
+
if (!ready.length) throw new Error("Workflow contains a dependency cycle or missing dependency")
|
|
104
|
+
|
|
105
|
+
const deterministicReady = ready.filter((task) => classifyWorkflowTask(task).kind === "deterministic")
|
|
106
|
+
const candidates = options.deterministicFirst && deterministicReady.length
|
|
107
|
+
? deterministicReady
|
|
108
|
+
: ready
|
|
109
|
+
|
|
110
|
+
const classified = candidates
|
|
111
|
+
.map((task) => {
|
|
112
|
+
const classification = classifyWorkflowTask(task)
|
|
113
|
+
return { task, classification, execution: executionMode(classification, options) }
|
|
114
|
+
})
|
|
115
|
+
.sort((a, b) => {
|
|
116
|
+
const order = { deterministic: 0, inline: 1, agent: 2 }
|
|
117
|
+
return order[a.execution] - order[b.execution] ||
|
|
118
|
+
a.classification.estimatedCost - b.classification.estimatedCost ||
|
|
119
|
+
a.task.id.localeCompare(b.task.id)
|
|
120
|
+
})
|
|
121
|
+
|
|
122
|
+
const selected = []
|
|
123
|
+
let waveCost = 0
|
|
124
|
+
for (const item of classified) {
|
|
125
|
+
if (selected.length >= options.maxConcurrent) break
|
|
126
|
+
if (selected.some((existing) => conflict(item.classification, existing.classification))) continue
|
|
127
|
+
if (!kindCapacity(selected, item.classification, options)) continue
|
|
128
|
+
if (selected.length && waveCost + item.classification.estimatedCost > options.maxWaveCost) continue
|
|
129
|
+
selected.push(item)
|
|
130
|
+
waveCost += item.classification.estimatedCost
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
if (!selected.length) selected.push(classified[0])
|
|
134
|
+
|
|
135
|
+
const tasksInWave = selected.map(({ task, classification, execution }) => ({
|
|
136
|
+
id: task.id,
|
|
137
|
+
kind: classification.kind,
|
|
138
|
+
execution,
|
|
139
|
+
estimatedCost: classification.estimatedCost,
|
|
140
|
+
writes: classification.writes,
|
|
141
|
+
files: classification.files,
|
|
142
|
+
spawnAgent: execution === "agent",
|
|
143
|
+
rationale:
|
|
144
|
+
execution === "deterministic"
|
|
145
|
+
? "deterministic tool/script work should not consume an agent slot"
|
|
146
|
+
: execution === "inline"
|
|
147
|
+
? "coordination cost exceeds expected benefit for this bounded unit"
|
|
148
|
+
: classification.kind === "vision"
|
|
149
|
+
? "visual judgment requires a bounded vision worker"
|
|
150
|
+
: "independent task size justifies isolated agent execution",
|
|
151
|
+
}))
|
|
152
|
+
|
|
153
|
+
waves.push({
|
|
154
|
+
index: waveIndex++,
|
|
155
|
+
tasks: tasksInWave,
|
|
156
|
+
totalEstimatedCost: tasksInWave.reduce((sum, item) => sum + item.estimatedCost, 0),
|
|
157
|
+
agentSlots: tasksInWave.filter((item) => item.spawnAgent).length,
|
|
158
|
+
visionAgentSlots: tasksInWave.filter((item) => item.spawnAgent && item.kind === "vision").length,
|
|
159
|
+
})
|
|
160
|
+
|
|
161
|
+
for (const { task } of selected) {
|
|
162
|
+
remaining.delete(task.id)
|
|
163
|
+
complete.add(task.id)
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
const flat = waves.flatMap((wave) => wave.tasks)
|
|
168
|
+
return {
|
|
169
|
+
schemaVersion: 2,
|
|
170
|
+
options,
|
|
171
|
+
waves,
|
|
172
|
+
taskCount: tasks.length,
|
|
173
|
+
agentTaskCount: flat.filter((task) => task.spawnAgent).length,
|
|
174
|
+
inlineTaskCount: flat.filter((task) => task.execution === "inline").length,
|
|
175
|
+
deterministicTaskCount: flat.filter((task) => task.execution === "deterministic").length,
|
|
176
|
+
visionAgentTaskCount: flat.filter((task) => task.spawnAgent && task.kind === "vision").length,
|
|
177
|
+
estimatedCoordinationSaved: flat.filter((task) => task.execution !== "agent").reduce((sum, task) => sum + task.estimatedCost, 0),
|
|
178
|
+
}
|
|
179
|
+
}
|
package/lib/eval-ablation.mjs
CHANGED
|
@@ -27,6 +27,9 @@ export function compareEvalSummaries(reference, candidate, options = {}) {
|
|
|
27
27
|
const minInitialInputReduction = Math.max(0, Number(options.minInitialInputReduction ?? 0.10))
|
|
28
28
|
const maxTotalTokenRatio = Math.max(0.01, Number(options.maxTotalTokenRatio || 1.05))
|
|
29
29
|
const maxDurationRatio = Math.max(0.01, Number(options.maxDurationRatio || 1.10))
|
|
30
|
+
const minCacheableRatio = options.minCacheableRatio == null ? null : Math.max(0, Math.min(1, Number(options.minCacheableRatio)))
|
|
31
|
+
const minEvidenceReuseRatio = options.minEvidenceReuseRatio == null ? null : Math.max(0, Math.min(1, Number(options.minEvidenceReuseRatio)))
|
|
32
|
+
const maxRepeatedStableRatio = options.maxRepeatedStableRatio == null ? null : Math.max(0, Number(options.maxRepeatedStableRatio))
|
|
30
33
|
|
|
31
34
|
const referencePassRate = finite(ref.passRate)
|
|
32
35
|
const candidatePassRate = finite(next.passRate)
|
|
@@ -36,6 +39,13 @@ export function compareEvalSummaries(reference, candidate, options = {}) {
|
|
|
36
39
|
const candidateTokens = finite(next.avgTokens)
|
|
37
40
|
const referenceDuration = finite(ref.avgDurationMs)
|
|
38
41
|
const candidateDuration = finite(next.avgDurationMs)
|
|
42
|
+
const referenceCacheableRatio = finite(ref.avgCacheableRatio)
|
|
43
|
+
const candidateCacheableRatio = finite(next.avgCacheableRatio)
|
|
44
|
+
const referenceEvidenceReuseRatio = finite(ref.avgEvidenceReuseRatio)
|
|
45
|
+
const candidateEvidenceReuseRatio = finite(next.avgEvidenceReuseRatio)
|
|
46
|
+
const candidateRepeatedStableChars = finite(next.avgRepeatedStableChars)
|
|
47
|
+
const referenceRepeatedStableRatio = finite(ref.avgRepeatedStableRatio)
|
|
48
|
+
const candidateRepeatedStableRatio = finite(next.avgRepeatedStableRatio)
|
|
39
49
|
|
|
40
50
|
const initialInputRatio = ratio(candidateInitialInput, referenceInitialInput)
|
|
41
51
|
const tokenRatio = ratio(candidateTokens, referenceTokens)
|
|
@@ -54,6 +64,18 @@ export function compareEvalSummaries(reference, candidate, options = {}) {
|
|
|
54
64
|
tokenRatio == null ? null : tokenRatio <= maxTotalTokenRatio,
|
|
55
65
|
durationBounded:
|
|
56
66
|
durationRatio == null ? null : durationRatio <= maxDurationRatio,
|
|
67
|
+
cacheableRatioTarget:
|
|
68
|
+
minCacheableRatio == null ? null :
|
|
69
|
+
candidateCacheableRatio == null ? false :
|
|
70
|
+
candidateCacheableRatio >= minCacheableRatio,
|
|
71
|
+
evidenceReuseTarget:
|
|
72
|
+
minEvidenceReuseRatio == null ? null :
|
|
73
|
+
candidateEvidenceReuseRatio == null ? false :
|
|
74
|
+
candidateEvidenceReuseRatio >= minEvidenceReuseRatio,
|
|
75
|
+
repeatedStableTarget:
|
|
76
|
+
maxRepeatedStableRatio == null ? null :
|
|
77
|
+
candidateRepeatedStableRatio == null ? false :
|
|
78
|
+
candidateRepeatedStableRatio <= maxRepeatedStableRatio,
|
|
57
79
|
}
|
|
58
80
|
|
|
59
81
|
const efficiencyEvidence = [
|
|
@@ -70,24 +92,37 @@ export function compareEvalSummaries(reference, candidate, options = {}) {
|
|
|
70
92
|
minInitialInputReduction,
|
|
71
93
|
maxTotalTokenRatio,
|
|
72
94
|
maxDurationRatio,
|
|
95
|
+
minCacheableRatio,
|
|
96
|
+
minEvidenceReuseRatio,
|
|
97
|
+
maxRepeatedStableRatio,
|
|
73
98
|
},
|
|
74
99
|
reference: {
|
|
75
100
|
passRate: referencePassRate,
|
|
76
101
|
avgInitialInputTokens: referenceInitialInput,
|
|
77
102
|
avgTokens: referenceTokens,
|
|
78
103
|
avgDurationMs: referenceDuration,
|
|
104
|
+
avgCacheableRatio: referenceCacheableRatio,
|
|
105
|
+
avgEvidenceReuseRatio: referenceEvidenceReuseRatio,
|
|
106
|
+
avgRepeatedStableRatio: referenceRepeatedStableRatio,
|
|
79
107
|
},
|
|
80
108
|
candidate: {
|
|
81
109
|
passRate: candidatePassRate,
|
|
82
110
|
avgInitialInputTokens: candidateInitialInput,
|
|
83
111
|
avgTokens: candidateTokens,
|
|
84
112
|
avgDurationMs: candidateDuration,
|
|
113
|
+
avgCacheableRatio: candidateCacheableRatio,
|
|
114
|
+
avgEvidenceReuseRatio: candidateEvidenceReuseRatio,
|
|
115
|
+
avgRepeatedStableChars: candidateRepeatedStableChars,
|
|
116
|
+
avgRepeatedStableRatio: candidateRepeatedStableRatio,
|
|
85
117
|
},
|
|
86
118
|
delta: {
|
|
87
119
|
passRate: delta(candidatePassRate, referencePassRate),
|
|
88
120
|
avgInitialInputTokens: delta(candidateInitialInput, referenceInitialInput),
|
|
89
121
|
avgTokens: delta(candidateTokens, referenceTokens),
|
|
90
122
|
avgDurationMs: delta(candidateDuration, referenceDuration),
|
|
123
|
+
avgCacheableRatio: delta(candidateCacheableRatio, referenceCacheableRatio),
|
|
124
|
+
avgEvidenceReuseRatio: delta(candidateEvidenceReuseRatio, referenceEvidenceReuseRatio),
|
|
125
|
+
avgRepeatedStableRatio: delta(candidateRepeatedStableRatio, referenceRepeatedStableRatio),
|
|
91
126
|
},
|
|
92
127
|
ratios: {
|
|
93
128
|
initialInput: initialInputRatio,
|
|
@@ -96,9 +131,16 @@ export function compareEvalSummaries(reference, candidate, options = {}) {
|
|
|
96
131
|
},
|
|
97
132
|
checks,
|
|
98
133
|
telemetrySufficient: checks.initialInputReduced !== null,
|
|
134
|
+
optionalTargetsSatisfied:
|
|
135
|
+
(checks.cacheableRatioTarget == null || checks.cacheableRatioTarget === true) &&
|
|
136
|
+
(checks.evidenceReuseTarget == null || checks.evidenceReuseTarget === true) &&
|
|
137
|
+
(checks.repeatedStableTarget == null || checks.repeatedStableTarget === true),
|
|
99
138
|
gateEligible:
|
|
100
139
|
checks.passRatePreserved === true &&
|
|
101
140
|
checks.initialInputReduced === true &&
|
|
102
|
-
efficiencyEvidence.every((value) => value === true)
|
|
141
|
+
efficiencyEvidence.every((value) => value === true) &&
|
|
142
|
+
(checks.cacheableRatioTarget == null || checks.cacheableRatioTarget === true) &&
|
|
143
|
+
(checks.evidenceReuseTarget == null || checks.evidenceReuseTarget === true) &&
|
|
144
|
+
(checks.repeatedStableTarget == null || checks.repeatedStableTarget === true),
|
|
103
145
|
}
|
|
104
146
|
}
|