opencode-agent-skill 11.0.0 → 13.0.0-beta.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +55 -0
- package/README.md +119 -14
- package/bin/ocskill.mjs +416 -83
- package/docs/DETERMINISTIC-TOOLS.md +1 -1
- package/docs/ENGINEERING-DESIGN.md +4 -4
- package/docs/EVALS.md +3 -3
- package/docs/GITHUB-RULESET.md +50 -0
- package/docs/NPM-PUBLISH.md +22 -8
- package/docs/OPENCODE-COMPAT.md +14 -4
- package/docs/TRACE-SCHEMA.md +1 -1
- package/docs/V11-PERCEPTION-ADAPTIVE.md +2 -2
- package/docs/V12-WEAK-MODEL-INTELLIGENCE.md +27 -0
- package/docs/V13-PARALLEL-WEAK-MODEL-RUNTIME.md +75 -0
- package/evals/repo-scale/tasks.json +62 -0
- package/global-config/AGENTS.md +4 -0
- package/global-config/commands/resume.md +4 -1
- package/global-config/commands/run.md +9 -7
- package/global-config/plugins/ues-router/capabilities.js +1 -0
- package/global-config/plugins/ues-router/command-runtime.js +79 -0
- package/global-config/plugins/ues-router/index.js +364 -43
- package/global-config/plugins/ues-router/parallel-runtime.js +271 -0
- package/global-config/plugins/ues-router/text-runtime.js +115 -0
- package/global-config/plugins/ues-router/verifier-runtime.js +33 -0
- package/global-config/skills/dynamic-workflow/SKILL.md +7 -6
- package/global-config/skills/dynamic-workflow/references/workflow.md +8 -6
- package/global-config/skills/engineering-orchestrator/references/long-horizon.md +7 -5
- package/lib/cli-utils.mjs +17 -4
- package/lib/context-engine-v11.mjs +4 -0
- package/lib/context-quality.mjs +59 -0
- package/lib/decision-policy.mjs +23 -0
- package/lib/installer.mjs +49 -16
- package/lib/model-config.mjs +13 -1
- package/lib/model-performance.mjs +113 -0
- package/lib/model-policy.mjs +9 -2
- package/lib/repo-scale-fixture.mjs +45 -0
- package/lib/task-engine.mjs +36 -12
- package/lib/task-graph.mjs +32 -0
- package/lib/text-encoding.mjs +74 -0
- package/lib/work-plan-scope.mjs +49 -0
- package/lib/worktree-sandbox.mjs +141 -12
- package/package.json +8 -4
- package/scripts/check-release-consistency.mjs +260 -0
- package/scripts/smoke-packed-install.mjs +39 -3
- package/scripts/smoke-plain-install.mjs +27 -7
- package/scripts/validate-repo-scale-suite.mjs +27 -0
- package/scripts/validate-v12-foundation.mjs +24 -0
- package/scripts/validate.mjs +18 -1
package/lib/installer.mjs
CHANGED
|
@@ -23,6 +23,15 @@ function validManagedPlugin(value) {
|
|
|
23
23
|
return value === ROUTER_PLUGIN_STATE
|
|
24
24
|
}
|
|
25
25
|
|
|
26
|
+
function validPromptAlias(value) {
|
|
27
|
+
if (typeof value !== "string" || !value.startsWith(SKILL_PREFIX)) return false
|
|
28
|
+
return validResourceID(value.slice(SKILL_PREFIX.length))
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
function promptAliasTemplateName(value) {
|
|
32
|
+
return value.slice(SKILL_PREFIX.length) + ".md"
|
|
33
|
+
}
|
|
34
|
+
|
|
26
35
|
function isOwnedPackageName(value) {
|
|
27
36
|
return !value || value === PACKAGE_NAME || value === LEGACY_PACKAGE_NAME
|
|
28
37
|
}
|
|
@@ -94,7 +103,7 @@ async function writeManagedFile(file, content, warnings) {
|
|
|
94
103
|
|
|
95
104
|
async function readManagedState(file, warnings = []) {
|
|
96
105
|
const raw = await readText(file)
|
|
97
|
-
if (!raw) return { kind: "empty", state: { skills: [], commands: [], agents: [], plugins: [] } }
|
|
106
|
+
if (!raw) return { kind: "empty", state: { skills: [], commands: [], promptAliases: [], agents: [], plugins: [] } }
|
|
98
107
|
|
|
99
108
|
try {
|
|
100
109
|
const parsed = JSON.parse(raw)
|
|
@@ -103,7 +112,7 @@ async function readManagedState(file, warnings = []) {
|
|
|
103
112
|
return {
|
|
104
113
|
kind: "foreign",
|
|
105
114
|
package: parsed.package,
|
|
106
|
-
state: { skills: [], commands: [], agents: [], plugins: [] },
|
|
115
|
+
state: { skills: [], commands: [], promptAliases: [], agents: [], plugins: [] },
|
|
107
116
|
}
|
|
108
117
|
}
|
|
109
118
|
return {
|
|
@@ -112,13 +121,14 @@ async function readManagedState(file, warnings = []) {
|
|
|
112
121
|
...parsed,
|
|
113
122
|
skills: Array.isArray(parsed.skills) ? parsed.skills.filter(validSkillID) : [],
|
|
114
123
|
commands: Array.isArray(parsed.commands) ? parsed.commands.filter(validManagedMarkdown) : [],
|
|
124
|
+
promptAliases: Array.isArray(parsed.promptAliases) ? parsed.promptAliases.filter(validPromptAlias) : [],
|
|
115
125
|
agents: Array.isArray(parsed.agents) ? parsed.agents.filter(validManagedMarkdown) : [],
|
|
116
126
|
plugins: Array.isArray(parsed.plugins) ? parsed.plugins.filter(validManagedPlugin) : [],
|
|
117
127
|
},
|
|
118
128
|
}
|
|
119
129
|
} catch {
|
|
120
130
|
warnings.push(`Invalid UES state file; run 'ocskill install' to rebuild it: ${file}`)
|
|
121
|
-
return { kind: "invalid", state: { skills: [], commands: [], agents: [], plugins: [] } }
|
|
131
|
+
return { kind: "invalid", state: { skills: [], commands: [], promptAliases: [], agents: [], plugins: [] } }
|
|
122
132
|
}
|
|
123
133
|
}
|
|
124
134
|
|
|
@@ -310,7 +320,7 @@ export async function installResources(options = {}) {
|
|
|
310
320
|
const stateFile = path.join(stateDir, "state.json")
|
|
311
321
|
const warnings = []
|
|
312
322
|
const openCodeMajor = options.openCodeMajor ?? detectOpenCodeMajor()
|
|
313
|
-
const installed = { skills: [], commands: [], agents: [], plugins: [] }
|
|
323
|
+
const installed = { skills: [], commands: [], promptAliases: [], agents: [], plugins: [] }
|
|
314
324
|
const previous = await readManagedState(stateFile, warnings)
|
|
315
325
|
const previousState = previous.state
|
|
316
326
|
const modelPolicy = await readModelPolicy(configDir)
|
|
@@ -366,25 +376,31 @@ export async function installResources(options = {}) {
|
|
|
366
376
|
|
|
367
377
|
const sourceCommands = path.join(sourceRoot, "global-config", "commands")
|
|
368
378
|
const commandEntries = await readdir(sourceCommands, { withFileTypes: true })
|
|
369
|
-
|
|
379
|
+
const validCommandEntries = []
|
|
370
380
|
for (const entry of commandEntries) {
|
|
371
381
|
if (!entry.isFile() || !entry.name.endsWith(".md")) continue
|
|
372
|
-
|
|
373
382
|
const id = entry.name.slice(0, -3)
|
|
374
383
|
if (!validResourceID(id)) {
|
|
375
384
|
warnings.push(`Skipped command with invalid resource ID: ${id}`)
|
|
376
385
|
continue
|
|
377
386
|
}
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
const source = await readFile(path.join(sourceCommands, entry.name), "utf8")
|
|
381
|
-
const normalizedSource = normalizeManagedMarker(source)
|
|
382
|
-
const content = hasManagedMarker(normalizedSource)
|
|
383
|
-
? normalizedSource
|
|
384
|
-
: `${normalizedSource.trimEnd()}\n\n${MANAGED_MARKER}\n`
|
|
387
|
+
validCommandEntries.push(entry)
|
|
388
|
+
}
|
|
385
389
|
|
|
386
|
-
|
|
387
|
-
|
|
390
|
+
if (openCodeMajor < 2) {
|
|
391
|
+
for (const entry of validCommandEntries) {
|
|
392
|
+
const id = entry.name.slice(0, -3)
|
|
393
|
+
const targetName = `${SKILL_PREFIX}${id}.md`
|
|
394
|
+
const targetFile = path.join(commandsTarget, targetName)
|
|
395
|
+
const source = await readFile(path.join(sourceCommands, entry.name), "utf8")
|
|
396
|
+
const normalizedSource = normalizeManagedMarker(source)
|
|
397
|
+
const content = hasManagedMarker(normalizedSource)
|
|
398
|
+
? normalizedSource
|
|
399
|
+
: `${normalizedSource.trimEnd()}\n\n${MANAGED_MARKER}\n`
|
|
400
|
+
|
|
401
|
+
if (await writeManagedFile(targetFile, content, warnings)) {
|
|
402
|
+
installed.commands.push(targetName)
|
|
403
|
+
}
|
|
388
404
|
}
|
|
389
405
|
}
|
|
390
406
|
|
|
@@ -425,6 +441,14 @@ export async function installResources(options = {}) {
|
|
|
425
441
|
if (existsSync(path.join(sourcePluginDir, "index.js"))) {
|
|
426
442
|
if (await installPluginDirectory(sourcePluginDir, targetPluginDir, warnings)) {
|
|
427
443
|
installed.plugins.push(ROUTER_PLUGIN_STATE)
|
|
444
|
+
|
|
445
|
+
const promptAliasDir = path.join(targetPluginDir, "command-templates")
|
|
446
|
+
await mkdir(promptAliasDir, { recursive: true })
|
|
447
|
+
for (const entry of validCommandEntries) {
|
|
448
|
+
const id = entry.name.slice(0, -3)
|
|
449
|
+
await cp(path.join(sourceCommands, entry.name), path.join(promptAliasDir, entry.name))
|
|
450
|
+
installed.promptAliases.push(`${SKILL_PREFIX}${id}`)
|
|
451
|
+
}
|
|
428
452
|
}
|
|
429
453
|
}
|
|
430
454
|
|
|
@@ -457,7 +481,7 @@ export async function installResources(options = {}) {
|
|
|
457
481
|
}
|
|
458
482
|
|
|
459
483
|
const state = {
|
|
460
|
-
schemaVersion:
|
|
484
|
+
schemaVersion: 3,
|
|
461
485
|
package: PACKAGE_NAME,
|
|
462
486
|
version: await packageVersion(),
|
|
463
487
|
configDir,
|
|
@@ -465,6 +489,7 @@ export async function installResources(options = {}) {
|
|
|
465
489
|
openCodeMajor,
|
|
466
490
|
skills: installed.skills.sort(),
|
|
467
491
|
commands: installed.commands.sort(),
|
|
492
|
+
promptAliases: installed.promptAliases.sort(),
|
|
468
493
|
agents: installed.agents.sort(),
|
|
469
494
|
plugins: installed.plugins.sort(),
|
|
470
495
|
}
|
|
@@ -611,11 +636,13 @@ export async function getStatus() {
|
|
|
611
636
|
|
|
612
637
|
state.skills = Array.isArray(state.skills) ? state.skills.filter(validSkillID) : []
|
|
613
638
|
state.commands = Array.isArray(state.commands) ? state.commands.filter(validManagedMarkdown) : []
|
|
639
|
+
state.promptAliases = Array.isArray(state.promptAliases) ? state.promptAliases.filter(validPromptAlias) : []
|
|
614
640
|
state.agents = Array.isArray(state.agents) ? state.agents.filter(validManagedMarkdown) : []
|
|
615
641
|
state.plugins = Array.isArray(state.plugins) ? state.plugins.filter(validManagedPlugin) : []
|
|
616
642
|
|
|
617
643
|
let skillsPresent = 0
|
|
618
644
|
let commandsPresent = 0
|
|
645
|
+
let promptAliasesPresent = 0
|
|
619
646
|
let agentsPresent = 0
|
|
620
647
|
let pluginsPresent = 0
|
|
621
648
|
|
|
@@ -625,6 +652,11 @@ export async function getStatus() {
|
|
|
625
652
|
for (const name of state.commands || []) {
|
|
626
653
|
if (existsSync(path.join(configDir, "commands", name))) commandsPresent += 1
|
|
627
654
|
}
|
|
655
|
+
for (const name of state.promptAliases || []) {
|
|
656
|
+
if (existsSync(path.join(configDir, "plugins", "ues-router", "command-templates", promptAliasTemplateName(name)))) {
|
|
657
|
+
promptAliasesPresent += 1
|
|
658
|
+
}
|
|
659
|
+
}
|
|
628
660
|
for (const name of state.agents || []) {
|
|
629
661
|
if (existsSync(path.join(configDir, "agents", name))) agentsPresent += 1
|
|
630
662
|
}
|
|
@@ -639,6 +671,7 @@ export async function getStatus() {
|
|
|
639
671
|
...state,
|
|
640
672
|
skillsPresent,
|
|
641
673
|
commandsPresent,
|
|
674
|
+
promptAliasesPresent,
|
|
642
675
|
agentsPresent,
|
|
643
676
|
pluginsPresent,
|
|
644
677
|
workflowPresent: agents.includes(AGENTS_BEGIN),
|
package/lib/model-config.mjs
CHANGED
|
@@ -3,6 +3,7 @@ import { mkdir, readFile, writeFile } from "node:fs/promises"
|
|
|
3
3
|
import path from "node:path"
|
|
4
4
|
import { defaultModelPolicy, resolveModel } from "./model-policy.mjs"
|
|
5
5
|
import { normalizeCapabilityProfile } from "./capability-registry.mjs"
|
|
6
|
+
import { normalizePerformanceHistory, recordPerformanceOutcome } from "./model-performance.mjs"
|
|
6
7
|
|
|
7
8
|
const TIERS = new Set(["light", "standard", "heavy"])
|
|
8
9
|
|
|
@@ -29,8 +30,10 @@ function normalize(policy) {
|
|
|
29
30
|
if (validModelID(model)) capabilities[model] = normalizeCapabilityProfile(profile)
|
|
30
31
|
}
|
|
31
32
|
|
|
33
|
+
const performance = normalizePerformanceHistory(input.performance || {})
|
|
34
|
+
|
|
32
35
|
return {
|
|
33
|
-
schemaVersion:
|
|
36
|
+
schemaVersion: 3,
|
|
34
37
|
enabled: input.enabled === true,
|
|
35
38
|
maxEscalations: Number.isInteger(input.maxEscalations)
|
|
36
39
|
? Math.max(0, Math.min(input.maxEscalations, 2))
|
|
@@ -38,6 +41,8 @@ function normalize(policy) {
|
|
|
38
41
|
tiers,
|
|
39
42
|
roleTiers,
|
|
40
43
|
capabilities,
|
|
44
|
+
performance,
|
|
45
|
+
performanceMinSamples: Number.isInteger(input.performanceMinSamples) ? Math.max(1, Math.min(input.performanceMinSamples, 20)) : base.performanceMinSamples,
|
|
41
46
|
}
|
|
42
47
|
}
|
|
43
48
|
|
|
@@ -64,6 +69,7 @@ export async function writeModelPolicy(configDir, patch = {}) {
|
|
|
64
69
|
tiers: { ...current.tiers, ...(patch.tiers || {}) },
|
|
65
70
|
roleTiers: { ...current.roleTiers, ...(patch.roleTiers || {}) },
|
|
66
71
|
capabilities: { ...(current.capabilities || {}), ...(patch.capabilities || {}) },
|
|
72
|
+
performance: patch.performance || current.performance || {},
|
|
67
73
|
})
|
|
68
74
|
const file = modelPolicyFile(configDir)
|
|
69
75
|
await mkdir(path.dirname(file), { recursive: true })
|
|
@@ -94,3 +100,9 @@ export function applyConfiguredModel(source, role, policy) {
|
|
|
94
100
|
}
|
|
95
101
|
return lines.join("\n")
|
|
96
102
|
}
|
|
103
|
+
|
|
104
|
+
export async function recordModelPerformance(configDir, outcome = {}) {
|
|
105
|
+
const current = await readModelPolicy(configDir)
|
|
106
|
+
const performance = recordPerformanceOutcome(current.performance || {}, outcome)
|
|
107
|
+
return writeModelPolicy(configDir, { performance })
|
|
108
|
+
}
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
export const MODEL_TASK_CLASSES = Object.freeze([
|
|
2
|
+
"general", "repo-scale", "debugging", "architecture", "security",
|
|
3
|
+
"migration", "frontend", "backend", "visual", "browser",
|
|
4
|
+
])
|
|
5
|
+
const KNOWN_TASK_CLASSES = new Set(MODEL_TASK_CLASSES)
|
|
6
|
+
|
|
7
|
+
function boundedNumber(value, fallback = 0, min = 0, max = Number.MAX_SAFE_INTEGER) {
|
|
8
|
+
const number = Number(value)
|
|
9
|
+
if (!Number.isFinite(number)) return fallback
|
|
10
|
+
return Math.max(min, Math.min(max, number))
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
export function inferTaskClass(text = "", facts = {}) {
|
|
14
|
+
const explicit = String(facts.taskClass || "").trim().toLowerCase()
|
|
15
|
+
if (KNOWN_TASK_CLASSES.has(explicit)) return explicit
|
|
16
|
+
const value = String(text || "").toLowerCase()
|
|
17
|
+
if (/(whole repo|entire project|large monorepo|repo[- ]scale|cross[- ]module|toàn bộ dự án|nhiều module)/.test(value)) return "repo-scale"
|
|
18
|
+
if (/(prompt injection|security|auth|authorization|permission|secret|credential|bảo mật|phân quyền)/.test(value)) return "security"
|
|
19
|
+
if (/(migration|schema|database|sql|backfill|migrate)/.test(value)) return "migration"
|
|
20
|
+
if (/(screenshot|visual|figma|pixel|responsive|storybook)/.test(value)) return "visual"
|
|
21
|
+
if (/(browser|playwright|e2e|web page|click flow)/.test(value)) return "browser"
|
|
22
|
+
if (/(root cause|debug|regression|crash|failing|bug|lỗi)/.test(value)) return "debugging"
|
|
23
|
+
if (/(architecture|architect|design decision|system design|kiến trúc)/.test(value)) return "architecture"
|
|
24
|
+
if (/(react|next\.js|vue|svelte|css|frontend|ui\b)/.test(value)) return "frontend"
|
|
25
|
+
if (/(api|service|node|python|java|dotnet|backend|server)/.test(value)) return "backend"
|
|
26
|
+
return "general"
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
export function normalizePerformanceRecord(record = {}) {
|
|
30
|
+
record = record && typeof record === "object" ? record : {}
|
|
31
|
+
const samples = Math.floor(boundedNumber(record.samples, 0, 0))
|
|
32
|
+
const successes = Math.floor(boundedNumber(record.successes, Math.round(samples * boundedNumber(record.passRate, 0, 0, 1)), 0, samples))
|
|
33
|
+
return {
|
|
34
|
+
samples,
|
|
35
|
+
successes,
|
|
36
|
+
passRate: samples ? successes / samples : 0,
|
|
37
|
+
avgRetries: boundedNumber(record.avgRetries, 0, 0, 100),
|
|
38
|
+
avgTokens: boundedNumber(record.avgTokens, 0, 0),
|
|
39
|
+
avgLatencyMs: boundedNumber(record.avgLatencyMs, 0, 0),
|
|
40
|
+
updatedAt: typeof record.updatedAt === "string" ? record.updatedAt : null,
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
export function normalizePerformanceHistory(history = {}) {
|
|
45
|
+
const output = {}
|
|
46
|
+
for (const [model, classes] of Object.entries(history || {})) {
|
|
47
|
+
if (!model || !classes || typeof classes !== "object") continue
|
|
48
|
+
const normalizedClasses = {}
|
|
49
|
+
for (const [taskClass, record] of Object.entries(classes)) {
|
|
50
|
+
if (!KNOWN_TASK_CLASSES.has(taskClass) && taskClass !== "overall") continue
|
|
51
|
+
normalizedClasses[taskClass] = normalizePerformanceRecord(record)
|
|
52
|
+
}
|
|
53
|
+
if (Object.keys(normalizedClasses).length) output[model] = normalizedClasses
|
|
54
|
+
}
|
|
55
|
+
return output
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
function mergeAverage(previousAverage, previousSamples, value) {
|
|
59
|
+
return previousSamples <= 0 ? value : ((previousAverage * previousSamples) + value) / (previousSamples + 1)
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export function recordPerformanceOutcome(history = {}, outcome = {}) {
|
|
63
|
+
const model = String(outcome.model || "").trim()
|
|
64
|
+
if (!model) throw new Error("model performance outcome requires model")
|
|
65
|
+
const taskClass = inferTaskClass(outcome.text || "", { taskClass: outcome.taskClass })
|
|
66
|
+
const normalized = normalizePerformanceHistory(history)
|
|
67
|
+
const current = normalizePerformanceRecord(normalized[model]?.[taskClass] || {})
|
|
68
|
+
const samples = current.samples
|
|
69
|
+
const passed = outcome.passed === true
|
|
70
|
+
const next = {
|
|
71
|
+
samples: samples + 1,
|
|
72
|
+
successes: current.successes + (passed ? 1 : 0),
|
|
73
|
+
passRate: 0,
|
|
74
|
+
avgRetries: mergeAverage(current.avgRetries, samples, boundedNumber(outcome.retries, 0, 0, 100)),
|
|
75
|
+
avgTokens: mergeAverage(current.avgTokens, samples, boundedNumber(outcome.tokens, 0, 0)),
|
|
76
|
+
avgLatencyMs: mergeAverage(current.avgLatencyMs, samples, boundedNumber(outcome.latencyMs, 0, 0)),
|
|
77
|
+
updatedAt: new Date().toISOString(),
|
|
78
|
+
}
|
|
79
|
+
next.passRate = next.successes / next.samples
|
|
80
|
+
return { ...normalized, [model]: { ...(normalized[model] || {}), [taskClass]: next } }
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
function performanceAdjustment(record, minSamples) {
|
|
84
|
+
const normalized = normalizePerformanceRecord(record)
|
|
85
|
+
if (normalized.samples <= 0) return { adjustment: 0, confidence: 0, record: normalized }
|
|
86
|
+
const confidence = Math.min(1, normalized.samples / Math.max(1, minSamples))
|
|
87
|
+
if (normalized.samples < minSamples) return { adjustment: 0, confidence, record: normalized }
|
|
88
|
+
const correctness = (normalized.passRate - 0.5) * 80
|
|
89
|
+
const retryPenalty = Math.min(20, normalized.avgRetries * 5)
|
|
90
|
+
const latencyPenalty = normalized.avgLatencyMs > 0 ? Math.min(10, Math.max(0, Math.log10(Math.max(1, normalized.avgLatencyMs / 1000)) * 3)) : 0
|
|
91
|
+
return { adjustment: (correctness - retryPenalty - latencyPenalty) * confidence, confidence, record: normalized }
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
export function rerankCapabilitySelection(selection = {}, history = {}, options = {}) {
|
|
95
|
+
const taskClass = inferTaskClass(options.text || "", { taskClass: options.taskClass })
|
|
96
|
+
const minSamples = Math.max(1, Number(options.minSamples || 3))
|
|
97
|
+
const normalized = normalizePerformanceHistory(history)
|
|
98
|
+
const candidates = (selection.candidates || []).map((candidate) => {
|
|
99
|
+
const record = normalized[candidate.id]?.[taskClass] || normalized[candidate.id]?.overall || null
|
|
100
|
+
const evidence = performanceAdjustment(record, minSamples)
|
|
101
|
+
return {
|
|
102
|
+
...candidate,
|
|
103
|
+
baseScore: Number(candidate.score || 0),
|
|
104
|
+
empiricalTaskClass: taskClass,
|
|
105
|
+
empiricalEvidence: evidence.record,
|
|
106
|
+
empiricalConfidence: Number(evidence.confidence.toFixed(4)),
|
|
107
|
+
adjustedScore: Number((Number(candidate.score || 0) + evidence.adjustment).toFixed(6)),
|
|
108
|
+
}
|
|
109
|
+
})
|
|
110
|
+
const eligible = candidates.filter((candidate) => candidate.eligible)
|
|
111
|
+
.sort((a, b) => b.adjustedScore - a.adjustedScore || b.baseScore - a.baseScore)
|
|
112
|
+
return { ...selection, selected: eligible[0] || null, candidates, taskClass, empirical: true }
|
|
113
|
+
}
|
package/lib/model-policy.mjs
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { inferTaskCapabilities, modelCandidatesFromPolicy, selectCapabilityCandidate } from "./capability-registry.mjs"
|
|
2
|
+
import { inferTaskClass, rerankCapabilitySelection } from "./model-performance.mjs"
|
|
2
3
|
|
|
3
4
|
const ROLE_TIERS = {
|
|
4
5
|
"codebase-mapper": "standard",
|
|
@@ -45,12 +46,14 @@ export function resolveModel(role, attempt, config = {}) {
|
|
|
45
46
|
|
|
46
47
|
export function defaultModelPolicy() {
|
|
47
48
|
return {
|
|
48
|
-
schemaVersion:
|
|
49
|
+
schemaVersion: 3,
|
|
49
50
|
enabled: false,
|
|
50
51
|
maxEscalations: 2,
|
|
51
52
|
tiers: { light: null, standard: null, heavy: null },
|
|
52
53
|
roleTiers: { ...ROLE_TIERS },
|
|
53
54
|
capabilities: {},
|
|
55
|
+
performance: {},
|
|
56
|
+
performanceMinSamples: 3,
|
|
54
57
|
}
|
|
55
58
|
}
|
|
56
59
|
|
|
@@ -90,10 +93,14 @@ export function resolveCapabilityModel(role, attempt, taskText = "", taskPolicy
|
|
|
90
93
|
})
|
|
91
94
|
const candidates = modelCandidatesFromPolicy(config)
|
|
92
95
|
.filter((candidate) => tierIndex(candidate.tier) >= tierIndex(base.tier))
|
|
93
|
-
const
|
|
96
|
+
const staticSelection = selectCapabilityCandidate(requirements, candidates, {
|
|
94
97
|
role,
|
|
95
98
|
preferredTier: base.tier,
|
|
96
99
|
})
|
|
100
|
+
const taskClass = inferTaskClass(taskText, facts)
|
|
101
|
+
const selection = rerankCapabilitySelection(staticSelection, config.performance || {}, {
|
|
102
|
+
taskClass, text: taskText, minSamples: config.performanceMinSamples || 3,
|
|
103
|
+
})
|
|
97
104
|
const capabilityEnforced = config.enabled === true && candidates.length > 0
|
|
98
105
|
if (capabilityEnforced && !selection.selected) {
|
|
99
106
|
return {
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
import { mkdir, writeFile } from "node:fs/promises"
|
|
2
|
+
import path from "node:path"
|
|
3
|
+
|
|
4
|
+
function moduleSource(packageIndex, moduleIndex) {
|
|
5
|
+
const previous = moduleIndex > 0
|
|
6
|
+
? 'import { value as previous } from "./mod' + String(moduleIndex - 1).padStart(3, "0") + '.mjs"\n'
|
|
7
|
+
: ""
|
|
8
|
+
return previous + "export const value = " + (moduleIndex > 0 ? "previous + 1" : packageIndex * 1000) +
|
|
9
|
+
"\nexport function compute(input) { return value + Number(input || 0) }\n"
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
export async function generateRepoScaleFixture(root, options = {}) {
|
|
13
|
+
const packageCount = Math.max(2, Number(options.packageCount || 6))
|
|
14
|
+
const modulesPerPackage = Math.max(5, Number(options.modulesPerPackage || 50))
|
|
15
|
+
await mkdir(root, { recursive: true })
|
|
16
|
+
await writeFile(path.join(root, "package.json"), JSON.stringify({
|
|
17
|
+
name: "ues-repo-scale-fixture", private: true, type: "module", workspaces: ["packages/*", "apps/*"],
|
|
18
|
+
}, null, 2) + "\n")
|
|
19
|
+
|
|
20
|
+
let generatedModules = 0
|
|
21
|
+
for (let p = 0; p < packageCount; p += 1) {
|
|
22
|
+
const pkgRoot = path.join(root, "packages", "pkg-" + p)
|
|
23
|
+
const src = path.join(pkgRoot, "src")
|
|
24
|
+
await mkdir(src, { recursive: true })
|
|
25
|
+
await writeFile(path.join(pkgRoot, "package.json"), JSON.stringify({
|
|
26
|
+
name: "@fixture/pkg-" + p, private: true, type: "module", exports: "./src/index.mjs",
|
|
27
|
+
}, null, 2) + "\n")
|
|
28
|
+
for (let m = 0; m < modulesPerPackage; m += 1) {
|
|
29
|
+
await writeFile(path.join(src, "mod" + String(m).padStart(3, "0") + ".mjs"), moduleSource(p, m))
|
|
30
|
+
generatedModules += 1
|
|
31
|
+
}
|
|
32
|
+
await writeFile(path.join(src, "index.mjs"),
|
|
33
|
+
'export { value, compute } from "./mod' + String(modulesPerPackage - 1).padStart(3, "0") + '.mjs"\n')
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
const apiRoot = path.join(root, "apps", "api", "src")
|
|
37
|
+
const webRoot = path.join(root, "apps", "web", "src")
|
|
38
|
+
const contracts = path.join(root, "contracts")
|
|
39
|
+
await mkdir(apiRoot, { recursive: true }); await mkdir(webRoot, { recursive: true }); await mkdir(contracts, { recursive: true })
|
|
40
|
+
await writeFile(path.join(contracts, "public-api.json"), JSON.stringify({ version: 1, fields: ["id", "name", "status"] }, null, 2) + "\n")
|
|
41
|
+
await writeFile(path.join(apiRoot, "service.mjs"), 'export function serialize(row) { return { id: row.id, name: row.name, status: row.status } }\n')
|
|
42
|
+
await writeFile(path.join(webRoot, "consumer.mjs"), 'export function render(item) { return item.name + ":" + item.status }\n')
|
|
43
|
+
await writeFile(path.join(root, "AGENTS.md"), "# Fixture instructions\n\nPreserve public contracts and update focused tests.\n")
|
|
44
|
+
return { root, packageCount, modulesPerPackage, generatedModules, totalPrimaryFiles: generatedModules + packageCount + 5 }
|
|
45
|
+
}
|
package/lib/task-engine.mjs
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import { closeSync, existsSync, openSync, readFileSync, readSync, readdirSync, statSync } from "node:fs"
|
|
2
|
-
import { mkdir,
|
|
2
|
+
import { mkdir, rename, rm, stat, writeFile } from "node:fs/promises"
|
|
3
3
|
import { spawnSync } from "node:child_process"
|
|
4
4
|
import { createHash, randomUUID } from "node:crypto"
|
|
5
5
|
import path from "node:path"
|
|
6
|
-
import { analyzePlan, taskFiles, validatePlan } from "./task-graph.mjs"
|
|
6
|
+
import { analyzePlan, taskFiles, taskVerificationCommands, validatePlan } from "./task-graph.mjs"
|
|
7
7
|
import { buildContextManifest } from "./context-manifest.mjs"
|
|
8
8
|
import { relevantAcceptedLearnings } from "./learning-engine.mjs"
|
|
9
9
|
import { validateVerificationReceipt } from "./evidence-receipt.mjs"
|
|
@@ -15,6 +15,9 @@ import { putEvidence } from "./evidence-store.mjs"
|
|
|
15
15
|
import { inferTaskCapabilities } from "./capability-registry.mjs"
|
|
16
16
|
import { buildPromptEnvelope } from "./prompt-cache.mjs"
|
|
17
17
|
import { externalizeContextExcerpts } from "./context-engine-v11.mjs"
|
|
18
|
+
import { assertActivePlanScope, persistPlanScope } from "./work-plan-scope.mjs"
|
|
19
|
+
import { classifyDecisionPolicy } from "./decision-policy.mjs"
|
|
20
|
+
import { readTextAuto } from "./text-encoding.mjs"
|
|
18
21
|
|
|
19
22
|
const WORK_DIR = ".ues-work"
|
|
20
23
|
const STATE_SCHEMA = 4
|
|
@@ -49,13 +52,15 @@ export function workPaths(root, slug) {
|
|
|
49
52
|
events: path.join(dir, "EVENTS.jsonl"),
|
|
50
53
|
tasks: path.join(dir, "tasks"),
|
|
51
54
|
reports: path.join(dir, "reports"),
|
|
55
|
+
plans: path.join(dir, "plans"),
|
|
56
|
+
activePlan: path.join(dir, "ACTIVE_PLAN.json"),
|
|
52
57
|
lock: path.join(dir, ".state-lock"),
|
|
53
58
|
}
|
|
54
59
|
}
|
|
55
60
|
|
|
56
61
|
async function readJson(file, fallback = null) {
|
|
57
62
|
try {
|
|
58
|
-
return JSON.parse(
|
|
63
|
+
return JSON.parse(readTextAuto(file))
|
|
59
64
|
} catch (error) {
|
|
60
65
|
if (error?.code === "ENOENT") return fallback
|
|
61
66
|
throw error
|
|
@@ -268,6 +273,7 @@ export async function initWork(root, slug, goal) {
|
|
|
268
273
|
|
|
269
274
|
await mkdir(paths.tasks, { recursive: true })
|
|
270
275
|
await mkdir(paths.reports, { recursive: true })
|
|
276
|
+
await mkdir(paths.plans, { recursive: true })
|
|
271
277
|
const createdAt = now()
|
|
272
278
|
const cleanGoal = String(goal || "").trim() || "Define the requested engineering outcome."
|
|
273
279
|
|
|
@@ -285,6 +291,7 @@ export async function initWork(root, slug, goal) {
|
|
|
285
291
|
planImportedAt: null,
|
|
286
292
|
planHash: null,
|
|
287
293
|
planApproval: null,
|
|
294
|
+
planScope: null,
|
|
288
295
|
integrationVerification: null,
|
|
289
296
|
tasks: {},
|
|
290
297
|
decisions: [],
|
|
@@ -334,7 +341,7 @@ function currentEvidencePolicy(state, plan) {
|
|
|
334
341
|
|
|
335
342
|
function taskBrief(task) {
|
|
336
343
|
const files = taskFiles(task)
|
|
337
|
-
return `# ${task.id}: ${task.title}\n\n## Summary\n\n${task.summary || "No summary provided."}\n\n## Dependencies\n\n${(task.dependsOn || []).map((id) => `- ${id}`).join("\n") || "- None"}\n\n## Files\n\n${files.map((file) => `- ${file}`).join("\n") || "- Scope must be established before editing."}\n\n## Acceptance criteria\n\n${(task.acceptance || []).map((item) => `- ${item}`).join("\n")}\n\n## Verification\n\n${(task.verification || []).map((item) => `- ${item}`).join("\n")}\n\n## Risk\n\n${task.risk || "medium"}\n`
|
|
344
|
+
return `# ${task.id}: ${task.title}\n\n## Summary\n\n${task.summary || "No summary provided."}\n\n## Dependencies\n\n${(task.dependsOn || []).map((id) => `- ${id}`).join("\n") || "- None"}\n\n## Files\n\n${files.map((file) => `- ${file}`).join("\n") || "- Scope must be established before editing."}\n\n## Acceptance criteria\n\n${(task.acceptance || []).map((item) => `- ${item}`).join("\n")}\n\n## Verification\n\n${(task.verification || []).map((item) => `- ${item}`).join("\n")}\n\n## Structured verification commands\n\n${taskVerificationCommands(task).map((item) => `- ${[item.command, ...item.args].join(" ")}`).join("\n") || "- None declared"}\n\n## Risk\n\n${task.risk || "medium"}\n`
|
|
338
345
|
}
|
|
339
346
|
|
|
340
347
|
function planIsApproved(state, plan) {
|
|
@@ -345,7 +352,7 @@ function planIsApproved(state, plan) {
|
|
|
345
352
|
|
|
346
353
|
export async function importPlan(root, slug, planInput) {
|
|
347
354
|
const plan = typeof planInput === "string"
|
|
348
|
-
? JSON.parse(
|
|
355
|
+
? JSON.parse(readTextAuto(path.resolve(planInput)))
|
|
349
356
|
: planInput
|
|
350
357
|
|
|
351
358
|
const validation = validatePlan(plan)
|
|
@@ -383,6 +390,7 @@ export async function importPlan(root, slug, planInput) {
|
|
|
383
390
|
}
|
|
384
391
|
|
|
385
392
|
const updatedAt = now()
|
|
393
|
+
const planScope = await persistPlanScope(loaded.paths, hash, plan, { importedAt: updatedAt, previousPlanHash: loaded.state.planHash || null })
|
|
386
394
|
const state = {
|
|
387
395
|
...loaded.state,
|
|
388
396
|
schemaVersion: STATE_SCHEMA,
|
|
@@ -392,6 +400,7 @@ export async function importPlan(root, slug, planInput) {
|
|
|
392
400
|
planImportedAt: updatedAt,
|
|
393
401
|
planHash: hash,
|
|
394
402
|
planApproval: { status: "pending", planHash: hash, at: updatedAt, evidence: null },
|
|
403
|
+
planScope,
|
|
395
404
|
integrationVerification: null,
|
|
396
405
|
checkpoint: null,
|
|
397
406
|
evidencePolicy: evidencePolicyForPlan(plan),
|
|
@@ -414,6 +423,7 @@ export async function approvePlan(root, slug, evidence, options = {}) {
|
|
|
414
423
|
const loaded = await loadWork(root, slug)
|
|
415
424
|
if (!loaded.plan) throw new Error("PLAN.json is missing")
|
|
416
425
|
const hash = planHash(loaded.plan)
|
|
426
|
+
await assertActivePlanScope(loaded.paths, loaded.state.planHash || hash)
|
|
417
427
|
if (loaded.state.planHash && loaded.state.planHash !== hash) {
|
|
418
428
|
throw new Error("PLAN.json changed after import; re-import it before approval")
|
|
419
429
|
}
|
|
@@ -525,7 +535,9 @@ export async function workStatus(root, slug) {
|
|
|
525
535
|
leaseExpiresAt: value.leaseExpiresAt || null,
|
|
526
536
|
}))
|
|
527
537
|
const taskEntries = Object.values(loaded.state.tasks || {})
|
|
528
|
-
const receiptBacked = taskEntries.filter((value) =>
|
|
538
|
+
const receiptBacked = taskEntries.filter((value) =>
|
|
539
|
+
["receipt-backed", "command-receipt-backed", "independent-agent-receipt-backed"].includes(value.evidenceStrength)
|
|
540
|
+
).length
|
|
529
541
|
const completed = taskEntries.filter((value) => value.status === "completed").length
|
|
530
542
|
|
|
531
543
|
return {
|
|
@@ -540,6 +552,7 @@ export async function workStatus(root, slug) {
|
|
|
540
552
|
blockers: loaded.state.blockers || [],
|
|
541
553
|
planApproval: loaded.state.planApproval || null,
|
|
542
554
|
integrationVerification: loaded.state.integrationVerification || null,
|
|
555
|
+
planScope: loaded.state.planScope || null,
|
|
543
556
|
evidence: {
|
|
544
557
|
receiptBacked,
|
|
545
558
|
completed,
|
|
@@ -557,6 +570,7 @@ export async function startTask(root, slug, taskID, options = {}) {
|
|
|
557
570
|
const result = await withWorkLock(root, slug, async () => {
|
|
558
571
|
const loaded = await loadWork(root, slug)
|
|
559
572
|
if (!loaded.plan) throw new Error("PLAN.json is missing; import a valid plan first")
|
|
573
|
+
await assertActivePlanScope(loaded.paths, loaded.state.planHash || planHash(loaded.plan))
|
|
560
574
|
if (!planIsApproved(loaded.state, loaded.plan)) {
|
|
561
575
|
throw new Error("plan is not approved; run ues-plan-checker and 'ocskill work approve-plan' first")
|
|
562
576
|
}
|
|
@@ -812,7 +826,13 @@ export async function completeTask(root, slug, taskID, options = {}) {
|
|
|
812
826
|
record.status = "completed"
|
|
813
827
|
record.completedAt = timestamp
|
|
814
828
|
record.report = path.relative(loaded.paths.root, reportPath).replaceAll("\\", "/")
|
|
815
|
-
|
|
829
|
+
const commandReceipts = acceptedReceipts.filter((item) => item.kind !== "agent-verifier")
|
|
830
|
+
const agentVerifierReceipts = acceptedReceipts.filter((item) => item.kind === "agent-verifier")
|
|
831
|
+
record.evidenceStrength = commandReceipts.length
|
|
832
|
+
? "command-receipt-backed"
|
|
833
|
+
: agentVerifierReceipts.length
|
|
834
|
+
? "independent-agent-receipt-backed"
|
|
835
|
+
: "narrative"
|
|
816
836
|
record.runId = null
|
|
817
837
|
record.owner = null
|
|
818
838
|
record.heartbeatAt = null
|
|
@@ -888,16 +908,19 @@ export async function failTask(root, slug, taskID, reason, options = {}) {
|
|
|
888
908
|
}
|
|
889
909
|
|
|
890
910
|
export async function addDecision(root, slug, decision) {
|
|
891
|
-
const
|
|
911
|
+
const input = decision && typeof decision === "object" ? decision : { text: decision }
|
|
912
|
+
const text = String(input.text || "").trim()
|
|
892
913
|
if (!text) throw new Error("decision text is required")
|
|
893
|
-
|
|
914
|
+
const policy = classifyDecisionPolicy(text, input.facts || input)
|
|
915
|
+
if (input.auto === true && !policy.autoResolvable) throw new Error("decision requires human approval before automatic resolution")
|
|
894
916
|
return withWorkLock(root, slug, async () => {
|
|
895
917
|
const loaded = await loadWork(root, slug)
|
|
896
918
|
loaded.state.decisions ??= []
|
|
897
919
|
const timestamp = now()
|
|
898
|
-
loaded.state.decisions.push({ at: timestamp, text })
|
|
920
|
+
loaded.state.decisions.push({ at: timestamp, text, policy, source: input.source || (input.auto === true ? "auto-ruling" : "human-or-agent") })
|
|
899
921
|
loaded.state.updatedAt = timestamp
|
|
900
922
|
await writeJson(loaded.paths.state, loaded.state)
|
|
923
|
+
await journal(loaded.paths, "decision.recorded", { text, policy })
|
|
901
924
|
return loaded.state
|
|
902
925
|
})
|
|
903
926
|
}
|
|
@@ -1071,14 +1094,15 @@ export async function finalizeWork(root, slug, evidence) {
|
|
|
1071
1094
|
async function buildContextPack(loaded, taskID) {
|
|
1072
1095
|
if (!loaded.plan) throw new Error("PLAN.json is missing")
|
|
1073
1096
|
const task = taskByID(loaded.plan, taskID)
|
|
1074
|
-
|
|
1097
|
+
let spec = ""
|
|
1098
|
+
try { spec = readTextAuto(loaded.paths.spec) } catch {}
|
|
1075
1099
|
const dependencyReports = {}
|
|
1076
1100
|
const dependencyReportRefs = {}
|
|
1077
1101
|
|
|
1078
1102
|
for (const dep of task.dependsOn || []) {
|
|
1079
1103
|
const file = path.join(loaded.paths.reports, `${dep}.md`)
|
|
1080
1104
|
if (!existsSync(file)) continue
|
|
1081
|
-
const report =
|
|
1105
|
+
const report = readTextAuto(file)
|
|
1082
1106
|
dependencyReports[dep] = report.slice(0, 6000)
|
|
1083
1107
|
const stored = await putEvidence(loaded.paths.root, report, {
|
|
1084
1108
|
kind: "dependency-report",
|
package/lib/task-graph.mjs
CHANGED
|
@@ -48,6 +48,20 @@ export function taskReadFiles(task) {
|
|
|
48
48
|
return [...new Set((files.read || []).map(normalizeFile).filter(Boolean))].sort()
|
|
49
49
|
}
|
|
50
50
|
|
|
51
|
+
|
|
52
|
+
export function taskVerificationCommands(task) {
|
|
53
|
+
if (!Array.isArray(task?.verificationCommands)) return []
|
|
54
|
+
return task.verificationCommands
|
|
55
|
+
.map((item) => {
|
|
56
|
+
if (!item || typeof item !== "object" || Array.isArray(item)) return null
|
|
57
|
+
const command = String(item.command || "").trim()
|
|
58
|
+
if (!command) return null
|
|
59
|
+
const args = Array.isArray(item.args) ? item.args.map((value) => String(value)) : []
|
|
60
|
+
return { command, args }
|
|
61
|
+
})
|
|
62
|
+
.filter(Boolean)
|
|
63
|
+
}
|
|
64
|
+
|
|
51
65
|
function taskMap(plan) {
|
|
52
66
|
return new Map((plan?.tasks || []).map((task) => [task.id, task]))
|
|
53
67
|
}
|
|
@@ -97,6 +111,24 @@ export function validatePlan(plan) {
|
|
|
97
111
|
errors.push(`${id || prefix}: verification must contain at least one concrete check`)
|
|
98
112
|
}
|
|
99
113
|
|
|
114
|
+
if (task.verificationCommands !== undefined) {
|
|
115
|
+
if (!Array.isArray(task.verificationCommands)) {
|
|
116
|
+
errors.push(`${id || prefix}.verificationCommands must be an array when provided`)
|
|
117
|
+
} else {
|
|
118
|
+
for (const [commandIndex, commandSpec] of task.verificationCommands.entries()) {
|
|
119
|
+
const commandPrefix = `${id || prefix}.verificationCommands[${commandIndex}]`
|
|
120
|
+
if (!commandSpec || typeof commandSpec !== "object" || Array.isArray(commandSpec)) {
|
|
121
|
+
errors.push(`${commandPrefix} must be an object`)
|
|
122
|
+
continue
|
|
123
|
+
}
|
|
124
|
+
if (!String(commandSpec.command || "").trim()) errors.push(`${commandPrefix}.command is required`)
|
|
125
|
+
if (commandSpec.args !== undefined && !Array.isArray(commandSpec.args)) {
|
|
126
|
+
errors.push(`${commandPrefix}.args must be an array when provided`)
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
|
|
100
132
|
const risk = task.risk ?? "medium"
|
|
101
133
|
if (!VALID_RISKS.has(risk)) errors.push(`${id || prefix}: invalid risk '${risk}'`)
|
|
102
134
|
|