opencode-agent-skill 11.0.0 → 13.0.0-beta.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/CHANGELOG.md +55 -0
  2. package/README.md +119 -14
  3. package/bin/ocskill.mjs +416 -83
  4. package/docs/DETERMINISTIC-TOOLS.md +1 -1
  5. package/docs/ENGINEERING-DESIGN.md +4 -4
  6. package/docs/EVALS.md +3 -3
  7. package/docs/GITHUB-RULESET.md +50 -0
  8. package/docs/NPM-PUBLISH.md +22 -8
  9. package/docs/OPENCODE-COMPAT.md +14 -4
  10. package/docs/TRACE-SCHEMA.md +1 -1
  11. package/docs/V11-PERCEPTION-ADAPTIVE.md +2 -2
  12. package/docs/V12-WEAK-MODEL-INTELLIGENCE.md +27 -0
  13. package/docs/V13-PARALLEL-WEAK-MODEL-RUNTIME.md +75 -0
  14. package/evals/repo-scale/tasks.json +62 -0
  15. package/global-config/AGENTS.md +4 -0
  16. package/global-config/commands/resume.md +4 -1
  17. package/global-config/commands/run.md +9 -7
  18. package/global-config/plugins/ues-router/capabilities.js +1 -0
  19. package/global-config/plugins/ues-router/command-runtime.js +79 -0
  20. package/global-config/plugins/ues-router/index.js +364 -43
  21. package/global-config/plugins/ues-router/parallel-runtime.js +271 -0
  22. package/global-config/plugins/ues-router/text-runtime.js +115 -0
  23. package/global-config/plugins/ues-router/verifier-runtime.js +33 -0
  24. package/global-config/skills/dynamic-workflow/SKILL.md +7 -6
  25. package/global-config/skills/dynamic-workflow/references/workflow.md +8 -6
  26. package/global-config/skills/engineering-orchestrator/references/long-horizon.md +7 -5
  27. package/lib/cli-utils.mjs +17 -4
  28. package/lib/context-engine-v11.mjs +4 -0
  29. package/lib/context-quality.mjs +59 -0
  30. package/lib/decision-policy.mjs +23 -0
  31. package/lib/installer.mjs +49 -16
  32. package/lib/model-config.mjs +13 -1
  33. package/lib/model-performance.mjs +113 -0
  34. package/lib/model-policy.mjs +9 -2
  35. package/lib/repo-scale-fixture.mjs +45 -0
  36. package/lib/task-engine.mjs +36 -12
  37. package/lib/task-graph.mjs +32 -0
  38. package/lib/text-encoding.mjs +74 -0
  39. package/lib/work-plan-scope.mjs +49 -0
  40. package/lib/worktree-sandbox.mjs +141 -12
  41. package/package.json +8 -4
  42. package/scripts/check-release-consistency.mjs +260 -0
  43. package/scripts/smoke-packed-install.mjs +39 -3
  44. package/scripts/smoke-plain-install.mjs +27 -7
  45. package/scripts/validate-repo-scale-suite.mjs +27 -0
  46. package/scripts/validate-v12-foundation.mjs +24 -0
  47. package/scripts/validate.mjs +18 -1
package/lib/installer.mjs CHANGED
@@ -23,6 +23,15 @@ function validManagedPlugin(value) {
23
23
  return value === ROUTER_PLUGIN_STATE
24
24
  }
25
25
 
26
+ function validPromptAlias(value) {
27
+ if (typeof value !== "string" || !value.startsWith(SKILL_PREFIX)) return false
28
+ return validResourceID(value.slice(SKILL_PREFIX.length))
29
+ }
30
+
31
+ function promptAliasTemplateName(value) {
32
+ return value.slice(SKILL_PREFIX.length) + ".md"
33
+ }
34
+
26
35
  function isOwnedPackageName(value) {
27
36
  return !value || value === PACKAGE_NAME || value === LEGACY_PACKAGE_NAME
28
37
  }
@@ -94,7 +103,7 @@ async function writeManagedFile(file, content, warnings) {
94
103
 
95
104
  async function readManagedState(file, warnings = []) {
96
105
  const raw = await readText(file)
97
- if (!raw) return { kind: "empty", state: { skills: [], commands: [], agents: [], plugins: [] } }
106
+ if (!raw) return { kind: "empty", state: { skills: [], commands: [], promptAliases: [], agents: [], plugins: [] } }
98
107
 
99
108
  try {
100
109
  const parsed = JSON.parse(raw)
@@ -103,7 +112,7 @@ async function readManagedState(file, warnings = []) {
103
112
  return {
104
113
  kind: "foreign",
105
114
  package: parsed.package,
106
- state: { skills: [], commands: [], agents: [], plugins: [] },
115
+ state: { skills: [], commands: [], promptAliases: [], agents: [], plugins: [] },
107
116
  }
108
117
  }
109
118
  return {
@@ -112,13 +121,14 @@ async function readManagedState(file, warnings = []) {
112
121
  ...parsed,
113
122
  skills: Array.isArray(parsed.skills) ? parsed.skills.filter(validSkillID) : [],
114
123
  commands: Array.isArray(parsed.commands) ? parsed.commands.filter(validManagedMarkdown) : [],
124
+ promptAliases: Array.isArray(parsed.promptAliases) ? parsed.promptAliases.filter(validPromptAlias) : [],
115
125
  agents: Array.isArray(parsed.agents) ? parsed.agents.filter(validManagedMarkdown) : [],
116
126
  plugins: Array.isArray(parsed.plugins) ? parsed.plugins.filter(validManagedPlugin) : [],
117
127
  },
118
128
  }
119
129
  } catch {
120
130
  warnings.push(`Invalid UES state file; run 'ocskill install' to rebuild it: ${file}`)
121
- return { kind: "invalid", state: { skills: [], commands: [], agents: [], plugins: [] } }
131
+ return { kind: "invalid", state: { skills: [], commands: [], promptAliases: [], agents: [], plugins: [] } }
122
132
  }
123
133
  }
124
134
 
@@ -310,7 +320,7 @@ export async function installResources(options = {}) {
310
320
  const stateFile = path.join(stateDir, "state.json")
311
321
  const warnings = []
312
322
  const openCodeMajor = options.openCodeMajor ?? detectOpenCodeMajor()
313
- const installed = { skills: [], commands: [], agents: [], plugins: [] }
323
+ const installed = { skills: [], commands: [], promptAliases: [], agents: [], plugins: [] }
314
324
  const previous = await readManagedState(stateFile, warnings)
315
325
  const previousState = previous.state
316
326
  const modelPolicy = await readModelPolicy(configDir)
@@ -366,25 +376,31 @@ export async function installResources(options = {}) {
366
376
 
367
377
  const sourceCommands = path.join(sourceRoot, "global-config", "commands")
368
378
  const commandEntries = await readdir(sourceCommands, { withFileTypes: true })
369
-
379
+ const validCommandEntries = []
370
380
  for (const entry of commandEntries) {
371
381
  if (!entry.isFile() || !entry.name.endsWith(".md")) continue
372
-
373
382
  const id = entry.name.slice(0, -3)
374
383
  if (!validResourceID(id)) {
375
384
  warnings.push(`Skipped command with invalid resource ID: ${id}`)
376
385
  continue
377
386
  }
378
- const targetName = `${SKILL_PREFIX}${id}.md`
379
- const targetFile = path.join(commandsTarget, targetName)
380
- const source = await readFile(path.join(sourceCommands, entry.name), "utf8")
381
- const normalizedSource = normalizeManagedMarker(source)
382
- const content = hasManagedMarker(normalizedSource)
383
- ? normalizedSource
384
- : `${normalizedSource.trimEnd()}\n\n${MANAGED_MARKER}\n`
387
+ validCommandEntries.push(entry)
388
+ }
385
389
 
386
- if (await writeManagedFile(targetFile, content, warnings)) {
387
- installed.commands.push(targetName)
390
+ if (openCodeMajor < 2) {
391
+ for (const entry of validCommandEntries) {
392
+ const id = entry.name.slice(0, -3)
393
+ const targetName = `${SKILL_PREFIX}${id}.md`
394
+ const targetFile = path.join(commandsTarget, targetName)
395
+ const source = await readFile(path.join(sourceCommands, entry.name), "utf8")
396
+ const normalizedSource = normalizeManagedMarker(source)
397
+ const content = hasManagedMarker(normalizedSource)
398
+ ? normalizedSource
399
+ : `${normalizedSource.trimEnd()}\n\n${MANAGED_MARKER}\n`
400
+
401
+ if (await writeManagedFile(targetFile, content, warnings)) {
402
+ installed.commands.push(targetName)
403
+ }
388
404
  }
389
405
  }
390
406
 
@@ -425,6 +441,14 @@ export async function installResources(options = {}) {
425
441
  if (existsSync(path.join(sourcePluginDir, "index.js"))) {
426
442
  if (await installPluginDirectory(sourcePluginDir, targetPluginDir, warnings)) {
427
443
  installed.plugins.push(ROUTER_PLUGIN_STATE)
444
+
445
+ const promptAliasDir = path.join(targetPluginDir, "command-templates")
446
+ await mkdir(promptAliasDir, { recursive: true })
447
+ for (const entry of validCommandEntries) {
448
+ const id = entry.name.slice(0, -3)
449
+ await cp(path.join(sourceCommands, entry.name), path.join(promptAliasDir, entry.name))
450
+ installed.promptAliases.push(`${SKILL_PREFIX}${id}`)
451
+ }
428
452
  }
429
453
  }
430
454
 
@@ -457,7 +481,7 @@ export async function installResources(options = {}) {
457
481
  }
458
482
 
459
483
  const state = {
460
- schemaVersion: 2,
484
+ schemaVersion: 3,
461
485
  package: PACKAGE_NAME,
462
486
  version: await packageVersion(),
463
487
  configDir,
@@ -465,6 +489,7 @@ export async function installResources(options = {}) {
465
489
  openCodeMajor,
466
490
  skills: installed.skills.sort(),
467
491
  commands: installed.commands.sort(),
492
+ promptAliases: installed.promptAliases.sort(),
468
493
  agents: installed.agents.sort(),
469
494
  plugins: installed.plugins.sort(),
470
495
  }
@@ -611,11 +636,13 @@ export async function getStatus() {
611
636
 
612
637
  state.skills = Array.isArray(state.skills) ? state.skills.filter(validSkillID) : []
613
638
  state.commands = Array.isArray(state.commands) ? state.commands.filter(validManagedMarkdown) : []
639
+ state.promptAliases = Array.isArray(state.promptAliases) ? state.promptAliases.filter(validPromptAlias) : []
614
640
  state.agents = Array.isArray(state.agents) ? state.agents.filter(validManagedMarkdown) : []
615
641
  state.plugins = Array.isArray(state.plugins) ? state.plugins.filter(validManagedPlugin) : []
616
642
 
617
643
  let skillsPresent = 0
618
644
  let commandsPresent = 0
645
+ let promptAliasesPresent = 0
619
646
  let agentsPresent = 0
620
647
  let pluginsPresent = 0
621
648
 
@@ -625,6 +652,11 @@ export async function getStatus() {
625
652
  for (const name of state.commands || []) {
626
653
  if (existsSync(path.join(configDir, "commands", name))) commandsPresent += 1
627
654
  }
655
+ for (const name of state.promptAliases || []) {
656
+ if (existsSync(path.join(configDir, "plugins", "ues-router", "command-templates", promptAliasTemplateName(name)))) {
657
+ promptAliasesPresent += 1
658
+ }
659
+ }
628
660
  for (const name of state.agents || []) {
629
661
  if (existsSync(path.join(configDir, "agents", name))) agentsPresent += 1
630
662
  }
@@ -639,6 +671,7 @@ export async function getStatus() {
639
671
  ...state,
640
672
  skillsPresent,
641
673
  commandsPresent,
674
+ promptAliasesPresent,
642
675
  agentsPresent,
643
676
  pluginsPresent,
644
677
  workflowPresent: agents.includes(AGENTS_BEGIN),
@@ -3,6 +3,7 @@ import { mkdir, readFile, writeFile } from "node:fs/promises"
3
3
  import path from "node:path"
4
4
  import { defaultModelPolicy, resolveModel } from "./model-policy.mjs"
5
5
  import { normalizeCapabilityProfile } from "./capability-registry.mjs"
6
+ import { normalizePerformanceHistory, recordPerformanceOutcome } from "./model-performance.mjs"
6
7
 
7
8
  const TIERS = new Set(["light", "standard", "heavy"])
8
9
 
@@ -29,8 +30,10 @@ function normalize(policy) {
29
30
  if (validModelID(model)) capabilities[model] = normalizeCapabilityProfile(profile)
30
31
  }
31
32
 
33
+ const performance = normalizePerformanceHistory(input.performance || {})
34
+
32
35
  return {
33
- schemaVersion: 2,
36
+ schemaVersion: 3,
34
37
  enabled: input.enabled === true,
35
38
  maxEscalations: Number.isInteger(input.maxEscalations)
36
39
  ? Math.max(0, Math.min(input.maxEscalations, 2))
@@ -38,6 +41,8 @@ function normalize(policy) {
38
41
  tiers,
39
42
  roleTiers,
40
43
  capabilities,
44
+ performance,
45
+ performanceMinSamples: Number.isInteger(input.performanceMinSamples) ? Math.max(1, Math.min(input.performanceMinSamples, 20)) : base.performanceMinSamples,
41
46
  }
42
47
  }
43
48
 
@@ -64,6 +69,7 @@ export async function writeModelPolicy(configDir, patch = {}) {
64
69
  tiers: { ...current.tiers, ...(patch.tiers || {}) },
65
70
  roleTiers: { ...current.roleTiers, ...(patch.roleTiers || {}) },
66
71
  capabilities: { ...(current.capabilities || {}), ...(patch.capabilities || {}) },
72
+ performance: patch.performance || current.performance || {},
67
73
  })
68
74
  const file = modelPolicyFile(configDir)
69
75
  await mkdir(path.dirname(file), { recursive: true })
@@ -94,3 +100,9 @@ export function applyConfiguredModel(source, role, policy) {
94
100
  }
95
101
  return lines.join("\n")
96
102
  }
103
+
104
+ export async function recordModelPerformance(configDir, outcome = {}) {
105
+ const current = await readModelPolicy(configDir)
106
+ const performance = recordPerformanceOutcome(current.performance || {}, outcome)
107
+ return writeModelPolicy(configDir, { performance })
108
+ }
@@ -0,0 +1,113 @@
1
+ export const MODEL_TASK_CLASSES = Object.freeze([
2
+ "general", "repo-scale", "debugging", "architecture", "security",
3
+ "migration", "frontend", "backend", "visual", "browser",
4
+ ])
5
+ const KNOWN_TASK_CLASSES = new Set(MODEL_TASK_CLASSES)
6
+
7
+ function boundedNumber(value, fallback = 0, min = 0, max = Number.MAX_SAFE_INTEGER) {
8
+ const number = Number(value)
9
+ if (!Number.isFinite(number)) return fallback
10
+ return Math.max(min, Math.min(max, number))
11
+ }
12
+
13
+ export function inferTaskClass(text = "", facts = {}) {
14
+ const explicit = String(facts.taskClass || "").trim().toLowerCase()
15
+ if (KNOWN_TASK_CLASSES.has(explicit)) return explicit
16
+ const value = String(text || "").toLowerCase()
17
+ if (/(whole repo|entire project|large monorepo|repo[- ]scale|cross[- ]module|toàn bộ dự án|nhiều module)/.test(value)) return "repo-scale"
18
+ if (/(prompt injection|security|auth|authorization|permission|secret|credential|bảo mật|phân quyền)/.test(value)) return "security"
19
+ if (/(migration|schema|database|sql|backfill|migrate)/.test(value)) return "migration"
20
+ if (/(screenshot|visual|figma|pixel|responsive|storybook)/.test(value)) return "visual"
21
+ if (/(browser|playwright|e2e|web page|click flow)/.test(value)) return "browser"
22
+ if (/(root cause|debug|regression|crash|failing|bug|lỗi)/.test(value)) return "debugging"
23
+ if (/(architecture|architect|design decision|system design|kiến trúc)/.test(value)) return "architecture"
24
+ if (/(react|next\.js|vue|svelte|css|frontend|ui\b)/.test(value)) return "frontend"
25
+ if (/(api|service|node|python|java|dotnet|backend|server)/.test(value)) return "backend"
26
+ return "general"
27
+ }
28
+
29
+ export function normalizePerformanceRecord(record = {}) {
30
+ record = record && typeof record === "object" ? record : {}
31
+ const samples = Math.floor(boundedNumber(record.samples, 0, 0))
32
+ const successes = Math.floor(boundedNumber(record.successes, Math.round(samples * boundedNumber(record.passRate, 0, 0, 1)), 0, samples))
33
+ return {
34
+ samples,
35
+ successes,
36
+ passRate: samples ? successes / samples : 0,
37
+ avgRetries: boundedNumber(record.avgRetries, 0, 0, 100),
38
+ avgTokens: boundedNumber(record.avgTokens, 0, 0),
39
+ avgLatencyMs: boundedNumber(record.avgLatencyMs, 0, 0),
40
+ updatedAt: typeof record.updatedAt === "string" ? record.updatedAt : null,
41
+ }
42
+ }
43
+
44
+ export function normalizePerformanceHistory(history = {}) {
45
+ const output = {}
46
+ for (const [model, classes] of Object.entries(history || {})) {
47
+ if (!model || !classes || typeof classes !== "object") continue
48
+ const normalizedClasses = {}
49
+ for (const [taskClass, record] of Object.entries(classes)) {
50
+ if (!KNOWN_TASK_CLASSES.has(taskClass) && taskClass !== "overall") continue
51
+ normalizedClasses[taskClass] = normalizePerformanceRecord(record)
52
+ }
53
+ if (Object.keys(normalizedClasses).length) output[model] = normalizedClasses
54
+ }
55
+ return output
56
+ }
57
+
58
+ function mergeAverage(previousAverage, previousSamples, value) {
59
+ return previousSamples <= 0 ? value : ((previousAverage * previousSamples) + value) / (previousSamples + 1)
60
+ }
61
+
62
+ export function recordPerformanceOutcome(history = {}, outcome = {}) {
63
+ const model = String(outcome.model || "").trim()
64
+ if (!model) throw new Error("model performance outcome requires model")
65
+ const taskClass = inferTaskClass(outcome.text || "", { taskClass: outcome.taskClass })
66
+ const normalized = normalizePerformanceHistory(history)
67
+ const current = normalizePerformanceRecord(normalized[model]?.[taskClass] || {})
68
+ const samples = current.samples
69
+ const passed = outcome.passed === true
70
+ const next = {
71
+ samples: samples + 1,
72
+ successes: current.successes + (passed ? 1 : 0),
73
+ passRate: 0,
74
+ avgRetries: mergeAverage(current.avgRetries, samples, boundedNumber(outcome.retries, 0, 0, 100)),
75
+ avgTokens: mergeAverage(current.avgTokens, samples, boundedNumber(outcome.tokens, 0, 0)),
76
+ avgLatencyMs: mergeAverage(current.avgLatencyMs, samples, boundedNumber(outcome.latencyMs, 0, 0)),
77
+ updatedAt: new Date().toISOString(),
78
+ }
79
+ next.passRate = next.successes / next.samples
80
+ return { ...normalized, [model]: { ...(normalized[model] || {}), [taskClass]: next } }
81
+ }
82
+
83
+ function performanceAdjustment(record, minSamples) {
84
+ const normalized = normalizePerformanceRecord(record)
85
+ if (normalized.samples <= 0) return { adjustment: 0, confidence: 0, record: normalized }
86
+ const confidence = Math.min(1, normalized.samples / Math.max(1, minSamples))
87
+ if (normalized.samples < minSamples) return { adjustment: 0, confidence, record: normalized }
88
+ const correctness = (normalized.passRate - 0.5) * 80
89
+ const retryPenalty = Math.min(20, normalized.avgRetries * 5)
90
+ const latencyPenalty = normalized.avgLatencyMs > 0 ? Math.min(10, Math.max(0, Math.log10(Math.max(1, normalized.avgLatencyMs / 1000)) * 3)) : 0
91
+ return { adjustment: (correctness - retryPenalty - latencyPenalty) * confidence, confidence, record: normalized }
92
+ }
93
+
94
+ export function rerankCapabilitySelection(selection = {}, history = {}, options = {}) {
95
+ const taskClass = inferTaskClass(options.text || "", { taskClass: options.taskClass })
96
+ const minSamples = Math.max(1, Number(options.minSamples || 3))
97
+ const normalized = normalizePerformanceHistory(history)
98
+ const candidates = (selection.candidates || []).map((candidate) => {
99
+ const record = normalized[candidate.id]?.[taskClass] || normalized[candidate.id]?.overall || null
100
+ const evidence = performanceAdjustment(record, minSamples)
101
+ return {
102
+ ...candidate,
103
+ baseScore: Number(candidate.score || 0),
104
+ empiricalTaskClass: taskClass,
105
+ empiricalEvidence: evidence.record,
106
+ empiricalConfidence: Number(evidence.confidence.toFixed(4)),
107
+ adjustedScore: Number((Number(candidate.score || 0) + evidence.adjustment).toFixed(6)),
108
+ }
109
+ })
110
+ const eligible = candidates.filter((candidate) => candidate.eligible)
111
+ .sort((a, b) => b.adjustedScore - a.adjustedScore || b.baseScore - a.baseScore)
112
+ return { ...selection, selected: eligible[0] || null, candidates, taskClass, empirical: true }
113
+ }
@@ -1,4 +1,5 @@
1
1
  import { inferTaskCapabilities, modelCandidatesFromPolicy, selectCapabilityCandidate } from "./capability-registry.mjs"
2
+ import { inferTaskClass, rerankCapabilitySelection } from "./model-performance.mjs"
2
3
 
3
4
  const ROLE_TIERS = {
4
5
  "codebase-mapper": "standard",
@@ -45,12 +46,14 @@ export function resolveModel(role, attempt, config = {}) {
45
46
 
46
47
  export function defaultModelPolicy() {
47
48
  return {
48
- schemaVersion: 2,
49
+ schemaVersion: 3,
49
50
  enabled: false,
50
51
  maxEscalations: 2,
51
52
  tiers: { light: null, standard: null, heavy: null },
52
53
  roleTiers: { ...ROLE_TIERS },
53
54
  capabilities: {},
55
+ performance: {},
56
+ performanceMinSamples: 3,
54
57
  }
55
58
  }
56
59
 
@@ -90,10 +93,14 @@ export function resolveCapabilityModel(role, attempt, taskText = "", taskPolicy
90
93
  })
91
94
  const candidates = modelCandidatesFromPolicy(config)
92
95
  .filter((candidate) => tierIndex(candidate.tier) >= tierIndex(base.tier))
93
- const selection = selectCapabilityCandidate(requirements, candidates, {
96
+ const staticSelection = selectCapabilityCandidate(requirements, candidates, {
94
97
  role,
95
98
  preferredTier: base.tier,
96
99
  })
100
+ const taskClass = inferTaskClass(taskText, facts)
101
+ const selection = rerankCapabilitySelection(staticSelection, config.performance || {}, {
102
+ taskClass, text: taskText, minSamples: config.performanceMinSamples || 3,
103
+ })
97
104
  const capabilityEnforced = config.enabled === true && candidates.length > 0
98
105
  if (capabilityEnforced && !selection.selected) {
99
106
  return {
@@ -0,0 +1,45 @@
1
+ import { mkdir, writeFile } from "node:fs/promises"
2
+ import path from "node:path"
3
+
4
+ function moduleSource(packageIndex, moduleIndex) {
5
+ const previous = moduleIndex > 0
6
+ ? 'import { value as previous } from "./mod' + String(moduleIndex - 1).padStart(3, "0") + '.mjs"\n'
7
+ : ""
8
+ return previous + "export const value = " + (moduleIndex > 0 ? "previous + 1" : packageIndex * 1000) +
9
+ "\nexport function compute(input) { return value + Number(input || 0) }\n"
10
+ }
11
+
12
+ export async function generateRepoScaleFixture(root, options = {}) {
13
+ const packageCount = Math.max(2, Number(options.packageCount || 6))
14
+ const modulesPerPackage = Math.max(5, Number(options.modulesPerPackage || 50))
15
+ await mkdir(root, { recursive: true })
16
+ await writeFile(path.join(root, "package.json"), JSON.stringify({
17
+ name: "ues-repo-scale-fixture", private: true, type: "module", workspaces: ["packages/*", "apps/*"],
18
+ }, null, 2) + "\n")
19
+
20
+ let generatedModules = 0
21
+ for (let p = 0; p < packageCount; p += 1) {
22
+ const pkgRoot = path.join(root, "packages", "pkg-" + p)
23
+ const src = path.join(pkgRoot, "src")
24
+ await mkdir(src, { recursive: true })
25
+ await writeFile(path.join(pkgRoot, "package.json"), JSON.stringify({
26
+ name: "@fixture/pkg-" + p, private: true, type: "module", exports: "./src/index.mjs",
27
+ }, null, 2) + "\n")
28
+ for (let m = 0; m < modulesPerPackage; m += 1) {
29
+ await writeFile(path.join(src, "mod" + String(m).padStart(3, "0") + ".mjs"), moduleSource(p, m))
30
+ generatedModules += 1
31
+ }
32
+ await writeFile(path.join(src, "index.mjs"),
33
+ 'export { value, compute } from "./mod' + String(modulesPerPackage - 1).padStart(3, "0") + '.mjs"\n')
34
+ }
35
+
36
+ const apiRoot = path.join(root, "apps", "api", "src")
37
+ const webRoot = path.join(root, "apps", "web", "src")
38
+ const contracts = path.join(root, "contracts")
39
+ await mkdir(apiRoot, { recursive: true }); await mkdir(webRoot, { recursive: true }); await mkdir(contracts, { recursive: true })
40
+ await writeFile(path.join(contracts, "public-api.json"), JSON.stringify({ version: 1, fields: ["id", "name", "status"] }, null, 2) + "\n")
41
+ await writeFile(path.join(apiRoot, "service.mjs"), 'export function serialize(row) { return { id: row.id, name: row.name, status: row.status } }\n')
42
+ await writeFile(path.join(webRoot, "consumer.mjs"), 'export function render(item) { return item.name + ":" + item.status }\n')
43
+ await writeFile(path.join(root, "AGENTS.md"), "# Fixture instructions\n\nPreserve public contracts and update focused tests.\n")
44
+ return { root, packageCount, modulesPerPackage, generatedModules, totalPrimaryFiles: generatedModules + packageCount + 5 }
45
+ }
@@ -1,9 +1,9 @@
1
1
  import { closeSync, existsSync, openSync, readFileSync, readSync, readdirSync, statSync } from "node:fs"
2
- import { mkdir, readFile, rename, rm, stat, writeFile } from "node:fs/promises"
2
+ import { mkdir, rename, rm, stat, writeFile } from "node:fs/promises"
3
3
  import { spawnSync } from "node:child_process"
4
4
  import { createHash, randomUUID } from "node:crypto"
5
5
  import path from "node:path"
6
- import { analyzePlan, taskFiles, validatePlan } from "./task-graph.mjs"
6
+ import { analyzePlan, taskFiles, taskVerificationCommands, validatePlan } from "./task-graph.mjs"
7
7
  import { buildContextManifest } from "./context-manifest.mjs"
8
8
  import { relevantAcceptedLearnings } from "./learning-engine.mjs"
9
9
  import { validateVerificationReceipt } from "./evidence-receipt.mjs"
@@ -15,6 +15,9 @@ import { putEvidence } from "./evidence-store.mjs"
15
15
  import { inferTaskCapabilities } from "./capability-registry.mjs"
16
16
  import { buildPromptEnvelope } from "./prompt-cache.mjs"
17
17
  import { externalizeContextExcerpts } from "./context-engine-v11.mjs"
18
+ import { assertActivePlanScope, persistPlanScope } from "./work-plan-scope.mjs"
19
+ import { classifyDecisionPolicy } from "./decision-policy.mjs"
20
+ import { readTextAuto } from "./text-encoding.mjs"
18
21
 
19
22
  const WORK_DIR = ".ues-work"
20
23
  const STATE_SCHEMA = 4
@@ -49,13 +52,15 @@ export function workPaths(root, slug) {
49
52
  events: path.join(dir, "EVENTS.jsonl"),
50
53
  tasks: path.join(dir, "tasks"),
51
54
  reports: path.join(dir, "reports"),
55
+ plans: path.join(dir, "plans"),
56
+ activePlan: path.join(dir, "ACTIVE_PLAN.json"),
52
57
  lock: path.join(dir, ".state-lock"),
53
58
  }
54
59
  }
55
60
 
56
61
  async function readJson(file, fallback = null) {
57
62
  try {
58
- return JSON.parse(await readFile(file, "utf8"))
63
+ return JSON.parse(readTextAuto(file))
59
64
  } catch (error) {
60
65
  if (error?.code === "ENOENT") return fallback
61
66
  throw error
@@ -268,6 +273,7 @@ export async function initWork(root, slug, goal) {
268
273
 
269
274
  await mkdir(paths.tasks, { recursive: true })
270
275
  await mkdir(paths.reports, { recursive: true })
276
+ await mkdir(paths.plans, { recursive: true })
271
277
  const createdAt = now()
272
278
  const cleanGoal = String(goal || "").trim() || "Define the requested engineering outcome."
273
279
 
@@ -285,6 +291,7 @@ export async function initWork(root, slug, goal) {
285
291
  planImportedAt: null,
286
292
  planHash: null,
287
293
  planApproval: null,
294
+ planScope: null,
288
295
  integrationVerification: null,
289
296
  tasks: {},
290
297
  decisions: [],
@@ -334,7 +341,7 @@ function currentEvidencePolicy(state, plan) {
334
341
 
335
342
  function taskBrief(task) {
336
343
  const files = taskFiles(task)
337
- return `# ${task.id}: ${task.title}\n\n## Summary\n\n${task.summary || "No summary provided."}\n\n## Dependencies\n\n${(task.dependsOn || []).map((id) => `- ${id}`).join("\n") || "- None"}\n\n## Files\n\n${files.map((file) => `- ${file}`).join("\n") || "- Scope must be established before editing."}\n\n## Acceptance criteria\n\n${(task.acceptance || []).map((item) => `- ${item}`).join("\n")}\n\n## Verification\n\n${(task.verification || []).map((item) => `- ${item}`).join("\n")}\n\n## Risk\n\n${task.risk || "medium"}\n`
344
+ return `# ${task.id}: ${task.title}\n\n## Summary\n\n${task.summary || "No summary provided."}\n\n## Dependencies\n\n${(task.dependsOn || []).map((id) => `- ${id}`).join("\n") || "- None"}\n\n## Files\n\n${files.map((file) => `- ${file}`).join("\n") || "- Scope must be established before editing."}\n\n## Acceptance criteria\n\n${(task.acceptance || []).map((item) => `- ${item}`).join("\n")}\n\n## Verification\n\n${(task.verification || []).map((item) => `- ${item}`).join("\n")}\n\n## Structured verification commands\n\n${taskVerificationCommands(task).map((item) => `- ${[item.command, ...item.args].join(" ")}`).join("\n") || "- None declared"}\n\n## Risk\n\n${task.risk || "medium"}\n`
338
345
  }
339
346
 
340
347
  function planIsApproved(state, plan) {
@@ -345,7 +352,7 @@ function planIsApproved(state, plan) {
345
352
 
346
353
  export async function importPlan(root, slug, planInput) {
347
354
  const plan = typeof planInput === "string"
348
- ? JSON.parse(await readFile(path.resolve(planInput), "utf8"))
355
+ ? JSON.parse(readTextAuto(path.resolve(planInput)))
349
356
  : planInput
350
357
 
351
358
  const validation = validatePlan(plan)
@@ -383,6 +390,7 @@ export async function importPlan(root, slug, planInput) {
383
390
  }
384
391
 
385
392
  const updatedAt = now()
393
+ const planScope = await persistPlanScope(loaded.paths, hash, plan, { importedAt: updatedAt, previousPlanHash: loaded.state.planHash || null })
386
394
  const state = {
387
395
  ...loaded.state,
388
396
  schemaVersion: STATE_SCHEMA,
@@ -392,6 +400,7 @@ export async function importPlan(root, slug, planInput) {
392
400
  planImportedAt: updatedAt,
393
401
  planHash: hash,
394
402
  planApproval: { status: "pending", planHash: hash, at: updatedAt, evidence: null },
403
+ planScope,
395
404
  integrationVerification: null,
396
405
  checkpoint: null,
397
406
  evidencePolicy: evidencePolicyForPlan(plan),
@@ -414,6 +423,7 @@ export async function approvePlan(root, slug, evidence, options = {}) {
414
423
  const loaded = await loadWork(root, slug)
415
424
  if (!loaded.plan) throw new Error("PLAN.json is missing")
416
425
  const hash = planHash(loaded.plan)
426
+ await assertActivePlanScope(loaded.paths, loaded.state.planHash || hash)
417
427
  if (loaded.state.planHash && loaded.state.planHash !== hash) {
418
428
  throw new Error("PLAN.json changed after import; re-import it before approval")
419
429
  }
@@ -525,7 +535,9 @@ export async function workStatus(root, slug) {
525
535
  leaseExpiresAt: value.leaseExpiresAt || null,
526
536
  }))
527
537
  const taskEntries = Object.values(loaded.state.tasks || {})
528
- const receiptBacked = taskEntries.filter((value) => value.evidenceStrength === "receipt-backed").length
538
+ const receiptBacked = taskEntries.filter((value) =>
539
+ ["receipt-backed", "command-receipt-backed", "independent-agent-receipt-backed"].includes(value.evidenceStrength)
540
+ ).length
529
541
  const completed = taskEntries.filter((value) => value.status === "completed").length
530
542
 
531
543
  return {
@@ -540,6 +552,7 @@ export async function workStatus(root, slug) {
540
552
  blockers: loaded.state.blockers || [],
541
553
  planApproval: loaded.state.planApproval || null,
542
554
  integrationVerification: loaded.state.integrationVerification || null,
555
+ planScope: loaded.state.planScope || null,
543
556
  evidence: {
544
557
  receiptBacked,
545
558
  completed,
@@ -557,6 +570,7 @@ export async function startTask(root, slug, taskID, options = {}) {
557
570
  const result = await withWorkLock(root, slug, async () => {
558
571
  const loaded = await loadWork(root, slug)
559
572
  if (!loaded.plan) throw new Error("PLAN.json is missing; import a valid plan first")
573
+ await assertActivePlanScope(loaded.paths, loaded.state.planHash || planHash(loaded.plan))
560
574
  if (!planIsApproved(loaded.state, loaded.plan)) {
561
575
  throw new Error("plan is not approved; run ues-plan-checker and 'ocskill work approve-plan' first")
562
576
  }
@@ -812,7 +826,13 @@ export async function completeTask(root, slug, taskID, options = {}) {
812
826
  record.status = "completed"
813
827
  record.completedAt = timestamp
814
828
  record.report = path.relative(loaded.paths.root, reportPath).replaceAll("\\", "/")
815
- record.evidenceStrength = acceptedReceipts.length ? "receipt-backed" : "narrative"
829
+ const commandReceipts = acceptedReceipts.filter((item) => item.kind !== "agent-verifier")
830
+ const agentVerifierReceipts = acceptedReceipts.filter((item) => item.kind === "agent-verifier")
831
+ record.evidenceStrength = commandReceipts.length
832
+ ? "command-receipt-backed"
833
+ : agentVerifierReceipts.length
834
+ ? "independent-agent-receipt-backed"
835
+ : "narrative"
816
836
  record.runId = null
817
837
  record.owner = null
818
838
  record.heartbeatAt = null
@@ -888,16 +908,19 @@ export async function failTask(root, slug, taskID, reason, options = {}) {
888
908
  }
889
909
 
890
910
  export async function addDecision(root, slug, decision) {
891
- const text = String(decision || "").trim()
911
+ const input = decision && typeof decision === "object" ? decision : { text: decision }
912
+ const text = String(input.text || "").trim()
892
913
  if (!text) throw new Error("decision text is required")
893
-
914
+ const policy = classifyDecisionPolicy(text, input.facts || input)
915
+ if (input.auto === true && !policy.autoResolvable) throw new Error("decision requires human approval before automatic resolution")
894
916
  return withWorkLock(root, slug, async () => {
895
917
  const loaded = await loadWork(root, slug)
896
918
  loaded.state.decisions ??= []
897
919
  const timestamp = now()
898
- loaded.state.decisions.push({ at: timestamp, text })
920
+ loaded.state.decisions.push({ at: timestamp, text, policy, source: input.source || (input.auto === true ? "auto-ruling" : "human-or-agent") })
899
921
  loaded.state.updatedAt = timestamp
900
922
  await writeJson(loaded.paths.state, loaded.state)
923
+ await journal(loaded.paths, "decision.recorded", { text, policy })
901
924
  return loaded.state
902
925
  })
903
926
  }
@@ -1071,14 +1094,15 @@ export async function finalizeWork(root, slug, evidence) {
1071
1094
  async function buildContextPack(loaded, taskID) {
1072
1095
  if (!loaded.plan) throw new Error("PLAN.json is missing")
1073
1096
  const task = taskByID(loaded.plan, taskID)
1074
- const spec = await readFile(loaded.paths.spec, "utf8").catch(() => "")
1097
+ let spec = ""
1098
+ try { spec = readTextAuto(loaded.paths.spec) } catch {}
1075
1099
  const dependencyReports = {}
1076
1100
  const dependencyReportRefs = {}
1077
1101
 
1078
1102
  for (const dep of task.dependsOn || []) {
1079
1103
  const file = path.join(loaded.paths.reports, `${dep}.md`)
1080
1104
  if (!existsSync(file)) continue
1081
- const report = await readFile(file, "utf8")
1105
+ const report = readTextAuto(file)
1082
1106
  dependencyReports[dep] = report.slice(0, 6000)
1083
1107
  const stored = await putEvidence(loaded.paths.root, report, {
1084
1108
  kind: "dependency-report",
@@ -48,6 +48,20 @@ export function taskReadFiles(task) {
48
48
  return [...new Set((files.read || []).map(normalizeFile).filter(Boolean))].sort()
49
49
  }
50
50
 
51
+
52
+ export function taskVerificationCommands(task) {
53
+ if (!Array.isArray(task?.verificationCommands)) return []
54
+ return task.verificationCommands
55
+ .map((item) => {
56
+ if (!item || typeof item !== "object" || Array.isArray(item)) return null
57
+ const command = String(item.command || "").trim()
58
+ if (!command) return null
59
+ const args = Array.isArray(item.args) ? item.args.map((value) => String(value)) : []
60
+ return { command, args }
61
+ })
62
+ .filter(Boolean)
63
+ }
64
+
51
65
  function taskMap(plan) {
52
66
  return new Map((plan?.tasks || []).map((task) => [task.id, task]))
53
67
  }
@@ -97,6 +111,24 @@ export function validatePlan(plan) {
97
111
  errors.push(`${id || prefix}: verification must contain at least one concrete check`)
98
112
  }
99
113
 
114
+ if (task.verificationCommands !== undefined) {
115
+ if (!Array.isArray(task.verificationCommands)) {
116
+ errors.push(`${id || prefix}.verificationCommands must be an array when provided`)
117
+ } else {
118
+ for (const [commandIndex, commandSpec] of task.verificationCommands.entries()) {
119
+ const commandPrefix = `${id || prefix}.verificationCommands[${commandIndex}]`
120
+ if (!commandSpec || typeof commandSpec !== "object" || Array.isArray(commandSpec)) {
121
+ errors.push(`${commandPrefix} must be an object`)
122
+ continue
123
+ }
124
+ if (!String(commandSpec.command || "").trim()) errors.push(`${commandPrefix}.command is required`)
125
+ if (commandSpec.args !== undefined && !Array.isArray(commandSpec.args)) {
126
+ errors.push(`${commandPrefix}.args must be an array when provided`)
127
+ }
128
+ }
129
+ }
130
+ }
131
+
100
132
  const risk = task.risk ?? "medium"
101
133
  if (!VALID_RISKS.has(risk)) errors.push(`${id || prefix}: invalid risk '${risk}'`)
102
134