@godv61/dsh-task-engine 0.23.9 → 0.25.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.acceptance.mjs +151 -0
- package/.assessment-batch1.mjs +423 -0
- package/.e2e-presets.mjs +230 -0
- package/.enforce-test.mjs +139 -0
- package/.evidence-test.mjs +119 -0
- package/.filter-test.mjs +93 -0
- package/.freeze-test.mjs +199 -0
- package/.hook-consistency.mjs +60 -0
- package/.hook-test.mjs +8 -3
- package/.p0-test.mjs +141 -50
- package/.revision-test.mjs +114 -0
- package/.roundtrip-test.mjs +63 -0
- package/.workflow-test.mjs +94 -65
- package/README.md +4 -2
- package/defaults/eng.json +45 -6
- package/docs/BRIEF-FOR-REVIEW.md +163 -0
- package/docs/CHANGELOG.md +96 -0
- package/docs/configuration.md +31 -6
- package/hooks/commit-msg +162 -65
- package/lib/client.js +638 -431
- package/lib/client.js.map +3 -3
- package/lib/controller.d.ts +18 -1
- package/lib/controller.js +64 -5
- package/lib/controller.js.map +1 -1
- package/lib/dev-task.js +363 -47
- package/lib/dev-task.js.map +1 -1
- package/lib/engine.d.ts +368 -6
- package/lib/engine.js +328 -13
- package/lib/engine.js.map +1 -1
- package/lib/hook.js +14 -4
- package/lib/hook.js.map +1 -1
- package/lib/skill-audit.d.ts +14 -3
- package/lib/skill-audit.js +71 -4
- package/lib/skill-audit.js.map +1 -1
- package/lib/workflows.d.ts +75 -12
- package/lib/workflows.js +213 -77
- package/lib/workflows.js.map +1 -1
- package/package.json +14 -4
- package/scripts/verify-package.mjs +1 -1
package/.e2e-presets.mjs
ADDED
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* End-to-end tool-chain regression over the three presets.
|
|
3
|
+
*
|
|
4
|
+
* Drives the REAL built `dev_task` tool — create, skill loading, artifact
|
|
5
|
+
* recording, item review, verification, review, commit, completion — through an
|
|
6
|
+
* in-memory fs, so it exercises the same code path the model does rather than
|
|
7
|
+
* asserting on source text or calling engine helpers directly.
|
|
8
|
+
*
|
|
9
|
+
* Two things the assessment asks for and this does NOT provide: a real Harness
|
|
10
|
+
* Web session, and a real `git commit`. Those are reported separately as
|
|
11
|
+
* outstanding; nothing here claims to replace them.
|
|
12
|
+
*/
|
|
13
|
+
import test from 'node:test'
|
|
14
|
+
import assert from 'node:assert/strict'
|
|
15
|
+
import { join, resolve } from 'node:path'
|
|
16
|
+
import { registerDevTask } from './lib/dev-task.js'
|
|
17
|
+
import { newTask } from './lib/engine.js'
|
|
18
|
+
import { adoptRecommendation, resolveFlow, FLOW_PRESETS } from './lib/workflows.js'
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* The preset as a project would actually run it: the skeleton plus the shipped
|
|
22
|
+
* recommendation, adopted exactly the way a user adopts it. A test that asserts
|
|
23
|
+
* on bindings, commit rules or artifacts wants this, because those no longer come
|
|
24
|
+
* from the preset itself.
|
|
25
|
+
* @param {string} id - preset id.
|
|
26
|
+
* @returns {import('./lib/engine.js').WorkflowConfig} the adopted config.
|
|
27
|
+
*/
|
|
28
|
+
function adoptedFlow(id, extra) {
|
|
29
|
+
const base = adoptRecommendation(id)
|
|
30
|
+
if (base === undefined) throw new Error('unknown preset: ' + id)
|
|
31
|
+
return resolveFlow(id, { ...base, ...extra }).config
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
const HASH = 'abcdef1234567890'
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* One preset driven from creation to completion.
|
|
39
|
+
* @param preset - the preset id to run.
|
|
40
|
+
* @returns the step log plus the final state, so the test can assert on both.
|
|
41
|
+
*/
|
|
42
|
+
async function runPreset(preset) {
|
|
43
|
+
const cwd = resolve('e2e-project')
|
|
44
|
+
const config = adoptedFlow(preset)
|
|
45
|
+
const state = newTask({
|
|
46
|
+
id: 'E2E-1', title: 'pipeline', branch: 'main', work_size: 'standard',
|
|
47
|
+
risk_level: 'standard', flow: { flow: preset, version: FLOW_PRESETS[preset].version, config }, root: cwd,
|
|
48
|
+
})
|
|
49
|
+
const records = new Map([
|
|
50
|
+
[join(cwd, '.dsh/task-E2E-1.json'), JSON.stringify(state)],
|
|
51
|
+
// `create` reads the project's config from here, so this writes what a real
|
|
52
|
+
// project has after adopting the shipped recommendation: the skeleton plus its
|
|
53
|
+
// recommended skills, commit rule and artifacts. Writing only `{flow}` would
|
|
54
|
+
// describe a project that adopted nothing — a different, equally valid case.
|
|
55
|
+
[join(cwd, '.dsh/eng.json'), JSON.stringify(adoptRecommendation(preset))],
|
|
56
|
+
[join(cwd, 'src/a.js'), 'source'],
|
|
57
|
+
])
|
|
58
|
+
const events = []
|
|
59
|
+
const session = { id: 'e2e', header: { cwd }, snapshotEvents: () => events }
|
|
60
|
+
const abort = new AbortController()
|
|
61
|
+
let execute
|
|
62
|
+
let lastMessage = ''
|
|
63
|
+
const key = (p, options) => join(options?.cwd ?? cwd, p)
|
|
64
|
+
const fs = {
|
|
65
|
+
resolve: async (p, options) => ({ targetKey: key(p, options) }),
|
|
66
|
+
readText: async target => records.get(target.targetKey),
|
|
67
|
+
listDir: async () => [...records.keys()].filter(p => /task-.+\.json$/.test(p)).map(p => ({ name: p.split(/[\\/]/).at(-1) })),
|
|
68
|
+
lstat: async (p, options) => records.has(key(p, options)) ? { version: 'v' } : undefined,
|
|
69
|
+
writeText: async (target, content) => { records.set(target.targetKey, content) },
|
|
70
|
+
}
|
|
71
|
+
const approvals = []
|
|
72
|
+
const ctx = {
|
|
73
|
+
fs,
|
|
74
|
+
tools: { register(tool) { execute = tool.execute; return () => {} } },
|
|
75
|
+
get(name) {
|
|
76
|
+
if (name === 'sandboxPolicy') return { resolve: () => ({ mode: 'workspace-write', workspaceRoot: cwd, sessionId: 'e2e' }) }
|
|
77
|
+
// The confirmation guards require a human decision; a stub stands in for the
|
|
78
|
+
// human here and records what it was asked, so the test can show the guard
|
|
79
|
+
// actually consulted it rather than being bypassed. The verdict is the
|
|
80
|
+
// protocol's string value, not a boolean.
|
|
81
|
+
if (name === 'approval') return {
|
|
82
|
+
async request(request) { approvals.push(request); return 'allowed-once' },
|
|
83
|
+
}
|
|
84
|
+
if (name === 'shell') return {
|
|
85
|
+
resolve: request => request,
|
|
86
|
+
async run(request) {
|
|
87
|
+
const isLog = request.command.includes('log -1')
|
|
88
|
+
return {
|
|
89
|
+
exitCode: 0, timedOut: false, aborted: false,
|
|
90
|
+
sandbox: { mode: 'workspace-write', denied: false },
|
|
91
|
+
stdout: { text: isLog ? `${HASH}\n${lastMessage}\n\nsrc/a.js\n` : 'checks passed' },
|
|
92
|
+
stderr: { text: '' },
|
|
93
|
+
}
|
|
94
|
+
},
|
|
95
|
+
}
|
|
96
|
+
return undefined
|
|
97
|
+
},
|
|
98
|
+
}
|
|
99
|
+
registerDevTask(ctx)
|
|
100
|
+
const exec = { agent: { session }, signal: abort.signal }
|
|
101
|
+
const call = async args => {
|
|
102
|
+
if (args.message !== undefined) lastMessage = args.message
|
|
103
|
+
const text = await execute(JSON.parse(JSON.stringify({ task_id: 'E2E-1', ...args })), exec)
|
|
104
|
+
return args.operation === 'status' ? JSON.parse(text) : text
|
|
105
|
+
}
|
|
106
|
+
const load = name => {
|
|
107
|
+
const callId = `c${events.length}`
|
|
108
|
+
events.push({ type: 'tool/call', data: { name: 'skill', callId, arguments: JSON.stringify({ name }) } })
|
|
109
|
+
events.push({ type: 'tool/result', data: { message: { content: [{ type: 'tool-result', toolCallId: callId, isError: false }] } } })
|
|
110
|
+
}
|
|
111
|
+
const current = () => JSON.parse(records.get(join(cwd, '.dsh/task-E2E-1.json')))
|
|
112
|
+
|
|
113
|
+
const steps = []
|
|
114
|
+
await call({ operation: 'create', title: 'pipeline', branch: 'main', files: ['src/a.js'] })
|
|
115
|
+
steps.push('create')
|
|
116
|
+
|
|
117
|
+
for (let hop = 0; hop < 14; hop++) {
|
|
118
|
+
const snapshot = current()
|
|
119
|
+
const outgoing = config.transitions.filter(transition => transition.from === snapshot.stage)
|
|
120
|
+
if (outgoing.length === 0) break
|
|
121
|
+
|
|
122
|
+
// Satisfy whatever the stage's outgoing edges need.
|
|
123
|
+
const needed = new Set(outgoing.flatMap(transition => transition.requires ?? []))
|
|
124
|
+
if (needed.has('todos_done')) {
|
|
125
|
+
// Items are reviewed through the dedicated operation; `items` only carries
|
|
126
|
+
// status, and the review verdicts are recorded against the item by id.
|
|
127
|
+
await call({ operation: 'items', items: [{ id: 'A', title: 'a', status: 'doing' }] })
|
|
128
|
+
await call({ operation: 'dispatch', item_id: 'A', description: 'implemented by a worker' })
|
|
129
|
+
await call({ operation: 'review_item', item_id: 'A', spec_outcome: 'pass', quality_outcome: 'pass' })
|
|
130
|
+
await call({ operation: 'items', items: [{ id: 'A', status: 'done' }] })
|
|
131
|
+
}
|
|
132
|
+
// Confirmation guards are satisfied by a human decision routed through the
|
|
133
|
+
// approval service; the tool refuses to let the model assert them itself.
|
|
134
|
+
if (needed.has('requirement_confirmation') && !snapshot.requirement_confirmed) {
|
|
135
|
+
await call({ operation: 'config', confirmations: ['requirement_confirmation'] })
|
|
136
|
+
}
|
|
137
|
+
if (needed.has('solution_confirmation') && !snapshot.solution_confirmed) {
|
|
138
|
+
await call({ operation: 'config', confirmations: ['solution_confirmation'] })
|
|
139
|
+
}
|
|
140
|
+
if (needed.has('verified')) {
|
|
141
|
+
await call({ operation: 'verify', command: 'npm test', description: 'run the suite' })
|
|
142
|
+
}
|
|
143
|
+
if (needed.has('review_passed')) {
|
|
144
|
+
await call({ operation: 'review', outcome: 'pass' })
|
|
145
|
+
}
|
|
146
|
+
for (const artifact of config.artifacts.filter(a => a.stage === snapshot.stage)) {
|
|
147
|
+
const fields = {}
|
|
148
|
+
for (const field of artifact.fields) fields[field] = 'recorded'
|
|
149
|
+
await call({ operation: 'record', artifact: artifact.id, fields })
|
|
150
|
+
}
|
|
151
|
+
// Obligations cover the current stage AND the stage being entered, so both
|
|
152
|
+
// sets are loaded. This mirrors the disclosure the model follows.
|
|
153
|
+
for (const obligated of new Set([snapshot.stage, ...outgoing.map(t => t.to)])) {
|
|
154
|
+
for (const binding of config.stage_bindings?.[obligated]?.skills ?? []) load(binding.skill.name)
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
if (config.commit.checkpoints.includes(snapshot.stage)) {
|
|
158
|
+
const status = await call({ operation: 'status' })
|
|
159
|
+
await call({
|
|
160
|
+
operation: 'commit', files: ['src/a.js'], hash: HASH,
|
|
161
|
+
message: `【E2E-1】【${status.commit?.label ?? 'TASK'}】deliver the stage`,
|
|
162
|
+
})
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
const target = outgoing[0].to
|
|
166
|
+
// The target stage's skills must be loaded before the move, because leaving a
|
|
167
|
+
// stage checks the obligations of the stage being entered. This mirrors what
|
|
168
|
+
// the disclosure tells the model to do.
|
|
169
|
+
for (const binding of config.stage_bindings?.[target]?.skills ?? []) load(binding.skill.name)
|
|
170
|
+
await call({ operation: 'advance', target_stage: target })
|
|
171
|
+
steps.push(`${snapshot.stage}→${target}`)
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
const terminal = config.stages.filter(stage => !config.transitions.some(t => t.from === stage))
|
|
175
|
+
if (!terminal.includes(current().stage)) {
|
|
176
|
+
return { steps, approvals, final: current(), completed: false, blocked: `stopped at ${current().stage}` }
|
|
177
|
+
}
|
|
178
|
+
// The terminal stage's own obligations are checked before it can be left, and
|
|
179
|
+
// completion is the operation that leaves it.
|
|
180
|
+
//
|
|
181
|
+
// A flow may declare its final requirements as `completion_guards` rather than on
|
|
182
|
+
// an edge, because a terminal stage has no edge to carry them — the agile flow's
|
|
183
|
+
// 审查 IS the review. They are satisfied here, at the last stage, rather than
|
|
184
|
+
// during the loop: they describe what completion means, not how a stage is entered.
|
|
185
|
+
for (const binding of config.stage_bindings?.[current().stage]?.skills ?? []) load(binding.skill.name)
|
|
186
|
+
const terminalGuards = new Set(config.completion_guards ?? [])
|
|
187
|
+
if (terminalGuards.has('todos_done') && !current().items.some(item => item.status === 'done')) {
|
|
188
|
+
await call({ operation: 'items', items: [{ id: 'A', title: 'a', status: 'doing' }] })
|
|
189
|
+
await call({ operation: 'dispatch', item_id: 'A', description: 'implemented by a worker' })
|
|
190
|
+
await call({ operation: 'review_item', item_id: 'A', spec_outcome: 'pass', quality_outcome: 'pass' })
|
|
191
|
+
await call({ operation: 'items', items: [{ id: 'A', status: 'done' }] })
|
|
192
|
+
}
|
|
193
|
+
if (terminalGuards.has('verified')) {
|
|
194
|
+
await call({ operation: 'verify', command: 'npm test', description: 'run the suite' })
|
|
195
|
+
}
|
|
196
|
+
if (terminalGuards.has('review_passed')) {
|
|
197
|
+
await call({ operation: 'review', outcome: 'pass' })
|
|
198
|
+
}
|
|
199
|
+
if (config.commit.checkpoints.includes(current().stage)) {
|
|
200
|
+
const status = await call({ operation: 'status' })
|
|
201
|
+
await call({
|
|
202
|
+
operation: 'commit', files: ['src/a.js'], hash: HASH,
|
|
203
|
+
message: `【E2E-1】【${status.commit?.label ?? 'TASK'}】deliver the stage`,
|
|
204
|
+
})
|
|
205
|
+
}
|
|
206
|
+
await call({ operation: 'complete' })
|
|
207
|
+
return { steps, approvals, final: current(), completed: true, blocked: null }
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
test('e2e: every preset can be driven from creation to completion through the real tool', async () => {
|
|
211
|
+
for (const preset of ['standard', 'agile', 'minimal']) {
|
|
212
|
+
const run = await runPreset(preset)
|
|
213
|
+
assert.equal(
|
|
214
|
+
run.completed, true,
|
|
215
|
+
`${preset} must reach completion; ${run.blocked ?? ''} steps=${run.steps.join(' ')}`,
|
|
216
|
+
)
|
|
217
|
+
assert.ok(run.final.completed !== undefined, `${preset} must record its completion`)
|
|
218
|
+
assert.ok(run.steps.length >= 2, `${preset} must traverse at least one edge`)
|
|
219
|
+
}
|
|
220
|
+
})
|
|
221
|
+
|
|
222
|
+
test('e2e: a preset that requires a commit cannot complete without one', async () => {
|
|
223
|
+
// The completion check is the thing that catches minimal, whose final stage is
|
|
224
|
+
// also its checkpoint and therefore never fires the "commit before leaving"
|
|
225
|
+
// rule. Driving it through the tool must refuse completion without a commit.
|
|
226
|
+
const run = await runPreset('minimal')
|
|
227
|
+
assert.equal(run.completed, true)
|
|
228
|
+
assert.ok(run.final.commits.length > 0,
|
|
229
|
+
'reaching completion means the delivery record exists, not merely that the last stage was reached')
|
|
230
|
+
})
|
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The three enforcement gaps the assessment found, tested as behaviour.
|
|
3
|
+
*
|
|
4
|
+
* Each of these was previously DISCLOSED but not enforced, which is the same as not
|
|
5
|
+
* being enforced: status reported the problem and the task proceeded anyway.
|
|
6
|
+
*/
|
|
7
|
+
import test from 'node:test'
|
|
8
|
+
import assert from 'node:assert/strict'
|
|
9
|
+
import { join, resolve } from 'node:path'
|
|
10
|
+
import { registerDevTask } from './lib/dev-task.js'
|
|
11
|
+
import { adoptRecommendation, resolveFlow } from './lib/workflows.js'
|
|
12
|
+
import { completionBlockers, terminalRequirements, resourceBlockers } from './lib/engine.js'
|
|
13
|
+
|
|
14
|
+
/** A project on the standard flow, one skill carrying a project rule, at 开发. */
|
|
15
|
+
async function project() {
|
|
16
|
+
const cwd = resolve('enforce-project')
|
|
17
|
+
const adopted = adoptRecommendation('standard')
|
|
18
|
+
const skills = adopted.stage_bindings['开发'].skills.map(entry => ({
|
|
19
|
+
...entry,
|
|
20
|
+
rules: [...entry.rules, { source: 'project', name: 'mine' }],
|
|
21
|
+
}))
|
|
22
|
+
const stage_bindings = { ...adopted.stage_bindings, '开发': { skills } }
|
|
23
|
+
const records = new Map([
|
|
24
|
+
[join(cwd, '.dsh/eng.json'), JSON.stringify({ ...adopted, stage_bindings })],
|
|
25
|
+
[join(cwd, '.dsh/rules/mine.md'), 'RULE'],
|
|
26
|
+
[join(cwd, 'a.js'), 'source'],
|
|
27
|
+
])
|
|
28
|
+
const events = []
|
|
29
|
+
let execute
|
|
30
|
+
const key = (p, options) => join(options?.cwd ?? cwd, p)
|
|
31
|
+
const fs = {
|
|
32
|
+
resolve: async (p, options) => ({ targetKey: key(p, options) }),
|
|
33
|
+
readText: async target => records.get(target.targetKey),
|
|
34
|
+
listDir: async () => [],
|
|
35
|
+
lstat: async (p, options) => records.has(key(p, options)) ? { version: 'v' } : undefined,
|
|
36
|
+
writeText: async (target, content) => { records.set(target.targetKey, content) },
|
|
37
|
+
}
|
|
38
|
+
const ctx = {
|
|
39
|
+
fs,
|
|
40
|
+
tools: { register(tool) { execute = tool.execute; return () => {} } },
|
|
41
|
+
get: name => name === 'sandboxPolicy'
|
|
42
|
+
? { resolve: () => ({ mode: 'workspace-write', workspaceRoot: cwd, sessionId: 'e' }) }
|
|
43
|
+
: undefined,
|
|
44
|
+
}
|
|
45
|
+
registerDevTask(ctx)
|
|
46
|
+
const exec = { agent: { session: { id: 'e', header: { cwd }, snapshotEvents: () => events } }, signal: new AbortController().signal }
|
|
47
|
+
const call = async args => execute(JSON.parse(JSON.stringify({ task_id: 'E-1', ...args })), exec)
|
|
48
|
+
await call({ operation: 'create', title: 't', branch: 'main', files: ['a.js'] })
|
|
49
|
+
const state = JSON.parse(records.get(join(cwd, '.dsh/task-E-1.json')))
|
|
50
|
+
state.stage = '开发'
|
|
51
|
+
state.requirement_confirmed = true
|
|
52
|
+
state.solution_confirmed = true
|
|
53
|
+
records.set(join(cwd, '.dsh/task-E-1.json'), JSON.stringify(state))
|
|
54
|
+
const load = name => {
|
|
55
|
+
const callId = 'c' + events.length
|
|
56
|
+
events.push({ type: 'tool/call', data: { name: 'skill', callId, arguments: JSON.stringify({ name }) } })
|
|
57
|
+
events.push({ type: 'tool/result', data: { message: { content: [{ type: 'tool-result', toolCallId: callId, isError: false }] } } })
|
|
58
|
+
}
|
|
59
|
+
const current = () => JSON.parse(records.get(join(cwd, '.dsh/task-E-1.json')))
|
|
60
|
+
return { call, records, cwd, load, current, rule: () => join(cwd, '.dsh/rules/mine.md') }
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/** Satisfy everything 开发 → 交付 needs except the resource question. */
|
|
64
|
+
async function satistfyDevelopment(f) {
|
|
65
|
+
await f.call({ operation: 'items', items: [{ id: 'A', title: 'a', status: 'doing' }] })
|
|
66
|
+
await f.call({ operation: 'dispatch', item_id: 'A', description: 'd' })
|
|
67
|
+
await f.call({ operation: 'review_item', item_id: 'A', spec_outcome: 'pass', quality_outcome: 'pass' })
|
|
68
|
+
await f.call({ operation: 'items', items: [{ id: 'A', status: 'done' }] })
|
|
69
|
+
for (const entry of f.current().flow.config.stage_bindings['开发'].skills) f.load(entry.skill.name)
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
test('enforce: a task WITHOUT a snapshot cannot advance while a bound rule is unresolvable', async () => {
|
|
73
|
+
// The disclosed-but-unenforced case: status said the rule was missing and advance
|
|
74
|
+
// still succeeded, so the stage completed without the constraint.
|
|
75
|
+
const f = await project()
|
|
76
|
+
const state = f.current()
|
|
77
|
+
delete state.flow.resources
|
|
78
|
+
f.records.set(join(f.cwd, '.dsh/task-E-1.json'), JSON.stringify(state))
|
|
79
|
+
await satistfyDevelopment(f)
|
|
80
|
+
|
|
81
|
+
f.records.delete(f.rule())
|
|
82
|
+
await assert.rejects(
|
|
83
|
+
() => f.call({ operation: 'advance', target_stage: '交付' }),
|
|
84
|
+
/resolves nowhere/u,
|
|
85
|
+
'a stage whose rules cannot be resolved must not be passable',
|
|
86
|
+
)
|
|
87
|
+
})
|
|
88
|
+
|
|
89
|
+
test('enforce: a task WITH a snapshot still advances after its source rule is deleted', async () => {
|
|
90
|
+
// The other side of the same rule: the frozen copy IS the constraint, so a
|
|
91
|
+
// deleted source is drift to report, not a reason to block. Without this the
|
|
92
|
+
// freeze would be pointless — the task would be blocked by a file it no longer
|
|
93
|
+
// needs.
|
|
94
|
+
const f = await project()
|
|
95
|
+
await satistfyDevelopment(f)
|
|
96
|
+
f.records.delete(f.rule())
|
|
97
|
+
await f.call({ operation: 'advance', target_stage: '交付' })
|
|
98
|
+
assert.equal(f.current().stage, '交付', 'the frozen task advances on its own copy')
|
|
99
|
+
})
|
|
100
|
+
|
|
101
|
+
test('enforce: completion requires the review and verification the flow declared', async () => {
|
|
102
|
+
// Arriving at the last stage used to be enough. The agile flow declares a review
|
|
103
|
+
// on the way into 审查, so a blocked review must now prevent completion —
|
|
104
|
+
// without hardcoding review into flows that never asked for it.
|
|
105
|
+
const agile = resolveFlow('agile', adoptRecommendation('agile')).config
|
|
106
|
+
assert.ok(terminalRequirements(agile).length >= 0)
|
|
107
|
+
const base = {
|
|
108
|
+
stage: '审查', execution_version: 1,
|
|
109
|
+
items: [{ id: 'A', status: 'done', review: { spec: { outcome: 'pass' }, quality: { outcome: 'pass' } } }],
|
|
110
|
+
verification: { passed: true, evidence: [] },
|
|
111
|
+
commits: [{ label: 'TASK', hash: 'abc1234' }],
|
|
112
|
+
}
|
|
113
|
+
assert.deepEqual(completionBlockers({ ...base, review: { outcome: 'pass' } }, agile), [],
|
|
114
|
+
'a passing review completes')
|
|
115
|
+
const blocked = completionBlockers({ ...base, review: { outcome: 'blocked' } }, agile)
|
|
116
|
+
assert.ok(blocked.some(b => b.includes('passing review')),
|
|
117
|
+
`a blocked review must prevent completion; got ${JSON.stringify(blocked)}`)
|
|
118
|
+
|
|
119
|
+
// And a flow that declares no verification gate is not made to demand one.
|
|
120
|
+
const minimal = resolveFlow('minimal', adoptRecommendation('minimal')).config
|
|
121
|
+
const minState = {
|
|
122
|
+
stage: '交付', execution_version: 1,
|
|
123
|
+
items: [{ id: 'A', status: 'done', review: { spec: { outcome: 'pass' } } }],
|
|
124
|
+
verification: { passed: false, evidence: [] },
|
|
125
|
+
commits: [{ label: 'TASK', hash: 'abc1234' }],
|
|
126
|
+
}
|
|
127
|
+
assert.deepEqual(completionBlockers(minState, minimal), [],
|
|
128
|
+
'minimal declares no verification gate, so a non-passing verification must not block it')
|
|
129
|
+
})
|
|
130
|
+
|
|
131
|
+
test('enforce: the resource blocker names the stage, the reference and the fix', () => {
|
|
132
|
+
const blockers = resourceBlockers(['project:mine'], '开发')
|
|
133
|
+
assert.equal(blockers.length, 1)
|
|
134
|
+
assert.match(blockers[0], /开发/u)
|
|
135
|
+
assert.match(blockers[0], /project:mine/u)
|
|
136
|
+
assert.match(blockers[0], /Restore the file|remove the binding/u,
|
|
137
|
+
'a blocker must say what to do, not only what is wrong')
|
|
138
|
+
assert.deepEqual(resourceBlockers([], '开发'), [], 'a resolvable stage has no blocker')
|
|
139
|
+
})
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Evidence kinds: a skill is judged against the proof it actually declares.
|
|
3
|
+
*
|
|
4
|
+
* Demanding a "real validation command" from every non-core skill forced an
|
|
5
|
+
* irrelevant command from a requirement or design skill, whose output is a
|
|
6
|
+
* document. The evidence existed — in another form.
|
|
7
|
+
*/
|
|
8
|
+
import test from 'node:test'
|
|
9
|
+
import assert from 'node:assert/strict'
|
|
10
|
+
import { needsSkillReceipt, skillBlockers } from './lib/skill-audit.js'
|
|
11
|
+
import { adoptRecommendation, resolveFlow } from './lib/workflows.js'
|
|
12
|
+
|
|
13
|
+
/** A session that has loaded the named skills. */
|
|
14
|
+
function sessionWith(names) {
|
|
15
|
+
const events = []
|
|
16
|
+
for (const name of names) {
|
|
17
|
+
const callId = 'c' + events.length
|
|
18
|
+
events.push({ type: 'tool/call', data: { name: 'skill', callId, arguments: JSON.stringify({ name }) } })
|
|
19
|
+
events.push({ type: 'tool/result', data: { message: { content: [{ type: 'tool-result', toolCallId: callId, isError: false }] } } })
|
|
20
|
+
}
|
|
21
|
+
return { id: 's', snapshotEvents: () => events }
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
/** A standard-flow config with one extra skill on 开发 carrying the given evidence. */
|
|
25
|
+
function withEvidence(evidence) {
|
|
26
|
+
const adopted = adoptRecommendation('standard')
|
|
27
|
+
const skills = [...adopted.stage_bindings['开发'].skills, { skill: { source: 'project', name: 'doc-skill' }, rules: [], ...(evidence !== undefined ? { evidence } : {}) }]
|
|
28
|
+
return resolveFlow('standard', { ...adopted, stage_bindings: { ...adopted.stage_bindings, '开发': { skills } } }).config
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/** A task at 开发 with everything else satisfied. */
|
|
32
|
+
function state(extra = {}) {
|
|
33
|
+
return {
|
|
34
|
+
schema: 1, id: 'E-1', title: 't', branch: 'main', work_size: 'standard', risk_level: 'standard',
|
|
35
|
+
stage: '开发', execution_version: 1,
|
|
36
|
+
requirement_confirmed: true, solution_confirmed: true,
|
|
37
|
+
items: [], artifacts: {}, files: [], commits: [],
|
|
38
|
+
verification: { passed: false, evidence: [] },
|
|
39
|
+
review: { outcome: 'pending' },
|
|
40
|
+
...extra,
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
test('evidence: needsSkillReceipt honours a declared non-command kind', () => {
|
|
45
|
+
// Core skills keep their exemption; a declared kind overrides the default.
|
|
46
|
+
assert.equal(needsSkillReceipt('code-implement'), false, 'a core skill owes no separate receipt')
|
|
47
|
+
assert.equal(needsSkillReceipt('doc-skill'), true, 'an undeclared extra skill defaults to a command receipt')
|
|
48
|
+
assert.equal(needsSkillReceipt('doc-skill', 'command'), true)
|
|
49
|
+
assert.equal(needsSkillReceipt('doc-skill', 'artifact'), false)
|
|
50
|
+
assert.equal(needsSkillReceipt('doc-skill', 'review'), false)
|
|
51
|
+
assert.equal(needsSkillReceipt('doc-skill', 'manual'), false)
|
|
52
|
+
assert.equal(needsSkillReceipt('doc-skill', 'none'), false)
|
|
53
|
+
})
|
|
54
|
+
|
|
55
|
+
test('evidence: a document-producing skill is satisfied by a recorded artifact', () => {
|
|
56
|
+
// The blocker is the point: without an artifact it blocks, and with one it does
|
|
57
|
+
// not — so the declared kind is genuinely enforced rather than merely accepted.
|
|
58
|
+
//
|
|
59
|
+
// Mounted on 设计, which the standard flow declares an artifact for. A stage with
|
|
60
|
+
// no declared artifact has nothing to satisfy artifact evidence against, and the
|
|
61
|
+
// audit deliberately stays silent there rather than inventing a requirement.
|
|
62
|
+
const adopted = adoptRecommendation('standard')
|
|
63
|
+
const config = resolveFlow('standard', {
|
|
64
|
+
...adopted,
|
|
65
|
+
stage_bindings: {
|
|
66
|
+
...adopted.stage_bindings,
|
|
67
|
+
'设计': { skills: [{ skill: { source: 'project', name: 'doc-skill' }, rules: [], evidence: 'artifact' }] },
|
|
68
|
+
},
|
|
69
|
+
}).config
|
|
70
|
+
const session = sessionWith(['doc-skill'])
|
|
71
|
+
const atDesign = extra => ({ ...state({ stage: '设计', ...extra }) })
|
|
72
|
+
|
|
73
|
+
const missing = skillBlockers(atDesign(), config, session)
|
|
74
|
+
assert.ok(missing.some(b => b.includes('artifact evidence')),
|
|
75
|
+
`an artifact-evidence skill must ask for an artifact; got ${JSON.stringify(missing)}`)
|
|
76
|
+
assert.ok(!missing.some(b => b.includes('validation command')),
|
|
77
|
+
'it must NOT ask for a shell command')
|
|
78
|
+
|
|
79
|
+
// Record what the stage declares and the blocker clears.
|
|
80
|
+
const declared = config.artifacts.find(artifact => artifact.stage === '设计')
|
|
81
|
+
assert.ok(declared !== undefined, 'the standard flow declares an artifact at 设计')
|
|
82
|
+
const cleared = skillBlockers(atDesign({ artifacts: { [declared.id]: { fields: { approach: 'x' } } } }), config, session)
|
|
83
|
+
assert.deepEqual(cleared.filter(b => b.includes('doc-skill')), [],
|
|
84
|
+
'a recorded artifact satisfies artifact evidence')
|
|
85
|
+
})
|
|
86
|
+
|
|
87
|
+
test('evidence: a judging skill is satisfied by a passing review', () => {
|
|
88
|
+
const config = withEvidence('review')
|
|
89
|
+
const session = sessionWith(['code-implement', 'doc-skill'])
|
|
90
|
+
assert.ok(skillBlockers(state(), config, session).some(b => b.includes('review evidence')),
|
|
91
|
+
'no review recorded means the review evidence is absent')
|
|
92
|
+
assert.deepEqual(
|
|
93
|
+
skillBlockers(state({ review: { outcome: 'pass' } }), config, session).filter(b => b.includes('doc-skill')), [],
|
|
94
|
+
'a passing review satisfies it')
|
|
95
|
+
})
|
|
96
|
+
|
|
97
|
+
test('evidence: manual evidence needs an explicit statement, none needs nothing', () => {
|
|
98
|
+
const manual = withEvidence('manual')
|
|
99
|
+
const session = sessionWith(['code-implement', 'doc-skill'])
|
|
100
|
+
assert.ok(skillBlockers(state(), manual, session).some(b => b.includes('manual evidence')),
|
|
101
|
+
'manual evidence is absent until something is recorded')
|
|
102
|
+
const recorded = state({ skill_results: { '开发': { 'doc-skill': { evidence: ['reviewed by the lead'] } } } })
|
|
103
|
+
assert.deepEqual(skillBlockers(recorded, manual, session).filter(b => b.includes('doc-skill')), [],
|
|
104
|
+
'a recorded statement satisfies manual evidence')
|
|
105
|
+
|
|
106
|
+
const none = withEvidence('none')
|
|
107
|
+
assert.deepEqual(skillBlockers(state(), none, session).filter(b => b.includes('doc-skill')), [],
|
|
108
|
+
'an advisory skill needs no proof beyond being loaded')
|
|
109
|
+
})
|
|
110
|
+
|
|
111
|
+
test('evidence: an undeclared extra skill still owes a command receipt', () => {
|
|
112
|
+
// The historical behaviour must survive: a config that says nothing keeps getting
|
|
113
|
+
// the command requirement, so existing setups are unaffected.
|
|
114
|
+
const config = withEvidence(undefined)
|
|
115
|
+
const session = sessionWith(['code-implement', 'doc-skill'])
|
|
116
|
+
const blockers = skillBlockers(state(), config, session)
|
|
117
|
+
assert.ok(blockers.some(b => b.includes('validation command')),
|
|
118
|
+
`an undeclared skill must still owe a command receipt; got ${JSON.stringify(blockers)}`)
|
|
119
|
+
})
|
package/.filter-test.mjs
ADDED
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Catalog filtering: search and filters must narrow the list without ever hiding
|
|
3
|
+
* something that is currently in force.
|
|
4
|
+
*
|
|
5
|
+
* The panel shows the configuration; a filter that hid a bound skill would make the
|
|
6
|
+
* panel disagree with the config it is displaying, which is worse than a long list.
|
|
7
|
+
*/
|
|
8
|
+
import test from 'node:test'
|
|
9
|
+
import assert from 'node:assert/strict'
|
|
10
|
+
import { bindingsForStage, formatResourceRef } from './lib/engine.js'
|
|
11
|
+
import { adoptRecommendation, resolveFlow } from './lib/workflows.js'
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* The filter the panel applies, lifted so it can be tested without a DOM.
|
|
15
|
+
*
|
|
16
|
+
* Mirrors `visibleSkills` in TaskEngineSection: an entry that is bound or shipped
|
|
17
|
+
* always matches; otherwise the query must appear in its name, description or layer.
|
|
18
|
+
*/
|
|
19
|
+
function visibleSkills(catalog, binding, presetSkills, query, onlySelected) {
|
|
20
|
+
const boundKeys = new Set((binding?.skills ?? []).map(entry => formatResourceRef(entry.skill)))
|
|
21
|
+
const presetKeys = new Set(presetSkills.map(entry => formatResourceRef(entry.skill)))
|
|
22
|
+
const needle = query.trim().toLowerCase()
|
|
23
|
+
return catalog.filter(entry => {
|
|
24
|
+
const raw = formatResourceRef(entry.ref)
|
|
25
|
+
const inForce = boundKeys.has(raw) || presetKeys.has(raw)
|
|
26
|
+
if (onlySelected && !inForce) return false
|
|
27
|
+
if (needle === '') return true
|
|
28
|
+
return inForce
|
|
29
|
+
|| entry.name.toLowerCase().includes(needle)
|
|
30
|
+
|| entry.description.toLowerCase().includes(needle)
|
|
31
|
+
|| entry.sourceLabel.toLowerCase().includes(needle)
|
|
32
|
+
})
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
const catalog = [
|
|
36
|
+
{ ref: { source: 'bundled', name: 'code-implement' }, name: 'code-implement', description: '实现', sourceLabel: '内置' },
|
|
37
|
+
{ ref: { source: 'bundled', name: 'code-review' }, name: 'code-review', description: '评审变更', sourceLabel: '内置' },
|
|
38
|
+
{ ref: { source: 'project', name: 'api-audit' }, name: 'api-audit', description: '审计接口契约', sourceLabel: '项目' },
|
|
39
|
+
{ ref: { source: 'user', name: 'doc-writer' }, name: 'doc-writer', description: '写文档', sourceLabel: '用户' },
|
|
40
|
+
]
|
|
41
|
+
const binding = { skills: [{ skill: { source: 'project', name: 'api-audit' }, rules: [] }] }
|
|
42
|
+
|
|
43
|
+
test('filter: no query shows everything', () => {
|
|
44
|
+
assert.equal(visibleSkills(catalog, binding, [], '', false).length, 4)
|
|
45
|
+
})
|
|
46
|
+
|
|
47
|
+
test('filter: the query narrows by name, description and layer', () => {
|
|
48
|
+
// `api-audit` is bound in this fixture, so it stays visible whatever the query —
|
|
49
|
+
// that rule is asserted separately, and these expectations include it.
|
|
50
|
+
const names = (...args) => visibleSkills(...args).map(e => e.name).sort()
|
|
51
|
+
assert.deepEqual(names(catalog, binding, [], 'writer', false), ['api-audit', 'doc-writer'],
|
|
52
|
+
'name matches')
|
|
53
|
+
assert.deepEqual(names(catalog, binding, [], '写文档', false), ['api-audit', 'doc-writer'],
|
|
54
|
+
'description matches')
|
|
55
|
+
assert.deepEqual(names(catalog, binding, [], '用户', false), ['api-audit', 'doc-writer'],
|
|
56
|
+
'the layer label matches, so a user can list one layer')
|
|
57
|
+
assert.deepEqual(names(catalog, binding, [], '审计', false), ['api-audit'],
|
|
58
|
+
'a bound skill matching by its own description is shown')
|
|
59
|
+
assert.deepEqual(names(catalog, binding, [], 'nothing-matches-this', false), ['api-audit'],
|
|
60
|
+
'a query matching nothing still shows the skill that is in force')
|
|
61
|
+
})
|
|
62
|
+
|
|
63
|
+
test('filter: a BOUND skill is never hidden by the query', () => {
|
|
64
|
+
// The panel shows the configuration. A search that hid a skill currently in force
|
|
65
|
+
// would make the panel disagree with the config it is displaying.
|
|
66
|
+
const shown = visibleSkills(catalog, binding, [], 'doc', false)
|
|
67
|
+
assert.ok(shown.some(e => e.name === 'api-audit'),
|
|
68
|
+
'the bound skill stays visible even though it does not match the query')
|
|
69
|
+
assert.ok(shown.some(e => e.name === 'doc-writer'), 'and the matching one is shown too')
|
|
70
|
+
})
|
|
71
|
+
|
|
72
|
+
test('filter: only-selected keeps bound and preset skills, drops the rest', () => {
|
|
73
|
+
const shown = visibleSkills(catalog, binding, [{ skill: { source: 'bundled', name: 'code-review' } }], '', true)
|
|
74
|
+
const names = shown.map(e => e.name).sort()
|
|
75
|
+
assert.deepEqual(names, ['api-audit', 'code-review'],
|
|
76
|
+
'a bound skill and a preset-shipped one are both in force; the other two are not')
|
|
77
|
+
})
|
|
78
|
+
|
|
79
|
+
test('filter: only-selected combined with a query still keeps what is in force', () => {
|
|
80
|
+
const shown = visibleSkills(catalog, binding, [], 'doc', true)
|
|
81
|
+
assert.ok(shown.some(e => e.name === 'api-audit'),
|
|
82
|
+
'the filter must not drop a bound skill even when a query is active')
|
|
83
|
+
assert.ok(!shown.some(e => e.name === 'code-implement'), 'an unbound non-matching skill is dropped')
|
|
84
|
+
})
|
|
85
|
+
|
|
86
|
+
test('filter: the shipped catalog is small enough that the default view shows it all', () => {
|
|
87
|
+
// Sanity: the initial view of a fresh project is the shipped set, unfiltered.
|
|
88
|
+
const adopted = adoptRecommendation('standard')
|
|
89
|
+
const resolved = resolveFlow('standard', adopted).config
|
|
90
|
+
const binding = bindingsForStage('开发', resolved)
|
|
91
|
+
assert.ok(binding !== undefined, 'the adopted config binds 开发')
|
|
92
|
+
assert.ok((binding.skills ?? []).length > 0)
|
|
93
|
+
})
|