@godv61/dsh-task-engine 0.23.8 → 0.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.acceptance.mjs +89 -0
- package/.assessment-batch1.mjs +403 -0
- package/.e2e-presets.mjs +195 -0
- package/.freeze-test.mjs +90 -0
- package/.hook-consistency.mjs +60 -0
- package/.p0-test.mjs +11 -4
- package/.preset-test.mjs +2 -2
- package/.revision-test.mjs +99 -0
- package/.workflow-test.mjs +64 -62
- package/defaults/eng.json +45 -6
- package/docs/BRIEF-FOR-REVIEW.md +163 -0
- package/docs/CHANGELOG.md +72 -0
- package/docs/configuration.md +31 -6
- package/hooks/commit-msg +104 -25
- package/lib/client.js +227 -106
- package/lib/client.js.map +2 -3
- package/lib/controller.d.ts +9 -0
- package/lib/controller.js +37 -3
- package/lib/controller.js.map +1 -1
- package/lib/dev-task.js +285 -43
- package/lib/dev-task.js.map +1 -1
- package/lib/engine.d.ts +278 -6
- package/lib/engine.js +264 -13
- package/lib/engine.js.map +1 -1
- package/lib/seed-preset.d.ts +5 -5
- package/lib/seed-preset.js +5 -5
- package/lib/skill-audit.js +5 -1
- package/lib/skill-audit.js.map +1 -1
- package/lib/workflows.js +87 -27
- package/lib/workflows.js.map +1 -1
- package/package.json +9 -3
- package/scripts/verify-package.mjs +3 -3
package/.e2e-presets.mjs
ADDED
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* End-to-end tool-chain regression over the three presets.
|
|
3
|
+
*
|
|
4
|
+
* Drives the REAL built `dev_task` tool — create, skill loading, artifact
|
|
5
|
+
* recording, item review, verification, review, commit, completion — through an
|
|
6
|
+
* in-memory fs, so it exercises the same code path the model does rather than
|
|
7
|
+
* asserting on source text or calling engine helpers directly.
|
|
8
|
+
*
|
|
9
|
+
* Two things the assessment asks for and this does NOT provide: a real Harness
|
|
10
|
+
* Web session, and a real `git commit`. Those are reported separately as
|
|
11
|
+
* outstanding; nothing here claims to replace them.
|
|
12
|
+
*/
|
|
13
|
+
import test from 'node:test'
|
|
14
|
+
import assert from 'node:assert/strict'
|
|
15
|
+
import { join, resolve } from 'node:path'
|
|
16
|
+
import { registerDevTask } from './lib/dev-task.js'
|
|
17
|
+
import { newTask } from './lib/engine.js'
|
|
18
|
+
import { resolveFlow, FLOW_PRESETS } from './lib/workflows.js'
|
|
19
|
+
|
|
20
|
+
const HASH = 'abcdef1234567890'
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* One preset driven from creation to completion.
|
|
24
|
+
* @param preset - the preset id to run.
|
|
25
|
+
* @returns the step log plus the final state, so the test can assert on both.
|
|
26
|
+
*/
|
|
27
|
+
async function runPreset(preset) {
|
|
28
|
+
const cwd = resolve('e2e-project')
|
|
29
|
+
const config = resolveFlow(preset, {}).config
|
|
30
|
+
const state = newTask({
|
|
31
|
+
id: 'E2E-1', title: 'pipeline', branch: 'main', work_size: 'standard',
|
|
32
|
+
risk_level: 'standard', flow: { flow: preset, version: FLOW_PRESETS[preset].version, config }, root: cwd,
|
|
33
|
+
})
|
|
34
|
+
const records = new Map([
|
|
35
|
+
[join(cwd, '.dsh/task-E2E-1.json'), JSON.stringify(state)],
|
|
36
|
+
// `create` reads the project's flow from here, so the preset under test must
|
|
37
|
+
// be the project's declared flow rather than one passed to the tool.
|
|
38
|
+
[join(cwd, '.dsh/eng.json'), JSON.stringify({ flow: preset })],
|
|
39
|
+
[join(cwd, 'src/a.js'), 'source'],
|
|
40
|
+
])
|
|
41
|
+
const events = []
|
|
42
|
+
const session = { id: 'e2e', header: { cwd }, snapshotEvents: () => events }
|
|
43
|
+
const abort = new AbortController()
|
|
44
|
+
let execute
|
|
45
|
+
let lastMessage = ''
|
|
46
|
+
const key = (p, options) => join(options?.cwd ?? cwd, p)
|
|
47
|
+
const fs = {
|
|
48
|
+
resolve: async (p, options) => ({ targetKey: key(p, options) }),
|
|
49
|
+
readText: async target => records.get(target.targetKey),
|
|
50
|
+
listDir: async () => [...records.keys()].filter(p => /task-.+\.json$/.test(p)).map(p => ({ name: p.split(/[\\/]/).at(-1) })),
|
|
51
|
+
lstat: async (p, options) => records.has(key(p, options)) ? { version: 'v' } : undefined,
|
|
52
|
+
writeText: async (target, content) => { records.set(target.targetKey, content) },
|
|
53
|
+
}
|
|
54
|
+
const approvals = []
|
|
55
|
+
const ctx = {
|
|
56
|
+
fs,
|
|
57
|
+
tools: { register(tool) { execute = tool.execute; return () => {} } },
|
|
58
|
+
get(name) {
|
|
59
|
+
if (name === 'sandboxPolicy') return { resolve: () => ({ mode: 'workspace-write', workspaceRoot: cwd, sessionId: 'e2e' }) }
|
|
60
|
+
// The confirmation guards require a human decision; a stub stands in for the
|
|
61
|
+
// human here and records what it was asked, so the test can show the guard
|
|
62
|
+
// actually consulted it rather than being bypassed. The verdict is the
|
|
63
|
+
// protocol's string value, not a boolean.
|
|
64
|
+
if (name === 'approval') return {
|
|
65
|
+
async request(request) { approvals.push(request); return 'allowed-once' },
|
|
66
|
+
}
|
|
67
|
+
if (name === 'shell') return {
|
|
68
|
+
resolve: request => request,
|
|
69
|
+
async run(request) {
|
|
70
|
+
const isLog = request.command.includes('log -1')
|
|
71
|
+
return {
|
|
72
|
+
exitCode: 0, timedOut: false, aborted: false,
|
|
73
|
+
sandbox: { mode: 'workspace-write', denied: false },
|
|
74
|
+
stdout: { text: isLog ? `${HASH}\n${lastMessage}\n\nsrc/a.js\n` : 'checks passed' },
|
|
75
|
+
stderr: { text: '' },
|
|
76
|
+
}
|
|
77
|
+
},
|
|
78
|
+
}
|
|
79
|
+
return undefined
|
|
80
|
+
},
|
|
81
|
+
}
|
|
82
|
+
registerDevTask(ctx)
|
|
83
|
+
const exec = { agent: { session }, signal: abort.signal }
|
|
84
|
+
const call = async args => {
|
|
85
|
+
if (args.message !== undefined) lastMessage = args.message
|
|
86
|
+
const text = await execute(JSON.parse(JSON.stringify({ task_id: 'E2E-1', ...args })), exec)
|
|
87
|
+
return args.operation === 'status' ? JSON.parse(text) : text
|
|
88
|
+
}
|
|
89
|
+
const load = name => {
|
|
90
|
+
const callId = `c${events.length}`
|
|
91
|
+
events.push({ type: 'tool/call', data: { name: 'skill', callId, arguments: JSON.stringify({ name }) } })
|
|
92
|
+
events.push({ type: 'tool/result', data: { message: { content: [{ type: 'tool-result', toolCallId: callId, isError: false }] } } })
|
|
93
|
+
}
|
|
94
|
+
const current = () => JSON.parse(records.get(join(cwd, '.dsh/task-E2E-1.json')))
|
|
95
|
+
|
|
96
|
+
const steps = []
|
|
97
|
+
await call({ operation: 'create', title: 'pipeline', branch: 'main', files: ['src/a.js'] })
|
|
98
|
+
steps.push('create')
|
|
99
|
+
|
|
100
|
+
for (let hop = 0; hop < 14; hop++) {
|
|
101
|
+
const snapshot = current()
|
|
102
|
+
const outgoing = config.transitions.filter(transition => transition.from === snapshot.stage)
|
|
103
|
+
if (outgoing.length === 0) break
|
|
104
|
+
|
|
105
|
+
// Satisfy whatever the stage's outgoing edges need.
|
|
106
|
+
const needed = new Set(outgoing.flatMap(transition => transition.requires ?? []))
|
|
107
|
+
if (needed.has('todos_done')) {
|
|
108
|
+
// Items are reviewed through the dedicated operation; `items` only carries
|
|
109
|
+
// status, and the review verdicts are recorded against the item by id.
|
|
110
|
+
await call({ operation: 'items', items: [{ id: 'A', title: 'a', status: 'doing' }] })
|
|
111
|
+
await call({ operation: 'dispatch', item_id: 'A', description: 'implemented by a worker' })
|
|
112
|
+
await call({ operation: 'review_item', item_id: 'A', spec_outcome: 'pass', quality_outcome: 'pass' })
|
|
113
|
+
await call({ operation: 'items', items: [{ id: 'A', status: 'done' }] })
|
|
114
|
+
}
|
|
115
|
+
// Confirmation guards are satisfied by a human decision routed through the
|
|
116
|
+
// approval service; the tool refuses to let the model assert them itself.
|
|
117
|
+
if (needed.has('requirement_confirmation') && !snapshot.requirement_confirmed) {
|
|
118
|
+
await call({ operation: 'config', confirmations: ['requirement_confirmation'] })
|
|
119
|
+
}
|
|
120
|
+
if (needed.has('solution_confirmation') && !snapshot.solution_confirmed) {
|
|
121
|
+
await call({ operation: 'config', confirmations: ['solution_confirmation'] })
|
|
122
|
+
}
|
|
123
|
+
if (needed.has('verified')) {
|
|
124
|
+
await call({ operation: 'verify', command: 'npm test', description: 'run the suite' })
|
|
125
|
+
}
|
|
126
|
+
if (needed.has('review_passed')) {
|
|
127
|
+
await call({ operation: 'review', outcome: 'pass' })
|
|
128
|
+
}
|
|
129
|
+
for (const artifact of config.artifacts.filter(a => a.stage === snapshot.stage)) {
|
|
130
|
+
const fields = {}
|
|
131
|
+
for (const field of artifact.fields) fields[field] = 'recorded'
|
|
132
|
+
await call({ operation: 'record', artifact: artifact.id, fields })
|
|
133
|
+
}
|
|
134
|
+
// Obligations cover the current stage AND the stage being entered, so both
|
|
135
|
+
// sets are loaded. This mirrors the disclosure the model follows.
|
|
136
|
+
for (const obligated of new Set([snapshot.stage, ...outgoing.map(t => t.to)])) {
|
|
137
|
+
for (const binding of config.stage_bindings?.[obligated]?.skills ?? []) load(binding.skill.name)
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
if (config.commit.checkpoints.includes(snapshot.stage)) {
|
|
141
|
+
const status = await call({ operation: 'status' })
|
|
142
|
+
await call({
|
|
143
|
+
operation: 'commit', files: ['src/a.js'], hash: HASH,
|
|
144
|
+
message: `【E2E-1】【${status.commit?.label ?? 'TASK'}】deliver the stage`,
|
|
145
|
+
})
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
const target = outgoing[0].to
|
|
149
|
+
// The target stage's skills must be loaded before the move, because leaving a
|
|
150
|
+
// stage checks the obligations of the stage being entered. This mirrors what
|
|
151
|
+
// the disclosure tells the model to do.
|
|
152
|
+
for (const binding of config.stage_bindings?.[target]?.skills ?? []) load(binding.skill.name)
|
|
153
|
+
await call({ operation: 'advance', target_stage: target })
|
|
154
|
+
steps.push(`${snapshot.stage}→${target}`)
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
const terminal = config.stages.filter(stage => !config.transitions.some(t => t.from === stage))
|
|
158
|
+
if (!terminal.includes(current().stage)) {
|
|
159
|
+
return { steps, approvals, final: current(), completed: false, blocked: `stopped at ${current().stage}` }
|
|
160
|
+
}
|
|
161
|
+
// The terminal stage's own obligations are checked before it can be left, and
|
|
162
|
+
// completion is the operation that leaves it.
|
|
163
|
+
for (const binding of config.stage_bindings?.[current().stage]?.skills ?? []) load(binding.skill.name)
|
|
164
|
+
if (config.commit.checkpoints.includes(current().stage)) {
|
|
165
|
+
const status = await call({ operation: 'status' })
|
|
166
|
+
await call({
|
|
167
|
+
operation: 'commit', files: ['src/a.js'], hash: HASH,
|
|
168
|
+
message: `【E2E-1】【${status.commit?.label ?? 'TASK'}】deliver the stage`,
|
|
169
|
+
})
|
|
170
|
+
}
|
|
171
|
+
await call({ operation: 'complete' })
|
|
172
|
+
return { steps, approvals, final: current(), completed: true, blocked: null }
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
test('e2e: every preset can be driven from creation to completion through the real tool', async () => {
|
|
176
|
+
for (const preset of ['standard', 'agile', 'minimal']) {
|
|
177
|
+
const run = await runPreset(preset)
|
|
178
|
+
assert.equal(
|
|
179
|
+
run.completed, true,
|
|
180
|
+
`${preset} must reach completion; ${run.blocked ?? ''} steps=${run.steps.join(' ')}`,
|
|
181
|
+
)
|
|
182
|
+
assert.ok(run.final.completed !== undefined, `${preset} must record its completion`)
|
|
183
|
+
assert.ok(run.steps.length >= 2, `${preset} must traverse at least one edge`)
|
|
184
|
+
}
|
|
185
|
+
})
|
|
186
|
+
|
|
187
|
+
test('e2e: a preset that requires a commit cannot complete without one', async () => {
|
|
188
|
+
// The completion check is the thing that catches minimal, whose final stage is
|
|
189
|
+
// also its checkpoint and therefore never fires the "commit before leaving"
|
|
190
|
+
// rule. Driving it through the tool must refuse completion without a commit.
|
|
191
|
+
const run = await runPreset('minimal')
|
|
192
|
+
assert.equal(run.completed, true)
|
|
193
|
+
assert.ok(run.final.commits.length > 0,
|
|
194
|
+
'reaching completion means the delivery record exists, not merely that the last stage was reached')
|
|
195
|
+
})
|
package/.freeze-test.mjs
ADDED
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Resource freezing: an in-flight task must not change because the library did.
|
|
3
|
+
*
|
|
4
|
+
* Names alone cannot keep a task stable. Before this, editing a rule that a
|
|
5
|
+
* running task used silently changed what that task was doing, and deleting one
|
|
6
|
+
* made a configured constraint disappear without the task recording anything.
|
|
7
|
+
*/
|
|
8
|
+
import test from 'node:test'
|
|
9
|
+
import assert from 'node:assert/strict'
|
|
10
|
+
import { join, resolve } from 'node:path'
|
|
11
|
+
import { registerDevTask } from './lib/dev-task.js'
|
|
12
|
+
import { hashText } from './lib/snapshot.js'
|
|
13
|
+
|
|
14
|
+
/** Create one task through the real tool and return its frozen snapshot. */
|
|
15
|
+
async function created(flow) {
|
|
16
|
+
const cwd = resolve('freeze-project')
|
|
17
|
+
const records = new Map([
|
|
18
|
+
[join(cwd, '.dsh/eng.json'), JSON.stringify({ flow })],
|
|
19
|
+
[join(cwd, 'a.js'), 'source'],
|
|
20
|
+
])
|
|
21
|
+
const events = []
|
|
22
|
+
const session = { id: 'f', header: { cwd }, snapshotEvents: () => events }
|
|
23
|
+
let execute
|
|
24
|
+
const key = (p, options) => join(options?.cwd ?? cwd, p)
|
|
25
|
+
const fs = {
|
|
26
|
+
resolve: async (p, options) => ({ targetKey: key(p, options) }),
|
|
27
|
+
readText: async target => records.get(target.targetKey),
|
|
28
|
+
listDir: async () => [],
|
|
29
|
+
lstat: async (p, options) => records.has(key(p, options)) ? { version: 'v' } : undefined,
|
|
30
|
+
writeText: async (target, content) => { records.set(target.targetKey, content) },
|
|
31
|
+
}
|
|
32
|
+
const ctx = {
|
|
33
|
+
fs,
|
|
34
|
+
tools: { register(tool) { execute = tool.execute; return () => {} } },
|
|
35
|
+
get: name => name === 'sandboxPolicy'
|
|
36
|
+
? { resolve: () => ({ mode: 'workspace-write', workspaceRoot: cwd, sessionId: 'f' }) }
|
|
37
|
+
: undefined,
|
|
38
|
+
}
|
|
39
|
+
registerDevTask(ctx)
|
|
40
|
+
const exec = { agent: { session }, signal: new AbortController().signal }
|
|
41
|
+
await execute({ task_id: 'FZ-1', operation: 'create', title: 'freeze', branch: 'main', files: ['a.js'] }, exec)
|
|
42
|
+
return JSON.parse(records.get(join(cwd, '.dsh/task-FZ-1.json')))
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
test('freeze: task creation captures every skill and rule body it will use', async () => {
|
|
46
|
+
const state = await created('standard')
|
|
47
|
+
const resources = state.flow.resources ?? []
|
|
48
|
+
assert.ok(resources.length > 0, 'creation must freeze the resolved resources')
|
|
49
|
+
|
|
50
|
+
// Every binding in the frozen config must be covered by a frozen body.
|
|
51
|
+
const frozen = new Set(resources.map(r => `${r.ref.source}:${r.ref.name}`))
|
|
52
|
+
for (const [stage, binding] of Object.entries(state.flow.config.stage_bindings ?? {})) {
|
|
53
|
+
for (const entry of binding.skills ?? []) {
|
|
54
|
+
assert.ok(frozen.has(`${entry.skill.source}:${entry.skill.name}`),
|
|
55
|
+
`${entry.skill.name} on ${stage} must be frozen`)
|
|
56
|
+
for (const rule of entry.rules) {
|
|
57
|
+
assert.ok(frozen.has(`${rule.source}:${rule.name}`),
|
|
58
|
+
`rule ${rule.name} for ${entry.skill.name} must be frozen`)
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
})
|
|
63
|
+
|
|
64
|
+
test('freeze: a frozen body carries its content and a hash of that content', async () => {
|
|
65
|
+
const state = await created('standard')
|
|
66
|
+
for (const resource of state.flow.resources ?? []) {
|
|
67
|
+
assert.ok(resource.content.length > 0, `${resource.ref.name} must carry its body`)
|
|
68
|
+
assert.equal(resource.hash, hashText(resource.content),
|
|
69
|
+
`${resource.ref.name} hash must describe the frozen body, so a later edit is detectable`)
|
|
70
|
+
}
|
|
71
|
+
})
|
|
72
|
+
|
|
73
|
+
test('freeze: shared resources are stored once, not duplicated per reference', async () => {
|
|
74
|
+
// security-redlines is referenced by two skills; the body must not be copied twice.
|
|
75
|
+
const state = await created('standard')
|
|
76
|
+
const resources = state.flow.resources ?? []
|
|
77
|
+
const hashes = resources.map(r => r.hash)
|
|
78
|
+
assert.equal(new Set(hashes).size, hashes.length,
|
|
79
|
+
'identical bodies must not be stored more than once')
|
|
80
|
+
assert.ok(resources.some(r => r.ref.name === 'security-redlines'),
|
|
81
|
+
'the shared rule must be present')
|
|
82
|
+
})
|
|
83
|
+
|
|
84
|
+
test('freeze: every preset freezes a complete, non-empty set', async () => {
|
|
85
|
+
for (const flow of ['standard', 'agile', 'minimal']) {
|
|
86
|
+
const state = await created(flow)
|
|
87
|
+
assert.ok((state.flow.resources ?? []).length >= 3,
|
|
88
|
+
`${flow} must freeze its resources; saw ${(state.flow.resources ?? []).length}`)
|
|
89
|
+
}
|
|
90
|
+
})
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The commit hook must agree with the tool about the same commit.
|
|
3
|
+
*
|
|
4
|
+
* The assessment requires one set of semantics shared by `status`, the tool's own
|
|
5
|
+
* gates and the Git hook, rather than three implementations that drift. The hook
|
|
6
|
+
* is built from the same engine source, so this checks the built artefact rather
|
|
7
|
+
* than the source relationship.
|
|
8
|
+
*/
|
|
9
|
+
import { readFileSync } from 'node:fs'
|
|
10
|
+
import { resolveFlow } from './lib/workflows.js'
|
|
11
|
+
import { validateCommitMessage, commitCheckpoint, verificationBlockers } from './lib/engine.js'
|
|
12
|
+
|
|
13
|
+
let bad = 0
|
|
14
|
+
const check = (label, ok, detail = '') => {
|
|
15
|
+
console.log(' ' + (ok ? '✅' : '❌') + ' ' + label + (detail ? ' — ' + detail : ''))
|
|
16
|
+
if (!ok) bad++
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
console.log('=== hook 与工具共享判定 ===')
|
|
20
|
+
const hook = readFileSync('hooks/commit-msg', 'utf8')
|
|
21
|
+
for (const fn of ['validateCommitMessage', 'commitCheckpoint', 'verificationHeldStages', 'verificationBlockers']) {
|
|
22
|
+
check('hook 内置 ' + fn, hook.includes(fn), hook.includes(fn) ? '' : 'hook 与工具语义会漂移')
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
console.log('')
|
|
26
|
+
console.log('=== 每种标签形状都能通过它自己预设的校验 ===')
|
|
27
|
+
for (const id of ['standard', 'agile', 'minimal']) {
|
|
28
|
+
const cfg = resolveFlow(id, {}).config
|
|
29
|
+
const labels = id === 'agile' ? ['TASK', 'T1', 'T2'] : ['TASK']
|
|
30
|
+
for (const label of labels) {
|
|
31
|
+
const message = `【T-1】【${label}】did the thing`
|
|
32
|
+
const r = validateCommitMessage(message, cfg)
|
|
33
|
+
check(`${id}「${label}」被接受`, r.ok, r.ok ? '' : (r.errors ?? []).join('; '))
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
console.log('')
|
|
38
|
+
console.log('=== 提交许可与验证要求一致(三个预设)===')
|
|
39
|
+
for (const id of ['standard', 'agile', 'minimal']) {
|
|
40
|
+
const cfg = resolveFlow(id, {}).config
|
|
41
|
+
const terminal = cfg.stages.filter(s => !cfg.transitions.some(t => t.from === s))[0]
|
|
42
|
+
const failing = {
|
|
43
|
+
stage: terminal, execution_version: 1,
|
|
44
|
+
items: [{ id: 'A', status: 'done', review: { spec: { outcome: 'pass' }, quality: { outcome: 'pass' } } }],
|
|
45
|
+
verification: { passed: false, evidence: [] }, commits: [],
|
|
46
|
+
}
|
|
47
|
+
const vb = verificationBlockers(failing, cfg)
|
|
48
|
+
const cp = commitCheckpoint(failing, cfg)
|
|
49
|
+
// 需要验证的流程:失败必须同时阻止提交;不需要验证的流程:两者都不应因验证而阻止
|
|
50
|
+
const requires = cfg.transitions.some(t => (t.requires ?? []).includes('verified'))
|
|
51
|
+
if (requires) {
|
|
52
|
+
check(`${id} 验证失败阻止提交`, vb.length > 0 && cp.allowed === false, `blockers=${vb.length} allowed=${cp.allowed}`)
|
|
53
|
+
} else {
|
|
54
|
+
check(`${id} 不发明验证要求`, vb.length === 0, JSON.stringify(vb))
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
console.log('')
|
|
59
|
+
console.log(bad === 0 ? '✅ hook 与工具语义一致' : `❌ ${bad} 项不一致`)
|
|
60
|
+
process.exit(bad === 0 ? 0 : 1)
|
package/.p0-test.mjs
CHANGED
|
@@ -108,10 +108,17 @@ const unknown = resolveFlow('nope')
|
|
|
108
108
|
assert(unknown.ok === false && unknown.code === 'UNKNOWN_FLOW', 'unknown flow returns UNKNOWN_FLOW')
|
|
109
109
|
assert(unknown.knownFlows.includes('standard'), 'known flow list is surfaced')
|
|
110
110
|
|
|
111
|
-
// ── 3. core bindings cannot be cancelled by an override
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
111
|
+
// ── 3. core skill bindings cannot be cancelled by an override ──────────────
|
|
112
|
+
// Rules moved onto the skill, so the invariant is about the SKILL surviving an
|
|
113
|
+
// override and that skill's own rule list staying intact. A stage has no rule
|
|
114
|
+
// list left to empty.
|
|
115
|
+
const merged = resolveFlow('standard', { stage_bindings: { '开发': {} } })
|
|
116
|
+
assert(merged.ok
|
|
117
|
+
&& merged.config.stage_bindings['开发'].skills.some(entry => entry.skill.name === 'code-implement'),
|
|
118
|
+
'an empty stage override keeps the core skill binding')
|
|
119
|
+
const implementBinding = merged.config.stage_bindings['开发'].skills.find(entry => entry.skill.name === 'code-implement')
|
|
120
|
+
assert(implementBinding.rules.map(rule => rule.name).includes('coding-conventions'),
|
|
121
|
+
'a skill keeps its own rules; there is no stage-level rule list for an override to empty')
|
|
115
122
|
|
|
116
123
|
// ── 4. newTask freezes the snapshot ────────────────────────────────────────
|
|
117
124
|
const snapshot = { flow: 'standard', version: 1, config: FLOW_PRESETS.standard.config }
|
package/.preset-test.mjs
CHANGED
|
@@ -26,7 +26,7 @@ import { derivedEngComposition, wasSeeded, applyPersona, personaFieldOf } from '
|
|
|
26
26
|
* The live derivation resolves `@deepseek-ai/dsh-agent-presets`, a PEER the
|
|
27
27
|
* running harness provides. In a consumer's install it may be absent or a
|
|
28
28
|
* different version, so the assertions that read a real `standard` skip when it
|
|
29
|
-
* is unavailable rather than failing
|
|
29
|
+
* is unavailable rather than failing —the fixture-driven cases below cover the
|
|
30
30
|
* logic itself and always run.
|
|
31
31
|
*/
|
|
32
32
|
const derived = derivedEngComposition()
|
|
@@ -48,7 +48,7 @@ test('the engineering persona replaces the shipped one', live, () => {
|
|
|
48
48
|
|
|
49
49
|
test('the derived persona row is well formed and never sets complete', live, () => {
|
|
50
50
|
// Which field is correct depends on the running harness, so assert the SHAPE
|
|
51
|
-
// (a block scalar under `config:`) rather than one field name
|
|
51
|
+
// (a block scalar under `config:`) rather than one field name —the field
|
|
52
52
|
// itself is pinned by the two fixtures below.
|
|
53
53
|
assert.match(derived.text, /- id: persona\n {2}name: '@deepseek-ai\/dsh-persona'\n {2}config:\n(?: {4}suffix: .*\n)? {4}(?:text|prefix): [|>]-/u,
|
|
54
54
|
'the persona row must carry a block scalar under config')
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Rework semantics: going backwards must invalidate exactly the superseded
|
|
3
|
+
* conclusions, no more and no less.
|
|
4
|
+
*
|
|
5
|
+
* The graph has only forward edges while the work does not, and before this the
|
|
6
|
+
* only way back was hand-editing the record — which left a confirmation,
|
|
7
|
+
* verification receipt or review verdict describing a superseded tree still
|
|
8
|
+
* reading as passing. The two halves of that are both tested: what must fall, and
|
|
9
|
+
* what must survive so work is not redone for no reason.
|
|
10
|
+
*/
|
|
11
|
+
import test from 'node:test'
|
|
12
|
+
import assert from 'node:assert/strict'
|
|
13
|
+
import { applyRevision, invalidatedBy, completionBlockers } from './lib/engine.js'
|
|
14
|
+
import { resolveFlow } from './lib/workflows.js'
|
|
15
|
+
|
|
16
|
+
/** A task carrying every conclusion, so a rework has something to invalidate. */
|
|
17
|
+
function loadedTask() {
|
|
18
|
+
return {
|
|
19
|
+
schema: 1,
|
|
20
|
+
id: 'R-1',
|
|
21
|
+
title: 'rework',
|
|
22
|
+
branch: 'main',
|
|
23
|
+
work_size: 'standard',
|
|
24
|
+
risk_level: 'standard',
|
|
25
|
+
stage: '代码审核',
|
|
26
|
+
executed: undefined,
|
|
27
|
+
requirement_confirmed: true,
|
|
28
|
+
solution_confirmed: true,
|
|
29
|
+
items: [{ id: 'A', title: 'a', status: 'done', review: { spec: { outcome: 'pass' }, quality: { outcome: 'pass' } } }],
|
|
30
|
+
verification: { passed: true, evidence: ['checks'], receipt: { exit_code: 0, timed_out: false, aborted: false } },
|
|
31
|
+
review: { outcome: 'pass' },
|
|
32
|
+
artifacts: {},
|
|
33
|
+
files: ['a.js'],
|
|
34
|
+
commits: [{ label: 'TASK', hash: 'abc1234' }],
|
|
35
|
+
execution_version: 1,
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
const rework = (kind, to) => ({ kind, reason: 'why', to, from: '代码审核', at: '2026-09-23T00:00:00.000Z', invalidated: invalidatedBy(kind) })
|
|
40
|
+
|
|
41
|
+
test('rework: the three kinds declare distinct scopes rather than one blanket rule', () => {
|
|
42
|
+
const requirement = invalidatedBy('requirement')
|
|
43
|
+
const solution = invalidatedBy('solution')
|
|
44
|
+
const defect = invalidatedBy('defect')
|
|
45
|
+
assert.ok(requirement.includes('requirement_confirmation'),
|
|
46
|
+
'a changed requirement must invalidate its own confirmation')
|
|
47
|
+
assert.ok(!solution.includes('requirement_confirmation'),
|
|
48
|
+
'a changed approach leaves the agreed requirement standing')
|
|
49
|
+
assert.ok(!defect.includes('requirement_confirmation') && !defect.includes('solution_confirmation'),
|
|
50
|
+
'a defect in the work does not un-agree what was agreed')
|
|
51
|
+
assert.ok(defect.includes('verification') && defect.includes('review'),
|
|
52
|
+
'a defect invalidates the evidence that the old implementation was correct')
|
|
53
|
+
})
|
|
54
|
+
|
|
55
|
+
test('rework: a changed requirement clears confirmations and downstream evidence', () => {
|
|
56
|
+
const state = loadedTask()
|
|
57
|
+
const outcome = applyRevision(state, rework('requirement', '需求评审'))
|
|
58
|
+
assert.equal(state.stage, '需求评审')
|
|
59
|
+
assert.equal(state.requirement_confirmed, false)
|
|
60
|
+
assert.equal(state.solution_confirmed, false)
|
|
61
|
+
assert.equal(state.verification.passed, false)
|
|
62
|
+
assert.equal(state.review.outcome, 'pending')
|
|
63
|
+
assert.equal(state.items[0].review, undefined)
|
|
64
|
+
assert.deepEqual(state.commits, [])
|
|
65
|
+
assert.equal(state.completed, undefined)
|
|
66
|
+
assert.ok(outcome.invalidated.includes('requirement_confirmation'))
|
|
67
|
+
})
|
|
68
|
+
|
|
69
|
+
test('rework: fixing a defect keeps the agreed requirement and solution', () => {
|
|
70
|
+
// The point of distinguishing kinds: a defect must not force the requirement to
|
|
71
|
+
// be re-agreed, which would make the lightest flows unusable for ordinary fixes.
|
|
72
|
+
const state = loadedTask()
|
|
73
|
+
applyRevision(state, rework('defect', '开发'))
|
|
74
|
+
assert.equal(state.requirement_confirmed, true, 'a defect does not un-agree the requirement')
|
|
75
|
+
assert.equal(state.solution_confirmed, true, 'nor the approach')
|
|
76
|
+
assert.equal(state.verification.passed, false, 'the evidence about the old implementation falls')
|
|
77
|
+
assert.deepEqual(state.commits, [], 'the commit described a superseded tree')
|
|
78
|
+
})
|
|
79
|
+
|
|
80
|
+
test('rework: clearing evidence returns an item to unfinished, not to done-without-audit', () => {
|
|
81
|
+
// A cleared review means the item is no longer proven done. Leaving it `done`
|
|
82
|
+
// would let todos_done pass on work whose audit had just been discarded.
|
|
83
|
+
const state = loadedTask()
|
|
84
|
+
applyRevision(state, rework('defect', '开发'))
|
|
85
|
+
assert.equal(state.items[0].status, 'doing')
|
|
86
|
+
const config = resolveFlow('standard', {}).config
|
|
87
|
+
assert.ok(completionBlockers({ ...state, stage: '完成' }, config).length > 0,
|
|
88
|
+
'a task with an unproven item must not be completable')
|
|
89
|
+
})
|
|
90
|
+
|
|
91
|
+
test('rework: history records what was invalidated, for audit', () => {
|
|
92
|
+
const state = loadedTask()
|
|
93
|
+
applyRevision(state, rework('solution', '设计'))
|
|
94
|
+
assert.equal(state.revisions.length, 1)
|
|
95
|
+
assert.equal(state.revisions[0].kind, 'solution')
|
|
96
|
+
assert.equal(state.revisions[0].from, '代码审核')
|
|
97
|
+
assert.equal(state.revisions[0].to, '设计')
|
|
98
|
+
assert.deepEqual(state.revisions[0].invalidated, invalidatedBy('solution'))
|
|
99
|
+
})
|
package/.workflow-test.mjs
CHANGED
|
@@ -9,9 +9,9 @@ import { resolveFlow } from './lib/workflows.js'
|
|
|
9
9
|
import Controller from './lib/controller.js'
|
|
10
10
|
import { registerShippedSkills } from './lib/shipped-skills.js'
|
|
11
11
|
|
|
12
|
-
function fixture(stage = '开发', extra = {}, services = {}) {
|
|
12
|
+
function fixture(stage = '开发', extra = {}, services = {}) {
|
|
13
13
|
const cwd = resolve('test-project')
|
|
14
|
-
const config = resolveFlow('standard', { stage_bindings: { 完成: { skills: ['software-testing'] } } }).config
|
|
14
|
+
const config = resolveFlow('standard', { stage_bindings: { 完成: { skills: [{ skill: { source: 'bundled', name: 'software-testing' }, rules: [] }] } } }).config
|
|
15
15
|
const state = newTask({ id: 'LIVE-1', title: 'EAM regression', branch: 'test', work_size: 'standard', risk_level: 'standard', flow: { flow: 'standard', version: 2, config }, root: cwd })
|
|
16
16
|
Object.assign(state, { stage, execution_version: 1, files: ['app.js'], requirement_confirmed: true, solution_confirmed: true, ...extra })
|
|
17
17
|
const records = new Map([[join(cwd, '.dsh/task-LIVE-1.json'), JSON.stringify(state)], [join(cwd, 'app.js'), 'source']])
|
|
@@ -28,8 +28,8 @@ function fixture(stage = '开发', extra = {}, services = {}) {
|
|
|
28
28
|
lstat: async (path, options) => records.has(key(path, options)) ? { version: 'v' } : undefined,
|
|
29
29
|
writeText: async (target, content) => { records.set(target.targetKey, content) },
|
|
30
30
|
}
|
|
31
|
-
const ctx = { fs, tools: { register(tool) { execute = tool.execute; return () => {} } }, get(name) {
|
|
32
|
-
if (Object.hasOwn(services, name)) return services[name]
|
|
31
|
+
const ctx = { fs, tools: { register(tool) { execute = tool.execute; return () => {} } }, get(name) {
|
|
32
|
+
if (Object.hasOwn(services, name)) return services[name]
|
|
33
33
|
if (name === 'sandboxPolicy') return { resolve(request) { assert.equal(request.session, session); return policy } }
|
|
34
34
|
if (name === 'shell') return { resolve(request) { runs.push(request); return request }, async run(request) {
|
|
35
35
|
const fail = request.command === 'fail'
|
|
@@ -85,66 +85,66 @@ test('派发立即反映进行中,重派不复用旧审核且保持单一进
|
|
|
85
85
|
assert.equal(f.state().items[0].review, undefined)
|
|
86
86
|
})
|
|
87
87
|
|
|
88
|
-
test('无任务 id 的 status 发现当前工作区任务并支持分支过滤', async () => {
|
|
88
|
+
test('无任务 id 的 status 发现当前工作区任务并支持分支过滤', async () => {
|
|
89
89
|
const f = fixture()
|
|
90
90
|
assert.equal(JSON.parse(await f.call({ operation: 'status', task_id: undefined })).tasks[0].id, 'LIVE-1')
|
|
91
91
|
assert.equal(JSON.parse(await f.call({ operation: 'status', task_id: undefined, branch: 'other' })).tasks.length, 0)
|
|
92
|
-
})
|
|
93
|
-
|
|
94
|
-
test('verify 升级审批不可用时不得先执行命令', async () => {
|
|
95
|
-
const f = fixture('交付')
|
|
96
|
-
const before = JSON.stringify(f.state())
|
|
97
|
-
await assert.rejects(f.call({ operation: 'verify', command: 'npm test', sandbox_permissions: 'danger-full-access', justification: 'Retry the denied test command' }), /approval/)
|
|
98
|
-
assert.equal(f.runs.length, 0, 'approval must precede shell execution')
|
|
99
|
-
assert.equal(JSON.stringify(f.state()), before)
|
|
100
|
-
})
|
|
101
|
-
|
|
102
|
-
test('verify 拒绝、取消及无效升级参数均不执行命令或改写台账', async () => {
|
|
103
|
-
for (const outcome of ['rejected', 'cancelled', 'unavailable']) {
|
|
104
|
-
const f = fixture('交付', {}, { approval: { request: async () => outcome } })
|
|
105
|
-
const before = JSON.stringify(f.state())
|
|
106
|
-
await assert.rejects(f.call({ operation: 'verify', command: 'npm test', sandbox_permissions: 'danger-full-access', justification: 'Retry denied command' }), /not approved/)
|
|
107
|
-
assert.equal(f.runs.length, 0)
|
|
108
|
-
assert.equal(JSON.stringify(f.state()), before)
|
|
109
|
-
}
|
|
110
|
-
for (const args of [{ sandbox_permissions: 'danger-full-access' }, { justification: 'orphan justification' }]) {
|
|
111
|
-
const f = fixture('交付')
|
|
112
|
-
await assert.rejects(f.call({ operation: 'verify', command: 'npm test', ...args }), /invalid escalation/)
|
|
113
|
-
assert.equal(f.runs.length, 0)
|
|
114
|
-
}
|
|
115
|
-
})
|
|
116
|
-
|
|
117
|
-
test('验证命令审批先于执行,只授权本次命令且回执写入不重复审批', async () => {
|
|
118
|
-
for (const operation of ['verify', 'skill_result']) {
|
|
119
|
-
const approvals = []
|
|
120
|
-
const f = fixture('完成', {}, { approval: { request: async request => {
|
|
121
|
-
assert.equal(f.runs.length, 0)
|
|
122
|
-
approvals.push(request)
|
|
123
|
-
return 'allowed-once'
|
|
124
|
-
} } })
|
|
125
|
-
f.load('software-testing')
|
|
126
|
-
await f.call({ operation, command: 'npm test', skill_name: 'software-testing', evidence: ['real checks'], sandbox_permissions: 'danger-full-access', justification: 'Retry denied test command' })
|
|
127
|
-
assert.equal(approvals.length, 1)
|
|
128
|
-
assert.match(approvals[0].reason, /npm test/)
|
|
129
|
-
assert.equal(f.runs[0].sandboxPolicy.mode, 'danger-full-access')
|
|
130
|
-
assert.equal(f.runs[0].sandboxPolicy.workspaceRoot, f.cwd)
|
|
131
|
-
assert.equal(f.runs[0].sandboxPolicy.sessionId, f.policy.sessionId)
|
|
132
|
-
assert.equal(f.runs[0].signal, f.exec.signal)
|
|
133
|
-
assert.equal(f.policy.mode, 'workspace-write')
|
|
134
|
-
await f.call({ operation: 'verify', command: 'npm test' })
|
|
135
|
-
assert.equal(f.runs[1].sandboxPolicy, f.policy)
|
|
136
|
-
assert.equal(approvals.length, 1)
|
|
137
|
-
}
|
|
138
|
-
})
|
|
139
|
-
|
|
140
|
-
test('skill_result 审批拒绝时不执行命令、不写入技能回执', async () => {
|
|
141
|
-
const f = fixture('完成', {}, { approval: { request: async () => 'rejected' } })
|
|
142
|
-
f.load('software-testing')
|
|
143
|
-
const before = JSON.stringify(f.state())
|
|
144
|
-
await assert.rejects(f.call({ operation: 'skill_result', skill_name: 'software-testing', command: 'npm test', evidence: ['test'], sandbox_permissions: 'danger-full-access', justification: 'Retry denied tests' }), /not approved/)
|
|
145
|
-
assert.equal(f.runs.length, 0)
|
|
146
|
-
assert.equal(JSON.stringify(f.state()), before)
|
|
147
|
-
})
|
|
92
|
+
})
|
|
93
|
+
|
|
94
|
+
test('verify 升级审批不可用时不得先执行命令', async () => {
|
|
95
|
+
const f = fixture('交付')
|
|
96
|
+
const before = JSON.stringify(f.state())
|
|
97
|
+
await assert.rejects(f.call({ operation: 'verify', command: 'npm test', sandbox_permissions: 'danger-full-access', justification: 'Retry the denied test command' }), /approval/)
|
|
98
|
+
assert.equal(f.runs.length, 0, 'approval must precede shell execution')
|
|
99
|
+
assert.equal(JSON.stringify(f.state()), before)
|
|
100
|
+
})
|
|
101
|
+
|
|
102
|
+
test('verify 拒绝、取消及无效升级参数均不执行命令或改写台账', async () => {
|
|
103
|
+
for (const outcome of ['rejected', 'cancelled', 'unavailable']) {
|
|
104
|
+
const f = fixture('交付', {}, { approval: { request: async () => outcome } })
|
|
105
|
+
const before = JSON.stringify(f.state())
|
|
106
|
+
await assert.rejects(f.call({ operation: 'verify', command: 'npm test', sandbox_permissions: 'danger-full-access', justification: 'Retry denied command' }), /not approved/)
|
|
107
|
+
assert.equal(f.runs.length, 0)
|
|
108
|
+
assert.equal(JSON.stringify(f.state()), before)
|
|
109
|
+
}
|
|
110
|
+
for (const args of [{ sandbox_permissions: 'danger-full-access' }, { justification: 'orphan justification' }]) {
|
|
111
|
+
const f = fixture('交付')
|
|
112
|
+
await assert.rejects(f.call({ operation: 'verify', command: 'npm test', ...args }), /invalid escalation/)
|
|
113
|
+
assert.equal(f.runs.length, 0)
|
|
114
|
+
}
|
|
115
|
+
})
|
|
116
|
+
|
|
117
|
+
test('验证命令审批先于执行,只授权本次命令且回执写入不重复审批', async () => {
|
|
118
|
+
for (const operation of ['verify', 'skill_result']) {
|
|
119
|
+
const approvals = []
|
|
120
|
+
const f = fixture('完成', {}, { approval: { request: async request => {
|
|
121
|
+
assert.equal(f.runs.length, 0)
|
|
122
|
+
approvals.push(request)
|
|
123
|
+
return 'allowed-once'
|
|
124
|
+
} } })
|
|
125
|
+
f.load('software-testing')
|
|
126
|
+
await f.call({ operation, command: 'npm test', skill_name: 'software-testing', evidence: ['real checks'], sandbox_permissions: 'danger-full-access', justification: 'Retry denied test command' })
|
|
127
|
+
assert.equal(approvals.length, 1)
|
|
128
|
+
assert.match(approvals[0].reason, /npm test/)
|
|
129
|
+
assert.equal(f.runs[0].sandboxPolicy.mode, 'danger-full-access')
|
|
130
|
+
assert.equal(f.runs[0].sandboxPolicy.workspaceRoot, f.cwd)
|
|
131
|
+
assert.equal(f.runs[0].sandboxPolicy.sessionId, f.policy.sessionId)
|
|
132
|
+
assert.equal(f.runs[0].signal, f.exec.signal)
|
|
133
|
+
assert.equal(f.policy.mode, 'workspace-write')
|
|
134
|
+
await f.call({ operation: 'verify', command: 'npm test' })
|
|
135
|
+
assert.equal(f.runs[1].sandboxPolicy, f.policy)
|
|
136
|
+
assert.equal(approvals.length, 1)
|
|
137
|
+
}
|
|
138
|
+
})
|
|
139
|
+
|
|
140
|
+
test('skill_result 审批拒绝时不执行命令、不写入技能回执', async () => {
|
|
141
|
+
const f = fixture('完成', {}, { approval: { request: async () => 'rejected' } })
|
|
142
|
+
f.load('software-testing')
|
|
143
|
+
const before = JSON.stringify(f.state())
|
|
144
|
+
await assert.rejects(f.call({ operation: 'skill_result', skill_name: 'software-testing', command: 'npm test', evidence: ['test'], sandbox_permissions: 'danger-full-access', justification: 'Retry denied tests' }), /not approved/)
|
|
145
|
+
assert.equal(f.runs.length, 0)
|
|
146
|
+
assert.equal(JSON.stringify(f.state()), before)
|
|
147
|
+
})
|
|
148
148
|
|
|
149
149
|
test('状态明确区分内置节点门禁与附加技能命令回执,避免重复验证', async () => {
|
|
150
150
|
const f = fixture('代码审核')
|
|
@@ -196,7 +196,9 @@ test('验证使用会话策略与取消信号,并明确返回失败及沙箱
|
|
|
196
196
|
})
|
|
197
197
|
|
|
198
198
|
test('只声明已测试、或加载失败,均不能冒充执行绑定技能', async () => {
|
|
199
|
-
|
|
199
|
+
// 代码审核坐在验证门之后,所以该阶段本就要求验证通过;
|
|
200
|
+
// 不给出验证状态会让验证阻塞先于本用例要测的技能阻塞。
|
|
201
|
+
const f = fixture('代码审核', { verification: { passed: true, evidence: ['checks passed'] } })
|
|
200
202
|
f.load('code-review'); f.load('code-commit'); f.load('software-testing', false)
|
|
201
203
|
await assert.rejects(f.call({ operation: 'skill_result', target_stage: '完成', skill_name: 'software-testing', command: 'check', evidence: ['tested'] }), /load skill/)
|
|
202
204
|
await assert.rejects(f.call({ operation: 'advance', target_stage: '完成' }), /software-testing/)
|