@gobing-ai/spur 0.3.86 → 0.3.87
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/config/plugin-scripts.json +4 -0
- package/config/templates/feature/default.md +2 -0
- package/config/templates/task/brainstorm.md +2 -2
- package/config/templates/task/feature-impl.md +2 -2
- package/config/templates/task/issue.md +2 -2
- package/config/templates/task/meta.md +2 -2
- package/config/templates/task/review.md +2 -2
- package/config/templates/task/standard.md +2 -2
- package/config/workflow-candidates.json +13 -35
- package/config/workflows/feature-lifecycle.yaml +6 -0
- package/config/workflows/feature-verification.yaml +6 -5
- package/config/workflows/idea-pipeline.yaml +72 -36
- package/package.json +1 -1
- package/plugins/sp/README.md +7 -4
- package/plugins/sp/commands/dev-idea.md +9 -2
- package/plugins/sp/commands/dev-refactor.md +33 -0
- package/plugins/sp/plugin.json +1 -1
- package/plugins/sp/references/roles.md +1 -1
- package/plugins/sp/scripts/idea-coverage-check.ts +168 -0
- package/plugins/sp/scripts/inline-pipeline-parity-check.ts +114 -3
- package/plugins/sp/scripts/inline-run-setup.ts +28 -18
- package/plugins/sp/skills/brainstorm/SKILL.md +4 -0
- package/plugins/sp/skills/code-refactoring/SKILL.md +155 -0
- package/plugins/sp/skills/code-refactoring/references/finding-schema.md +74 -0
- package/plugins/sp/skills/code-refactoring/references/fix-ladder.md +52 -0
- package/plugins/sp/skills/code-refactoring/references/focus-detection.md +44 -0
- package/plugins/sp/skills/code-refactoring/references/refactor-finding.schema.json +95 -0
- package/plugins/sp/skills/spec-decomposition/references/decomposition.md +23 -16
- package/plugins/sp/skills/spur-cli/references/features/acceptance-criteria.md +4 -2
- package/plugins/sp/skills/spur-cli/references/features.md +6 -1
- package/plugins/sp/skills/spur-cli/references/workflows/authoring-workflows.md +2 -2
- package/plugins/sp/skills/spur-cli/references/workflows/operations.md +3 -3
- package/plugins/sp/skills/spur-cli/references/workflows/workflow-fit-and-tuning.md +1 -1
- package/plugins/sp/skills/spur-dev/references/ac-style-guide.md +24 -0
- package/plugins/sp/skills/spur-dev/references/dev-operations.md +18 -2
- package/plugins/sp/skills/spur-dev/references/flag-glossary.md +22 -3
- package/plugins/sp/skills/spur-dev/references/idea-evaluation.md +11 -1
- package/plugins/sp/skills/spur-dev/references/inline-pipeline-driver.md +33 -1
- package/plugins/sp/skills/taste-refactoring-api/SKILL.md +44 -1
- package/plugins/sp/skills/taste-refactoring-api/references/protocol-modes.md +31 -0
- package/plugins/sp/skills/taste-refactoring-architect/SKILL.md +43 -0
- package/plugins/sp/skills/taste-refactoring-tests/SKILL.md +42 -0
- package/plugins/sp/skills/taste-refactoring-ui/SKILL.md +43 -0
- package/schemas/task-batch.schema.json +2 -2
- package/spur.js +124 -23
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
#!/usr/bin/env bun
|
|
2
|
+
/**
|
|
3
|
+
* idea-coverage-check — requirement-inventory ↔ AC coverage gate for the idea pipeline
|
|
4
|
+
* (task 0887 R4).
|
|
5
|
+
*
|
|
6
|
+
* Cross-checks the `## Requirement inventory` section of the idea-evaluation report
|
|
7
|
+
* (R3) against the `# covers: I<n>, ...` comment lines of the generated acceptance
|
|
8
|
+
* criteria (R4): every inventory item that is not explicitly `[deferred: ...]` must be
|
|
9
|
+
* covered by at least one scenario. An `[unclear: ...]` marker does not exempt an item —
|
|
10
|
+
* it still demands coverage or an explicit deferral.
|
|
11
|
+
*
|
|
12
|
+
* Writes PASS/FAIL to `.spur/run/<run-id>-idea-coverage.status` and always exits 0
|
|
13
|
+
* (soft action, task 0769 pattern): the ac-generate guards in idea-pipeline.yaml read
|
|
14
|
+
* the status file, so a missing or failing checker fails closed through the guard, not
|
|
15
|
+
* through the exit code.
|
|
16
|
+
*
|
|
17
|
+
* Ships with the plugin to arbitrary projects, so it stays node-builtin-only —
|
|
18
|
+
* no workspace imports.
|
|
19
|
+
*
|
|
20
|
+
* Usage:
|
|
21
|
+
* bun plugins/sp/scripts/idea-coverage-check.ts --run-id <id> --report <path>
|
|
22
|
+
* --ac <path> [--out <path>]
|
|
23
|
+
*/
|
|
24
|
+
|
|
25
|
+
import { existsSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs';
|
|
26
|
+
import { dirname, join } from 'node:path';
|
|
27
|
+
|
|
28
|
+
interface ParsedArgs {
|
|
29
|
+
runId: string;
|
|
30
|
+
reportPath: string;
|
|
31
|
+
acPath: string;
|
|
32
|
+
outPath: string;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/** `- **I3** — ask text` / `- I3. ask text` — the R3 inventory item form. */
|
|
36
|
+
const INVENTORY_ITEM_RE = /^\s*[-*]\s+\**I(\d+)\**\s*[.:—-]?\s+(.*)$/;
|
|
37
|
+
|
|
38
|
+
/** `# covers: I1, I3` — the R4 scenario coverage comment. */
|
|
39
|
+
const COVERS_RE = /^\s*#\s*covers:\s*(.+)$/i;
|
|
40
|
+
|
|
41
|
+
/** Scenario header — the anchor a covers-comment attaches to. */
|
|
42
|
+
const SCENARIO_RE = /^\s*Scenario(?:\s+Outline)?:/;
|
|
43
|
+
|
|
44
|
+
function parseArgs(argv: string[]): ParsedArgs {
|
|
45
|
+
let runId = '';
|
|
46
|
+
let reportPath = '';
|
|
47
|
+
let acPath = '';
|
|
48
|
+
let outPath = '';
|
|
49
|
+
for (let i = 0; i < argv.length; i++) {
|
|
50
|
+
const value = argv[i + 1];
|
|
51
|
+
switch (argv[i]) {
|
|
52
|
+
case '--run-id':
|
|
53
|
+
runId = value ?? '';
|
|
54
|
+
i++;
|
|
55
|
+
break;
|
|
56
|
+
case '--report':
|
|
57
|
+
reportPath = value ?? '';
|
|
58
|
+
i++;
|
|
59
|
+
break;
|
|
60
|
+
case '--ac':
|
|
61
|
+
acPath = value ?? '';
|
|
62
|
+
i++;
|
|
63
|
+
break;
|
|
64
|
+
case '--out':
|
|
65
|
+
outPath = value ?? '';
|
|
66
|
+
i++;
|
|
67
|
+
break;
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
if (runId === '' || reportPath === '' || acPath === '') {
|
|
71
|
+
process.stderr.write(
|
|
72
|
+
'usage: idea-coverage-check.ts --run-id <id> --report <path> --ac <path> [--out <path>]\n',
|
|
73
|
+
);
|
|
74
|
+
process.exit(2);
|
|
75
|
+
}
|
|
76
|
+
return {
|
|
77
|
+
runId,
|
|
78
|
+
reportPath,
|
|
79
|
+
acPath,
|
|
80
|
+
outPath: outPath !== '' ? outPath : join('.spur', 'run', `${runId}-idea-coverage.status`),
|
|
81
|
+
};
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* Inventory ids from the report's `## Requirement inventory` section, split into
|
|
86
|
+
* covered-owing (no `[deferred:` marker) and exempt (deferred) ids. An `[unclear:`
|
|
87
|
+
* marker is informational — the item still owes coverage.
|
|
88
|
+
*/
|
|
89
|
+
function parseInventory(report: string): { owing: Set<string>; deferred: Set<string>; hasSection: boolean } {
|
|
90
|
+
const lines = report.split('\n');
|
|
91
|
+
const start = lines.findIndex((line) => /^#{1,6}\s*Requirement inventory\s*$/i.test(line));
|
|
92
|
+
if (start === -1) return { owing: new Set(), deferred: new Set(), hasSection: false };
|
|
93
|
+
const owing = new Set<string>();
|
|
94
|
+
const deferred = new Set<string>();
|
|
95
|
+
for (let i = start + 1; i < lines.length; i++) {
|
|
96
|
+
if (/^#{1,6}\s/.test(lines[i] ?? '')) break;
|
|
97
|
+
const match = INVENTORY_ITEM_RE.exec(lines[i] ?? '');
|
|
98
|
+
if (match === null) continue;
|
|
99
|
+
const id = `I${match[1]}`;
|
|
100
|
+
if (/\[\s*deferred\s*:/i.test(match[2] ?? '')) deferred.add(id);
|
|
101
|
+
else owing.add(id);
|
|
102
|
+
}
|
|
103
|
+
return { owing, deferred, hasSection: true };
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/** Coverage map from the AC content: scenario-count per `I<n>` id. */
|
|
107
|
+
function parseCoverage(ac: string): Map<string, number> {
|
|
108
|
+
const covered = new Map<string, number>();
|
|
109
|
+
let inScenario = false;
|
|
110
|
+
for (const line of ac.split('\n')) {
|
|
111
|
+
if (SCENARIO_RE.test(line)) {
|
|
112
|
+
inScenario = true;
|
|
113
|
+
continue;
|
|
114
|
+
}
|
|
115
|
+
if (!inScenario) continue;
|
|
116
|
+
const match = COVERS_RE.exec(line);
|
|
117
|
+
if (match === null) continue;
|
|
118
|
+
for (const raw of (match[1] ?? '').split(',')) {
|
|
119
|
+
const id = raw.trim().toUpperCase();
|
|
120
|
+
if (/^I\d+$/.test(id)) covered.set(id, (covered.get(id) ?? 0) + 1);
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
return covered;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
function writeStatus(outPath: string, verdict: 'PASS' | 'FAIL', runId: string, detail: string): void {
|
|
127
|
+
mkdirSync(dirname(outPath), { recursive: true });
|
|
128
|
+
writeFileSync(outPath, `${verdict}\n`);
|
|
129
|
+
// Sibling reason file: the feature-check HITL prompt surfaces WHY next to the bare
|
|
130
|
+
// status letter, without guards having to parse a multi-line status file.
|
|
131
|
+
writeFileSync(`${outPath}.reason`, `${verdict} run=${runId} ${detail}\n`);
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
function main(): void {
|
|
135
|
+
const args = parseArgs(process.argv.slice(2));
|
|
136
|
+
|
|
137
|
+
if (!existsSync(args.reportPath) || !existsSync(args.acPath)) {
|
|
138
|
+
const detail = `missing input (report=${existsSync(args.reportPath) ? 'ok' : 'absent'}, ac=${existsSync(args.acPath) ? 'ok' : 'absent'})`;
|
|
139
|
+
writeStatus(args.outPath, 'FAIL', args.runId, detail);
|
|
140
|
+
process.stdout.write(`idea-coverage-check FAIL run=${args.runId} ${detail}\n`);
|
|
141
|
+
return;
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
const inventory = parseInventory(readFileSync(args.reportPath, 'utf8'));
|
|
145
|
+
const covered = parseCoverage(readFileSync(args.acPath, 'utf8'));
|
|
146
|
+
|
|
147
|
+
if (!inventory.hasSection || inventory.owing.size + inventory.deferred.size === 0) {
|
|
148
|
+
const detail = `no Requirement inventory items in ${args.reportPath}`;
|
|
149
|
+
writeStatus(args.outPath, 'FAIL', args.runId, detail);
|
|
150
|
+
process.stdout.write(`idea-coverage-check FAIL run=${args.runId} ${detail}\n`);
|
|
151
|
+
return;
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
const uncovered = [...inventory.owing].filter((id) => (covered.get(id) ?? 0) === 0).sort();
|
|
155
|
+
const shape = `inventory=${inventory.owing.size + inventory.deferred.size} (deferred=${inventory.deferred.size})`;
|
|
156
|
+
if (uncovered.length > 0) {
|
|
157
|
+
const detail = `${shape} covered=${inventory.owing.size - uncovered.length} uncovered=${uncovered.join(',')}`;
|
|
158
|
+
writeStatus(args.outPath, 'FAIL', args.runId, detail);
|
|
159
|
+
process.stdout.write(`idea-coverage-check FAIL run=${args.runId} ${detail}\n`);
|
|
160
|
+
return;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
const detail = `${shape} all covered`;
|
|
164
|
+
writeStatus(args.outPath, 'PASS', args.runId, detail);
|
|
165
|
+
process.stdout.write(`idea-coverage-check PASS run=${args.runId} ${detail}\n`);
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
main();
|
|
@@ -11,8 +11,14 @@
|
|
|
11
11
|
* check is a symmetric set diff: an element present in one and absent in the
|
|
12
12
|
* other fails the check and names the element.
|
|
13
13
|
*
|
|
14
|
-
* The set is defined in {@link DOCUMENTED} below
|
|
15
|
-
*
|
|
14
|
+
* The set is defined in {@link DOCUMENTED} below. Task 0881 (0877 R7): the harness
|
|
15
|
+
* enumerates state from BOTH reference sets — this constant and the driver
|
|
16
|
+
* markdown's mirror list — plus the YAML union, as a three-way symmetric diff, so
|
|
17
|
+
* removing a kind from either reference cannot escape parity. It also scans task
|
|
18
|
+
* frontmatter `dependencies[]` for spurious edges: a referenced wbs with no task
|
|
19
|
+
* file (post-0875, dependency edges no longer move the planning digest —
|
|
20
|
+
* `packages/app/src/services/task-readiness.ts` — so over-declared edges would
|
|
21
|
+
* otherwise be silently unbound).
|
|
16
22
|
*
|
|
17
23
|
* Usage:
|
|
18
24
|
* bun plugins/sp/scripts/inline-pipeline-parity-check.ts
|
|
@@ -49,6 +55,91 @@ const DOCUMENTED = {
|
|
|
49
55
|
/** Directory of workflow definitions the driver is responsible for. */
|
|
50
56
|
const WORKFLOW_DIR = join('config', 'workflows');
|
|
51
57
|
|
|
58
|
+
/** The driver reference doc — the second reference set (0881). */
|
|
59
|
+
const DRIVER_REF = join('plugins', 'sp', 'skills', 'spur-dev', 'references', 'inline-pipeline-driver.md');
|
|
60
|
+
|
|
61
|
+
/** Task corpus dirs scanned for spurious `dependencies[]` edges (0881). */
|
|
62
|
+
const TASK_DIRS = ['docs/tasks', 'docs/tasks5'];
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Parse the driver reference's mirrored kind lists (`**Actions:** … · `kind` …`) into
|
|
66
|
+
* reference sets. Returns null when the doc is absent (fixture roots).
|
|
67
|
+
*/
|
|
68
|
+
function collectMarkdownReferenceKinds(root: string): { actions: Set<string>; guards: Set<string> } | null {
|
|
69
|
+
let md: string;
|
|
70
|
+
try {
|
|
71
|
+
md = readFileSync(join(root, DRIVER_REF), 'utf8');
|
|
72
|
+
} catch {
|
|
73
|
+
return null;
|
|
74
|
+
}
|
|
75
|
+
const read = (label: RegExp): Set<string> => {
|
|
76
|
+
const line = md.split('\n').find((l) => label.test(l));
|
|
77
|
+
const out = new Set<string>();
|
|
78
|
+
if (!line) return out;
|
|
79
|
+
for (const m of line.matchAll(/`([^`]+)`/g)) out.add(m[1] ?? '');
|
|
80
|
+
out.delete('');
|
|
81
|
+
return out;
|
|
82
|
+
};
|
|
83
|
+
return { actions: read(/^\*\*Actions:/), guards: read(/^\*\*Guards/) };
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* Scan task frontmatter `dependencies[]` for spurious edges: a referenced wbs that
|
|
88
|
+
* resolves to no task file in the corpus (0881 — post-0875 these edges no longer
|
|
89
|
+
* move the planning digest, so over-declared ones are otherwise silently unbound).
|
|
90
|
+
*/
|
|
91
|
+
function checkTaskDependencyEdges(root: string, errors: string[]): number {
|
|
92
|
+
const wbss = new Set<string>();
|
|
93
|
+
const depFiles: { path: string; deps: string[] }[] = [];
|
|
94
|
+
for (const dir of TASK_DIRS) {
|
|
95
|
+
const abs = join(root, dir);
|
|
96
|
+
let entries: string[] = [];
|
|
97
|
+
try {
|
|
98
|
+
entries = readdirSync(abs);
|
|
99
|
+
} catch {
|
|
100
|
+
continue;
|
|
101
|
+
}
|
|
102
|
+
for (const e of entries) {
|
|
103
|
+
const m = /^(\d+)_.*\.md$/.exec(e);
|
|
104
|
+
if (!m?.[1]) continue;
|
|
105
|
+
wbss.add(m[1]);
|
|
106
|
+
const raw = readFileSync(join(abs, e), 'utf8');
|
|
107
|
+
const fm = /^---\n([\s\S]*?)\n---/.exec(raw);
|
|
108
|
+
if (!fm?.[1]) continue;
|
|
109
|
+
const deps: string[] = [];
|
|
110
|
+
const inline = /^dependencies:\s*\[(.*)\]/m.exec(fm[1]);
|
|
111
|
+
if (inline?.[1]) {
|
|
112
|
+
for (const part of inline[1].split(',')) {
|
|
113
|
+
const w = part.trim().replace(/["']/g, '');
|
|
114
|
+
if (w !== '') deps.push(w);
|
|
115
|
+
}
|
|
116
|
+
} else {
|
|
117
|
+
const block = /^dependencies:\s*$/m.exec(fm[1]);
|
|
118
|
+
if (block) {
|
|
119
|
+
for (const line of fm[1].split('\n')) {
|
|
120
|
+
const item = /^\s*-\s*["']?(\d+)["']?\s*$/.exec(line);
|
|
121
|
+
if (item?.[1]) deps.push(item[1]);
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
if (deps.length > 0) depFiles.push({ path: join(dir, e), deps });
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
let count = 0;
|
|
129
|
+
for (const { path, deps } of depFiles) {
|
|
130
|
+
for (const dep of deps) {
|
|
131
|
+
// Only wbs-shaped values are machine dependency edges; prose refs in the
|
|
132
|
+
// legacy corpus are descriptive text, never consumed as edges.
|
|
133
|
+
if (!/^\d{3,4}$/.test(dep)) continue;
|
|
134
|
+
if (!wbss.has(dep)) {
|
|
135
|
+
errors.push(`spurious dependency edge: ${path} depends on ${dep} but no task with wbs ${dep} exists`);
|
|
136
|
+
count += 1;
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
return count;
|
|
141
|
+
}
|
|
142
|
+
|
|
52
143
|
/** Walk a state list and yield every `kind:` value found in `onEnter` action
|
|
53
144
|
* lists. Skips the top-level workflow `kind:` (e.g. `state-machine`). */
|
|
54
145
|
function collectActionKinds(states: unknown): Set<string> {
|
|
@@ -170,6 +261,26 @@ async function main(): Promise<number> {
|
|
|
170
261
|
errors.push(`guard kind "${x}" documented in inline-pipeline-driver.md but never used in any workflow`);
|
|
171
262
|
}
|
|
172
263
|
|
|
264
|
+
// Three-way parity with the markdown mirror (0881): a kind dropped from either
|
|
265
|
+
// reference set — constant or doc — is caught, not just YAML drift.
|
|
266
|
+
const ref = collectMarkdownReferenceKinds(root);
|
|
267
|
+
if (ref !== null) {
|
|
268
|
+
for (const x of diff(ref.actions, DOCUMENTED.actions).onlyInA)
|
|
269
|
+
errors.push(`action kind "${x}" in the driver markdown but absent from the DOCUMENTED set`);
|
|
270
|
+
for (const x of diff(ref.actions, DOCUMENTED.actions).onlyInB)
|
|
271
|
+
errors.push(`action kind "${x}" in DOCUMENTED but deleted from the driver markdown reference`);
|
|
272
|
+
for (const x of diff(ref.guards, DOCUMENTED.guards).onlyInA)
|
|
273
|
+
errors.push(`guard kind "${x}" in the driver markdown but absent from the DOCUMENTED set`);
|
|
274
|
+
for (const x of diff(ref.guards, DOCUMENTED.guards).onlyInB)
|
|
275
|
+
errors.push(`guard kind "${x}" in DOCUMENTED but deleted from the driver markdown reference`);
|
|
276
|
+
for (const x of diff(ref.actions, unionActions).onlyInB)
|
|
277
|
+
errors.push(`action kind "${x}" in the references but never used in any workflow`);
|
|
278
|
+
for (const x of diff(ref.guards, unionGuards).onlyInB)
|
|
279
|
+
errors.push(`guard kind "${x}" in the references but never used in any workflow`);
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
const spuriousEdges = checkTaskDependencyEdges(root, errors);
|
|
283
|
+
|
|
173
284
|
if (errors.length > 0) {
|
|
174
285
|
process.stderr.write(`inline-pipeline-parity-check: ${errors.length} divergence(s)\n`);
|
|
175
286
|
for (const e of errors) process.stderr.write(` - ${e}\n`);
|
|
@@ -177,7 +288,7 @@ async function main(): Promise<number> {
|
|
|
177
288
|
}
|
|
178
289
|
|
|
179
290
|
process.stdout.write(
|
|
180
|
-
`inline-pipeline-parity-check: ok (${unionActions.size} actions, ${unionGuards.size} guards agree across ${files.length} workflows)\n`,
|
|
291
|
+
`inline-pipeline-parity-check: ok (${unionActions.size} actions, ${unionGuards.size} guards agree across ${files.length} workflows and both reference sets; ${spuriousEdges} spurious dependency edges)\n`,
|
|
181
292
|
);
|
|
182
293
|
return 0;
|
|
183
294
|
}
|
|
@@ -31,7 +31,7 @@
|
|
|
31
31
|
* bun plugins/sp/scripts/inline-run-setup.ts --run-id <id> --file <definition> [--spur-bin <path>]
|
|
32
32
|
* bun plugins/sp/scripts/inline-run-setup.ts --fingerprint --task-file <path> [--feature-file <path>] [--spur-bin <path>]
|
|
33
33
|
* bun plugins/sp/scripts/inline-run-setup.ts --action --run-id <id> --node <state> --kind <kind> \
|
|
34
|
-
* --status <done|failed
|
|
34
|
+
* --status <done|failed> --ok <true|false> --duration-ms <n> [--spur-bin <path>]
|
|
35
35
|
* bun plugins/sp/scripts/inline-run-setup.ts --close --run-id <id> --status <done|failed|paused> [--spur-bin <path>]
|
|
36
36
|
*
|
|
37
37
|
* The `--fingerprint` mode prints the engine's proof-input digest for the given spec files and
|
|
@@ -78,7 +78,7 @@ function usage(): never {
|
|
|
78
78
|
);
|
|
79
79
|
console.error(
|
|
80
80
|
' bun plugins/sp/scripts/inline-run-setup.ts --action --run-id <id> --node <state> --kind <kind> ' +
|
|
81
|
-
'--status <done|failed
|
|
81
|
+
'--status <done|failed> --ok <true|false> --duration-ms <n> [--spur-bin <path>]',
|
|
82
82
|
);
|
|
83
83
|
console.error(
|
|
84
84
|
' bun plugins/sp/scripts/inline-run-setup.ts --close --run-id <id> --status <done|failed|paused> [--spur-bin <path>]',
|
|
@@ -200,10 +200,12 @@ async function printFingerprint(taskFile: string, featureFile: string, spurBin:
|
|
|
200
200
|
/** Terminal statuses the inline driver may declare when closing its run row. */
|
|
201
201
|
const CLOSE_STATUSES = new Set(['done', 'failed', 'paused']);
|
|
202
202
|
|
|
203
|
-
/**
|
|
204
|
-
const ACTION_STATUSES = new Set(['
|
|
203
|
+
/** Finalize statuses — a finish emission is terminal, so only done|failed are valid (0868 #4). */
|
|
204
|
+
const ACTION_STATUSES = new Set(['done', 'failed']);
|
|
205
205
|
|
|
206
206
|
/** Input for the ADR-117 emission modes (`--action` / `--close`). */
|
|
207
|
+
import type { WorkflowActionTraceWriter } from '@gobing-ai/app';
|
|
208
|
+
|
|
207
209
|
interface TraceModeInput {
|
|
208
210
|
readonly runId: string;
|
|
209
211
|
readonly close: boolean;
|
|
@@ -260,15 +262,15 @@ async function runTraceMode(input: TraceModeInput): Promise<number> {
|
|
|
260
262
|
);
|
|
261
263
|
}
|
|
262
264
|
|
|
265
|
+
// Compile-time link (0868 finding #5): the writer half is typed by the real packages/app
|
|
266
|
+
// export (type-only import, erased at runtime) so a signature drift breaks THIS file's
|
|
267
|
+
// typecheck instead of hiding behind the hand-declared cast.
|
|
263
268
|
const app = (await import(entry)) as {
|
|
264
269
|
openInlineRunProjectDb: (workdir: string) => Promise<{ adapter: unknown; close: () => void }>;
|
|
265
270
|
createWorkflowActionTraceWriter: (
|
|
266
271
|
db: unknown,
|
|
267
272
|
recordFailure?: (failure: unknown) => void,
|
|
268
|
-
) =>
|
|
269
|
-
recordAction: (boundary: Record<string, unknown>) => Promise<{ ok: boolean; actionId?: string }>;
|
|
270
|
-
closeRun: (runId: string, status: string) => Promise<{ ok: boolean }>;
|
|
271
|
-
};
|
|
273
|
+
) => WorkflowActionTraceWriter;
|
|
272
274
|
};
|
|
273
275
|
|
|
274
276
|
let projectDb: { adapter: unknown; close: () => void } | undefined;
|
|
@@ -281,16 +283,24 @@ async function runTraceMode(input: TraceModeInput): Promise<number> {
|
|
|
281
283
|
`trace-emission-failed operation=${detail.operation ?? operation} run=${input.runId}: ${detail.error ?? 'unknown error'}`,
|
|
282
284
|
);
|
|
283
285
|
});
|
|
284
|
-
const result =
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
286
|
+
const result = (
|
|
287
|
+
input.close
|
|
288
|
+
? await writer.closeRun(input.runId, input.status)
|
|
289
|
+
: await writer.recordAction({
|
|
290
|
+
runId: input.runId,
|
|
291
|
+
node: input.node,
|
|
292
|
+
kind: input.kind,
|
|
293
|
+
status: input.status,
|
|
294
|
+
ok: input.ok,
|
|
295
|
+
durationMs: input.durationMs,
|
|
296
|
+
})
|
|
297
|
+
) as Record<string, unknown>;
|
|
298
|
+
if (result.ok !== true && result.failure !== undefined) {
|
|
299
|
+
// One stdout shape for emission failures (0868 finding #1): flatten the guard's
|
|
300
|
+
// nested failure object to the same `{ok, runId, error}` the direct paths emit.
|
|
301
|
+
const failure = result.failure as { error?: string };
|
|
302
|
+
return fail(failure.error ?? 'unknown trace emission failure');
|
|
303
|
+
}
|
|
294
304
|
process.stdout.write(`${JSON.stringify({ ...result, runId: input.runId })}\n`);
|
|
295
305
|
return 0;
|
|
296
306
|
} catch (error) {
|
|
@@ -197,6 +197,10 @@ When brainstorm runs under `idea-pipeline` discovery, it MUST also emit a filled
|
|
|
197
197
|
[`spur-dev/references/idea-evaluation.md`](../spur-dev/references/idea-evaluation.md):
|
|
198
198
|
|
|
199
199
|
- Enhanced idea statement (sidecar — does **not** overwrite the operator's original idea text)
|
|
200
|
+
- **Requirement inventory** — mandatory `## Requirement inventory` section: numbered `I<n>` items
|
|
201
|
+
quoting or paraphrasing the source lines from the run's verbatim idea artifact
|
|
202
|
+
(`.spur/run/<run-id>-idea-input.md`, 0887 R1/R3), with `[unclear: ...]` markers where the ask is
|
|
203
|
+
ambiguous and `[deferred: <reason>]` markers for explicit out-of-scope items
|
|
200
204
|
- Urgency and necessity scores (0–5) with one-line rationales
|
|
201
205
|
- Premises, pros, cons, better alternatives (if any)
|
|
202
206
|
- Recommendation (`proceed` | `reshape` | `drop`) + stakes
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: code-refactoring
|
|
3
|
+
description: "Refactoring coordinator with a preservation contract: routes the four taste lenses (api, architect, tests, ui), normalizes findings to a shared schema and P1–P4 severities, gates applies, runs a fix ladder with revert-on-regression. Use for 'dev-refactor', 'safe refactor', 'refactor while keeping every feature'."
|
|
4
|
+
license: Apache-2.0
|
|
5
|
+
metadata:
|
|
6
|
+
author: spur
|
|
7
|
+
version: "1.0"
|
|
8
|
+
platforms: "claude-code,codex,openclaw,opencode,antigravity"
|
|
9
|
+
category: execution
|
|
10
|
+
interactions:
|
|
11
|
+
- technique
|
|
12
|
+
operations:
|
|
13
|
+
- refactor
|
|
14
|
+
openclaw:
|
|
15
|
+
emoji: "♻️"
|
|
16
|
+
see_also:
|
|
17
|
+
- sp:taste-refactoring-api
|
|
18
|
+
- sp:taste-refactoring-architect
|
|
19
|
+
- sp:taste-refactoring-tests
|
|
20
|
+
- sp:taste-refactoring-ui
|
|
21
|
+
- sp:code-simplification
|
|
22
|
+
- sp:code-review
|
|
23
|
+
---
|
|
24
|
+
|
|
25
|
+
# code-refactoring — the lens-routed refactoring coordinator
|
|
26
|
+
|
|
27
|
+
Own everything a cheaper executor must get right **mechanically**: routing, finding schema,
|
|
28
|
+
severity normalization, preservation classification, gate policy, apply loop, artifacts. The four
|
|
29
|
+
taste lenses keep their principles and judgments; this skill turns their prose into gated,
|
|
30
|
+
checkable, reversible work.
|
|
31
|
+
|
|
32
|
+
The product is **analysis first, edits second**: with the default `--fix none` the run writes two
|
|
33
|
+
artifacts and performs no edit at all.
|
|
34
|
+
|
|
35
|
+
Backs `/sp:dev-refactor`. Authority: `docs/design/dev-refactor-command.md` (§2 phases, §4–§8
|
|
36
|
+
contracts, §11 invariants).
|
|
37
|
+
|
|
38
|
+
## Inputs
|
|
39
|
+
|
|
40
|
+
| Input | Meaning | Default |
|
|
41
|
+
| --- | --- | --- |
|
|
42
|
+
| `--scope <path>` | Path bound; no edit may land outside it. | working tree (recent changes) |
|
|
43
|
+
| `--focus <api\|architect\|tests\|ui\|auto>` | Lens set; comma list allowed. | `auto` |
|
|
44
|
+
| `--fix <none\|blockers-first\|all>` | Apply policy. | `none` |
|
|
45
|
+
| `--check <cmd>` | Baseline and per-fix verification command. | project gate (`bun run spur-check` when present) |
|
|
46
|
+
| `--auto` | Skips **objective** gates only. | off |
|
|
47
|
+
| free-text description | Steering for the lenses (e.g. "pagination consistency"). | — |
|
|
48
|
+
|
|
49
|
+
## Stop rules (design §11 — verbatim, non-negotiable)
|
|
50
|
+
|
|
51
|
+
1. **No edit outside `--scope`; no edit before a green baseline.**
|
|
52
|
+
2. **`cutting` / `breaking` findings are applied only after an explicit operator `yes`.**
|
|
53
|
+
3. **Tests are never removed or weakened by an `auto` fix.**
|
|
54
|
+
4. **Every finding carries `file:line` evidence inside scope.**
|
|
55
|
+
5. **The command file contains no orchestration logic** — this skill owns the orchestration, the
|
|
56
|
+
command stays a thin wrapper.
|
|
57
|
+
|
|
58
|
+
Violating any stop rule ends the run as a failure; do not "finish" a partial apply.
|
|
59
|
+
|
|
60
|
+
## Phase 0 — Resolve
|
|
61
|
+
|
|
62
|
+
1. Resolve `--scope` to a concrete path; list the in-scope files.
|
|
63
|
+
2. Resolve the lens set: if `--focus auto`, classify per
|
|
64
|
+
[focus-detection.md](./references/focus-detection.md) (ordered globs, first match per file,
|
|
65
|
+
union across files); otherwise use the given lens list.
|
|
66
|
+
3. **Report the lens set before any lens runs** (one line, per focus-detection.md).
|
|
67
|
+
4. Run the `--check` command for a **green baseline**. Red baseline = hard stop: write the report,
|
|
68
|
+
apply nothing, end the run.
|
|
69
|
+
|
|
70
|
+
## Phase 1 — Analyze (dispatch lenses)
|
|
71
|
+
|
|
72
|
+
For each lens in the resolved set, dispatch the taste skill and collect native findings:
|
|
73
|
+
|
|
74
|
+
```text
|
|
75
|
+
Skill(sp:taste-refactoring-tests) # if tests in lens set
|
|
76
|
+
Skill(sp:taste-refactoring-ui) # if ui in lens set
|
|
77
|
+
Skill(sp:taste-refactoring-api) # if api in lens set
|
|
78
|
+
Skill(sp:taste-refactoring-architect) # if architect in lens set
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
Pass each lens the scope path and the free-text description. Read the lens's `## Spur contract`
|
|
82
|
+
section, then map its native findings into the shared schema
|
|
83
|
+
([finding-schema.md](./references/finding-schema.md)): lens-native rung is kept in `rung`,
|
|
84
|
+
severity is translated through the §5 map, and every finding carries a **preserved-behavior
|
|
85
|
+
inventory** input from the lens (endpoints / modules / test-protected behaviors / UI controls in
|
|
86
|
+
scope) before any proposal.
|
|
87
|
+
|
|
88
|
+
Classification rules when mapping (enforced by the structural check):
|
|
89
|
+
|
|
90
|
+
- `preservation`: does the proposal remove a user-visible feature/test/endpoint/control
|
|
91
|
+
(`cutting`), change a contract or observable behavior for a consumer (`breaking`), or keep
|
|
92
|
+
behavior identical (`preserving`)?
|
|
93
|
+
- `fix_eligibility`: `auto` only for mechanical, behavior-preserving, checkable changes;
|
|
94
|
+
`confirm` when an operator answer is needed; `suggest` for report-only plans.
|
|
95
|
+
- A `cutting`/`breaking` finding is never below P2 and never `auto`.
|
|
96
|
+
|
|
97
|
+
## Phase 2 — Merge
|
|
98
|
+
|
|
99
|
+
1. Dedupe by `(evidence file, line span, rung)`: identical span + rung from two lenses is one
|
|
100
|
+
finding — keep the higher severity, keep the first lens's `id` (renumber gaps afterward).
|
|
101
|
+
2. Rank P1 → P4 (then focus order tests, ui, api, architect).
|
|
102
|
+
3. Re-assert scope: every `evidence[].file` is inside `--scope`; drop or fail findings that are
|
|
103
|
+
not (stop rule 4).
|
|
104
|
+
|
|
105
|
+
## Phase 3 — Gate
|
|
106
|
+
|
|
107
|
+
| Gate | Class | `--auto` | Headless (`--agent auto\|name`) |
|
|
108
|
+
| --- | --- | --- | --- |
|
|
109
|
+
| Confirm resolved scope + lens set | objective | skipped | skipped |
|
|
110
|
+
| Confirm apply batch of `auto` findings | objective | skipped | skipped |
|
|
111
|
+
| Each `cutting` / `breaking` finding | **taste** | **pauses** | no pause: `status: deferred`, listed as SUGGEST in the report |
|
|
112
|
+
| Baseline check red | hard stop | stop, report | stop, report |
|
|
113
|
+
|
|
114
|
+
- **Objective gates** ask yes/no about facts (scope, lens set, batch contents); `--auto` answers
|
|
115
|
+
them from the resolved state.
|
|
116
|
+
- **Taste gates** — every `cutting` or `breaking` finding — **always pause for an explicit
|
|
117
|
+
operator answer** in an interactive session, regardless of `--auto`. Under a headless executor
|
|
118
|
+
there is no operator to ask: set `status: deferred` and list the finding as SUGGEST in the
|
|
119
|
+
report. A headless run never applies a cutting/breaking finding.
|
|
120
|
+
- A declined taste gate marks the finding `status: rejected` (operator "no") — never silently
|
|
121
|
+
dropped.
|
|
122
|
+
|
|
123
|
+
## Phase 4 — Apply (fix ladder)
|
|
124
|
+
|
|
125
|
+
Execute `--fix` per [fix-ladder.md](./references/fix-ladder.md): green baseline first, one finding
|
|
126
|
+
at a time, re-run `--check` after each, revert only the failed finding's own edits (reverse-apply
|
|
127
|
+
its hunks — never a file-level checkout; see fix-ladder.md) and mark
|
|
128
|
+
`reverted`. `none` skips this phase entirely (both artifacts still written).
|
|
129
|
+
|
|
130
|
+
## Phase 5 — Report
|
|
131
|
+
|
|
132
|
+
Run the structural check from [finding-schema.md](./references/finding-schema.md), then write both
|
|
133
|
+
artifacts:
|
|
134
|
+
|
|
135
|
+
1. `.spur/run/<run-id>-refactor-findings.json` — the bare findings array (validated schema).
|
|
136
|
+
2. `.spur/run/<run-id>-refactor-report.md` — containing:
|
|
137
|
+
- the resolved **lens set** (and per-file counts for `auto`),
|
|
138
|
+
- a **P1–P4 findings table** with `file:line`,
|
|
139
|
+
- a **preservation summary** (preserving / cutting / breaking counts),
|
|
140
|
+
- **applied / reverted / deferred** lists (plus rejected when taste gates declined findings).
|
|
141
|
+
|
|
142
|
+
`<run-id>` is the enclosing pipeline run id when invoked inside one, otherwise
|
|
143
|
+
`refactor-<yyyymmdd>-<hhmmss>`. `--fix none` still writes both artifacts.
|
|
144
|
+
|
|
145
|
+
## Cheaper executor — minimum reading
|
|
146
|
+
|
|
147
|
+
A reduced-context executor may run this skill with exactly these files:
|
|
148
|
+
|
|
149
|
+
1. this `SKILL.md` (phases, gates, stop rules),
|
|
150
|
+
2. `references/finding-schema.md` (fields, severity map, structural check),
|
|
151
|
+
3. `references/fix-ladder.md` (apply loop, revert rule),
|
|
152
|
+
4. `references/focus-detection.md` (globs, report-before-run),
|
|
153
|
+
5. each selected lens's `## Spur contract` section only.
|
|
154
|
+
|
|
155
|
+
Do not skip the structural check or the stop rules; everything else is compressible.
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
# Refactor finding schema
|
|
2
|
+
|
|
3
|
+
Every finding produced by a taste lens is normalized into one shared object before the coordinator
|
|
4
|
+
merges, gates, or applies anything. The findings artifact is a **bare JSON array** of these objects
|
|
5
|
+
at `.spur/run/<run-id>-refactor-findings.json`, machine-checked by
|
|
6
|
+
[`refactor-finding.schema.json`](./refactor-finding.schema.json) (JSON Schema draft-07).
|
|
7
|
+
|
|
8
|
+
Authority: `docs/design/dev-refactor-command.md` §4 (fields) and §5 (severity map).
|
|
9
|
+
|
|
10
|
+
## Fields
|
|
11
|
+
|
|
12
|
+
| Field | Type | Values / notes |
|
|
13
|
+
| --- | --- | --- |
|
|
14
|
+
| `id` | string | `RF-<focus>-<nnn>` (e.g. `RF-api-001`) |
|
|
15
|
+
| `focus` | enum | `api` `architect` `tests` `ui` — the lens that produced the finding |
|
|
16
|
+
| `severity` | enum | `P1` `P2` `P3` `P4` (see the map below; `P0` does not exist in this schema) |
|
|
17
|
+
| `rung` | string | lens-native rung kept verbatim (`A0–A7`, `T0–T7`, api compatibility class, ui pass name) |
|
|
18
|
+
| `title` | string | one line |
|
|
19
|
+
| `evidence` | `{file, line}[]` | at least one `file:line` inside `--scope` |
|
|
20
|
+
| `preservation` | enum | `preserving` (behavior identical) · `cutting` (a user-visible feature/test/endpoint/control is removed) · `breaking` (contract or behavior changes for a consumer) |
|
|
21
|
+
| `fix_eligibility` | enum | `auto` (mechanical, behavior-preserving, checkable) · `confirm` (needs an operator answer) · `suggest` (report only) |
|
|
22
|
+
| `proposal` | string | what to change, imperative |
|
|
23
|
+
| `verify` | string | command or check that proves the fix (defaults to the `--check` command) |
|
|
24
|
+
| `status` | enum | `open` `applied` `reverted` `deferred` `rejected` |
|
|
25
|
+
|
|
26
|
+
Severity authority: `plugins/sp/agents/super-reviewer.md` (P1 blocker · P2 major · P3 minor ·
|
|
27
|
+
P4 advisory). The map below is the only place lens-native severities are translated.
|
|
28
|
+
|
|
29
|
+
## Lens-native → P1–P4 severity map
|
|
30
|
+
|
|
31
|
+
| Lens | Native scale | → P1 (blocker) | → P2 (major) | → P3 (minor) | → P4 (advisory) |
|
|
32
|
+
| --- | --- | --- | --- | --- | --- |
|
|
33
|
+
| tests | P0–P3 | P0 | P1 | P2 | P3 |
|
|
34
|
+
| ui | P0–P3 | P0 | P1 | P2 | P3 |
|
|
35
|
+
| architect | A-ladder + 7-axis scores | any axis ≤1 with a correctness/safety consequence | axis ≤2 or A5–A7 seam problems | A3–A4 | A0–A2, ADR candidates, deferred questions |
|
|
36
|
+
| api | compatibility class | `breaking` change already shipped or contract ambiguity that corrupts data | `risky` inconsistency across ≥2 endpoints | `additive` cleanups | naming/docs |
|
|
37
|
+
|
|
38
|
+
Rule (structural, enforced by the check below): a `cutting` or `breaking` finding is **never below
|
|
39
|
+
P2** and **never `fix_eligibility: auto`**.
|
|
40
|
+
|
|
41
|
+
## Structural check (no new dependency)
|
|
42
|
+
|
|
43
|
+
Run this documented `bun -e` snippet before writing the report — it asserts required keys and enum
|
|
44
|
+
membership and fails loudly on the two hard rules above (it must reject `severity: P0` and any
|
|
45
|
+
`cutting`/`breaking` finding marked `fix_eligibility: auto`):
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
bun -e '
|
|
49
|
+
const f = require("fs");
|
|
50
|
+
const FINDINGS = JSON.parse(f.readFileSync(process.argv[process.argv.length - 1], "utf8"));
|
|
51
|
+
const REQ = ["id", "focus", "severity", "rung", "title", "evidence", "preservation", "fix_eligibility", "proposal", "verify", "status"];
|
|
52
|
+
const ENUM = {
|
|
53
|
+
focus: ["api", "architect", "tests", "ui"],
|
|
54
|
+
severity: ["P1", "P2", "P3", "P4"],
|
|
55
|
+
preservation: ["preserving", "cutting", "breaking"],
|
|
56
|
+
fix_eligibility: ["auto", "confirm", "suggest"],
|
|
57
|
+
status: ["open", "applied", "reverted", "deferred", "rejected"],
|
|
58
|
+
};
|
|
59
|
+
if (!Array.isArray(FINDINGS)) throw new Error("findings artifact must be a bare JSON array");
|
|
60
|
+
for (const x of FINDINGS) {
|
|
61
|
+
const id = x.id ?? "(no id)";
|
|
62
|
+
for (const k of REQ) if (!(k in x)) throw new Error(`${id}: missing required key ${k}`);
|
|
63
|
+
for (const [k, vals] of Object.entries(ENUM)) if (!vals.includes(x[k])) throw new Error(`${id}: ${k}=${JSON.stringify(x[k])} not in ${vals.join("|")}`);
|
|
64
|
+
if (!Array.isArray(x.evidence) || x.evidence.length < 1 || !x.evidence.every((e) => e && typeof e.file === "string" && e.file.length > 0 && Number.isInteger(e.line) && e.line >= 1)) throw new Error(`${id}: evidence must be a non-empty [{file, line}] array`);
|
|
65
|
+
if ((x.preservation === "cutting" || x.preservation === "breaking") && x.severity !== "P1" && x.severity !== "P2") throw new Error(`${id}: cutting/breaking is never below P2`);
|
|
66
|
+
if ((x.preservation === "cutting" || x.preservation === "breaking") && x.fix_eligibility === "auto") throw new Error(`${id}: cutting/breaking is never fix_eligibility auto`);
|
|
67
|
+
}
|
|
68
|
+
console.log(`refactor-findings: ${FINDINGS.length} finding(s) structurally valid`);
|
|
69
|
+
' .spur/run/<run-id>-refactor-findings.json
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
The coordinator runs this check after mapping and again after every status change, before the
|
|
73
|
+
report is written. A failed check is a hard stop — the artifact is not written and the run reports
|
|
74
|
+
the failure instead of applying anything.
|