@popoverai/dotrequirements 0.26.0 → 0.26.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/codebase-to-spec/area-name.d.ts +13 -0
- package/dist/codebase-to-spec/area-name.js +18 -0
- package/dist/codebase-to-spec/cache.d.ts +31 -0
- package/dist/codebase-to-spec/cache.js +29 -1
- package/dist/codebase-to-spec/compose.d.ts +7 -4
- package/dist/codebase-to-spec/compose.js +10 -21
- package/dist/codebase-to-spec/dispatch.js +1 -1
- package/dist/codebase-to-spec/present.js +11 -7
- package/dist/codebase-to-spec/prompts/planner-initial.d.ts +3 -2
- package/dist/codebase-to-spec/prompts/planner-initial.js +3 -2
- package/dist/codebase-to-spec/renumber.d.ts +52 -0
- package/dist/codebase-to-spec/renumber.js +105 -0
- package/dist/codebase-to-spec/schemas.d.ts +35 -240
- package/dist/codebase-to-spec/schemas.js +5 -173
- package/dist/codebase-to-spec/validate.js +3 -2
- package/dist/commands/acceptance-test.js +4 -2
- package/dist/commands/ai-setup.js +97 -59
- package/dist/commands/codebase-to-spec/dispatch-editor.js +1 -1
- package/dist/commands/codebase-to-spec/dispatch-spec.js +1 -1
- package/dist/commands/codebase-to-spec/index.js +7 -103
- package/dist/commands/codebase-to-spec/pack.js +8 -1
- package/dist/commands/codebase-to-spec/present-orchestrator.d.ts +3 -1
- package/dist/commands/codebase-to-spec/present-orchestrator.js +11 -4
- package/dist/commands/get.js +6 -2
- package/dist/commands/init.js +7 -5
- package/dist/commands/link-resolution.d.ts +3 -1
- package/dist/commands/link-resolution.js +4 -2
- package/dist/commands/pull.js +36 -3
- package/dist/commands/push.js +54 -16
- package/dist/commands/report.js +18 -3
- package/dist/commands/review-test.js +16 -8
- package/dist/commands/tests-for.js +13 -13
- package/dist/commands/validate.js +14 -14
- package/dist/harness/cache.d.ts +19 -3
- package/dist/harness/cache.js +38 -12
- package/dist/harness/finalize.js +33 -1
- package/dist/harness/index.js +16 -9
- package/dist/harness/requirementsLoader.js +12 -0
- package/dist/harness/tracking.d.ts +17 -2
- package/dist/harness/tracking.js +83 -9
- package/dist/mcp/handlers/authoring.js +13 -4
- package/dist/mcp/handlers/get.js +7 -3
- package/dist/mcp/handlers/push.js +59 -13
- package/dist/mcp/handlers/review.d.ts +1 -0
- package/dist/mcp/handlers/review.js +58 -15
- package/dist/mcp/handlers/test-mapping.js +47 -12
- package/dist/mcp/handlers/types.d.ts +14 -0
- package/dist/mcp/handlers/types.js +27 -0
- package/dist/mcp/index.js +4 -0
- package/dist/push/core.d.ts +50 -0
- package/dist/push/core.js +149 -11
- package/dist/push/index.d.ts +1 -1
- package/dist/push/index.js +1 -1
- package/dist/requirements/cloud-coverage.d.ts +12 -2
- package/dist/requirements/cloud-coverage.js +30 -3
- package/dist/requirements/grep.d.ts +7 -2
- package/dist/requirements/grep.js +75 -47
- package/dist/schema/builder.d.ts +1 -1
- package/dist/schema/builder.js +13 -0
- package/dist/schema/conversions.d.ts +7 -2
- package/dist/schema/conversions.js +13 -4
- package/dist/schema/parser-core.d.ts +28 -0
- package/dist/schema/parser-core.js +80 -9
- package/dist/schema/parser.d.ts +8 -26
- package/dist/schema/parser.js +23 -251
- package/dist/schema/resolver.js +18 -8
- package/dist/templates/skills/codebase-to-spec/SKILL.md +10 -4
- package/dist/utils/env.js +17 -1
- package/dist/utils/oauth-flow.js +8 -0
- package/dist/utils/project-settings.d.ts +4 -0
- package/dist/utils/project-settings.js +14 -1
- package/package.json +1 -1
- package/dist/codebase-to-spec/edit-loop.d.ts +0 -54
- package/dist/codebase-to-spec/edit-loop.js +0 -195
- package/dist/codebase-to-spec/editor.d.ts +0 -54
- package/dist/codebase-to-spec/editor.js +0 -74
- package/dist/codebase-to-spec/fan-out.d.ts +0 -63
- package/dist/codebase-to-spec/fan-out.js +0 -215
- package/dist/codebase-to-spec/outline-review-loop.d.ts +0 -51
- package/dist/codebase-to-spec/outline-review-loop.js +0 -187
- package/dist/codebase-to-spec/planner.d.ts +0 -41
- package/dist/codebase-to-spec/planner.js +0 -76
- package/dist/codebase-to-spec/prompts/outline-reviewer.d.ts +0 -12
- package/dist/codebase-to-spec/prompts/outline-reviewer.js +0 -89
- package/dist/codebase-to-spec/slice.d.ts +0 -49
- package/dist/codebase-to-spec/slice.js +0 -111
- package/dist/codebase-to-spec/specifier.d.ts +0 -60
- package/dist/codebase-to-spec/specifier.js +0 -85
- package/dist/codebase-to-spec/summary.d.ts +0 -51
- package/dist/codebase-to-spec/summary.js +0 -183
- package/dist/commands/codebase-to-spec/compose.d.ts +0 -14
- package/dist/commands/codebase-to-spec/compose.js +0 -57
- package/dist/commands/codebase-to-spec/edit-loop.d.ts +0 -16
- package/dist/commands/codebase-to-spec/edit-loop.js +0 -83
- package/dist/commands/codebase-to-spec/fan-out.d.ts +0 -19
- package/dist/commands/codebase-to-spec/fan-out.js +0 -77
- package/dist/commands/codebase-to-spec/plan-loop.d.ts +0 -26
- package/dist/commands/codebase-to-spec/plan-loop.js +0 -105
- package/dist/commands/codebase-to-spec/present.d.ts +0 -26
- package/dist/commands/codebase-to-spec/present.js +0 -97
- package/dist/commands/codebase-to-spec/run.d.ts +0 -20
- package/dist/commands/codebase-to-spec/run.js +0 -86
- package/dist/commands/codebase-to-spec/specify-area.d.ts +0 -18
- package/dist/commands/codebase-to-spec/specify-area.js +0 -82
|
@@ -1,215 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Specifier fan-out orchestrator.
|
|
3
|
-
*
|
|
4
|
-
* For each area in the outline, extract a slice, spawn a specifier worker,
|
|
5
|
-
* and write the resulting partial to the cache. Workers run in parallel up
|
|
6
|
-
* to a bounded concurrency limit. On per-area failure (no partial written,
|
|
7
|
-
* empty partial), the worker is retried once; on second failure, the area
|
|
8
|
-
* is recorded as failed for surfacing to the user.
|
|
9
|
-
*
|
|
10
|
-
* Requirements covered:
|
|
11
|
-
* - CTS-SPEC-1.2: bounded concurrent execution
|
|
12
|
-
* - CTS-SPEC-5: failed/empty specifier output is detected and retried
|
|
13
|
-
* - CTS-OBSERVE-2.0: failure details are logged with area name + cause
|
|
14
|
-
* - CTS-RESUME-1.2: existing non-empty partial means the specifier is skipped
|
|
15
|
-
*/
|
|
16
|
-
import { existsSync, mkdirSync, statSync, writeFileSync } from "node:fs";
|
|
17
|
-
import { dirname } from "node:path";
|
|
18
|
-
import { extractSlice } from "./slice.js";
|
|
19
|
-
import { runSpecifier } from "./specifier.js";
|
|
20
|
-
export const DEFAULT_FAN_OUT_CONCURRENCY = 8;
|
|
21
|
-
/**
|
|
22
|
-
* Convert an area name to a stable filesystem-safe identifier.
|
|
23
|
-
* E.g., "Reading & Editing Files" → "reading-editing-files"
|
|
24
|
-
*/
|
|
25
|
-
export function sanitizeAreaName(name) {
|
|
26
|
-
return (name
|
|
27
|
-
.replace(/[^A-Za-z0-9]+/g, "-")
|
|
28
|
-
.replace(/^-+|-+$/g, "")
|
|
29
|
-
.toLowerCase() || "area");
|
|
30
|
-
}
|
|
31
|
-
/**
|
|
32
|
-
* Run a bounded-concurrency loop over items. Workers pull items off a shared
|
|
33
|
-
* counter; at most `limit` workers run at a time.
|
|
34
|
-
*/
|
|
35
|
-
async function withLimit(items, limit, fn) {
|
|
36
|
-
let next = 0;
|
|
37
|
-
async function worker() {
|
|
38
|
-
while (true) {
|
|
39
|
-
const i = next++;
|
|
40
|
-
if (i >= items.length)
|
|
41
|
-
return;
|
|
42
|
-
await fn(items[i], i);
|
|
43
|
-
}
|
|
44
|
-
}
|
|
45
|
-
const workerCount = Math.min(Math.max(1, limit), items.length);
|
|
46
|
-
const workers = Array.from({ length: workerCount }, () => worker());
|
|
47
|
-
await Promise.all(workers);
|
|
48
|
-
}
|
|
49
|
-
function partialIsUsable(path) {
|
|
50
|
-
if (!existsSync(path))
|
|
51
|
-
return false;
|
|
52
|
-
try {
|
|
53
|
-
return statSync(path).size > 0;
|
|
54
|
-
}
|
|
55
|
-
catch {
|
|
56
|
-
return false;
|
|
57
|
-
}
|
|
58
|
-
}
|
|
59
|
-
/**
|
|
60
|
-
* Run the fan-out: extract slices, invoke specifiers, retry on failure.
|
|
61
|
-
*/
|
|
62
|
-
export async function runFanOut(options) {
|
|
63
|
-
const { outline, fullPackPath, partialPathFor, slicePathFor, validateCommand, styleCheckCommand, addDirs, concurrency = DEFAULT_FAN_OUT_CONCURRENCY, runner, model, progress, } = options;
|
|
64
|
-
if (!existsSync(fullPackPath)) {
|
|
65
|
-
throw new Error(`Missing full pack at ${fullPackPath}`);
|
|
66
|
-
}
|
|
67
|
-
// Sanitization is lossy ("Read & Write" and "read/write" both become
|
|
68
|
-
// "read-write"). If two areas collide, they'd share a partial file path and
|
|
69
|
-
// silently overwrite each other downstream. Catch this up front.
|
|
70
|
-
const sanitizedNames = new Map();
|
|
71
|
-
for (const area of outline.areas) {
|
|
72
|
-
const key = sanitizeAreaName(area.name);
|
|
73
|
-
const bucket = sanitizedNames.get(key) ?? [];
|
|
74
|
-
bucket.push(area.name);
|
|
75
|
-
sanitizedNames.set(key, bucket);
|
|
76
|
-
}
|
|
77
|
-
const collisions = [...sanitizedNames.entries()].filter(([, names]) => names.length > 1);
|
|
78
|
-
if (collisions.length > 0) {
|
|
79
|
-
const detail = collisions
|
|
80
|
-
.map(([key, names]) => ` "${key}" ← ${names.map((n) => `"${n}"`).join(", ")}`)
|
|
81
|
-
.join("\n");
|
|
82
|
-
throw new Error(`Outline contains area names that collide after sanitization. Rename them so each produces a unique partial file path:\n${detail}`);
|
|
83
|
-
}
|
|
84
|
-
const outcomes = outline.areas.map((area) => ({
|
|
85
|
-
area,
|
|
86
|
-
status: "failed",
|
|
87
|
-
attempts: 0,
|
|
88
|
-
partialPath: partialPathFor(sanitizeAreaName(area.name)),
|
|
89
|
-
}));
|
|
90
|
-
// Ensure partials directory exists for each area.
|
|
91
|
-
for (const outcome of outcomes) {
|
|
92
|
-
mkdirSync(dirname(outcome.partialPath), { recursive: true });
|
|
93
|
-
}
|
|
94
|
-
await withLimit(outcomes, concurrency, async (outcome) => {
|
|
95
|
-
const sanitized = sanitizeAreaName(outcome.area.name);
|
|
96
|
-
const slicePath = slicePathFor(sanitized);
|
|
97
|
-
// CTS-RESUME-1.2: skip if a non-empty partial exists from a prior run.
|
|
98
|
-
if (partialIsUsable(outcome.partialPath)) {
|
|
99
|
-
outcome.status = "skipped-resume";
|
|
100
|
-
outcome.attempts = 0;
|
|
101
|
-
progress?.emit({
|
|
102
|
-
stage: "specify",
|
|
103
|
-
step: outcome.area.prefix,
|
|
104
|
-
message: `(resume) skipping ${outcome.area.name} — partial exists`,
|
|
105
|
-
});
|
|
106
|
-
return;
|
|
107
|
-
}
|
|
108
|
-
// Extract this area's slice from the full pack.
|
|
109
|
-
if (outcome.area.files.length === 0) {
|
|
110
|
-
// No files assigned — write a stub partial so compose has something.
|
|
111
|
-
writeFileSync(outcome.partialPath, "_(no files assigned to this area)_\n", "utf-8");
|
|
112
|
-
outcome.status = "failed";
|
|
113
|
-
outcome.attempts = 0;
|
|
114
|
-
outcome.failureCause = "no files assigned in outline";
|
|
115
|
-
progress?.emit({
|
|
116
|
-
stage: "specify",
|
|
117
|
-
step: outcome.area.prefix,
|
|
118
|
-
message: `!! ${outcome.area.name} has no files; marking as failed`,
|
|
119
|
-
});
|
|
120
|
-
return;
|
|
121
|
-
}
|
|
122
|
-
let sliceMatched = 0;
|
|
123
|
-
try {
|
|
124
|
-
const sliceResult = extractSlice(fullPackPath, slicePath, outcome.area.files);
|
|
125
|
-
sliceMatched = sliceResult.matched;
|
|
126
|
-
}
|
|
127
|
-
catch (err) {
|
|
128
|
-
outcome.failureCause = `slice extraction failed: ${err instanceof Error ? err.message : String(err)}`;
|
|
129
|
-
progress?.emit({
|
|
130
|
-
stage: "specify",
|
|
131
|
-
step: outcome.area.prefix,
|
|
132
|
-
message: `!! ${outcome.failureCause}`,
|
|
133
|
-
});
|
|
134
|
-
return;
|
|
135
|
-
}
|
|
136
|
-
if (sliceMatched === 0) {
|
|
137
|
-
progress?.emit({
|
|
138
|
-
stage: "specify",
|
|
139
|
-
step: outcome.area.prefix,
|
|
140
|
-
message: `(warning) slice for ${outcome.area.name} matched 0 files; specifier will need to fall back to full-pack reads`,
|
|
141
|
-
});
|
|
142
|
-
}
|
|
143
|
-
// Attempt 1
|
|
144
|
-
progress?.emit({
|
|
145
|
-
stage: "specify",
|
|
146
|
-
step: outcome.area.prefix,
|
|
147
|
-
message: `Running specifier for ${outcome.area.name} (attempt 1/2)`,
|
|
148
|
-
});
|
|
149
|
-
outcome.attempts = 1;
|
|
150
|
-
const r1 = await runSpecifier({
|
|
151
|
-
outline,
|
|
152
|
-
area: outcome.area,
|
|
153
|
-
slicePath,
|
|
154
|
-
fullPackPath,
|
|
155
|
-
partialPath: outcome.partialPath,
|
|
156
|
-
validateCommand,
|
|
157
|
-
styleCheckCommand,
|
|
158
|
-
addDirs,
|
|
159
|
-
runner,
|
|
160
|
-
model,
|
|
161
|
-
});
|
|
162
|
-
if (r1.wrotePartial) {
|
|
163
|
-
outcome.status = "completed";
|
|
164
|
-
progress?.emit({
|
|
165
|
-
stage: "specify",
|
|
166
|
-
step: outcome.area.prefix,
|
|
167
|
-
message: `${outcome.area.name} done (${r1.partialBytes} bytes)`,
|
|
168
|
-
});
|
|
169
|
-
return;
|
|
170
|
-
}
|
|
171
|
-
// Attempt 2 (retry once on no/empty partial)
|
|
172
|
-
progress?.emit({
|
|
173
|
-
stage: "specify",
|
|
174
|
-
step: outcome.area.prefix,
|
|
175
|
-
message: `!! ${outcome.area.name} produced no partial; retrying (attempt 2/2)`,
|
|
176
|
-
});
|
|
177
|
-
outcome.attempts = 2;
|
|
178
|
-
const r2 = await runSpecifier({
|
|
179
|
-
outline,
|
|
180
|
-
area: outcome.area,
|
|
181
|
-
slicePath,
|
|
182
|
-
fullPackPath,
|
|
183
|
-
partialPath: outcome.partialPath,
|
|
184
|
-
validateCommand,
|
|
185
|
-
styleCheckCommand,
|
|
186
|
-
addDirs,
|
|
187
|
-
runner,
|
|
188
|
-
model,
|
|
189
|
-
});
|
|
190
|
-
if (r2.wrotePartial) {
|
|
191
|
-
outcome.status = "completed";
|
|
192
|
-
progress?.emit({
|
|
193
|
-
stage: "specify",
|
|
194
|
-
step: outcome.area.prefix,
|
|
195
|
-
message: `${outcome.area.name} done on retry (${r2.partialBytes} bytes)`,
|
|
196
|
-
});
|
|
197
|
-
return;
|
|
198
|
-
}
|
|
199
|
-
// Two failures — record and continue. CTS-SPEC-5 says other areas still proceed.
|
|
200
|
-
outcome.status = "failed";
|
|
201
|
-
outcome.failureCause =
|
|
202
|
-
r2.exitCode !== 0
|
|
203
|
-
? `specifier exited with code ${r2.exitCode}: ${r2.stderr.slice(0, 200)}`
|
|
204
|
-
: "specifier wrote no partial (or empty partial) on both attempts";
|
|
205
|
-
writeFileSync(outcome.partialPath, `_(missing partial — specifier failed after retry: ${outcome.failureCause})_\n`, "utf-8");
|
|
206
|
-
progress?.emit({
|
|
207
|
-
stage: "specify",
|
|
208
|
-
step: outcome.area.prefix,
|
|
209
|
-
message: `!! ${outcome.area.name} failed: ${outcome.failureCause}`,
|
|
210
|
-
});
|
|
211
|
-
});
|
|
212
|
-
const allSucceeded = outcomes.every((o) => o.status === "completed" || o.status === "skipped-resume");
|
|
213
|
-
return { outcomes, allSucceeded };
|
|
214
|
-
}
|
|
215
|
-
//# sourceMappingURL=fan-out.js.map
|
|
@@ -1,51 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Outline review loop: planner → reviewer → revise → reviewer → ... → approved.
|
|
3
|
-
*
|
|
4
|
-
* The reviewer runs in a SINGLE stateful claude session across turns (so it
|
|
5
|
-
* can compare prior outlines to the latest one). The planner is stateless —
|
|
6
|
-
* each invocation is fresh.
|
|
7
|
-
*
|
|
8
|
-
* Requirements covered:
|
|
9
|
-
* - CTS-PLAN-2: stateful reviewer session
|
|
10
|
-
* - CTS-PLAN-3: approved → proceed
|
|
11
|
-
* - CTS-PLAN-4: approved-with-revisions → one revision pass, then proceed
|
|
12
|
-
* - CTS-PLAN-5: requires-another-review → revision loop
|
|
13
|
-
* - CTS-PLAN-6: max-turns + convergence nudge
|
|
14
|
-
*/
|
|
15
|
-
import { type ClaudeRunner } from "./claude.js";
|
|
16
|
-
import type { ProgressEmitter } from "./progress.js";
|
|
17
|
-
import { type Outline, type OutlineReview } from "./schemas.js";
|
|
18
|
-
export declare const DEFAULT_OUTLINE_LOOP_MAX_TURNS = 3;
|
|
19
|
-
export interface OutlineReviewLoopOptions {
|
|
20
|
-
overviewPath: string;
|
|
21
|
-
addDirs: string[];
|
|
22
|
-
maxTurns?: number;
|
|
23
|
-
runner?: ClaudeRunner;
|
|
24
|
-
model?: string;
|
|
25
|
-
progress?: ProgressEmitter;
|
|
26
|
-
/** Override session id (for tests and reproducibility). Otherwise random. */
|
|
27
|
-
sessionId?: string;
|
|
28
|
-
}
|
|
29
|
-
export interface OutlineReviewLoopResult {
|
|
30
|
-
/** The final outline after the loop converges (or hits max-turns). */
|
|
31
|
-
outline: Outline;
|
|
32
|
-
/** Per-turn outline snapshots (turn 1 is the initial, then each revision). */
|
|
33
|
-
outlinesByTurn: Outline[];
|
|
34
|
-
/** Per-turn reviewer results. */
|
|
35
|
-
reviewsByTurn: OutlineReview[];
|
|
36
|
-
/** The final reviewer verdict. */
|
|
37
|
-
finalVerdict: OutlineReview["verdict"];
|
|
38
|
-
/** Number of review turns used (= length of reviewsByTurn). */
|
|
39
|
-
turnsUsed: number;
|
|
40
|
-
/** True if the loop hit max-turns without converging. */
|
|
41
|
-
hitMaxTurns: boolean;
|
|
42
|
-
}
|
|
43
|
-
/**
|
|
44
|
-
* Run the outline review loop end-to-end.
|
|
45
|
-
*
|
|
46
|
-
* Returns the final outline (after applying mechanical revisions if the
|
|
47
|
-
* verdict was `approved-with-revisions`), the per-turn trace, and the final
|
|
48
|
-
* verdict.
|
|
49
|
-
*/
|
|
50
|
-
export declare function runOutlineReviewLoop(options: OutlineReviewLoopOptions): Promise<OutlineReviewLoopResult>;
|
|
51
|
-
//# sourceMappingURL=outline-review-loop.d.ts.map
|
|
@@ -1,187 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Outline review loop: planner → reviewer → revise → reviewer → ... → approved.
|
|
3
|
-
*
|
|
4
|
-
* The reviewer runs in a SINGLE stateful claude session across turns (so it
|
|
5
|
-
* can compare prior outlines to the latest one). The planner is stateless —
|
|
6
|
-
* each invocation is fresh.
|
|
7
|
-
*
|
|
8
|
-
* Requirements covered:
|
|
9
|
-
* - CTS-PLAN-2: stateful reviewer session
|
|
10
|
-
* - CTS-PLAN-3: approved → proceed
|
|
11
|
-
* - CTS-PLAN-4: approved-with-revisions → one revision pass, then proceed
|
|
12
|
-
* - CTS-PLAN-5: requires-another-review → revision loop
|
|
13
|
-
* - CTS-PLAN-6: max-turns + convergence nudge
|
|
14
|
-
*/
|
|
15
|
-
import { randomUUID } from "node:crypto";
|
|
16
|
-
import { runClaude } from "./claude.js";
|
|
17
|
-
import { runInitialPlanner, runRevisingPlanner, } from "./planner.js";
|
|
18
|
-
import { OUTLINE_REVIEWER_PROMPT } from "./prompts/outline-reviewer.js";
|
|
19
|
-
import { OUTLINE_REVIEW_JSON_SCHEMA, parseOutlineReview, } from "./schemas.js";
|
|
20
|
-
export const DEFAULT_OUTLINE_LOOP_MAX_TURNS = 3;
|
|
21
|
-
function reviewerUserMessageInitial(overviewPath, outline) {
|
|
22
|
-
return [
|
|
23
|
-
`Initial outline review.`,
|
|
24
|
-
``,
|
|
25
|
-
`The compressed packed codebase is at: ${overviewPath}`,
|
|
26
|
-
``,
|
|
27
|
-
`## Outline`,
|
|
28
|
-
`\`\`\`json`,
|
|
29
|
-
JSON.stringify(outline, null, 2),
|
|
30
|
-
`\`\`\``,
|
|
31
|
-
``,
|
|
32
|
-
`Read the codebase if you need to verify file paths or coverage gaps, then produce your critique JSON.`,
|
|
33
|
-
].join("\n");
|
|
34
|
-
}
|
|
35
|
-
function reviewerUserMessageRevised(turn, overviewPath, outline, nudge) {
|
|
36
|
-
const parts = [
|
|
37
|
-
`Review turn ${turn}. The planner has revised the outline based on your prior feedback.`,
|
|
38
|
-
``,
|
|
39
|
-
];
|
|
40
|
-
if (nudge) {
|
|
41
|
-
parts.push(nudge, ``);
|
|
42
|
-
}
|
|
43
|
-
parts.push(`The codebase pack at ${overviewPath} is unchanged.`, ``, `## Revised outline`, `\`\`\`json`, JSON.stringify(outline, null, 2), `\`\`\``, ``, `Compare this outline to the prior one. Report on what was fixed, what is still open, and what (if anything) was introduced as a new issue. Then produce your critique JSON.`);
|
|
44
|
-
return parts.join("\n");
|
|
45
|
-
}
|
|
46
|
-
function convergenceNudge(turn, maxTurns) {
|
|
47
|
-
return `CONVERGENCE NUDGE: This is review turn ${turn} of ${maxTurns} maximum. Please prioritize convergence. Your job is still to only approve a genuinely-ready outline, but if it is close, lean toward \`approved-with-revisions\` over \`requires-another-review\`. Reserve \`requires-another-review\` for genuine structural problems that the planner has not addressed despite prior feedback.`;
|
|
48
|
-
}
|
|
49
|
-
async function callReviewer(args) {
|
|
50
|
-
const result = await args.runner({
|
|
51
|
-
systemPrompt: OUTLINE_REVIEWER_PROMPT,
|
|
52
|
-
userMessage: args.userMessage,
|
|
53
|
-
model: args.model,
|
|
54
|
-
tools: ["Read", "Grep"],
|
|
55
|
-
addDirs: args.addDirs,
|
|
56
|
-
jsonSchema: OUTLINE_REVIEW_JSON_SCHEMA,
|
|
57
|
-
session: {
|
|
58
|
-
kind: args.isFirstTurn ? "fresh" : "resume",
|
|
59
|
-
sessionId: args.sessionId,
|
|
60
|
-
},
|
|
61
|
-
});
|
|
62
|
-
if (result.exitCode !== 0) {
|
|
63
|
-
throw new Error(`Outline reviewer failed (exit ${result.exitCode}): ${result.stderr || result.stdout}`);
|
|
64
|
-
}
|
|
65
|
-
// On parse failure, retry the same turn once before failing.
|
|
66
|
-
// CTS-PLAN-2 requirement: "on validation failure the CLI logs the parse error
|
|
67
|
-
// and retries the same turn once before failing the loop"
|
|
68
|
-
try {
|
|
69
|
-
return parseOutlineReview(result.stdout);
|
|
70
|
-
}
|
|
71
|
-
catch (err) {
|
|
72
|
-
// Retry once
|
|
73
|
-
const retry = await args.runner({
|
|
74
|
-
systemPrompt: OUTLINE_REVIEWER_PROMPT,
|
|
75
|
-
userMessage: `${args.userMessage}\n\n(Your prior output failed schema validation: ${err instanceof Error ? err.message : String(err)}. Please try again, emitting JSON only.)`,
|
|
76
|
-
model: args.model,
|
|
77
|
-
tools: ["Read", "Grep"],
|
|
78
|
-
addDirs: args.addDirs,
|
|
79
|
-
jsonSchema: OUTLINE_REVIEW_JSON_SCHEMA,
|
|
80
|
-
session: { kind: "resume", sessionId: args.sessionId },
|
|
81
|
-
});
|
|
82
|
-
if (retry.exitCode !== 0) {
|
|
83
|
-
throw new Error(`Outline reviewer retry failed (exit ${retry.exitCode}): ${retry.stderr || retry.stdout}`);
|
|
84
|
-
}
|
|
85
|
-
return parseOutlineReview(retry.stdout);
|
|
86
|
-
}
|
|
87
|
-
}
|
|
88
|
-
/**
|
|
89
|
-
* Run the outline review loop end-to-end.
|
|
90
|
-
*
|
|
91
|
-
* Returns the final outline (after applying mechanical revisions if the
|
|
92
|
-
* verdict was `approved-with-revisions`), the per-turn trace, and the final
|
|
93
|
-
* verdict.
|
|
94
|
-
*/
|
|
95
|
-
export async function runOutlineReviewLoop(options) {
|
|
96
|
-
const { overviewPath, addDirs, maxTurns = DEFAULT_OUTLINE_LOOP_MAX_TURNS, runner = runClaude, model, progress, sessionId = randomUUID(), } = options;
|
|
97
|
-
const plannerCtx = { overviewPath, addDirs, runner, model };
|
|
98
|
-
// Turn 1 — initial planner + initial reviewer
|
|
99
|
-
progress?.emit({
|
|
100
|
-
stage: "plan",
|
|
101
|
-
step: "planner-initial",
|
|
102
|
-
message: "Generating initial outline",
|
|
103
|
-
});
|
|
104
|
-
const outline1 = await runInitialPlanner(plannerCtx);
|
|
105
|
-
const outlinesByTurn = [outline1];
|
|
106
|
-
const reviewsByTurn = [];
|
|
107
|
-
progress?.emit({
|
|
108
|
-
stage: "outline-review",
|
|
109
|
-
step: "turn-1",
|
|
110
|
-
message: `Reviewing initial outline (${outline1.areas.length} areas)`,
|
|
111
|
-
});
|
|
112
|
-
let currentOutline = outline1;
|
|
113
|
-
let review = await callReviewer({
|
|
114
|
-
runner,
|
|
115
|
-
sessionId,
|
|
116
|
-
isFirstTurn: true,
|
|
117
|
-
userMessage: reviewerUserMessageInitial(overviewPath, outline1),
|
|
118
|
-
addDirs,
|
|
119
|
-
model,
|
|
120
|
-
});
|
|
121
|
-
reviewsByTurn.push(review);
|
|
122
|
-
progress?.emit({
|
|
123
|
-
stage: "outline-review",
|
|
124
|
-
step: "turn-1",
|
|
125
|
-
message: `Verdict: ${review.verdict}`,
|
|
126
|
-
data: { verdict: review.verdict },
|
|
127
|
-
});
|
|
128
|
-
let turn = 1;
|
|
129
|
-
while (review.verdict === "requires-another-review" && turn < maxTurns) {
|
|
130
|
-
turn++;
|
|
131
|
-
progress?.emit({
|
|
132
|
-
stage: "plan",
|
|
133
|
-
step: `revise-turn-${turn}`,
|
|
134
|
-
message: "Revising outline based on critique",
|
|
135
|
-
});
|
|
136
|
-
currentOutline = await runRevisingPlanner(plannerCtx, currentOutline, review);
|
|
137
|
-
outlinesByTurn.push(currentOutline);
|
|
138
|
-
const nudge = turn >= maxTurns - 1 && turn !== maxTurns
|
|
139
|
-
? convergenceNudge(turn, maxTurns)
|
|
140
|
-
: undefined;
|
|
141
|
-
progress?.emit({
|
|
142
|
-
stage: "outline-review",
|
|
143
|
-
step: `turn-${turn}`,
|
|
144
|
-
message: `Reviewing revised outline (turn ${turn}/${maxTurns})`,
|
|
145
|
-
data: { turn, maxTurns, nudge: !!nudge },
|
|
146
|
-
});
|
|
147
|
-
review = await callReviewer({
|
|
148
|
-
runner,
|
|
149
|
-
sessionId,
|
|
150
|
-
isFirstTurn: false,
|
|
151
|
-
userMessage: reviewerUserMessageRevised(turn, overviewPath, currentOutline, nudge),
|
|
152
|
-
addDirs,
|
|
153
|
-
model,
|
|
154
|
-
});
|
|
155
|
-
reviewsByTurn.push(review);
|
|
156
|
-
progress?.emit({
|
|
157
|
-
stage: "outline-review",
|
|
158
|
-
step: `turn-${turn}`,
|
|
159
|
-
message: `Verdict: ${review.verdict}`,
|
|
160
|
-
data: { verdict: review.verdict },
|
|
161
|
-
});
|
|
162
|
-
}
|
|
163
|
-
const hitMaxTurns = review.verdict === "requires-another-review" && turn >= maxTurns;
|
|
164
|
-
// If the verdict ended on approved-with-revisions, run the planner once
|
|
165
|
-
// more to apply the listed revisions. Same agent as the revise loop; the
|
|
166
|
-
// difference is that we proceed to fan-out instead of looping back to
|
|
167
|
-
// review.
|
|
168
|
-
if (review.verdict === "approved-with-revisions" &&
|
|
169
|
-
review.revisions.length > 0) {
|
|
170
|
-
progress?.emit({
|
|
171
|
-
stage: "plan",
|
|
172
|
-
step: "apply-revisions",
|
|
173
|
-
message: `Applying ${review.revisions.length} revision(s)`,
|
|
174
|
-
});
|
|
175
|
-
currentOutline = await runRevisingPlanner(plannerCtx, currentOutline, review);
|
|
176
|
-
outlinesByTurn.push(currentOutline);
|
|
177
|
-
}
|
|
178
|
-
return {
|
|
179
|
-
outline: currentOutline,
|
|
180
|
-
outlinesByTurn,
|
|
181
|
-
reviewsByTurn,
|
|
182
|
-
finalVerdict: review.verdict,
|
|
183
|
-
turnsUsed: turn,
|
|
184
|
-
hitMaxTurns,
|
|
185
|
-
};
|
|
186
|
-
}
|
|
187
|
-
//# sourceMappingURL=outline-review-loop.js.map
|
|
@@ -1,41 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Planner functions (initial and revise).
|
|
3
|
-
*
|
|
4
|
-
* Two modes correspond to two places the planner is invoked:
|
|
5
|
-
* - initial: first outline from the compressed pack (CTS-PLAN-1)
|
|
6
|
-
* - revise: address reviewer feedback — works for both
|
|
7
|
-
* `requires-another-review` (CTS-PLAN-5) and
|
|
8
|
-
* `approved-with-revisions` (CTS-PLAN-4). The orchestration
|
|
9
|
-
* decides whether to loop back to the reviewer or proceed to
|
|
10
|
-
* fan-out.
|
|
11
|
-
*
|
|
12
|
-
* Each mode is stateless — a fresh claude -p session per call.
|
|
13
|
-
*/
|
|
14
|
-
import { type ClaudeRunner } from "./claude.js";
|
|
15
|
-
import { type Outline, type OutlineReview } from "./schemas.js";
|
|
16
|
-
export interface PlannerContext {
|
|
17
|
-
/** Path to the compressed pack file (readable by the spawned agent). */
|
|
18
|
-
overviewPath: string;
|
|
19
|
-
/** Directories the agent should have read access to (e.g., the cache dir). */
|
|
20
|
-
addDirs: string[];
|
|
21
|
-
/** Optional claude runner override (for tests). */
|
|
22
|
-
runner?: ClaudeRunner;
|
|
23
|
-
/** Optional model override. */
|
|
24
|
-
model?: string;
|
|
25
|
-
}
|
|
26
|
-
/**
|
|
27
|
-
* Run the planner in initial mode and return a parsed outline.
|
|
28
|
-
* Requirement: CTS-PLAN-1
|
|
29
|
-
*/
|
|
30
|
-
export declare function runInitialPlanner(ctx: PlannerContext): Promise<Outline>;
|
|
31
|
-
/**
|
|
32
|
-
* Run the planner in revise mode — produce a new outline that addresses the
|
|
33
|
-
* reviewer's output. Handles both verdicts that trigger another planner pass:
|
|
34
|
-
* `requires-another-review` (categorized findings, judgment-based) and
|
|
35
|
-
* `approved-with-revisions` (explicit revisions list). The same prompt handles
|
|
36
|
-
* both shapes; the prompt switches behavior based on what's in the review.
|
|
37
|
-
*
|
|
38
|
-
* Requirements: CTS-PLAN-4, CTS-PLAN-5
|
|
39
|
-
*/
|
|
40
|
-
export declare function runRevisingPlanner(ctx: PlannerContext, priorOutline: Outline, review: OutlineReview): Promise<Outline>;
|
|
41
|
-
//# sourceMappingURL=planner.d.ts.map
|
|
@@ -1,76 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Planner functions (initial and revise).
|
|
3
|
-
*
|
|
4
|
-
* Two modes correspond to two places the planner is invoked:
|
|
5
|
-
* - initial: first outline from the compressed pack (CTS-PLAN-1)
|
|
6
|
-
* - revise: address reviewer feedback — works for both
|
|
7
|
-
* `requires-another-review` (CTS-PLAN-5) and
|
|
8
|
-
* `approved-with-revisions` (CTS-PLAN-4). The orchestration
|
|
9
|
-
* decides whether to loop back to the reviewer or proceed to
|
|
10
|
-
* fan-out.
|
|
11
|
-
*
|
|
12
|
-
* Each mode is stateless — a fresh claude -p session per call.
|
|
13
|
-
*/
|
|
14
|
-
import { runClaude } from "./claude.js";
|
|
15
|
-
import { PLANNER_INITIAL_PROMPT } from "./prompts/planner-initial.js";
|
|
16
|
-
import { PLANNER_REVISE_PROMPT } from "./prompts/planner-revise.js";
|
|
17
|
-
import { OUTLINE_JSON_SCHEMA, parseOutline, } from "./schemas.js";
|
|
18
|
-
/**
|
|
19
|
-
* Run the planner in initial mode and return a parsed outline.
|
|
20
|
-
* Requirement: CTS-PLAN-1
|
|
21
|
-
*/
|
|
22
|
-
export async function runInitialPlanner(ctx) {
|
|
23
|
-
const runner = ctx.runner ?? runClaude;
|
|
24
|
-
const result = await runner({
|
|
25
|
-
systemPrompt: PLANNER_INITIAL_PROMPT,
|
|
26
|
-
userMessage: `The compressed packed codebase is at: ${ctx.overviewPath}\n\nRead it and produce the outline JSON described in your system prompt.`,
|
|
27
|
-
model: ctx.model,
|
|
28
|
-
tools: ["Read"],
|
|
29
|
-
addDirs: ctx.addDirs,
|
|
30
|
-
jsonSchema: OUTLINE_JSON_SCHEMA,
|
|
31
|
-
});
|
|
32
|
-
if (result.exitCode !== 0) {
|
|
33
|
-
throw new Error(`Planner failed (exit ${result.exitCode}): ${result.stderr || result.stdout}`);
|
|
34
|
-
}
|
|
35
|
-
return parseOutline(result.stdout);
|
|
36
|
-
}
|
|
37
|
-
/**
|
|
38
|
-
* Run the planner in revise mode — produce a new outline that addresses the
|
|
39
|
-
* reviewer's output. Handles both verdicts that trigger another planner pass:
|
|
40
|
-
* `requires-another-review` (categorized findings, judgment-based) and
|
|
41
|
-
* `approved-with-revisions` (explicit revisions list). The same prompt handles
|
|
42
|
-
* both shapes; the prompt switches behavior based on what's in the review.
|
|
43
|
-
*
|
|
44
|
-
* Requirements: CTS-PLAN-4, CTS-PLAN-5
|
|
45
|
-
*/
|
|
46
|
-
export async function runRevisingPlanner(ctx, priorOutline, review) {
|
|
47
|
-
const runner = ctx.runner ?? runClaude;
|
|
48
|
-
const userMessage = [
|
|
49
|
-
`The compressed packed codebase is at: ${ctx.overviewPath}`,
|
|
50
|
-
``,
|
|
51
|
-
`## Prior outline`,
|
|
52
|
-
`\`\`\`json`,
|
|
53
|
-
JSON.stringify(priorOutline, null, 2),
|
|
54
|
-
`\`\`\``,
|
|
55
|
-
``,
|
|
56
|
-
`## Reviewer output`,
|
|
57
|
-
`\`\`\`json`,
|
|
58
|
-
JSON.stringify(review, null, 2),
|
|
59
|
-
`\`\`\``,
|
|
60
|
-
``,
|
|
61
|
-
`Produce a revised outline JSON that addresses the reviewer's output. Output JSON only.`,
|
|
62
|
-
].join("\n");
|
|
63
|
-
const result = await runner({
|
|
64
|
-
systemPrompt: PLANNER_REVISE_PROMPT,
|
|
65
|
-
userMessage,
|
|
66
|
-
model: ctx.model,
|
|
67
|
-
tools: ["Read"],
|
|
68
|
-
addDirs: ctx.addDirs,
|
|
69
|
-
jsonSchema: OUTLINE_JSON_SCHEMA,
|
|
70
|
-
});
|
|
71
|
-
if (result.exitCode !== 0) {
|
|
72
|
-
throw new Error(`Revising planner failed (exit ${result.exitCode}): ${result.stderr || result.stdout}`);
|
|
73
|
-
}
|
|
74
|
-
return parseOutline(result.stdout);
|
|
75
|
-
}
|
|
76
|
-
//# sourceMappingURL=planner.js.map
|
|
@@ -1,12 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* System prompt for the outline reviewer.
|
|
3
|
-
*
|
|
4
|
-
* Stateful across review turns (same conversation, same session-id). Reads
|
|
5
|
-
* the outline and the compressed pack, produces a JSON object with
|
|
6
|
-
* categorized findings and a verdict.
|
|
7
|
-
*
|
|
8
|
-
* Requirements covered:
|
|
9
|
-
* - CTS-PLAN-2: Outline reviewer critiques the outline in a stateful session
|
|
10
|
-
*/
|
|
11
|
-
export declare const OUTLINE_REVIEWER_PROMPT = "You are an outline reviewer for a codebase-to-spec pipeline. The pipeline takes a codebase, generates a planning outline (areas of behavior), then fans out specifier agents to produce detailed requirements per area. Your job: gate fan-out on outline quality. Bad outlines \u2192 wasted specifier work.\n\nThis is a **stateful conversation**. Across turns, you may receive multiple revisions of the outline, each addressing prior feedback. Track what you asked for and whether the planner addressed it.\n\nOutput a JSON object matching the supplied schema. No prose, no markdown fences.\n\n## What you receive on each turn\n\n- The first turn includes the compressed packed codebase and the planner's first outline JSON.\n- Each subsequent turn includes a revised outline JSON. The codebase is unchanged.\n- Some turns may include a convergence nudge \u2014 read and respect it.\n\n## What counts as a customer\n\nThe outline's summary should describe what the system is and who it's for. The \"who\" is the customer \u2014 the kind of person whose needs shape what counts as behavior.\n\nA useful customer description is **specific enough to shape behavior** \u2014 it goes one level deeper than generic categories like \"end-user,\" \"administrator,\" or \"developer.\"\n\n- Not \"an end-user\" but \"a shopper\" or \"a guest checking out without an account.\"\n- Not \"an administrator\" but \"a store manager who fulfills orders\" and \"a business owner who runs reports.\"\n- Not \"a developer\" but \"a React developer integrating an eCommerce SDK,\" \"a Python data engineer building ETL pipelines,\" or \"a distributed-systems engineer wiring up a message broker.\"\n\nMost large or sprawling codebases serve more than one customer. A system whose outline reads as if built for a single customer when the code clearly serves several is missing something \u2014 and surfacing that gap is one of the most useful things you can do.\n\n**Customers consume the software; contributors work on it.** A developer can be a legitimate customer when they consume the software being specified \u2014 a React developer integrating an SDK, a Python data engineer using a library, an operator running a CLI in CI, a contributor to an open-source project who uses it as much as they extend it. But a developer whose only role is to *work on this codebase* \u2014 described as \"a contributor,\" \"an internal maintainer,\" \"a stage author building the next feature,\" or similar \u2014 is not a customer. They're the audience for code comments and architecture docs, not for behavioral requirements. Data contracts between architectural components can still be (and often should be) specified, but think about whose experience they matter for. If an interface mediates business logic between a front-end and a server, that logic should be specified in terms of what it does for the user, not what it does for the front-end developer.\n\n## What to evaluate\n\nThe outline names its customers in the summary and breaks the system into areas. Your evaluation has two parts.\n\n### Part 1 \u2014 Per-area outcome checks\n\nFor each area, apply these four criteria:\n\n1. **Relevant to a customer.** Each area must declare at least one customer in its `customers` field. Apply the consume-vs-work-on test from \"What counts as a customer\" above: a customer is someone who *uses* the software being specified, not someone who works on its codebase. An area whose `customers` field is empty, missing, or contains only developers-of-this-codebase (contributors, maintainers, stage authors) is a `framing_error` \u2014 propose either reframing the area for a real customer or dropping it. When the area's description is written in mechanism-only voice (\"the subprocess wrapper that spawns...\", \"how the pipeline generates...\") with no customer named, that's the same finding even if a customer name was bolted on after the fact.\n2. **Speaks the customer's vocabulary.** Would the relevant customer go looking for this behavior under this area's name? The same name might be right for one kind of customer and wrong for another \u2014 what matters is whether it matches whom this area serves.\n3. **Groups a collection of functionality.** Does the area cover multiple related behaviors with a shared customer-meaningful purpose? An area with one isolated function, or a \"miscellaneous\" bucket of unrelated things, fails this test.\n4. **Has customer-observable outcomes.** Could the relevant customer verify whether the behavior is present or absent (return value, visible UI state, logged event, thrown error)? \"The system manages memory efficiently\" is true but not customer-observable.\n\nFailures of criterion 1, 2, or 4 are `framing_errors`. Failures of criterion 3 are `granularity_issues` \u2014 which also covers sizing problems more broadly.\n\n#### On sizing\n\nThink of organizing a big box of 100 crayons.\n\n- One drawer for all 100 crayons \u2192 impossible to find what you need.\n- 100 drawers, one crayon each \u2192 you've recreated the same problem with different semantics.\n- Organize by ROYGBIV \u2192 each drawer is a meaningful group, and you can find any crayon quickly.\n\nThe same logic applies to areas. An area too broad covers fundamentally distinct concerns; an area too narrow fragments what should hang together. Two areas that describe the same thing from different angles are a sizing problem \u2014 they should be one area, or split along a different axis. The right sizing for *this* codebase is whatever lets each area be coherent on its own and the whole set be complete.\n\n### Part 2 \u2014 Coverage and file assignment\n\n- What behavior is in the codebase but missing from any area? \u2192 `coverage_gaps`. Gaps matter most when they map to something a real customer would expect.\n- Are file paths actually present in the pack as written? Are files assigned to areas where they don't fit? Are public-contract docs (README, package metadata, LICENSE, CHANGELOG) unassigned? \u2192 `file_assignment_issues`.\n\n## On second and later turns\n\nAlso evaluate:\n\n- Did the revision address what you asked for in the prior turn? Be honest if it did or didn't.\n- Did the revision introduce new problems? Sometimes fixing one gap creates another.\n\n## Findings must be actionable\n\nA finding is only useful if the planner can act on it. Two principles:\n\n**Show your reasoning.** If your finding rests on a judgment about who the system is for, what the customer would want, or how an area should be reshaped, surface that reasoning. \"This outline doesn't read like it's for any specific customer\" gives the planner nothing to act on. \"I think this is most plausibly for a React developer integrating an eCommerce SDK; areas X and Y are organized around backend storage rather than what that developer would reach for; suggest reframing as Z\" does. Same pattern for any other \"this feels off\" finding \u2014 propose the alternative.\n\n**Be specific.** Cite paths and area names by exact spelling. \"Could be more comprehensive\" is not actionable. \"The outline has no area covering [behavior X], visible in [file Y]\" is.\n\n## Verdict types\n\nThe `verdict` field is exactly one of:\n\n- `approved` \u2014 outline is ready for fan-out as-is. No findings, or findings are negligible. Reserve for genuinely good outlines.\n- `approved-with-revisions` \u2014 outline is fundamentally sound and ready for fan-out, but includes specific small revisions that should be applied first. List the revisions in the `revisions` array. The planner will apply them mechanically without further review. Use for inline tweaks: rename a file path, split one bloated area into two, add a missing public-contract file to an area, tighten the customer description in the summary.\n- `requires-another-review` \u2014 outline has meaningful issues that need a structural fix, not just tweaks. Coverage gaps for whole subsystems, framing errors at the area level, customer set in the summary wrong or incomplete in ways that ripple through area design. The planner needs to think again, not just tweak.";
|
|
12
|
-
//# sourceMappingURL=outline-reviewer.d.ts.map
|