@jwilger/pi-development-system 0.54.0 → 0.55.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -20,6 +20,7 @@ import { createRequestApprovalTool } from "../src/gates/request-approval-tool.ts
20
20
  import { registerTestGuard } from "../src/gates/test-guard.ts";
21
21
  import { createJevHolder } from "../src/jev/holder.ts";
22
22
  import { createIntakeTool } from "../src/planning/intake-tool.ts";
23
+ import { createTaskCheckTool } from "../src/planning/task-check-tool.ts";
23
24
  import { createReviewRecordTool, createReviewStartTool } from "../src/review/review-tools.ts";
24
25
  import { registerCiCommand } from "../src/state/ci-command.ts";
25
26
  import { loadConfig } from "../src/state/config.ts";
@@ -140,6 +141,7 @@ export function createDevelopmentSystem(pi: ExtensionAPI) {
140
141
  pi.registerTool(createModelsTool());
141
142
  pi.registerTool(createRouteTaskTool({ jev: (ctx) => jevHolder.forContext(ctx) }));
142
143
  pi.registerTool(createIntakeTool({ state, jev: (ctx) => jevHolder.forContext(ctx) }));
144
+ pi.registerTool(createTaskCheckTool({ jev: (ctx) => jevHolder.forContext(ctx) }));
143
145
  const reviewDeps = {
144
146
  state,
145
147
  jev: (ctx: ExtensionContext) => jevHolder.forContext(ctx),
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@jwilger/pi-development-system",
3
- "version": "0.54.0",
3
+ "version": "0.55.0",
4
4
  "description": "A pi extension package representing a seasoned approach to software development using a full AI SDLC.",
5
5
  "keywords": [
6
6
  "pi-package"
@@ -0,0 +1,82 @@
1
+ import type { ClassifierBoolQuestion } from "@earendil-works/pi-ai";
2
+ import { redactSecrets } from "../../core/redact.ts";
3
+ import { err, ok, type Result } from "../../core/result.ts";
4
+ import type { TaskRecord } from "../../planning/task-record.ts";
5
+ import type { Jev, JevError } from "../client.ts";
6
+
7
+ const vague = (instructions: string): ClassifierBoolQuestion => ({
8
+ type: "bool",
9
+ instructions,
10
+ criteria: {
11
+ true: "A weaker model implementing this alone would have to guess",
12
+ false: "It is specific enough to act on without guessing",
13
+ },
14
+ });
15
+
16
+ /** One narrow judgement per aspect; `tooBig` is separate because the fix is to split, not to add detail. */
17
+ export const READINESS_QUESTIONS = {
18
+ goal: vague(
19
+ "`task.goal` should be one observable sentence: something a person or test could see change. Is it vague, a restatement of the title, or about activity rather than outcome?",
20
+ ),
21
+ interfaces: vague(
22
+ "`task.interfaces` should give exact signatures or contracts for what is created or changed. Is it missing signatures, hand-wavy ('a helper', 'some function') or inconsistent with `task.files`?",
23
+ ),
24
+ firstFailingTest: vague(
25
+ "`task.firstFailingTest` should name a test file and test and say what it asserts. Does it fail to say what would be asserted, or assert something unrelated to `task.goal`?",
26
+ ),
27
+ steps: vague(
28
+ "`task.steps` should be small steps a reviewer could reject independently. Are any steps large, combined ('implement and wire everything'), or out of order with the test-first step?",
29
+ ),
30
+ tooBig: {
31
+ type: "bool",
32
+ instructions:
33
+ "Is this more than one task: does it touch unrelated areas, or need more than one failing test to describe, so that it should be split before anyone implements it?",
34
+ criteria: {
35
+ true: "It should be split into two or more task records",
36
+ false: "It is one coherent task a single implementer can finish",
37
+ },
38
+ },
39
+ } as const satisfies Record<string, ClassifierBoolQuestion>;
40
+
41
+ type Readiness = "ready" | "needs-detail" | "too-big";
42
+ export type ReadinessJudgement = { readiness: Readiness; missing: string[] };
43
+
44
+ /** Probability at or above which an aspect counts as vague. */
45
+ const VAGUE_AT = 0.6;
46
+
47
+ const clip = (text: string, max: number): string =>
48
+ redactSecrets(text.slice(0, max * 2)).slice(0, max);
49
+
50
+ export async function judgeTaskReadiness(
51
+ jev: Jev,
52
+ record: TaskRecord,
53
+ ): Promise<Result<ReadinessJudgement, JevError>> {
54
+ const asked = await jev.ask(
55
+ {
56
+ task: {
57
+ title: clip(record.title, 200),
58
+ goal: clip(record.goal, 600),
59
+ files: record.files.slice(0, 30).map((f) => clip(f, 200)),
60
+ interfaces: clip(record.interfaces, 1500),
61
+ firstFailingTest: clip(record.firstFailingTest, 600),
62
+ steps: record.steps.slice(0, 10).map((s) => clip(s, 300)),
63
+ },
64
+ },
65
+ READINESS_QUESTIONS,
66
+ );
67
+ if (!asked.ok) return asked;
68
+ const probability = new Map<string, number>();
69
+ for (const key of Object.keys(READINESS_QUESTIONS)) {
70
+ const answer = asked.value[key];
71
+ if (answer?.type !== "bool" || !Number.isFinite(answer.probability)) {
72
+ return err({ kind: "provider", message: `missing ${key} answer` });
73
+ }
74
+ probability.set(key, answer.probability);
75
+ }
76
+ if ((probability.get("tooBig") ?? 0) >= VAGUE_AT)
77
+ return ok({ readiness: "too-big", missing: [] });
78
+ const missing = Object.keys(READINESS_QUESTIONS).filter(
79
+ (k) => k !== "tooBig" && (probability.get(k) ?? 0) >= VAGUE_AT,
80
+ );
81
+ return ok({ readiness: missing.length > 0 ? "needs-detail" : "ready", missing });
82
+ }
@@ -0,0 +1,73 @@
1
+ import { readFile } from "node:fs/promises";
2
+ import { isAbsolute, relative, resolve } from "node:path";
3
+ import type { ExtensionContext, ToolDefinition } from "@earendil-works/pi-coding-agent";
4
+ import { type Static, Type } from "typebox";
5
+ import { isParseError } from "../core/types.ts";
6
+ import type { Jev } from "../jev/client.ts";
7
+ import { judgeTaskReadiness } from "../jev/questions/readiness.ts";
8
+ import { parseTaskRecord } from "./task-record.ts";
9
+
10
+ const Parameters = Type.Object({
11
+ path: Type.String({
12
+ description: "Path to the task record markdown file, inside the repository.",
13
+ }),
14
+ });
15
+
16
+ const reply = (text: string, isError = false) => ({
17
+ content: [{ type: "text" as const, text }],
18
+ details: undefined,
19
+ isError,
20
+ });
21
+
22
+ const insideRepo = (cwd: string, target: string): boolean => {
23
+ const rel = relative(cwd, target);
24
+ return rel !== "" && !rel.startsWith("..") && !isAbsolute(rel);
25
+ };
26
+
27
+ /** `devsys_task_check`: structural readiness is deterministic; Jev adds "specific enough" and "one task". */
28
+ export function createTaskCheckTool(deps: {
29
+ jev: (ctx: ExtensionContext) => Jev;
30
+ }): ToolDefinition<typeof Parameters> {
31
+ return {
32
+ name: "devsys_task_check",
33
+ label: "Check task record",
34
+ description:
35
+ "Check a task record file against the task-record format and judge whether a weaker implementer could act on it. " +
36
+ "Run before handing a task record to an implementer subagent.",
37
+ promptSnippet: "Check a task record is ready for an implementer",
38
+ parameters: Parameters,
39
+ exposure: "direct",
40
+ async execute(_id, params: Static<typeof Parameters>, _signal, _onUpdate, ctx) {
41
+ const target = resolve(ctx.cwd, params.path);
42
+ if (!insideRepo(ctx.cwd, target))
43
+ return reply(`${params.path} is outside the repository`, true);
44
+ let markdown: string;
45
+ try {
46
+ markdown = await readFile(target, "utf8");
47
+ } catch (cause) {
48
+ return reply(
49
+ `cannot read ${params.path}: ${cause instanceof Error ? cause.message : String(cause)}`,
50
+ true,
51
+ );
52
+ }
53
+ const record = parseTaskRecord(markdown);
54
+ if (isParseError(record))
55
+ return reply(`${params.path} is not ready: ${record.message}`, true);
56
+ const judged = await judgeTaskReadiness(deps.jev(ctx), record);
57
+ if (!judged.ok) {
58
+ return reply(
59
+ `${record.id}: structure ok. Jev unavailable (${judged.error.kind}), so specificity and size were not judged; review the record yourself.`,
60
+ );
61
+ }
62
+ const { readiness, missing } = judged.value;
63
+ if (readiness === "ready")
64
+ return reply(`${record.id}: ready. Structure ok and specific enough for an implementer.`);
65
+ if (readiness === "too-big") {
66
+ return reply(
67
+ `${record.id}: too-big. Split it into task records that each need one failing test, then check each.`,
68
+ );
69
+ }
70
+ return reply(`${record.id}: needs-detail. Make these more specific: ${missing.join(", ")}.`);
71
+ },
72
+ };
73
+ }
@@ -0,0 +1,121 @@
1
+ import { type ParseError, parseError } from "../core/types.ts";
2
+
3
+ /** Appendix C of the plan: the one-task-per-implementer format a weaker model can follow. */
4
+ export type TaskRecord = {
5
+ readonly id: string;
6
+ readonly title: string;
7
+ readonly goal: string;
8
+ readonly files: readonly string[];
9
+ readonly interfaces: string;
10
+ readonly firstFailingTest: string;
11
+ readonly steps: readonly string[];
12
+ readonly run: string;
13
+ readonly expected: string;
14
+ readonly outOfScope: string;
15
+ };
16
+
17
+ const SECTIONS = [
18
+ "Goal",
19
+ "Files",
20
+ "Interfaces",
21
+ "First failing test",
22
+ "Steps",
23
+ "Run",
24
+ "Expected",
25
+ "Out of scope",
26
+ ] as const;
27
+ type Section = (typeof SECTIONS)[number];
28
+
29
+ const HEADER = /^## (\S+) [—-] (.+)$/;
30
+ const SECTION_LINE = /^\*\*([^:*]+):\*\*\s*(.*)$/;
31
+ const STEP = /^\s*(?:\d+[.)]|[-*])\s+(.*)$/;
32
+ const STEPS_MIN = 3;
33
+ const STEPS_MAX = 7;
34
+ /** Words that stand in for a command or an observable result. */
35
+ const PLACEHOLDER = /^(?:n\/?a|none|todo|tbc|-|…|\.\.\.|works?|it works|passes|ok)\.?$/i;
36
+
37
+ const isSection = (name: string): name is Section => SECTIONS.some((s) => s === name);
38
+
39
+ /** Splits the body into section texts keyed by name; text may continue over several lines. */
40
+ function splitSections(lines: readonly string[]): Map<Section, string> {
41
+ const found = new Map<Section, string[]>();
42
+ let current: string[] | undefined;
43
+ for (const line of lines) {
44
+ const match = SECTION_LINE.exec(line);
45
+ const name = match?.[1]?.trim();
46
+ if (match !== null && name !== undefined && isSection(name)) {
47
+ current = match[2] === "" ? [] : [match[2] ?? ""];
48
+ found.set(name, current);
49
+ } else current?.push(line);
50
+ }
51
+ return new Map([...found].map(([name, text]) => [name, text.join("\n").trim()]));
52
+ }
53
+
54
+ const stripTicks = (text: string): string => text.replace(/^`+|`+$/g, "").trim();
55
+
56
+ const stepsOf = (text: string): string[] =>
57
+ text.split("\n").flatMap((line) => {
58
+ const item = STEP.exec(line)?.[1]?.trim();
59
+ return item === undefined || item === "" ? [] : [item];
60
+ });
61
+
62
+ function concrete(name: "Run" | "Expected", text: string): string | undefined {
63
+ const bare = stripTicks(text);
64
+ if (PLACEHOLDER.test(bare))
65
+ return `${name} must be a concrete ${name === "Run" ? "command" : "observable result"}, not "${bare}"`;
66
+ if (name === "Expected" && bare.split(/\s+/).length < 2) {
67
+ return "Expected must say what is observed (at least two words, e.g. a count, output or exit status)";
68
+ }
69
+ return undefined;
70
+ }
71
+
72
+ const get = (sections: ReadonlyMap<Section, string>, s: Section): string => sections.get(s) ?? "";
73
+
74
+ /** Every readiness problem in the sections, so the author fixes them in one pass. */
75
+ function problemsOf(markdown: string, sections: ReadonlyMap<Section, string>): string[] {
76
+ const problems: string[] = [];
77
+ const missing = SECTIONS.filter((s) => get(sections, s) === "");
78
+ if (missing.length > 0) problems.push(`missing section: ${missing.join(", ")}`);
79
+ if (/\bTBD\b/.test(markdown)) problems.push("contains TBD; decide it or split the task");
80
+ const steps = stepsOf(get(sections, "Steps"));
81
+ if (!missing.includes("Steps") && (steps.length < STEPS_MIN || steps.length > STEPS_MAX)) {
82
+ problems.push(`Steps must be ${STEPS_MIN}-${STEPS_MAX} steps, found ${steps.length}`);
83
+ }
84
+ for (const name of ["Run", "Expected"] as const) {
85
+ const why = missing.includes(name) ? undefined : concrete(name, get(sections, name));
86
+ if (why !== undefined) problems.push(why);
87
+ }
88
+ return problems;
89
+ }
90
+
91
+ const filesOf = (text: string): string[] =>
92
+ text
93
+ .split(/[,\n]/)
94
+ .map((f) => stripTicks(f.trim()))
95
+ .filter((f) => f !== "");
96
+
97
+ /** Parses one task record, reporting every problem together. */
98
+ export function parseTaskRecord(markdown: string): TaskRecord | ParseError {
99
+ const lines = markdown.split("\n");
100
+ const headerAt = lines.findIndex((l) => l.startsWith("## "));
101
+ const header = headerAt < 0 ? null : HEADER.exec(lines[headerAt] ?? "");
102
+ if (header === null) {
103
+ return parseError("header must be `## <id> — <title>` (an id, an em dash, then the title)");
104
+ }
105
+ const sections = splitSections(lines.slice(headerAt + 1));
106
+ const problems = problemsOf(markdown, sections);
107
+ if (problems.length > 0) return parseError(problems.join("; "));
108
+ const text = (s: Section): string => get(sections, s);
109
+ return {
110
+ id: header[1] ?? "",
111
+ title: header[2]?.trim() ?? "",
112
+ goal: text("Goal"),
113
+ files: filesOf(text("Files")),
114
+ interfaces: text("Interfaces"),
115
+ firstFailingTest: text("First failing test"),
116
+ steps: stepsOf(text("Steps")),
117
+ run: stripTicks(text("Run")),
118
+ expected: text("Expected"),
119
+ outOfScope: text("Out of scope"),
120
+ };
121
+ }