@jwilger/pi-development-system 0.33.0 → 0.35.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/extensions/development-system.ts +10 -0
- package/package.json +1 -1
- package/prompts/devsys-review.md +1 -1
- package/skills/code-review/SKILL.md +1 -1
- package/src/context/turn-verifier.ts +114 -0
- package/src/core/review-packet.ts +2 -1
- package/src/jev/questions/turn.ts +81 -0
- package/src/review/review-tools.ts +1 -1
|
@@ -3,6 +3,7 @@ import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-a
|
|
|
3
3
|
import { appendContextTail, renderContextTail } from "../src/context/context-tail.ts";
|
|
4
4
|
import { renderStatus, renderStatusLine, STATUS_KEY } from "../src/context/status.ts";
|
|
5
5
|
import { applyPromptSection } from "../src/context/system-prompt.ts";
|
|
6
|
+
import { DEFAULT_VERIFIER_MAX, registerTurnVerifier } from "../src/context/turn-verifier.ts";
|
|
6
7
|
import { type Exec, timeoutAsFailure } from "../src/core/exec.ts";
|
|
7
8
|
import { detectProfiles } from "../src/core/profile.ts";
|
|
8
9
|
import { createApprovalStore } from "../src/gates/approvals.ts";
|
|
@@ -101,6 +102,15 @@ export function createDevelopmentSystem(pi: ExtensionAPI) {
|
|
|
101
102
|
});
|
|
102
103
|
registerTestGuard({ pi, state, approvals, jev: (ctx) => jevHolder.forContext(ctx) });
|
|
103
104
|
registerTestEvidence({ pi, state });
|
|
105
|
+
registerTurnVerifier({
|
|
106
|
+
pi,
|
|
107
|
+
state,
|
|
108
|
+
jev: (ctx) => jevHolder.forContext(ctx),
|
|
109
|
+
maxPerSession: async (ctx) => {
|
|
110
|
+
const config = await loadConfig(ctx.cwd);
|
|
111
|
+
return config.ok ? config.value.verifier.maxPerSession : DEFAULT_VERIFIER_MAX;
|
|
112
|
+
},
|
|
113
|
+
});
|
|
104
114
|
registerRedFirstGuard({ pi, state });
|
|
105
115
|
registerLintSuppressionGuard({ pi, state });
|
|
106
116
|
pi.registerTool(createRecordDepartureTool({ pi, state }));
|
package/package.json
CHANGED
package/prompts/devsys-review.md
CHANGED
|
@@ -6,5 +6,5 @@ Review the work in progress with the code-review skill.
|
|
|
6
6
|
|
|
7
7
|
1. Call `devsys_review_start` (slice and diff range: $ARGUMENTS; omit either to use the active slice and `HEAD`; only the `HEAD` range clears the commit gate).
|
|
8
8
|
2. Run the `agent_spawn` payload it returns, unchanged.
|
|
9
|
-
3. Pass the reviewer's packet verbatim to `devsys_review_record` with the same `diffDigest` and `diffRange` the start reply gave.
|
|
9
|
+
3. Pass the reviewer's packet verbatim to `devsys_review_record` with the same `slice`, `diffDigest` and `diffRange` the start reply gave.
|
|
10
10
|
4. Report `review: N/R clean` and the next action. If the next action is `fix-findings`, list each blocking and should-fix finding with its `path:line`.
|
|
@@ -18,7 +18,7 @@ Review is a soft gate (`review.unsatisfied`): skipping it needs a recorded depar
|
|
|
18
18
|
2. Run that `agent_spawn` unchanged. The reviewer is a top-level agent, not a
|
|
19
19
|
child of this conversation, so it does not inherit your context.
|
|
20
20
|
3. Pass the reviewer's packet, verbatim, to `devsys_review_record` with the
|
|
21
|
-
`diffDigest` the start reply gave, and the same `diffRange` if you gave one
|
|
21
|
+
`slice` and `diffDigest` the start reply gave, and the same `diffRange` if you gave one
|
|
22
22
|
(`packets: [...]`; all lens packets of one
|
|
23
23
|
round in one call). A packet is recorded once, in the round its header names;
|
|
24
24
|
if the diff changed since the start, the round is refused: start again.
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
import type { AgentMessage } from "@earendil-works/pi-agent-core";
|
|
2
|
+
import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
3
|
+
import { redactSecrets } from "../core/redact.ts";
|
|
4
|
+
import { summarizeOutput } from "../core/test-runner.ts";
|
|
5
|
+
import type { Phase } from "../core/types.ts";
|
|
6
|
+
import type { Jev } from "../jev/client.ts";
|
|
7
|
+
import { judgeTurn, type ToolEvidence } from "../jev/questions/turn.ts";
|
|
8
|
+
import type { SessionState } from "../state/session-state.ts";
|
|
9
|
+
|
|
10
|
+
export const VERIFIER_THRESHOLD = 0.75;
|
|
11
|
+
export const DEFAULT_VERIFIER_MAX = 6;
|
|
12
|
+
export const VERIFIER_ENTRY_TYPE = "devsys-verifier";
|
|
13
|
+
const VERIFIED_PHASES: ReadonlySet<Phase> = new Set(["implementing", "reviewing", "delivering"]);
|
|
14
|
+
const MAX_EVIDENCE = 60;
|
|
15
|
+
|
|
16
|
+
export type TurnVerifierDeps = {
|
|
17
|
+
pi: ExtensionAPI;
|
|
18
|
+
state: SessionState;
|
|
19
|
+
jev(ctx: ExtensionContext): Jev;
|
|
20
|
+
maxPerSession(ctx: ExtensionContext): number | Promise<number>;
|
|
21
|
+
};
|
|
22
|
+
|
|
23
|
+
type Parts = { text: string; callsTools: boolean };
|
|
24
|
+
|
|
25
|
+
/** Text and tool-call presence of an assistant message; undefined for any other role. */
|
|
26
|
+
export function assistantParts(message: AgentMessage): Parts | undefined {
|
|
27
|
+
if (message.role !== "assistant") return undefined;
|
|
28
|
+
const text = message.content.flatMap((c) => (c.type === "text" ? [c.text] : [])).join("\n");
|
|
29
|
+
return { text, callsTools: message.content.some((c) => c.type === "toolCall") };
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/** True when the last non-empty line of the message ends with a question mark. */
|
|
33
|
+
export function asksUser(text: string): boolean {
|
|
34
|
+
const last = text
|
|
35
|
+
.split("\n")
|
|
36
|
+
.map((l) => l.trim())
|
|
37
|
+
.filter((l) => l !== "")
|
|
38
|
+
.at(-1);
|
|
39
|
+
return last !== undefined && /\?[\s)"'*_`]*$/.test(last);
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
const claimMessage = (): string =>
|
|
43
|
+
"Development system: your last message claims something was run, passed or done, but no tool evidence in this run shows it. Run the verification now, or restate without the claim.";
|
|
44
|
+
|
|
45
|
+
const driftMessage = (slice: string): string =>
|
|
46
|
+
`Development system: your last message describes work beyond the active slice "${slice}". Either return to the slice, or call devsys_record_departure (gate scope.expansion) to record why the scope is growing.`;
|
|
47
|
+
|
|
48
|
+
const note = (content: string) => ({
|
|
49
|
+
type: "custom_message" as const,
|
|
50
|
+
customType: VERIFIER_ENTRY_TYPE,
|
|
51
|
+
content,
|
|
52
|
+
display: true,
|
|
53
|
+
});
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* At turn end, asks Jev whether the final message claims results no tool call supports, or wanders
|
|
57
|
+
* from the active slice, and forces ONE corrective continuation. Bounded: never on a turn that
|
|
58
|
+
* still calls tools, never on a question to the user, never twice in a row, never more than
|
|
59
|
+
* `maxPerSession` times. Jev trouble means silence, not a block.
|
|
60
|
+
*/
|
|
61
|
+
export function registerTurnVerifier(deps: TurnVerifierDeps): void {
|
|
62
|
+
let evidence: ToolEvidence[] = [];
|
|
63
|
+
let corrected = 0;
|
|
64
|
+
let justCorrected = false;
|
|
65
|
+
|
|
66
|
+
deps.pi.on("session_start", () => {
|
|
67
|
+
corrected = 0;
|
|
68
|
+
justCorrected = false;
|
|
69
|
+
evidence = [];
|
|
70
|
+
});
|
|
71
|
+
deps.pi.on("agent_start", () => {
|
|
72
|
+
evidence = [];
|
|
73
|
+
justCorrected = false;
|
|
74
|
+
});
|
|
75
|
+
deps.pi.on("tool_result", (event) => {
|
|
76
|
+
const text = event.content.flatMap((c) => (c.type === "text" ? [c.text] : [])).join("\n");
|
|
77
|
+
const summary = redactSecrets(summarizeOutput(text));
|
|
78
|
+
const item: ToolEvidence =
|
|
79
|
+
event.toolName === "bash"
|
|
80
|
+
? { tool: "bash", summary, exitCode: event.isError ? 1 : 0 }
|
|
81
|
+
: { tool: event.toolName, summary };
|
|
82
|
+
evidence = [...evidence, item].slice(-MAX_EVIDENCE);
|
|
83
|
+
});
|
|
84
|
+
|
|
85
|
+
deps.pi.on("turn_end", async (event, ctx) => {
|
|
86
|
+
const state = deps.state.get();
|
|
87
|
+
if (!VERIFIED_PHASES.has(state.phase)) return undefined;
|
|
88
|
+
const parts = assistantParts(event.message);
|
|
89
|
+
if (parts === undefined || parts.callsTools || parts.text.trim() === "") return undefined;
|
|
90
|
+
if (justCorrected) {
|
|
91
|
+
justCorrected = false;
|
|
92
|
+
return undefined;
|
|
93
|
+
}
|
|
94
|
+
if (asksUser(parts.text) || corrected >= (await deps.maxPerSession(ctx))) return undefined;
|
|
95
|
+
const jev = deps.jev(ctx);
|
|
96
|
+
if (jev.availability() === "offline") return undefined;
|
|
97
|
+
const judged = await judgeTurn(jev, {
|
|
98
|
+
assistantText: parts.text,
|
|
99
|
+
toolEvidence: evidence,
|
|
100
|
+
...(state.activeSlice === undefined ? {} : { activeSlice: state.activeSlice }),
|
|
101
|
+
});
|
|
102
|
+
if (!judged.ok) return undefined;
|
|
103
|
+
const entries = [
|
|
104
|
+
...(judged.value.unverifiedClaim >= VERIFIER_THRESHOLD ? [note(claimMessage())] : []),
|
|
105
|
+
...(judged.value.driftFromSlice >= VERIFIER_THRESHOLD && state.activeSlice !== undefined
|
|
106
|
+
? [note(driftMessage(state.activeSlice))]
|
|
107
|
+
: []),
|
|
108
|
+
];
|
|
109
|
+
if (entries.length === 0) return undefined;
|
|
110
|
+
corrected += 1;
|
|
111
|
+
justCorrected = true;
|
|
112
|
+
return { continue: true, entries };
|
|
113
|
+
});
|
|
114
|
+
}
|
|
@@ -18,7 +18,8 @@ const HEADER = new RegExp(
|
|
|
18
18
|
// Reviewers add text after the location (`fn`, a second location); the first backticked one is the finding's.
|
|
19
19
|
const FINDING = /^-\s*\[([^\]]+)\]\s+(\S+)\s+(?:`([^`]+)`[^—–]*?\s+)?[—–-]\s+(.+)$/;
|
|
20
20
|
// An indented line under a finding is its detail (trigger, fix), unless it is itself a finding.
|
|
21
|
-
const isDetail = (line: string): boolean =>
|
|
21
|
+
const isDetail = (line: string): boolean =>
|
|
22
|
+
/^\s+/.test(line) && !/^\s*(?:[-*+]|\d+[.)])\s*\[/.test(line);
|
|
22
23
|
|
|
23
24
|
// `until` ends a section; by default any heading does. Findings ends only at a known section so a stray
|
|
24
25
|
// heading inside it (### Nits) makes its lines errors instead of silently cutting the findings off.
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
import type { ClassifierBoolQuestion } from "@earendil-works/pi-ai";
|
|
2
|
+
import { redactSecrets } from "../../core/redact.ts";
|
|
3
|
+
import { err, ok, type Result } from "../../core/result.ts";
|
|
4
|
+
import type { Jev, JevError } from "../client.ts";
|
|
5
|
+
|
|
6
|
+
export const CLAIM_QUESTION: ClassifierBoolQuestion = {
|
|
7
|
+
type: "bool",
|
|
8
|
+
instructions:
|
|
9
|
+
'`assistantText` is the last message an AI coding agent wrote. `toolEvidence` lists the tool calls it made this turn (tool, summary, exit code). Does the message state as fact that something was run, passed, fixed, built, committed, pushed or verified, where `toolEvidence` does not show it? A claim that is plainly about a past turn, a plan, an intention, a question or a hedge ("I will run", "should pass") is not a claim of fact.',
|
|
10
|
+
criteria: {
|
|
11
|
+
true: "It asserts a completed action or a passing result that no tool call this turn supports (or that a failing tool call contradicts)",
|
|
12
|
+
false:
|
|
13
|
+
"Every stated result is backed by the tool evidence, or the message only plans, asks or hedges",
|
|
14
|
+
},
|
|
15
|
+
};
|
|
16
|
+
|
|
17
|
+
export const DRIFT_QUESTION: ClassifierBoolQuestion = {
|
|
18
|
+
type: "bool",
|
|
19
|
+
instructions:
|
|
20
|
+
"`activeSlice` names the one slice of work the agent agreed to do. Given `assistantText` and `toolEvidence`, is the agent doing substantial work that the slice does not cover: new features, extra refactors, unrelated files, or changing scope? Housekeeping needed to do the slice (tests, docs for it, commits) is not drift.",
|
|
21
|
+
criteria: {
|
|
22
|
+
true: "The agent is doing or proposing work beyond the slice's stated scope",
|
|
23
|
+
false: "The work stays within the slice, or is the housekeeping it needs",
|
|
24
|
+
},
|
|
25
|
+
};
|
|
26
|
+
|
|
27
|
+
export type ToolEvidence = { tool: string; summary: string; exitCode?: number };
|
|
28
|
+
export type TurnInput = {
|
|
29
|
+
assistantText: string;
|
|
30
|
+
toolEvidence: readonly ToolEvidence[];
|
|
31
|
+
activeSlice?: string;
|
|
32
|
+
};
|
|
33
|
+
export type TurnJudgement = { unverifiedClaim: number; driftFromSlice: number };
|
|
34
|
+
|
|
35
|
+
const TEXT_MAX = 4000;
|
|
36
|
+
const SUMMARY_MAX = 240;
|
|
37
|
+
const EVIDENCE_MAX = 30;
|
|
38
|
+
|
|
39
|
+
const clip = (text: string, max: number): string =>
|
|
40
|
+
redactSecrets(text.slice(0, max * 2)).slice(0, max);
|
|
41
|
+
|
|
42
|
+
const evidenceLine = (e: ToolEvidence): string =>
|
|
43
|
+
`${clip(e.tool, 40)}${e.exitCode === undefined ? "" : ` exit ${e.exitCode}`}: ${clip(e.summary, SUMMARY_MAX)}`;
|
|
44
|
+
|
|
45
|
+
const probability = (
|
|
46
|
+
answer: { type: string; probability?: number } | undefined,
|
|
47
|
+
name: string,
|
|
48
|
+
): Result<number, JevError> => {
|
|
49
|
+
if (answer?.type !== "bool" || typeof answer.probability !== "number") {
|
|
50
|
+
return err({ kind: "provider", message: `missing ${name} answer` });
|
|
51
|
+
}
|
|
52
|
+
return Number.isFinite(answer.probability)
|
|
53
|
+
? ok(answer.probability)
|
|
54
|
+
: err({ kind: "provider", message: `malformed ${name} answer` });
|
|
55
|
+
};
|
|
56
|
+
|
|
57
|
+
/** Two independent narrow questions; drift is only asked when a slice is active. */
|
|
58
|
+
export async function judgeTurn(
|
|
59
|
+
jev: Jev,
|
|
60
|
+
input: TurnInput,
|
|
61
|
+
): Promise<Result<TurnJudgement, JevError>> {
|
|
62
|
+
const slice = input.activeSlice === undefined ? undefined : clip(input.activeSlice, 200);
|
|
63
|
+
const state = {
|
|
64
|
+
assistantText: clip(input.assistantText, TEXT_MAX),
|
|
65
|
+
toolEvidence: input.toolEvidence.slice(-EVIDENCE_MAX).map(evidenceLine),
|
|
66
|
+
...(slice === undefined ? {} : { activeSlice: slice }),
|
|
67
|
+
};
|
|
68
|
+
const asked = await jev.ask(
|
|
69
|
+
state,
|
|
70
|
+
slice === undefined
|
|
71
|
+
? { claim: CLAIM_QUESTION }
|
|
72
|
+
: { claim: CLAIM_QUESTION, drift: DRIFT_QUESTION },
|
|
73
|
+
);
|
|
74
|
+
if (!asked.ok) return asked;
|
|
75
|
+
const claim = probability(asked.value.claim, "claim");
|
|
76
|
+
if (!claim.ok) return claim;
|
|
77
|
+
if (slice === undefined) return ok({ unverifiedClaim: claim.value, driftFromSlice: 0 });
|
|
78
|
+
const drift = probability(asked.value.drift, "drift");
|
|
79
|
+
if (!drift.ok) return drift;
|
|
80
|
+
return ok({ unverifiedClaim: claim.value, driftFromSlice: drift.value });
|
|
81
|
+
}
|
|
@@ -157,7 +157,7 @@ export function createReviewStartTool(
|
|
|
157
157
|
[
|
|
158
158
|
`${reviewLabel(review)}; round ${round}; diff ${snap.value.digest}. ${basis}`,
|
|
159
159
|
`lenses: ${lenses.join(", ")}`,
|
|
160
|
-
`Spawn the reviewer with agent_spawn using exactly this payload, then pass its packet to devsys_review_record with diffDigest "${snap.value.digest}"${range === "HEAD" ? "" : ` and diffRange "${range}"`}:`,
|
|
160
|
+
`Spawn the reviewer with agent_spawn using exactly this payload, then pass its packet to devsys_review_record with slice "${slice}", diffDigest "${snap.value.digest}"${range === "HEAD" ? "" : ` and diffRange "${range}"`}:`,
|
|
161
161
|
JSON.stringify(spawn),
|
|
162
162
|
].join("\n"),
|
|
163
163
|
);
|