pi-code 1.0.47 → 1.0.49
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -2
- package/extensions/claude-rules.ts +2 -1
- package/extensions/commands.ts +2 -2
- package/extensions/context-imports.ts +13 -12
- package/extensions/git-checkpoint.ts +26 -0
- package/extensions/goal.ts +522 -0
- package/extensions/hooks/index.ts +2 -7
- package/extensions/internal/command-file.ts +22 -11
- package/extensions/internal/goal-evaluator.ts +262 -0
- package/extensions/internal/html-markdown.ts +37 -9
- package/extensions/internal/model-lookup.ts +20 -0
- package/extensions/internal/output-guard.ts +6 -5
- package/extensions/internal/strip-comments.ts +3 -3
- package/extensions/memory.ts +26 -1
- package/extensions/plan-mode/index.ts +3 -1
- package/extensions/question.ts +1 -1
- package/extensions/subagent/background.ts +33 -10
- package/extensions/todo.ts +6 -2
- package/extensions/web.ts +2 -6
- package/package.json +2 -2
package/README.md
CHANGED
|
@@ -11,7 +11,7 @@
|
|
|
11
11
|
[](https://sonarcloud.io/summary/new_code?id=ilovepixelart_pi-code)
|
|
12
12
|
[](https://sonarcloud.io/summary/new_code?id=ilovepixelart_pi-code)
|
|
13
13
|
|
|
14
|
-
Claude Code experience for the [pi](https://pi.dev) coding agent, in one package. Point pi at a project that already has a `.claude/` directory and it reads your existing config: rules, commands, skills, hooks, output styles, MCP servers, and agents. It also adds the Claude Code features pi lacks: a todo overlay, checkpoints, memory, web search, and
|
|
14
|
+
Claude Code experience for the [pi](https://pi.dev) coding agent, in one package. Point pi at a project that already has a `.claude/` directory and it reads your existing config: rules, commands, skills, hooks, output styles, MCP servers, and agents. It also adds the Claude Code features pi lacks: a todo overlay, checkpoints, memory, web search, subagents, and goals.
|
|
15
15
|
|
|
16
16
|
What a repository ships is treated as untrusted until you approve it: project MCP servers, hooks, agents, rules, output styles, commands and skills load only once you say yes.
|
|
17
17
|
|
|
@@ -53,9 +53,10 @@ Each topic links to its own doc with the full contract and any divergences from
|
|
|
53
53
|
- **[Statusline](docs/statusline.md)** — your Claude `statusLine` command with the documented stdin JSON.
|
|
54
54
|
- **[WebSearch / WebFetch](docs/web.md)** — key-free search and SSRF-guarded fetch.
|
|
55
55
|
- **[Claude plugins](docs/plugins.md)** — installed marketplace plugins: commands, agents, hooks, MCP servers, styles, skills.
|
|
56
|
+
- **[Goal](docs/goal.md)**: `/goal <condition>` keeps the session working until a separate model check confirms the condition holds, with status, clear, block cap, background-work deferral, and resume.
|
|
56
57
|
- **[Session extras](docs/session-extras.md)** — project trust, plan mode, todos, checkpoints/rewind, AskUserQuestion, notifications, think keywords, session titles, `/context`, `/init`.
|
|
57
58
|
|
|
58
|
-
Slash commands: `/init`, `/context`, `/memory`, `/todos`, `/rewind`, `/tasks`, `/agents`, `/plan`, `/mcp`, `/hooks`, and `/output-style`, alongside your own `/dir:name` commands, `/skill:name` skills, `/plugin:name` plugin commands, and each connected server's `/mcp__server__prompt` prompts.
|
|
59
|
+
Slash commands: `/init`, `/context`, `/goal`, `/memory`, `/todos`, `/rewind`, `/tasks`, `/agents`, `/plan`, `/mcp`, `/hooks`, and `/output-style`, alongside your own `/dir:name` commands, `/skill:name` skills, `/plugin:name` plugin commands, and each connected server's `/mcp__server__prompt` prompts.
|
|
59
60
|
|
|
60
61
|
pi has no general permission system, so most of what Claude routes through a permission prompt maps to hard behavior here: `allowed-tools` restricts the turn's tool set instead of pre-approving calls, and a hook that times out on PreToolUse or UserPromptSubmit fails closed. A hook's `permissionDecision: "ask"` is the exception: it shows a confirm dialog and lets the call through when you approve (a headless run has no dialog, so it blocks). Where a Claude restriction cannot be expressed at all (an argument-scoped grant in an agent's `tools:`), the definition is rejected rather than widened.
|
|
61
62
|
|
|
@@ -68,7 +68,8 @@ function parsePaths(frontmatter: string): string[] {
|
|
|
68
68
|
|
|
69
69
|
/** Split YAML-ish frontmatter off the front of a rule file, extracting `paths`. */
|
|
70
70
|
export function parseFrontmatter(content: string): Frontmatter {
|
|
71
|
-
|
|
71
|
+
// A leading byte order mark (Windows editors add one) is part of the header, not the body.
|
|
72
|
+
const match = /^\uFEFF?---\r?\n([\s\S]*?)\r?\n---\r?\n?/.exec(content)
|
|
72
73
|
if (!match) return { paths: [], body: content }
|
|
73
74
|
return { paths: parsePaths(match[1]), body: content.slice(match[0].length) }
|
|
74
75
|
}
|
package/extensions/commands.ts
CHANGED
|
@@ -197,7 +197,7 @@ function commandVars(ctx: { cwd: string }, filePath: string, plugin?: CommandPlu
|
|
|
197
197
|
|
|
198
198
|
/** The exec seam expandCommand runs spans through; pi itself satisfies it. */
|
|
199
199
|
interface SpanRunner {
|
|
200
|
-
exec(command: string, args: string[], options?: { cwd?: string; timeout?: number }): Promise<{ stdout: string; stderr: string; code: number }>
|
|
200
|
+
exec(command: string, args: string[], options?: { cwd?: string; timeout?: number }): Promise<{ stdout: string; stderr: string; code: number; killed?: boolean }>
|
|
201
201
|
}
|
|
202
202
|
|
|
203
203
|
/**
|
|
@@ -231,7 +231,7 @@ export async function expandCommand(runner: SpanRunner, parsed: ParsedCommand, a
|
|
|
231
231
|
// pwsh cannot merge a native command's stderr in-script (spanExec sets
|
|
232
232
|
// mergeStreams), so it is appended here; the sh script merges via 2>&1.
|
|
233
233
|
const stdout = run.mergeStreams ? result.stdout + result.stderr : result.stdout
|
|
234
|
-
return { stdout, stderr: result.stderr, code: result.code }
|
|
234
|
+
return { stdout, stderr: result.stderr, code: result.code, killed: result.killed }
|
|
235
235
|
}
|
|
236
236
|
: async () => ({ stdout: SHELL_DISABLED_PLACEHOLDER, stderr: '', code: 0 })
|
|
237
237
|
let expanded = await expandDynamicContent(withVars, ctx.cwd, exec, parsed.shell === 'powershell' ? 'powershell' : 'bash')
|
|
@@ -72,12 +72,13 @@ import type { ExtensionAPI } from '@earendil-works/pi-coding-agent'
|
|
|
72
72
|
import { claudeConfigDir } from './internal/config-dir.js'
|
|
73
73
|
import { type InstructionLoadEvent, memoryTypeForPath, publishInstructionLoad } from './internal/instruction-events.js'
|
|
74
74
|
import { managedSettingsPath, readManagedSettings } from './internal/managed-settings.js'
|
|
75
|
+
import { sliceBytes } from './internal/output-guard.js'
|
|
75
76
|
import { globToRegExpSource } from './internal/path-rules.js'
|
|
76
77
|
import { isProjectApproved, isProjectApprovedSilently } from './internal/project-approval.js'
|
|
77
78
|
import { ancestorFiles, findNearestFile, repoRoot } from './internal/project-root.js'
|
|
78
79
|
import { claudeSettingsChain } from './internal/settings-chain.js'
|
|
79
80
|
import { statToken } from './internal/stat-token.js'
|
|
80
|
-
import { fenceMarker, stripBlockComments } from './internal/strip-comments.js'
|
|
81
|
+
import { type Fence, fenceMarker, stepFence, stripBlockComments } from './internal/strip-comments.js'
|
|
81
82
|
|
|
82
83
|
/** Claude documents "a maximum depth of four hops" for recursive imports. */
|
|
83
84
|
const MAX_IMPORT_DEPTH = 4
|
|
@@ -145,16 +146,14 @@ export const createImportBudget = (): ImportBudget => ({ files: MAX_IMPORT_FILES
|
|
|
145
146
|
* imports neither in fenced code blocks (backtick or tilde) nor in inline spans. */
|
|
146
147
|
function importTargets(content: string): string[] {
|
|
147
148
|
const targets: string[] = []
|
|
148
|
-
//
|
|
149
|
-
// example may
|
|
150
|
-
let fence:
|
|
149
|
+
// CommonMark fences: closed only by the same character in a run at least as long
|
|
150
|
+
// as the opener, so a backtick example may hold tilde lines or shorter fences.
|
|
151
|
+
let fence: Fence | null = null
|
|
151
152
|
for (const line of content.split('\n')) {
|
|
152
|
-
const
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
}
|
|
157
|
-
if (fence !== null) continue
|
|
153
|
+
const trimmed = line.trimStart()
|
|
154
|
+
const step = stepFence(fence, trimmed, fenceMarker(trimmed))
|
|
155
|
+
fence = step.fence
|
|
156
|
+
if (step.fenced) continue
|
|
158
157
|
// Backreference so a multi-backtick span (``literal `@x` backticks``) strips whole.
|
|
159
158
|
const withoutSpans = line.replace(/(`+)[^`]*?\1/g, '')
|
|
160
159
|
for (const match of withoutSpans.matchAll(/(^|\s)@(\S+)/g)) targets.push(match[2])
|
|
@@ -222,8 +221,10 @@ function collectFrom(scan: ImportScan, content: string, fromDir: string, depth:
|
|
|
222
221
|
const file = readImport(target, fromDir, scan.home, scan.allowedRoots, scan.seen, scan.isExcluded)
|
|
223
222
|
if (!file) continue
|
|
224
223
|
scan.budget.files -= 1
|
|
225
|
-
|
|
226
|
-
|
|
224
|
+
// The budget is bytes: a string slice counts UTF-16 units and lets CJK text through
|
|
225
|
+
// at three times the budget without ever reaching the truncation marker.
|
|
226
|
+
const kept = sliceBytes(file.body, scan.budget.bytes)
|
|
227
|
+
scan.budget.bytes -= Buffer.byteLength(kept)
|
|
227
228
|
const body = kept.length < file.body.length ? `${kept.trim()}\n${IMPORT_TRUNCATED_MARKER}` : kept.trim()
|
|
228
229
|
// Comments are stripped before the scan for further imports, so a
|
|
229
230
|
// commented-out @import stays dead at every depth, matching the top level
|
|
@@ -194,6 +194,32 @@ export default function gitCheckpointExtension(pi: ExtensionAPI) {
|
|
|
194
194
|
}
|
|
195
195
|
// Written on every start, so repos that predate the sidecar pick it up too.
|
|
196
196
|
rememberWorkTree(shadowDir, ctx.cwd)
|
|
197
|
+
await mirrorLocalExcludes(ctx.cwd)
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
/** git reads ignore rules from the tree's .gitignore files, the user's global excludes,
|
|
201
|
+
* and $GIT_DIR/info/exclude. The shadow is the GIT_DIR here, so the repo's own
|
|
202
|
+
* .git/info/exclude (where secrets and scratch that must never be committed live)
|
|
203
|
+
* would be snapshotted and restored. Mirror it into the shadow on every start; the
|
|
204
|
+
* global excludes stay untouched (core.excludesFile is single-valued, so pointing it
|
|
205
|
+
* at the repo file would replace them). */
|
|
206
|
+
async function mirrorLocalExcludes(cwd: string): Promise<void> {
|
|
207
|
+
if (!shadowDir) return
|
|
208
|
+
// Resolved through git so a linked worktree maps to its common dir; outside a repo
|
|
209
|
+
// git exits 128 and there is nothing to mirror.
|
|
210
|
+
const located = await pi.exec('git', ['rev-parse', '--git-path', 'info/exclude'], { cwd })
|
|
211
|
+
const target = path.join(shadowDir, 'info', 'exclude')
|
|
212
|
+
try {
|
|
213
|
+
const source = located.code === 0 ? path.resolve(cwd, located.stdout.trim()) : undefined
|
|
214
|
+
if (source && fs.existsSync(source)) {
|
|
215
|
+
fs.mkdirSync(path.dirname(target), { recursive: true })
|
|
216
|
+
fs.copyFileSync(source, target)
|
|
217
|
+
} else {
|
|
218
|
+
fs.rmSync(target, { force: true })
|
|
219
|
+
}
|
|
220
|
+
} catch {
|
|
221
|
+
// A failed mirror only means local excludes are not honored this session.
|
|
222
|
+
}
|
|
197
223
|
}
|
|
198
224
|
|
|
199
225
|
/** `checkout -f <ref> -- .` errors when the ref's tree holds no files, so an empty
|
|
@@ -0,0 +1,522 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Goal Extension
|
|
3
|
+
*
|
|
4
|
+
* Claude Code's /goal: a completion condition the session keeps working toward. After
|
|
5
|
+
* each turn a separate model judges the condition against the conversation and returns
|
|
6
|
+
* met, not yet met (its reason becomes the next turn's guidance), or impossible. Claude
|
|
7
|
+
* implements it as a session-scoped prompt-based Stop hook; pi-code runs the same loop
|
|
8
|
+
* on agent_end next to the hooks extension's Stop path:
|
|
9
|
+
* - `/goal <condition>` sets (or replaces) the goal and starts a turn with Claude's
|
|
10
|
+
* kickoff directive; `/goal` shows status; `/goal clear` (stop/off/reset/none/cancel)
|
|
11
|
+
* removes it. The condition is capped at 4,000 characters.
|
|
12
|
+
* - The evaluator is the session model, or ANTHROPIC_DEFAULT_HAIKU_MODEL resolved against
|
|
13
|
+
* the models this user can run. It reads the branch transcript trimmed to half its
|
|
14
|
+
* context window and cannot run tools, so the condition must be provable from output.
|
|
15
|
+
* - The Stop hooks' consecutive-block cap (CLAUDE_CODE_STOP_HOOK_BLOCK_CAP, default 8)
|
|
16
|
+
* bounds a stalled loop: that many not-met verdicts in a row on turns that used no tool
|
|
17
|
+
* pause the loop with a warning, goal still set, until the next user prompt.
|
|
18
|
+
* - Evaluation is skipped while a subagent is still running (tracked from the subagent
|
|
19
|
+
* extension's bus events; pi has no other background work). A check-in turn is
|
|
20
|
+
* injected once the wait reaches CLAUDE_CODE_GOAL_CHECKIN_MINUTES (30, doubling up to
|
|
21
|
+
* four times the first interval, at most three idle check-ins between user prompts;
|
|
22
|
+
* 0 turns check-ins off).
|
|
23
|
+
* - A turn ended by the user (Esc) or by an error is not evaluated, and an Esc during
|
|
24
|
+
* the evaluation cancels it with the goal left set. An unrecoverable
|
|
25
|
+
* error (authentication, credits, context overflow, model unavailable) clears the goal
|
|
26
|
+
* with a warning once the run settles; a transient one leaves it set.
|
|
27
|
+
* - The active goal persists as a session entry and is restored on resume or reload with
|
|
28
|
+
* the turn count, timer, and token baseline reset; an achieved, failed, or cleared goal
|
|
29
|
+
* is not restored.
|
|
30
|
+
* - Gated as Claude documents: unavailable when hooks are restricted (disableAllHooks or
|
|
31
|
+
* allowManagedHooksOnly) or the project is not trusted, with the reason shown.
|
|
32
|
+
* - Headless (`pi -p "/goal ..."`), the command holds the process open until the goal
|
|
33
|
+
* resolves or pauses, and warnings also go to stderr since there is no notify surface.
|
|
34
|
+
*
|
|
35
|
+
* Docs: https://code.claude.com/docs/en/goal.md
|
|
36
|
+
*/
|
|
37
|
+
|
|
38
|
+
import * as os from 'node:os'
|
|
39
|
+
import type { ExtensionAPI, ExtensionContext } from '@earendil-works/pi-coding-agent'
|
|
40
|
+
|
|
41
|
+
import { hookFiles, readSettingsDisableAllHooks, stopHookBlockCap } from './hooks/index.js'
|
|
42
|
+
import {
|
|
43
|
+
checkinIntervalMs,
|
|
44
|
+
checkinText,
|
|
45
|
+
classifyUnrecoverable,
|
|
46
|
+
EVALUATOR_SYSTEM,
|
|
47
|
+
evaluatorPrompt,
|
|
48
|
+
formatAchievedGoal,
|
|
49
|
+
formatActiveGoal,
|
|
50
|
+
formatDuration,
|
|
51
|
+
formatTokens,
|
|
52
|
+
GOAL_CONDITION_MAX_CHARS,
|
|
53
|
+
type GoalSummary,
|
|
54
|
+
type GoalVerdict,
|
|
55
|
+
isClearAlias,
|
|
56
|
+
kickoffPrompt,
|
|
57
|
+
MAX_IDLE_CHECKINS,
|
|
58
|
+
NO_GOAL_TEXT,
|
|
59
|
+
parseVerdict,
|
|
60
|
+
type RunningWork,
|
|
61
|
+
renderTranscript,
|
|
62
|
+
summaryText,
|
|
63
|
+
} from './internal/goal-evaluator.js'
|
|
64
|
+
import { readManagedSettings } from './internal/managed-settings.js'
|
|
65
|
+
import { completeText } from './internal/model-complete.js'
|
|
66
|
+
import { resolveModelOverride } from './internal/model-lookup.js'
|
|
67
|
+
import { isProjectApprovedSilently } from './internal/project-approval.js'
|
|
68
|
+
import { isSubagentPhaseEvent, SUBAGENT_CHANNEL } from './internal/subagent-events.js'
|
|
69
|
+
|
|
70
|
+
/** Session entry type the goal state persists under, and the custom message type its
|
|
71
|
+
* transcript lines (kickoff, verdicts, check-ins) carry. */
|
|
72
|
+
const GOAL_ENTRY = 'goal'
|
|
73
|
+
const GOAL_MESSAGE = 'goal'
|
|
74
|
+
/** Claude's prompt-hook default; a slow evaluator is an error, not a stall. */
|
|
75
|
+
const EVALUATOR_TIMEOUT_MS = 30_000
|
|
76
|
+
/** A verdict is one JSON object; the cap stops a runaway reply. */
|
|
77
|
+
const EVALUATOR_MAX_TOKENS = 512
|
|
78
|
+
/** Claude trims the transcript to half the evaluator's context window. */
|
|
79
|
+
const TRANSCRIPT_WINDOW_SHARE = 0.5
|
|
80
|
+
const CHARS_PER_TOKEN = 4
|
|
81
|
+
const DEFAULT_CONTEXT_WINDOW = 200_000
|
|
82
|
+
/** The `◎ goal <elapsed>` indicator re-renders on this cadence while a goal is active. */
|
|
83
|
+
const INDICATOR_REFRESH_MS = 60_000
|
|
84
|
+
|
|
85
|
+
const HOOKS_GATE = "/goal can't run while hooks are restricted (disableAllHooks or allowManagedHooksOnly is set in settings or by policy)."
|
|
86
|
+
const TRUST_GATE = '/goal is only available in trusted workspaces. Restart, accept the trust dialog, and try again.'
|
|
87
|
+
|
|
88
|
+
type GoalState = 'active' | 'cleared' | 'achieved' | 'failed'
|
|
89
|
+
|
|
90
|
+
interface GoalEntry {
|
|
91
|
+
state: GoalState
|
|
92
|
+
condition: string
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
interface ActiveGoal {
|
|
96
|
+
condition: string
|
|
97
|
+
setAt: number
|
|
98
|
+
iterations: number
|
|
99
|
+
lastReason?: string
|
|
100
|
+
/** Session token total when the goal was set; spend is measured from here. */
|
|
101
|
+
tokensAtStart: number
|
|
102
|
+
/** Evaluator calls are not on the session total; they count toward the goal's spend. */
|
|
103
|
+
evaluatorTokens: number
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
interface TurnMessage {
|
|
107
|
+
role?: string
|
|
108
|
+
stopReason?: string
|
|
109
|
+
errorMessage?: string
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
interface TokenUsage {
|
|
113
|
+
totalTokens?: number
|
|
114
|
+
input?: number
|
|
115
|
+
output?: number
|
|
116
|
+
cacheRead?: number
|
|
117
|
+
cacheWrite?: number
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
function usageTokens(usage: TokenUsage | undefined): number {
|
|
121
|
+
if (!usage) return 0
|
|
122
|
+
return usage.totalTokens ?? (usage.input ?? 0) + (usage.output ?? 0) + (usage.cacheRead ?? 0) + (usage.cacheWrite ?? 0)
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
function lastAssistant(messages: readonly unknown[]): TurnMessage | undefined {
|
|
126
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
127
|
+
const message = messages[i] as TurnMessage
|
|
128
|
+
if (message?.role === 'assistant') return message
|
|
129
|
+
}
|
|
130
|
+
return undefined
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
type SessionReader = Pick<ExtensionContext, 'sessionManager'>
|
|
134
|
+
|
|
135
|
+
/** The branch as the evaluator reads it: messages plus the goal's own transcript lines,
|
|
136
|
+
* which live in custom_message entries. */
|
|
137
|
+
function branchMessages(ctx: SessionReader): unknown[] {
|
|
138
|
+
const messages: unknown[] = []
|
|
139
|
+
for (const entry of ctx.sessionManager.getBranch()) {
|
|
140
|
+
if (entry.type === 'message') messages.push(entry.message)
|
|
141
|
+
else if (entry.type === 'custom_message') messages.push({ role: 'custom', customType: entry.customType, content: entry.content })
|
|
142
|
+
}
|
|
143
|
+
return messages
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/** Cumulative assistant usage on the branch: the session's token spend so far. */
|
|
147
|
+
function sessionTokens(ctx: SessionReader): number {
|
|
148
|
+
let total = 0
|
|
149
|
+
for (const entry of ctx.sessionManager.getBranch()) {
|
|
150
|
+
if (entry.type === 'message' && entry.message.role === 'assistant') total += usageTokens(entry.message.usage as TokenUsage | undefined)
|
|
151
|
+
}
|
|
152
|
+
return total
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/** The newest goal entry on the branch, which is the goal's persisted state. */
|
|
156
|
+
function lastGoalEntry(ctx: SessionReader): GoalEntry | undefined {
|
|
157
|
+
const branch = ctx.sessionManager.getBranch()
|
|
158
|
+
for (let i = branch.length - 1; i >= 0; i--) {
|
|
159
|
+
const entry = branch[i]
|
|
160
|
+
if (entry.type !== 'custom' || entry.customType !== GOAL_ENTRY) continue
|
|
161
|
+
const data = entry.data as Partial<GoalEntry> | undefined
|
|
162
|
+
return typeof data?.condition === 'string' && typeof data.state === 'string' ? { state: data.state, condition: data.condition } : undefined
|
|
163
|
+
}
|
|
164
|
+
return undefined
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/** Claude's availability rule: /goal rides the hooks system, so a hooks restriction or an
|
|
168
|
+
* untrusted workspace refuses it with the reason. Checked in Claude's order. */
|
|
169
|
+
function goalUnavailable(ctx: Pick<ExtensionContext, 'cwd' | 'isProjectTrusted' | 'ui'>): string | undefined {
|
|
170
|
+
const managed = readManagedSettings()
|
|
171
|
+
const trusted = isProjectApprovedSilently(ctx)
|
|
172
|
+
const restricted = managed.disableAllHooks === true || managed.allowManagedHooksOnly === true || readSettingsDisableAllHooks(hookFiles(ctx.cwd, os.homedir(), trusted))
|
|
173
|
+
if (restricted) return HOOKS_GATE
|
|
174
|
+
if (!trusted) return TRUST_GATE
|
|
175
|
+
return undefined
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/** Claude evaluates on its small fast model, ANTHROPIC_DEFAULT_HAIKU_MODEL overriding it;
|
|
179
|
+
* pi has no such tier, so the session model stands in unless the override names one of
|
|
180
|
+
* the models this user can run. */
|
|
181
|
+
function evaluatorModel(ctx: ExtensionContext): ExtensionContext['model'] {
|
|
182
|
+
return resolveModelOverride(ctx, process.env.ANTHROPIC_DEFAULT_HAIKU_MODEL?.trim() || undefined)
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
function transcriptBudget(model: { contextWindow?: number }): number {
|
|
186
|
+
return Math.floor((model.contextWindow || DEFAULT_CONTEXT_WINDOW) * TRANSCRIPT_WINDOW_SHARE * CHARS_PER_TOKEN)
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
export default function goalExtension(pi: ExtensionAPI) {
|
|
190
|
+
let goal: ActiveGoal | undefined
|
|
191
|
+
/** The last goal achieved this session, for the status view after it clears. */
|
|
192
|
+
let achieved: GoalSummary | undefined
|
|
193
|
+
/** The live context, for timer callbacks that fire outside any pi event. */
|
|
194
|
+
let sessionCtx: ExtensionContext | undefined
|
|
195
|
+
/** Bumped whenever the goal is set, cleared, or the session changes, so an evaluation
|
|
196
|
+
* that resolves after any of those is dropped instead of steering the wrong goal. */
|
|
197
|
+
let generation = 0
|
|
198
|
+
/** Whether the run now ending executed a tool: Claude's "progress" for the block cap. */
|
|
199
|
+
let turnUsedTools = false
|
|
200
|
+
/** Not-met verdicts in a row on turns that used no tool. */
|
|
201
|
+
let noProgressStreak = 0
|
|
202
|
+
/** The error the last run ended on, judged once the run settles (past pi's retries). */
|
|
203
|
+
let lastTurnError: string | undefined
|
|
204
|
+
/** Subagents still running, by id, from the subagent extension's bus events. */
|
|
205
|
+
const running = new Map<string, string>()
|
|
206
|
+
let deferredSince: number | undefined
|
|
207
|
+
let checkinsDelivered = 0
|
|
208
|
+
let idleCheckins = 0
|
|
209
|
+
/** A check-in came due while a turn was running: deliver it at the next turn end. */
|
|
210
|
+
let checkinDue = false
|
|
211
|
+
let checkinTimer: ReturnType<typeof setTimeout> | undefined
|
|
212
|
+
let indicatorTimer: ReturnType<typeof setInterval> | undefined
|
|
213
|
+
|
|
214
|
+
/** A goal line in the transcript, which the model also reads. A timer can outlive the
|
|
215
|
+
* session that armed it, and sendMessage throws on a disposed one. */
|
|
216
|
+
function send(content: string, options: { triggerTurn: boolean; deliverAs?: 'followUp' }): void {
|
|
217
|
+
try {
|
|
218
|
+
pi.sendMessage({ customType: GOAL_MESSAGE, content, display: true }, options)
|
|
219
|
+
} catch {
|
|
220
|
+
// Disposed session: nothing left to tell.
|
|
221
|
+
}
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
/** A line for the user. Headless runs have no notify surface, and a refusal or a
|
|
225
|
+
* loop that stops there without a word would look like a hang, so it also goes to
|
|
226
|
+
* stderr (where Claude's -p mode prints its goal messages). */
|
|
227
|
+
function tell(ctx: ExtensionContext, text: string, level: 'info' | 'warning' | 'error'): void {
|
|
228
|
+
ctx.ui.notify(text, level)
|
|
229
|
+
if (!ctx.hasUI) process.stderr.write(`${text}\n`)
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
function currentSummary(ctx: SessionReader): GoalSummary | undefined {
|
|
233
|
+
if (!goal) return undefined
|
|
234
|
+
return { condition: goal.condition, durationMs: Date.now() - goal.setAt, iterations: goal.iterations, tokens: sessionTokens(ctx) - goal.tokensAtStart + goal.evaluatorTokens, lastReason: goal.lastReason }
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
function showIndicator(ctx: ExtensionContext): void {
|
|
238
|
+
if (!goal) return
|
|
239
|
+
ctx.ui.setStatus('goal', ctx.ui.theme.fg('accent', `◎ goal ${formatDuration(Date.now() - goal.setAt)}`))
|
|
240
|
+
if (!indicatorTimer) {
|
|
241
|
+
indicatorTimer = setInterval(() => {
|
|
242
|
+
if (sessionCtx) showIndicator(sessionCtx)
|
|
243
|
+
}, INDICATOR_REFRESH_MS)
|
|
244
|
+
indicatorTimer.unref?.()
|
|
245
|
+
}
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
function stopIndicator(ctx: ExtensionContext): void {
|
|
249
|
+
clearInterval(indicatorTimer)
|
|
250
|
+
indicatorTimer = undefined
|
|
251
|
+
ctx.ui.setStatus('goal', undefined)
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
function stopCheckin(): void {
|
|
255
|
+
clearTimeout(checkinTimer)
|
|
256
|
+
checkinTimer = undefined
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
function endDeferral(): void {
|
|
260
|
+
deferredSince = undefined
|
|
261
|
+
checkinDue = false
|
|
262
|
+
stopCheckin()
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
/** Make `condition` the active goal with fresh counters; the entry is the caller's. */
|
|
266
|
+
function activate(ctx: ExtensionContext, condition: string): void {
|
|
267
|
+
goal = { condition, setAt: Date.now(), iterations: 0, tokensAtStart: sessionTokens(ctx), evaluatorTokens: 0 }
|
|
268
|
+
generation += 1
|
|
269
|
+
noProgressStreak = 0
|
|
270
|
+
checkinsDelivered = 0
|
|
271
|
+
idleCheckins = 0
|
|
272
|
+
endDeferral()
|
|
273
|
+
showIndicator(ctx)
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
/** Drop the active goal, recording how it ended; returns its condition. */
|
|
277
|
+
function endGoal(ctx: ExtensionContext, state: Exclude<GoalState, 'active'>): string | undefined {
|
|
278
|
+
const ended = goal
|
|
279
|
+
if (!ended) return undefined
|
|
280
|
+
goal = undefined
|
|
281
|
+
generation += 1
|
|
282
|
+
endDeferral()
|
|
283
|
+
stopIndicator(ctx)
|
|
284
|
+
pi.appendEntry(GOAL_ENTRY, { state, condition: ended.condition } satisfies GoalEntry)
|
|
285
|
+
return ended.condition
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
function finish(ctx: ExtensionContext, state: 'achieved' | 'failed', reason: string): void {
|
|
289
|
+
const summary = currentSummary(ctx)
|
|
290
|
+
if (!summary) return
|
|
291
|
+
if (state === 'achieved') achieved = summary
|
|
292
|
+
endGoal(ctx, state)
|
|
293
|
+
const head = state === 'achieved' ? `Goal achieved (${summaryText(summary)}): ${summary.condition}` : `Goal could not be achieved (${summaryText(summary)}): ${summary.condition}`
|
|
294
|
+
send(reason ? `${head}\nEvaluator: ${reason}` : head, { triggerTurn: false })
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
/** Claude feeds a not-met reason back as the next turn, under the Stop hooks' cap. */
|
|
298
|
+
function continueGoal(ctx: ExtensionContext, active: ActiveGoal, reason: string): void {
|
|
299
|
+
noProgressStreak = turnUsedTools ? 0 : noProgressStreak + 1
|
|
300
|
+
const cap = stopHookBlockCap()
|
|
301
|
+
showIndicator(ctx)
|
|
302
|
+
if (noProgressStreak >= cap) {
|
|
303
|
+
noProgressStreak = 0
|
|
304
|
+
tell(ctx, `Goal paused after ${cap} turns in a row without tool use; it stays set and evaluation resumes after your next prompt.`, 'warning')
|
|
305
|
+
return
|
|
306
|
+
}
|
|
307
|
+
const summary = currentSummary(ctx)
|
|
308
|
+
const spent = summary ? ` · ${formatDuration(summary.durationMs)} · ${formatTokens(summary.tokens)} tokens` : ''
|
|
309
|
+
send(`Goal not yet met (turn ${active.iterations}${spent}): ${reason}\nGoal: ${active.condition}`, { triggerTurn: true })
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
async function askEvaluator(ctx: ExtensionContext, active: ActiveGoal, model: NonNullable<ExtensionContext['model']>): Promise<GoalVerdict> {
|
|
313
|
+
const transcript = renderTranscript(branchMessages(ctx), transcriptBudget(model))
|
|
314
|
+
// Esc during the evaluation aborts the run's signal; the call must die with it, or
|
|
315
|
+
// a late verdict would queue the next goal turn into a run the user just stopped.
|
|
316
|
+
const deadline = AbortSignal.timeout(EVALUATOR_TIMEOUT_MS)
|
|
317
|
+
const signal = ctx.signal ? AbortSignal.any([ctx.signal, deadline]) : deadline
|
|
318
|
+
// Claude runs its evaluator with thinking disabled: the verdict is mechanical.
|
|
319
|
+
// completeText requests no thinking level, which is the same for pi.
|
|
320
|
+
const { text, usage } = await completeText(model, evaluatorPrompt(transcript, active.condition), { system: EVALUATOR_SYSTEM, maxTokens: EVALUATOR_MAX_TOKENS, signal })
|
|
321
|
+
active.evaluatorTokens += usageTokens(usage as TokenUsage | undefined)
|
|
322
|
+
const verdict = parseVerdict(text)
|
|
323
|
+
if (!verdict) throw new Error(`unreadable verdict: ${text.slice(0, 200)}`)
|
|
324
|
+
return verdict
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
async function evaluate(ctx: ExtensionContext): Promise<void> {
|
|
328
|
+
const active = goal
|
|
329
|
+
if (!active) return
|
|
330
|
+
const startedGeneration = generation
|
|
331
|
+
const model = evaluatorModel(ctx)
|
|
332
|
+
if (!model) {
|
|
333
|
+
tell(ctx, 'Goal evaluator has no model to run on; the goal stays set.', 'warning')
|
|
334
|
+
return
|
|
335
|
+
}
|
|
336
|
+
let verdict: GoalVerdict
|
|
337
|
+
try {
|
|
338
|
+
verdict = await askEvaluator(ctx, active, model)
|
|
339
|
+
} catch (error) {
|
|
340
|
+
// A user interrupt is not an evaluator failure: the goal stays, nothing to say.
|
|
341
|
+
if (ctx.signal?.aborted) return
|
|
342
|
+
// No verdict is a hook error in Claude's terms: the turn ends and the goal stays.
|
|
343
|
+
if (generation === startedGeneration) tell(ctx, `Goal evaluator error: ${error instanceof Error ? error.message : String(error)}. The goal stays set; the next turn is evaluated again.`, 'warning')
|
|
344
|
+
return
|
|
345
|
+
}
|
|
346
|
+
// Interrupted, cleared, or replaced during the await: this verdict must not act.
|
|
347
|
+
if (ctx.signal?.aborted || generation !== startedGeneration) return
|
|
348
|
+
active.iterations += 1
|
|
349
|
+
active.lastReason = verdict.reason
|
|
350
|
+
if (verdict.ok) finish(ctx, 'achieved', verdict.reason)
|
|
351
|
+
else if (verdict.impossible) finish(ctx, 'failed', verdict.reason)
|
|
352
|
+
else continueGoal(ctx, active, verdict.reason)
|
|
353
|
+
}
|
|
354
|
+
|
|
355
|
+
function deliverCheckin(paused: boolean): void {
|
|
356
|
+
if (!goal) return
|
|
357
|
+
checkinDue = false
|
|
358
|
+
checkinsDelivered += 1
|
|
359
|
+
const work: RunningWork[] = [...running].map(([id, agentType]) => ({ id, agentType }))
|
|
360
|
+
send(checkinText(goal.condition, Date.now() - (deferredSince ?? Date.now()), work, paused), { triggerTurn: true })
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
function onCheckinDue(): void {
|
|
364
|
+
checkinTimer = undefined
|
|
365
|
+
if (!goal || !sessionCtx) return
|
|
366
|
+
// Claude delivers a due check-in at the next turn end when a turn is running, and
|
|
367
|
+
// only starts idle turns for it up to the per-prompt cap.
|
|
368
|
+
if (!sessionCtx.isIdle() || idleCheckins >= MAX_IDLE_CHECKINS) {
|
|
369
|
+
checkinDue = true
|
|
370
|
+
return
|
|
371
|
+
}
|
|
372
|
+
idleCheckins += 1
|
|
373
|
+
deliverCheckin(idleCheckins >= MAX_IDLE_CHECKINS)
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
function armCheckin(): void {
|
|
377
|
+
if (checkinTimer) return
|
|
378
|
+
const ms = checkinIntervalMs(process.env, checkinsDelivered)
|
|
379
|
+
if (ms <= 0) return
|
|
380
|
+
checkinTimer = setTimeout(onCheckinDue, ms)
|
|
381
|
+
checkinTimer.unref?.()
|
|
382
|
+
}
|
|
383
|
+
|
|
384
|
+
/** Background work keeps the goal waiting: no verdict this turn, a check-in later. */
|
|
385
|
+
function deferEvaluation(): void {
|
|
386
|
+
deferredSince ??= Date.now()
|
|
387
|
+
if (checkinDue) {
|
|
388
|
+
deliverCheckin(false)
|
|
389
|
+
return
|
|
390
|
+
}
|
|
391
|
+
armCheckin()
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
function statusText(ctx: SessionReader): string {
|
|
395
|
+
const summary = currentSummary(ctx)
|
|
396
|
+
if (summary) return formatActiveGoal(summary)
|
|
397
|
+
if (achieved) return formatAchievedGoal(achieved)
|
|
398
|
+
return NO_GOAL_TEXT
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
pi.registerCommand('goal', {
|
|
402
|
+
description: 'Set a goal the session keeps working toward until a separate check confirms it is met; /goal shows status, /goal clear stops',
|
|
403
|
+
handler: async (args, ctx) => {
|
|
404
|
+
sessionCtx = ctx
|
|
405
|
+
const condition = args.trim()
|
|
406
|
+
if (condition === '') {
|
|
407
|
+
tell(ctx, statusText(ctx), 'info')
|
|
408
|
+
return
|
|
409
|
+
}
|
|
410
|
+
if (isClearAlias(condition)) {
|
|
411
|
+
const cleared = endGoal(ctx, 'cleared')
|
|
412
|
+
if (cleared === undefined) tell(ctx, 'No goal set', 'info')
|
|
413
|
+
else send(`Goal cleared: ${cleared}`, { triggerTurn: false })
|
|
414
|
+
return
|
|
415
|
+
}
|
|
416
|
+
if (condition.length > GOAL_CONDITION_MAX_CHARS) {
|
|
417
|
+
tell(ctx, `Goal condition is limited to ${GOAL_CONDITION_MAX_CHARS} characters (got ${condition.length})`, 'error')
|
|
418
|
+
return
|
|
419
|
+
}
|
|
420
|
+
const blocked = goalUnavailable(ctx)
|
|
421
|
+
if (blocked) {
|
|
422
|
+
tell(ctx, blocked, 'error')
|
|
423
|
+
return
|
|
424
|
+
}
|
|
425
|
+
// A new goal replaces the current one, as Claude documents; the entry written here
|
|
426
|
+
// is the state a resume restores.
|
|
427
|
+
activate(ctx, condition)
|
|
428
|
+
pi.appendEntry(GOAL_ENTRY, { state: 'active', condition } satisfies GoalEntry)
|
|
429
|
+
tell(ctx, `Goal set: ${condition}`, 'info')
|
|
430
|
+
// Setting a goal starts a turn immediately with the condition as the directive; a
|
|
431
|
+
// send during streaming queues it as a follow-up turn.
|
|
432
|
+
send(kickoffPrompt(condition), ctx.isIdle() ? { triggerTurn: true } : { triggerTurn: true, deliverAs: 'followUp' })
|
|
433
|
+
// A headless run (`pi -p "/goal ..."`) exits as soon as the command returns, before
|
|
434
|
+
// the kickoff turn runs. pi marks the run active synchronously on the send, and the
|
|
435
|
+
// continuations queued at agent_end extend that run, so waiting here holds the
|
|
436
|
+
// process open until the goal resolves, as Claude's -p mode does. The TUI's main
|
|
437
|
+
// loop returns to the editor on its own, so it must not block on the loop.
|
|
438
|
+
if (!ctx.hasUI) await ctx.waitForIdle()
|
|
439
|
+
},
|
|
440
|
+
})
|
|
441
|
+
|
|
442
|
+
pi.events.on(SUBAGENT_CHANNEL, (data) => {
|
|
443
|
+
if (!isSubagentPhaseEvent(data)) return
|
|
444
|
+
if (data.phase === 'start') running.set(data.agentId, data.agentType)
|
|
445
|
+
else running.delete(data.agentId)
|
|
446
|
+
})
|
|
447
|
+
|
|
448
|
+
pi.on('session_start', (_event, ctx) => {
|
|
449
|
+
// One extension instance serves every session: drop the previous session's goal and
|
|
450
|
+
// timers before reading this session's persisted state.
|
|
451
|
+
sessionCtx = ctx
|
|
452
|
+
goal = undefined
|
|
453
|
+
achieved = undefined
|
|
454
|
+
generation += 1
|
|
455
|
+
turnUsedTools = false
|
|
456
|
+
noProgressStreak = 0
|
|
457
|
+
lastTurnError = undefined
|
|
458
|
+
checkinsDelivered = 0
|
|
459
|
+
idleCheckins = 0
|
|
460
|
+
endDeferral()
|
|
461
|
+
stopIndicator(ctx)
|
|
462
|
+
const entry = lastGoalEntry(ctx)
|
|
463
|
+
if (entry?.state !== 'active') return
|
|
464
|
+
// Claude restores a still-active goal on resume with its counters reset.
|
|
465
|
+
activate(ctx, entry.condition)
|
|
466
|
+
tell(ctx, `Goal restored: ${entry.condition}`, 'info')
|
|
467
|
+
})
|
|
468
|
+
|
|
469
|
+
pi.on('session_shutdown', () => {
|
|
470
|
+
stopCheckin()
|
|
471
|
+
clearInterval(indicatorTimer)
|
|
472
|
+
indicatorTimer = undefined
|
|
473
|
+
})
|
|
474
|
+
|
|
475
|
+
pi.on('agent_start', () => {
|
|
476
|
+
turnUsedTools = false
|
|
477
|
+
})
|
|
478
|
+
|
|
479
|
+
pi.on('tool_execution_start', () => {
|
|
480
|
+
turnUsedTools = true
|
|
481
|
+
})
|
|
482
|
+
|
|
483
|
+
pi.on('input', (event) => {
|
|
484
|
+
// Only genuine user input is progress: a goal continuation or a subagent prompt
|
|
485
|
+
// arrives with source 'extension'. Claude resets the block cap and the idle
|
|
486
|
+
// check-in allowance on the user's next prompt.
|
|
487
|
+
if (event.source === 'extension') return
|
|
488
|
+
noProgressStreak = 0
|
|
489
|
+
idleCheckins = 0
|
|
490
|
+
})
|
|
491
|
+
|
|
492
|
+
// On agent_end rather than agent_settled, for the reason hooks.ts gives: a peer
|
|
493
|
+
// extension can hold its agent_end handler on a dialog, which would starve a settle-
|
|
494
|
+
// based evaluation; and a continuation sent here is picked up as the run's own
|
|
495
|
+
// continuation rather than a fresh prompt.
|
|
496
|
+
pi.on('agent_end', async (event, ctx) => {
|
|
497
|
+
sessionCtx = ctx
|
|
498
|
+
const last = lastAssistant(event.messages)
|
|
499
|
+
lastTurnError = last?.stopReason === 'error' ? (last.errorMessage ?? 'unknown error') : undefined
|
|
500
|
+
// Claude runs no Stop evaluation after a user interrupt; a failed turn is judged once
|
|
501
|
+
// the run settles, so neither is evaluated here and the goal stays set.
|
|
502
|
+
if (!goal || last?.stopReason === 'aborted' || lastTurnError !== undefined) return
|
|
503
|
+
if (running.size > 0) {
|
|
504
|
+
deferEvaluation()
|
|
505
|
+
return
|
|
506
|
+
}
|
|
507
|
+
endDeferral()
|
|
508
|
+
await evaluate(ctx)
|
|
509
|
+
})
|
|
510
|
+
|
|
511
|
+
pi.on('agent_settled', (_event, ctx) => {
|
|
512
|
+
const message = lastTurnError
|
|
513
|
+
lastTurnError = undefined
|
|
514
|
+
if (!goal || message === undefined) return
|
|
515
|
+
const kind = classifyUnrecoverable(message)
|
|
516
|
+
if (!kind) return
|
|
517
|
+
endGoal(ctx, 'cleared')
|
|
518
|
+
const text = `Goal cleared after an unrecoverable error (${kind}): "${message}". Run /goal again to continue.`
|
|
519
|
+
send(text, { triggerTurn: false })
|
|
520
|
+
tell(ctx, text, 'warning')
|
|
521
|
+
})
|
|
522
|
+
}
|