pi-code 1.0.47 → 1.0.49

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -11,7 +11,7 @@
11
11
  [![Maintainability Rating](https://sonarcloud.io/api/project_badges/measure?project=ilovepixelart_pi-code&metric=sqale_rating)](https://sonarcloud.io/summary/new_code?id=ilovepixelart_pi-code)
12
12
  [![Security Rating](https://sonarcloud.io/api/project_badges/measure?project=ilovepixelart_pi-code&metric=security_rating)](https://sonarcloud.io/summary/new_code?id=ilovepixelart_pi-code)
13
13
 
14
- Claude Code experience for the [pi](https://pi.dev) coding agent, in one package. Point pi at a project that already has a `.claude/` directory and it reads your existing config: rules, commands, skills, hooks, output styles, MCP servers, and agents. It also adds the Claude Code features pi lacks: a todo overlay, checkpoints, memory, web search, and subagents.
14
+ Claude Code experience for the [pi](https://pi.dev) coding agent, in one package. Point pi at a project that already has a `.claude/` directory and it reads your existing config: rules, commands, skills, hooks, output styles, MCP servers, and agents. It also adds the Claude Code features pi lacks: a todo overlay, checkpoints, memory, web search, subagents, and goals.
15
15
 
16
16
  What a repository ships is treated as untrusted until you approve it: project MCP servers, hooks, agents, rules, output styles, commands and skills load only once you say yes.
17
17
 
@@ -53,9 +53,10 @@ Each topic links to its own doc with the full contract and any divergences from
53
53
  - **[Statusline](docs/statusline.md)** — your Claude `statusLine` command with the documented stdin JSON.
54
54
  - **[WebSearch / WebFetch](docs/web.md)** — key-free search and SSRF-guarded fetch.
55
55
  - **[Claude plugins](docs/plugins.md)** — installed marketplace plugins: commands, agents, hooks, MCP servers, styles, skills.
56
+ - **[Goal](docs/goal.md)**: `/goal <condition>` keeps the session working until a separate model check confirms the condition holds, with status, clear, block cap, background-work deferral, and resume.
56
57
  - **[Session extras](docs/session-extras.md)** — project trust, plan mode, todos, checkpoints/rewind, AskUserQuestion, notifications, think keywords, session titles, `/context`, `/init`.
57
58
 
58
- Slash commands: `/init`, `/context`, `/memory`, `/todos`, `/rewind`, `/tasks`, `/agents`, `/plan`, `/mcp`, `/hooks`, and `/output-style`, alongside your own `/dir:name` commands, `/skill:name` skills, `/plugin:name` plugin commands, and each connected server's `/mcp__server__prompt` prompts.
59
+ Slash commands: `/init`, `/context`, `/goal`, `/memory`, `/todos`, `/rewind`, `/tasks`, `/agents`, `/plan`, `/mcp`, `/hooks`, and `/output-style`, alongside your own `/dir:name` commands, `/skill:name` skills, `/plugin:name` plugin commands, and each connected server's `/mcp__server__prompt` prompts.
59
60
 
60
61
  pi has no general permission system, so most of what Claude routes through a permission prompt maps to hard behavior here: `allowed-tools` restricts the turn's tool set instead of pre-approving calls, and a hook that times out on PreToolUse or UserPromptSubmit fails closed. A hook's `permissionDecision: "ask"` is the exception: it shows a confirm dialog and lets the call through when you approve (a headless run has no dialog, so it blocks). Where a Claude restriction cannot be expressed at all (an argument-scoped grant in an agent's `tools:`), the definition is rejected rather than widened.
61
62
 
@@ -68,7 +68,8 @@ function parsePaths(frontmatter: string): string[] {
68
68
 
69
69
  /** Split YAML-ish frontmatter off the front of a rule file, extracting `paths`. */
70
70
  export function parseFrontmatter(content: string): Frontmatter {
71
- const match = /^---\r?\n([\s\S]*?)\r?\n---\r?\n?/.exec(content)
71
+ // A leading byte order mark (Windows editors add one) is part of the header, not the body.
72
+ const match = /^\uFEFF?---\r?\n([\s\S]*?)\r?\n---\r?\n?/.exec(content)
72
73
  if (!match) return { paths: [], body: content }
73
74
  return { paths: parsePaths(match[1]), body: content.slice(match[0].length) }
74
75
  }
@@ -197,7 +197,7 @@ function commandVars(ctx: { cwd: string }, filePath: string, plugin?: CommandPlu
197
197
 
198
198
  /** The exec seam expandCommand runs spans through; pi itself satisfies it. */
199
199
  interface SpanRunner {
200
- exec(command: string, args: string[], options?: { cwd?: string; timeout?: number }): Promise<{ stdout: string; stderr: string; code: number }>
200
+ exec(command: string, args: string[], options?: { cwd?: string; timeout?: number }): Promise<{ stdout: string; stderr: string; code: number; killed?: boolean }>
201
201
  }
202
202
 
203
203
  /**
@@ -231,7 +231,7 @@ export async function expandCommand(runner: SpanRunner, parsed: ParsedCommand, a
231
231
  // pwsh cannot merge a native command's stderr in-script (spanExec sets
232
232
  // mergeStreams), so it is appended here; the sh script merges via 2>&1.
233
233
  const stdout = run.mergeStreams ? result.stdout + result.stderr : result.stdout
234
- return { stdout, stderr: result.stderr, code: result.code }
234
+ return { stdout, stderr: result.stderr, code: result.code, killed: result.killed }
235
235
  }
236
236
  : async () => ({ stdout: SHELL_DISABLED_PLACEHOLDER, stderr: '', code: 0 })
237
237
  let expanded = await expandDynamicContent(withVars, ctx.cwd, exec, parsed.shell === 'powershell' ? 'powershell' : 'bash')
@@ -72,12 +72,13 @@ import type { ExtensionAPI } from '@earendil-works/pi-coding-agent'
72
72
  import { claudeConfigDir } from './internal/config-dir.js'
73
73
  import { type InstructionLoadEvent, memoryTypeForPath, publishInstructionLoad } from './internal/instruction-events.js'
74
74
  import { managedSettingsPath, readManagedSettings } from './internal/managed-settings.js'
75
+ import { sliceBytes } from './internal/output-guard.js'
75
76
  import { globToRegExpSource } from './internal/path-rules.js'
76
77
  import { isProjectApproved, isProjectApprovedSilently } from './internal/project-approval.js'
77
78
  import { ancestorFiles, findNearestFile, repoRoot } from './internal/project-root.js'
78
79
  import { claudeSettingsChain } from './internal/settings-chain.js'
79
80
  import { statToken } from './internal/stat-token.js'
80
- import { fenceMarker, stripBlockComments } from './internal/strip-comments.js'
81
+ import { type Fence, fenceMarker, stepFence, stripBlockComments } from './internal/strip-comments.js'
81
82
 
82
83
  /** Claude documents "a maximum depth of four hops" for recursive imports. */
83
84
  const MAX_IMPORT_DEPTH = 4
@@ -145,16 +146,14 @@ export const createImportBudget = (): ImportBudget => ({ files: MAX_IMPORT_FILES
145
146
  * imports neither in fenced code blocks (backtick or tilde) nor in inline spans. */
146
147
  function importTargets(content: string): string[] {
147
148
  const targets: string[] = []
148
- // A fence only closes with the character that opened it: a backtick-fenced
149
- // example may legitimately contain tilde-fence lines, and vice versa.
150
- let fence: string | null = null
149
+ // CommonMark fences: closed only by the same character in a run at least as long
150
+ // as the opener, so a backtick example may hold tilde lines or shorter fences.
151
+ let fence: Fence | null = null
151
152
  for (const line of content.split('\n')) {
152
- const marker = fenceMarker(line.trimStart())
153
- if (marker !== null && (fence === null || fence === marker)) {
154
- fence = fence === null ? marker : null
155
- continue
156
- }
157
- if (fence !== null) continue
153
+ const trimmed = line.trimStart()
154
+ const step = stepFence(fence, trimmed, fenceMarker(trimmed))
155
+ fence = step.fence
156
+ if (step.fenced) continue
158
157
  // Backreference so a multi-backtick span (``literal `@x` backticks``) strips whole.
159
158
  const withoutSpans = line.replace(/(`+)[^`]*?\1/g, '')
160
159
  for (const match of withoutSpans.matchAll(/(^|\s)@(\S+)/g)) targets.push(match[2])
@@ -222,8 +221,10 @@ function collectFrom(scan: ImportScan, content: string, fromDir: string, depth:
222
221
  const file = readImport(target, fromDir, scan.home, scan.allowedRoots, scan.seen, scan.isExcluded)
223
222
  if (!file) continue
224
223
  scan.budget.files -= 1
225
- const kept = file.body.slice(0, scan.budget.bytes)
226
- scan.budget.bytes -= kept.length
224
+ // The budget is bytes: a string slice counts UTF-16 units and lets CJK text through
225
+ // at three times the budget without ever reaching the truncation marker.
226
+ const kept = sliceBytes(file.body, scan.budget.bytes)
227
+ scan.budget.bytes -= Buffer.byteLength(kept)
227
228
  const body = kept.length < file.body.length ? `${kept.trim()}\n${IMPORT_TRUNCATED_MARKER}` : kept.trim()
228
229
  // Comments are stripped before the scan for further imports, so a
229
230
  // commented-out @import stays dead at every depth, matching the top level
@@ -194,6 +194,32 @@ export default function gitCheckpointExtension(pi: ExtensionAPI) {
194
194
  }
195
195
  // Written on every start, so repos that predate the sidecar pick it up too.
196
196
  rememberWorkTree(shadowDir, ctx.cwd)
197
+ await mirrorLocalExcludes(ctx.cwd)
198
+ }
199
+
200
+ /** git reads ignore rules from the tree's .gitignore files, the user's global excludes,
201
+ * and $GIT_DIR/info/exclude. The shadow is the GIT_DIR here, so the repo's own
202
+ * .git/info/exclude (where secrets and scratch that must never be committed live)
203
+ * would be snapshotted and restored. Mirror it into the shadow on every start; the
204
+ * global excludes stay untouched (core.excludesFile is single-valued, so pointing it
205
+ * at the repo file would replace them). */
206
+ async function mirrorLocalExcludes(cwd: string): Promise<void> {
207
+ if (!shadowDir) return
208
+ // Resolved through git so a linked worktree maps to its common dir; outside a repo
209
+ // git exits 128 and there is nothing to mirror.
210
+ const located = await pi.exec('git', ['rev-parse', '--git-path', 'info/exclude'], { cwd })
211
+ const target = path.join(shadowDir, 'info', 'exclude')
212
+ try {
213
+ const source = located.code === 0 ? path.resolve(cwd, located.stdout.trim()) : undefined
214
+ if (source && fs.existsSync(source)) {
215
+ fs.mkdirSync(path.dirname(target), { recursive: true })
216
+ fs.copyFileSync(source, target)
217
+ } else {
218
+ fs.rmSync(target, { force: true })
219
+ }
220
+ } catch {
221
+ // A failed mirror only means local excludes are not honored this session.
222
+ }
197
223
  }
198
224
 
199
225
  /** `checkout -f <ref> -- .` errors when the ref's tree holds no files, so an empty
@@ -0,0 +1,522 @@
1
+ /**
2
+ * Goal Extension
3
+ *
4
+ * Claude Code's /goal: a completion condition the session keeps working toward. After
5
+ * each turn a separate model judges the condition against the conversation and returns
6
+ * met, not yet met (its reason becomes the next turn's guidance), or impossible. Claude
7
+ * implements it as a session-scoped prompt-based Stop hook; pi-code runs the same loop
8
+ * on agent_end next to the hooks extension's Stop path:
9
+ * - `/goal <condition>` sets (or replaces) the goal and starts a turn with Claude's
10
+ * kickoff directive; `/goal` shows status; `/goal clear` (stop/off/reset/none/cancel)
11
+ * removes it. The condition is capped at 4,000 characters.
12
+ * - The evaluator is the session model, or ANTHROPIC_DEFAULT_HAIKU_MODEL resolved against
13
+ * the models this user can run. It reads the branch transcript trimmed to half its
14
+ * context window and cannot run tools, so the condition must be provable from output.
15
+ * - The Stop hooks' consecutive-block cap (CLAUDE_CODE_STOP_HOOK_BLOCK_CAP, default 8)
16
+ * bounds a stalled loop: that many not-met verdicts in a row on turns that used no tool
17
+ * pause the loop with a warning, goal still set, until the next user prompt.
18
+ * - Evaluation is skipped while a subagent is still running (tracked from the subagent
19
+ * extension's bus events; pi has no other background work). A check-in turn is
20
+ * injected once the wait reaches CLAUDE_CODE_GOAL_CHECKIN_MINUTES (30, doubling up to
21
+ * four times the first interval, at most three idle check-ins between user prompts;
22
+ * 0 turns check-ins off).
23
+ * - A turn ended by the user (Esc) or by an error is not evaluated, and an Esc during
24
+ * the evaluation cancels it with the goal left set. An unrecoverable
25
+ * error (authentication, credits, context overflow, model unavailable) clears the goal
26
+ * with a warning once the run settles; a transient one leaves it set.
27
+ * - The active goal persists as a session entry and is restored on resume or reload with
28
+ * the turn count, timer, and token baseline reset; an achieved, failed, or cleared goal
29
+ * is not restored.
30
+ * - Gated as Claude documents: unavailable when hooks are restricted (disableAllHooks or
31
+ * allowManagedHooksOnly) or the project is not trusted, with the reason shown.
32
+ * - Headless (`pi -p "/goal ..."`), the command holds the process open until the goal
33
+ * resolves or pauses, and warnings also go to stderr since there is no notify surface.
34
+ *
35
+ * Docs: https://code.claude.com/docs/en/goal.md
36
+ */
37
+
38
+ import * as os from 'node:os'
39
+ import type { ExtensionAPI, ExtensionContext } from '@earendil-works/pi-coding-agent'
40
+
41
+ import { hookFiles, readSettingsDisableAllHooks, stopHookBlockCap } from './hooks/index.js'
42
+ import {
43
+ checkinIntervalMs,
44
+ checkinText,
45
+ classifyUnrecoverable,
46
+ EVALUATOR_SYSTEM,
47
+ evaluatorPrompt,
48
+ formatAchievedGoal,
49
+ formatActiveGoal,
50
+ formatDuration,
51
+ formatTokens,
52
+ GOAL_CONDITION_MAX_CHARS,
53
+ type GoalSummary,
54
+ type GoalVerdict,
55
+ isClearAlias,
56
+ kickoffPrompt,
57
+ MAX_IDLE_CHECKINS,
58
+ NO_GOAL_TEXT,
59
+ parseVerdict,
60
+ type RunningWork,
61
+ renderTranscript,
62
+ summaryText,
63
+ } from './internal/goal-evaluator.js'
64
+ import { readManagedSettings } from './internal/managed-settings.js'
65
+ import { completeText } from './internal/model-complete.js'
66
+ import { resolveModelOverride } from './internal/model-lookup.js'
67
+ import { isProjectApprovedSilently } from './internal/project-approval.js'
68
+ import { isSubagentPhaseEvent, SUBAGENT_CHANNEL } from './internal/subagent-events.js'
69
+
70
+ /** Session entry type the goal state persists under, and the custom message type its
71
+ * transcript lines (kickoff, verdicts, check-ins) carry. */
72
+ const GOAL_ENTRY = 'goal'
73
+ const GOAL_MESSAGE = 'goal'
74
+ /** Claude's prompt-hook default; a slow evaluator is an error, not a stall. */
75
+ const EVALUATOR_TIMEOUT_MS = 30_000
76
+ /** A verdict is one JSON object; the cap stops a runaway reply. */
77
+ const EVALUATOR_MAX_TOKENS = 512
78
+ /** Claude trims the transcript to half the evaluator's context window. */
79
+ const TRANSCRIPT_WINDOW_SHARE = 0.5
80
+ const CHARS_PER_TOKEN = 4
81
+ const DEFAULT_CONTEXT_WINDOW = 200_000
82
+ /** The `◎ goal <elapsed>` indicator re-renders on this cadence while a goal is active. */
83
+ const INDICATOR_REFRESH_MS = 60_000
84
+
85
+ const HOOKS_GATE = "/goal can't run while hooks are restricted (disableAllHooks or allowManagedHooksOnly is set in settings or by policy)."
86
+ const TRUST_GATE = '/goal is only available in trusted workspaces. Restart, accept the trust dialog, and try again.'
87
+
88
+ type GoalState = 'active' | 'cleared' | 'achieved' | 'failed'
89
+
90
+ interface GoalEntry {
91
+ state: GoalState
92
+ condition: string
93
+ }
94
+
95
+ interface ActiveGoal {
96
+ condition: string
97
+ setAt: number
98
+ iterations: number
99
+ lastReason?: string
100
+ /** Session token total when the goal was set; spend is measured from here. */
101
+ tokensAtStart: number
102
+ /** Evaluator calls are not on the session total; they count toward the goal's spend. */
103
+ evaluatorTokens: number
104
+ }
105
+
106
+ interface TurnMessage {
107
+ role?: string
108
+ stopReason?: string
109
+ errorMessage?: string
110
+ }
111
+
112
+ interface TokenUsage {
113
+ totalTokens?: number
114
+ input?: number
115
+ output?: number
116
+ cacheRead?: number
117
+ cacheWrite?: number
118
+ }
119
+
120
+ function usageTokens(usage: TokenUsage | undefined): number {
121
+ if (!usage) return 0
122
+ return usage.totalTokens ?? (usage.input ?? 0) + (usage.output ?? 0) + (usage.cacheRead ?? 0) + (usage.cacheWrite ?? 0)
123
+ }
124
+
125
+ function lastAssistant(messages: readonly unknown[]): TurnMessage | undefined {
126
+ for (let i = messages.length - 1; i >= 0; i--) {
127
+ const message = messages[i] as TurnMessage
128
+ if (message?.role === 'assistant') return message
129
+ }
130
+ return undefined
131
+ }
132
+
133
+ type SessionReader = Pick<ExtensionContext, 'sessionManager'>
134
+
135
+ /** The branch as the evaluator reads it: messages plus the goal's own transcript lines,
136
+ * which live in custom_message entries. */
137
+ function branchMessages(ctx: SessionReader): unknown[] {
138
+ const messages: unknown[] = []
139
+ for (const entry of ctx.sessionManager.getBranch()) {
140
+ if (entry.type === 'message') messages.push(entry.message)
141
+ else if (entry.type === 'custom_message') messages.push({ role: 'custom', customType: entry.customType, content: entry.content })
142
+ }
143
+ return messages
144
+ }
145
+
146
+ /** Cumulative assistant usage on the branch: the session's token spend so far. */
147
+ function sessionTokens(ctx: SessionReader): number {
148
+ let total = 0
149
+ for (const entry of ctx.sessionManager.getBranch()) {
150
+ if (entry.type === 'message' && entry.message.role === 'assistant') total += usageTokens(entry.message.usage as TokenUsage | undefined)
151
+ }
152
+ return total
153
+ }
154
+
155
+ /** The newest goal entry on the branch, which is the goal's persisted state. */
156
+ function lastGoalEntry(ctx: SessionReader): GoalEntry | undefined {
157
+ const branch = ctx.sessionManager.getBranch()
158
+ for (let i = branch.length - 1; i >= 0; i--) {
159
+ const entry = branch[i]
160
+ if (entry.type !== 'custom' || entry.customType !== GOAL_ENTRY) continue
161
+ const data = entry.data as Partial<GoalEntry> | undefined
162
+ return typeof data?.condition === 'string' && typeof data.state === 'string' ? { state: data.state, condition: data.condition } : undefined
163
+ }
164
+ return undefined
165
+ }
166
+
167
+ /** Claude's availability rule: /goal rides the hooks system, so a hooks restriction or an
168
+ * untrusted workspace refuses it with the reason. Checked in Claude's order. */
169
+ function goalUnavailable(ctx: Pick<ExtensionContext, 'cwd' | 'isProjectTrusted' | 'ui'>): string | undefined {
170
+ const managed = readManagedSettings()
171
+ const trusted = isProjectApprovedSilently(ctx)
172
+ const restricted = managed.disableAllHooks === true || managed.allowManagedHooksOnly === true || readSettingsDisableAllHooks(hookFiles(ctx.cwd, os.homedir(), trusted))
173
+ if (restricted) return HOOKS_GATE
174
+ if (!trusted) return TRUST_GATE
175
+ return undefined
176
+ }
177
+
178
+ /** Claude evaluates on its small fast model, ANTHROPIC_DEFAULT_HAIKU_MODEL overriding it;
179
+ * pi has no such tier, so the session model stands in unless the override names one of
180
+ * the models this user can run. */
181
+ function evaluatorModel(ctx: ExtensionContext): ExtensionContext['model'] {
182
+ return resolveModelOverride(ctx, process.env.ANTHROPIC_DEFAULT_HAIKU_MODEL?.trim() || undefined)
183
+ }
184
+
185
+ function transcriptBudget(model: { contextWindow?: number }): number {
186
+ return Math.floor((model.contextWindow || DEFAULT_CONTEXT_WINDOW) * TRANSCRIPT_WINDOW_SHARE * CHARS_PER_TOKEN)
187
+ }
188
+
189
+ export default function goalExtension(pi: ExtensionAPI) {
190
+ let goal: ActiveGoal | undefined
191
+ /** The last goal achieved this session, for the status view after it clears. */
192
+ let achieved: GoalSummary | undefined
193
+ /** The live context, for timer callbacks that fire outside any pi event. */
194
+ let sessionCtx: ExtensionContext | undefined
195
+ /** Bumped whenever the goal is set, cleared, or the session changes, so an evaluation
196
+ * that resolves after any of those is dropped instead of steering the wrong goal. */
197
+ let generation = 0
198
+ /** Whether the run now ending executed a tool: Claude's "progress" for the block cap. */
199
+ let turnUsedTools = false
200
+ /** Not-met verdicts in a row on turns that used no tool. */
201
+ let noProgressStreak = 0
202
+ /** The error the last run ended on, judged once the run settles (past pi's retries). */
203
+ let lastTurnError: string | undefined
204
+ /** Subagents still running, by id, from the subagent extension's bus events. */
205
+ const running = new Map<string, string>()
206
+ let deferredSince: number | undefined
207
+ let checkinsDelivered = 0
208
+ let idleCheckins = 0
209
+ /** A check-in came due while a turn was running: deliver it at the next turn end. */
210
+ let checkinDue = false
211
+ let checkinTimer: ReturnType<typeof setTimeout> | undefined
212
+ let indicatorTimer: ReturnType<typeof setInterval> | undefined
213
+
214
+ /** A goal line in the transcript, which the model also reads. A timer can outlive the
215
+ * session that armed it, and sendMessage throws on a disposed one. */
216
+ function send(content: string, options: { triggerTurn: boolean; deliverAs?: 'followUp' }): void {
217
+ try {
218
+ pi.sendMessage({ customType: GOAL_MESSAGE, content, display: true }, options)
219
+ } catch {
220
+ // Disposed session: nothing left to tell.
221
+ }
222
+ }
223
+
224
+ /** A line for the user. Headless runs have no notify surface, and a refusal or a
225
+ * loop that stops there without a word would look like a hang, so it also goes to
226
+ * stderr (where Claude's -p mode prints its goal messages). */
227
+ function tell(ctx: ExtensionContext, text: string, level: 'info' | 'warning' | 'error'): void {
228
+ ctx.ui.notify(text, level)
229
+ if (!ctx.hasUI) process.stderr.write(`${text}\n`)
230
+ }
231
+
232
+ function currentSummary(ctx: SessionReader): GoalSummary | undefined {
233
+ if (!goal) return undefined
234
+ return { condition: goal.condition, durationMs: Date.now() - goal.setAt, iterations: goal.iterations, tokens: sessionTokens(ctx) - goal.tokensAtStart + goal.evaluatorTokens, lastReason: goal.lastReason }
235
+ }
236
+
237
+ function showIndicator(ctx: ExtensionContext): void {
238
+ if (!goal) return
239
+ ctx.ui.setStatus('goal', ctx.ui.theme.fg('accent', `◎ goal ${formatDuration(Date.now() - goal.setAt)}`))
240
+ if (!indicatorTimer) {
241
+ indicatorTimer = setInterval(() => {
242
+ if (sessionCtx) showIndicator(sessionCtx)
243
+ }, INDICATOR_REFRESH_MS)
244
+ indicatorTimer.unref?.()
245
+ }
246
+ }
247
+
248
+ function stopIndicator(ctx: ExtensionContext): void {
249
+ clearInterval(indicatorTimer)
250
+ indicatorTimer = undefined
251
+ ctx.ui.setStatus('goal', undefined)
252
+ }
253
+
254
+ function stopCheckin(): void {
255
+ clearTimeout(checkinTimer)
256
+ checkinTimer = undefined
257
+ }
258
+
259
+ function endDeferral(): void {
260
+ deferredSince = undefined
261
+ checkinDue = false
262
+ stopCheckin()
263
+ }
264
+
265
+ /** Make `condition` the active goal with fresh counters; the entry is the caller's. */
266
+ function activate(ctx: ExtensionContext, condition: string): void {
267
+ goal = { condition, setAt: Date.now(), iterations: 0, tokensAtStart: sessionTokens(ctx), evaluatorTokens: 0 }
268
+ generation += 1
269
+ noProgressStreak = 0
270
+ checkinsDelivered = 0
271
+ idleCheckins = 0
272
+ endDeferral()
273
+ showIndicator(ctx)
274
+ }
275
+
276
+ /** Drop the active goal, recording how it ended; returns its condition. */
277
+ function endGoal(ctx: ExtensionContext, state: Exclude<GoalState, 'active'>): string | undefined {
278
+ const ended = goal
279
+ if (!ended) return undefined
280
+ goal = undefined
281
+ generation += 1
282
+ endDeferral()
283
+ stopIndicator(ctx)
284
+ pi.appendEntry(GOAL_ENTRY, { state, condition: ended.condition } satisfies GoalEntry)
285
+ return ended.condition
286
+ }
287
+
288
+ function finish(ctx: ExtensionContext, state: 'achieved' | 'failed', reason: string): void {
289
+ const summary = currentSummary(ctx)
290
+ if (!summary) return
291
+ if (state === 'achieved') achieved = summary
292
+ endGoal(ctx, state)
293
+ const head = state === 'achieved' ? `Goal achieved (${summaryText(summary)}): ${summary.condition}` : `Goal could not be achieved (${summaryText(summary)}): ${summary.condition}`
294
+ send(reason ? `${head}\nEvaluator: ${reason}` : head, { triggerTurn: false })
295
+ }
296
+
297
+ /** Claude feeds a not-met reason back as the next turn, under the Stop hooks' cap. */
298
+ function continueGoal(ctx: ExtensionContext, active: ActiveGoal, reason: string): void {
299
+ noProgressStreak = turnUsedTools ? 0 : noProgressStreak + 1
300
+ const cap = stopHookBlockCap()
301
+ showIndicator(ctx)
302
+ if (noProgressStreak >= cap) {
303
+ noProgressStreak = 0
304
+ tell(ctx, `Goal paused after ${cap} turns in a row without tool use; it stays set and evaluation resumes after your next prompt.`, 'warning')
305
+ return
306
+ }
307
+ const summary = currentSummary(ctx)
308
+ const spent = summary ? ` · ${formatDuration(summary.durationMs)} · ${formatTokens(summary.tokens)} tokens` : ''
309
+ send(`Goal not yet met (turn ${active.iterations}${spent}): ${reason}\nGoal: ${active.condition}`, { triggerTurn: true })
310
+ }
311
+
312
+ async function askEvaluator(ctx: ExtensionContext, active: ActiveGoal, model: NonNullable<ExtensionContext['model']>): Promise<GoalVerdict> {
313
+ const transcript = renderTranscript(branchMessages(ctx), transcriptBudget(model))
314
+ // Esc during the evaluation aborts the run's signal; the call must die with it, or
315
+ // a late verdict would queue the next goal turn into a run the user just stopped.
316
+ const deadline = AbortSignal.timeout(EVALUATOR_TIMEOUT_MS)
317
+ const signal = ctx.signal ? AbortSignal.any([ctx.signal, deadline]) : deadline
318
+ // Claude runs its evaluator with thinking disabled: the verdict is mechanical.
319
+ // completeText requests no thinking level, which is the same for pi.
320
+ const { text, usage } = await completeText(model, evaluatorPrompt(transcript, active.condition), { system: EVALUATOR_SYSTEM, maxTokens: EVALUATOR_MAX_TOKENS, signal })
321
+ active.evaluatorTokens += usageTokens(usage as TokenUsage | undefined)
322
+ const verdict = parseVerdict(text)
323
+ if (!verdict) throw new Error(`unreadable verdict: ${text.slice(0, 200)}`)
324
+ return verdict
325
+ }
326
+
327
+ async function evaluate(ctx: ExtensionContext): Promise<void> {
328
+ const active = goal
329
+ if (!active) return
330
+ const startedGeneration = generation
331
+ const model = evaluatorModel(ctx)
332
+ if (!model) {
333
+ tell(ctx, 'Goal evaluator has no model to run on; the goal stays set.', 'warning')
334
+ return
335
+ }
336
+ let verdict: GoalVerdict
337
+ try {
338
+ verdict = await askEvaluator(ctx, active, model)
339
+ } catch (error) {
340
+ // A user interrupt is not an evaluator failure: the goal stays, nothing to say.
341
+ if (ctx.signal?.aborted) return
342
+ // No verdict is a hook error in Claude's terms: the turn ends and the goal stays.
343
+ if (generation === startedGeneration) tell(ctx, `Goal evaluator error: ${error instanceof Error ? error.message : String(error)}. The goal stays set; the next turn is evaluated again.`, 'warning')
344
+ return
345
+ }
346
+ // Interrupted, cleared, or replaced during the await: this verdict must not act.
347
+ if (ctx.signal?.aborted || generation !== startedGeneration) return
348
+ active.iterations += 1
349
+ active.lastReason = verdict.reason
350
+ if (verdict.ok) finish(ctx, 'achieved', verdict.reason)
351
+ else if (verdict.impossible) finish(ctx, 'failed', verdict.reason)
352
+ else continueGoal(ctx, active, verdict.reason)
353
+ }
354
+
355
+ function deliverCheckin(paused: boolean): void {
356
+ if (!goal) return
357
+ checkinDue = false
358
+ checkinsDelivered += 1
359
+ const work: RunningWork[] = [...running].map(([id, agentType]) => ({ id, agentType }))
360
+ send(checkinText(goal.condition, Date.now() - (deferredSince ?? Date.now()), work, paused), { triggerTurn: true })
361
+ }
362
+
363
+ function onCheckinDue(): void {
364
+ checkinTimer = undefined
365
+ if (!goal || !sessionCtx) return
366
+ // Claude delivers a due check-in at the next turn end when a turn is running, and
367
+ // only starts idle turns for it up to the per-prompt cap.
368
+ if (!sessionCtx.isIdle() || idleCheckins >= MAX_IDLE_CHECKINS) {
369
+ checkinDue = true
370
+ return
371
+ }
372
+ idleCheckins += 1
373
+ deliverCheckin(idleCheckins >= MAX_IDLE_CHECKINS)
374
+ }
375
+
376
+ function armCheckin(): void {
377
+ if (checkinTimer) return
378
+ const ms = checkinIntervalMs(process.env, checkinsDelivered)
379
+ if (ms <= 0) return
380
+ checkinTimer = setTimeout(onCheckinDue, ms)
381
+ checkinTimer.unref?.()
382
+ }
383
+
384
+ /** Background work keeps the goal waiting: no verdict this turn, a check-in later. */
385
+ function deferEvaluation(): void {
386
+ deferredSince ??= Date.now()
387
+ if (checkinDue) {
388
+ deliverCheckin(false)
389
+ return
390
+ }
391
+ armCheckin()
392
+ }
393
+
394
+ function statusText(ctx: SessionReader): string {
395
+ const summary = currentSummary(ctx)
396
+ if (summary) return formatActiveGoal(summary)
397
+ if (achieved) return formatAchievedGoal(achieved)
398
+ return NO_GOAL_TEXT
399
+ }
400
+
401
+ pi.registerCommand('goal', {
402
+ description: 'Set a goal the session keeps working toward until a separate check confirms it is met; /goal shows status, /goal clear stops',
403
+ handler: async (args, ctx) => {
404
+ sessionCtx = ctx
405
+ const condition = args.trim()
406
+ if (condition === '') {
407
+ tell(ctx, statusText(ctx), 'info')
408
+ return
409
+ }
410
+ if (isClearAlias(condition)) {
411
+ const cleared = endGoal(ctx, 'cleared')
412
+ if (cleared === undefined) tell(ctx, 'No goal set', 'info')
413
+ else send(`Goal cleared: ${cleared}`, { triggerTurn: false })
414
+ return
415
+ }
416
+ if (condition.length > GOAL_CONDITION_MAX_CHARS) {
417
+ tell(ctx, `Goal condition is limited to ${GOAL_CONDITION_MAX_CHARS} characters (got ${condition.length})`, 'error')
418
+ return
419
+ }
420
+ const blocked = goalUnavailable(ctx)
421
+ if (blocked) {
422
+ tell(ctx, blocked, 'error')
423
+ return
424
+ }
425
+ // A new goal replaces the current one, as Claude documents; the entry written here
426
+ // is the state a resume restores.
427
+ activate(ctx, condition)
428
+ pi.appendEntry(GOAL_ENTRY, { state: 'active', condition } satisfies GoalEntry)
429
+ tell(ctx, `Goal set: ${condition}`, 'info')
430
+ // Setting a goal starts a turn immediately with the condition as the directive; a
431
+ // send during streaming queues it as a follow-up turn.
432
+ send(kickoffPrompt(condition), ctx.isIdle() ? { triggerTurn: true } : { triggerTurn: true, deliverAs: 'followUp' })
433
+ // A headless run (`pi -p "/goal ..."`) exits as soon as the command returns, before
434
+ // the kickoff turn runs. pi marks the run active synchronously on the send, and the
435
+ // continuations queued at agent_end extend that run, so waiting here holds the
436
+ // process open until the goal resolves, as Claude's -p mode does. The TUI's main
437
+ // loop returns to the editor on its own, so it must not block on the loop.
438
+ if (!ctx.hasUI) await ctx.waitForIdle()
439
+ },
440
+ })
441
+
442
+ pi.events.on(SUBAGENT_CHANNEL, (data) => {
443
+ if (!isSubagentPhaseEvent(data)) return
444
+ if (data.phase === 'start') running.set(data.agentId, data.agentType)
445
+ else running.delete(data.agentId)
446
+ })
447
+
448
+ pi.on('session_start', (_event, ctx) => {
449
+ // One extension instance serves every session: drop the previous session's goal and
450
+ // timers before reading this session's persisted state.
451
+ sessionCtx = ctx
452
+ goal = undefined
453
+ achieved = undefined
454
+ generation += 1
455
+ turnUsedTools = false
456
+ noProgressStreak = 0
457
+ lastTurnError = undefined
458
+ checkinsDelivered = 0
459
+ idleCheckins = 0
460
+ endDeferral()
461
+ stopIndicator(ctx)
462
+ const entry = lastGoalEntry(ctx)
463
+ if (entry?.state !== 'active') return
464
+ // Claude restores a still-active goal on resume with its counters reset.
465
+ activate(ctx, entry.condition)
466
+ tell(ctx, `Goal restored: ${entry.condition}`, 'info')
467
+ })
468
+
469
+ pi.on('session_shutdown', () => {
470
+ stopCheckin()
471
+ clearInterval(indicatorTimer)
472
+ indicatorTimer = undefined
473
+ })
474
+
475
+ pi.on('agent_start', () => {
476
+ turnUsedTools = false
477
+ })
478
+
479
+ pi.on('tool_execution_start', () => {
480
+ turnUsedTools = true
481
+ })
482
+
483
+ pi.on('input', (event) => {
484
+ // Only genuine user input is progress: a goal continuation or a subagent prompt
485
+ // arrives with source 'extension'. Claude resets the block cap and the idle
486
+ // check-in allowance on the user's next prompt.
487
+ if (event.source === 'extension') return
488
+ noProgressStreak = 0
489
+ idleCheckins = 0
490
+ })
491
+
492
+ // On agent_end rather than agent_settled, for the reason hooks.ts gives: a peer
493
+ // extension can hold its agent_end handler on a dialog, which would starve a settle-
494
+ // based evaluation; and a continuation sent here is picked up as the run's own
495
+ // continuation rather than a fresh prompt.
496
+ pi.on('agent_end', async (event, ctx) => {
497
+ sessionCtx = ctx
498
+ const last = lastAssistant(event.messages)
499
+ lastTurnError = last?.stopReason === 'error' ? (last.errorMessage ?? 'unknown error') : undefined
500
+ // Claude runs no Stop evaluation after a user interrupt; a failed turn is judged once
501
+ // the run settles, so neither is evaluated here and the goal stays set.
502
+ if (!goal || last?.stopReason === 'aborted' || lastTurnError !== undefined) return
503
+ if (running.size > 0) {
504
+ deferEvaluation()
505
+ return
506
+ }
507
+ endDeferral()
508
+ await evaluate(ctx)
509
+ })
510
+
511
+ pi.on('agent_settled', (_event, ctx) => {
512
+ const message = lastTurnError
513
+ lastTurnError = undefined
514
+ if (!goal || message === undefined) return
515
+ const kind = classifyUnrecoverable(message)
516
+ if (!kind) return
517
+ endGoal(ctx, 'cleared')
518
+ const text = `Goal cleared after an unrecoverable error (${kind}): "${message}". Run /goal again to continue.`
519
+ send(text, { triggerTurn: false })
520
+ tell(ctx, text, 'warning')
521
+ })
522
+ }