pi-crew 0.10.4 → 0.10.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. package/CHANGELOG.md +233 -0
  2. package/agents/analyst.md +37 -2
  3. package/agents/cold-verifier.md +10 -1
  4. package/agents/councillor-critic.md +39 -0
  5. package/agents/councillor-pragmatist.md +39 -0
  6. package/agents/councillor-skeptic.md +41 -0
  7. package/agents/critic.md +40 -2
  8. package/agents/designer.md +58 -0
  9. package/agents/executor.md +39 -2
  10. package/agents/explorer.md +38 -2
  11. package/agents/librarian.md +49 -0
  12. package/agents/oracle.md +54 -0
  13. package/agents/orchestrator.md +48 -0
  14. package/agents/planner.md +41 -2
  15. package/agents/reviewer.md +39 -2
  16. package/agents/security-reviewer.md +43 -2
  17. package/agents/test-engineer.md +48 -2
  18. package/agents/verifier.md +14 -1
  19. package/agents/writer.md +32 -2
  20. package/dist/index.mjs +1297 -853
  21. package/package.json +1 -1
  22. package/skills/async-worker-recovery/SKILL.md +4 -1
  23. package/skills/child-pi-spawning/SKILL.md +4 -1
  24. package/skills/context-artifact-hygiene/SKILL.md +4 -1
  25. package/skills/council/SKILL.md +24 -45
  26. package/skills/delegation-patterns/SKILL.md +18 -1
  27. package/skills/distill-persona/SKILL.md +4 -1
  28. package/skills/distill-software/SKILL.md +4 -1
  29. package/skills/event-log-tracing/SKILL.md +4 -1
  30. package/skills/git-master/SKILL.md +4 -1
  31. package/skills/iterative-audit/SKILL.md +4 -1
  32. package/skills/live-agent-lifecycle/SKILL.md +4 -1
  33. package/skills/mailbox-interactive/SKILL.md +4 -1
  34. package/skills/model-routing-context/SKILL.md +10 -1
  35. package/skills/multi-perspective-review/SKILL.md +18 -1
  36. package/skills/observability-reliability/SKILL.md +4 -1
  37. package/skills/orchestration/SKILL.md +18 -1
  38. package/skills/ownership-session-security/SKILL.md +4 -1
  39. package/skills/pi-extension-lifecycle/SKILL.md +4 -1
  40. package/skills/post-mortem/SKILL.md +4 -1
  41. package/skills/read-only-explorer/SKILL.md +4 -1
  42. package/skills/real-test-pi-crew/SKILL.md +165 -12
  43. package/skills/requirements-to-task-packet/SKILL.md +10 -1
  44. package/skills/research/SKILL.md +4 -1
  45. package/skills/resource-discovery-config/SKILL.md +10 -1
  46. package/skills/runtime-state-reader/SKILL.md +4 -1
  47. package/skills/safe-bash/SKILL.md +4 -1
  48. package/skills/scrutinize/SKILL.md +24 -1
  49. package/skills/secure-agent-orchestration-review/SKILL.md +4 -1
  50. package/skills/state-mutation-locking/SKILL.md +4 -1
  51. package/skills/systematic-debugging/SKILL.md +4 -1
  52. package/skills/verification-before-done/SKILL.md +18 -1
  53. package/skills/widget-rendering/SKILL.md +4 -1
  54. package/skills/workspace-isolation/SKILL.md +4 -1
  55. package/skills/worktree-isolation/SKILL.md +4 -1
  56. package/src/config/config-validation.ts +1 -0
  57. package/src/config/types.ts +8 -0
  58. package/src/errors.ts +1 -1
  59. package/src/extension/context-status-injection.ts +2 -2
  60. package/src/extension/knowledge-injection.ts +19 -7
  61. package/src/extension/post-init-skill-check.ts +32 -0
  62. package/src/extension/register.ts +9 -1
  63. package/src/extension/registration/hook-registration.ts +20 -3
  64. package/src/extension/registration/tool-loop-guard.ts +243 -0
  65. package/src/extension/team-tool/handle-settings.ts +10 -0
  66. package/src/extension/team-tool/run.ts +42 -1
  67. package/src/extension/team-tool-types.ts +6 -0
  68. package/src/prompt/prompt-runtime.ts +25 -6
  69. package/src/runtime/async-runner.ts +75 -11
  70. package/src/runtime/background-runner.ts +73 -7
  71. package/src/runtime/broker/crew-broker-client.ts +45 -2
  72. package/src/runtime/broker/crew-broker.ts +22 -27
  73. package/src/runtime/broker/protocol/request-parsers.ts +10 -2
  74. package/src/runtime/broker/stdin-handshake.ts +87 -0
  75. package/src/runtime/broker/wait-push.ts +45 -0
  76. package/src/runtime/detached-run-results.ts +25 -1
  77. package/src/runtime/foreground-watchdog.ts +24 -5
  78. package/src/runtime/live-session/live-session-runtime.ts +1 -1
  79. package/src/runtime/model/model-scope.ts +2 -2
  80. package/src/runtime/run-tracker.ts +74 -19
  81. package/src/runtime/skill-instructions.ts +20 -4
  82. package/src/runtime/task-runner/child-executor.ts +1 -1
  83. package/src/runtime/task-runner/prompt-builder.ts +22 -9
  84. package/src/schema/config-schema.ts +1 -0
  85. package/src/skills/discover-skills.ts +2 -2
  86. package/src/ui/settings-overlay.ts +40 -0
  87. package/src/utils/frontmatter.ts +7 -1
  88. package/src/utils/ndjson.ts +9 -1
@@ -28,7 +28,7 @@
28
28
  * - `emitContext` already wraps handlers in try/catch and emits errors instead
29
29
  * of crashing the loop (Pi `runner.ts:933`), so a throw here can't break the
30
30
  * agent — but we also guard defensively.
31
- * - Opt-out: `runtime.reliability.ambientStatusInjection: false` in config.
31
+ * - Opt-out: `reliability.ambientStatusInjection: false` in config.
32
32
  */
33
33
 
34
34
  import type { AgentMessage } from "@earendil-works/pi-agent-core";
@@ -163,7 +163,7 @@ export function handleContextEvent(event: ContextEvent, cwd: string, sessionId?:
163
163
  * Register the ambient-status `context` event handler. Reads the project cwd
164
164
  * from the session context on each call (crew state is per-project).
165
165
  *
166
- * Pass `enabled: false` (from `runtime.reliability.ambientStatusInjection`) to
166
+ * Pass `enabled: false` (from `reliability.ambientStatusInjection`) to
167
167
  * disable the feature without unwiring the handler.
168
168
  */
169
169
  export function registerContextStatusInjection(pi: ExtensionAPI, opts: { enabled?: boolean } = {}): void {
@@ -437,13 +437,21 @@ export function buildKnowledgeFragment(cwd: string, query?: KnowledgeQuery): str
437
437
 
438
438
  /**
439
439
  * Register the knowledge-injection hook. Appends project knowledge to the
440
- * MAIN session's system prompt on `before_agent_start`. This hook does NOT
441
- * fire for crew workers: they are spawned with `--no-extensions`, so the
442
- * extension layer (and this hook) never loads in their process. Workers
443
- * instead receive knowledge via `buildKnowledgeFragment(task.cwd)` injected
444
- * into their prompt stablePrefix by `prompt-builder.ts`. Do NOT "fix" this
445
- * perceived gap by making the hook reach workers — it would cause
446
- * double-injection. (Verified by research workflow 2026-06-28.)
440
+ * MAIN session's system prompt on `before_agent_start`.
441
+ *
442
+ * Worker knowledge path (ARCH-2 corrected): children are NOT spawned with
443
+ * `--no-extensions` anymore — `pi-args.ts` runs extension discovery like
444
+ * the main session, but a child only loads an extension when the agent's
445
+ * frontmatter declares it (`extensions:`; builtin pi-crew agents declare
446
+ * none, so in practice workers never load pi-crew). Workers instead receive
447
+ * knowledge via `buildKnowledgeFragment(task.cwd)` injected into their
448
+ * prompt stablePrefix by `prompt-builder.ts`.
449
+ *
450
+ * Defense-in-depth: if an agent DOES declare the pi-crew extension, this
451
+ * hook would fire in the worker process and double-inject knowledge (once
452
+ * here, once via prompt-builder). The `PI_CREW_KIND=subagent` early-return
453
+ * below keeps main-session hooks main-session-only regardless of how the
454
+ * child was spawned. Do NOT remove it.
447
455
  *
448
456
  * The hook calls buildKnowledgeFragment(cwd) with NO query — so the main
449
457
  * session gets conventions-only (no session-log noise), which is the right
@@ -452,6 +460,10 @@ export function buildKnowledgeFragment(cwd: string, query?: KnowledgeQuery): str
452
460
  */
453
461
  export function registerKnowledgeInjection(pi: ExtensionAPI): void {
454
462
  pi.on("before_agent_start", (event: BeforeAgentStartEvent) => {
463
+ // ARCH-2: never fire in child worker processes — knowledge reaches
464
+ // workers via prompt-builder's stablePrefix fragment; firing here too
465
+ // would double-inject.
466
+ if (process.env.PI_CREW_KIND === "subagent") return;
455
467
  const options =
456
468
  (
457
469
  event as BeforeAgentStartEvent & {
@@ -0,0 +1,32 @@
1
+ import { renderSkillInstructions } from "../runtime/skill-instructions.ts";
2
+
3
+ export type SkillCheckResult = {
4
+ total: number;
5
+ resolved: number;
6
+ missing: string[];
7
+ severity: "ok" | "warn" | "error";
8
+ message: string;
9
+ };
10
+
11
+ export async function runPostInitSkillCheck(cwd: string): Promise<SkillCheckResult> {
12
+ const result = renderSkillInstructions({ cwd, role: "executor" });
13
+ const total = result.names.length;
14
+ const missingMatches = result.block.match(/Skill '([^']+)' was selected but no SKILL\.md file was found/g);
15
+ const missing = missingMatches ? missingMatches.map((m) => m.match(/'([^']+)'/)![1]) : [];
16
+ const resolved = total - missing.length;
17
+
18
+ let severity: "ok" | "warn" | "error";
19
+ let message: string;
20
+ if (resolved === total) {
21
+ severity = "ok";
22
+ message = `All ${total} default skills resolved`;
23
+ } else if (resolved === 0) {
24
+ severity = "error";
25
+ message = `0/${total} default skills resolved — likely bundle stale. Run \`npm run build:bundle\`.`;
26
+ } else {
27
+ severity = "warn";
28
+ message = `${resolved}/${total} default skills resolved — degraded: ${missing.join(", ")}`;
29
+ }
30
+
31
+ return { total, resolved, missing, severity, message };
32
+ }
@@ -32,6 +32,7 @@ import { registerCrewShortcuts } from "./crew-shortcuts.ts";
32
32
  import { registerCrewVibes } from "./crew-vibes/index.ts";
33
33
  import { registerKnowledgeInjection } from "./knowledge-injection.ts";
34
34
  import { registerCrewMessageRenderers } from "./message-renderers.ts";
35
+ import { runPostInitSkillCheck } from "./post-init-skill-check.ts";
35
36
  import { registerPiCommands } from "./registration/command-registration.ts";
36
37
  import { buildRegistrationContext } from "./registration/context-builder.ts";
37
38
  import { importCrashRecovery, purgeStaleActiveRunIndexSyncIfLoaded } from "./registration/crash-recovery-cache.ts";
@@ -50,7 +51,7 @@ export { __test__subagentSpawnParams };
50
51
  /**
51
52
  * Pi extension entry point. See module-level docstring for the full pipeline.
52
53
  */
53
- export function registerPiTeams(pi: ExtensionAPI): void {
54
+ export async function registerPiTeams(pi: ExtensionAPI): Promise<void> {
54
55
  resetTimings();
55
56
  time("register:start");
56
57
 
@@ -127,6 +128,13 @@ export function registerPiTeams(pi: ExtensionAPI): void {
127
128
  } catch (err) {
128
129
  console.warn("[pi-crew] crew-vibes initialization failed:", err instanceof Error ? err.message : err);
129
130
  }
131
+
132
+ const skillCheck = await runPostInitSkillCheck(process.cwd());
133
+ if (skillCheck.severity === "error") {
134
+ console.error(`[pi-crew] ${skillCheck.message}`);
135
+ } else if (skillCheck.severity === "warn") {
136
+ console.warn(`[pi-crew] ${skillCheck.message}`);
137
+ }
130
138
  }
131
139
 
132
140
  // Bottom-of-file import: keeps compaction-guard out of the top-level
@@ -16,13 +16,14 @@
16
16
  */
17
17
  import * as fs from "node:fs";
18
18
  import * as path from "node:path";
19
- import { fileURLToPath } from "node:url";
20
19
  import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
21
20
  import { asRecord, loadConfig } from "../../config/config.ts";
22
21
  import { buildValidationBlocker, extractPathFromInput, validateWrittenFile } from "../../runtime/per-write-validator.ts";
22
+ import { packageRoot } from "../../utils/paths.ts";
23
23
  import { resolveRealContainedPath } from "../../utils/safe-paths.ts";
24
24
  import { shouldBlockDestructiveTeamAction } from "../team-tool/destructive-gate.ts";
25
25
  import type { RegistrationContext } from "./registration-types.ts";
26
+ import { installToolLoopGuard } from "./tool-loop-guard.ts";
26
27
 
27
28
  /**
28
29
  * Register all non-lifecycle event hooks on the ExtensionAPI.
@@ -35,6 +36,22 @@ export function installPiHooks(pi: ExtensionAPI, ctx: RegistrationContext): void
35
36
  installResourcesDiscoverHook(pi, ctx);
36
37
  installToolCallHook(pi, ctx);
37
38
  installToolResultHook(pi, ctx);
39
+ installToolLoopGuardIfEnabled(pi, ctx);
40
+ }
41
+
42
+ /**
43
+ * ARCH-1: dispatch loop guard — warn at 3 identical results, block read-only
44
+ * tools at 5. Toggle via reliability.loopGuard (default on), mirroring
45
+ * perWriteValidation.
46
+ */
47
+ function installToolLoopGuardIfEnabled(pi: ExtensionAPI, ctx: RegistrationContext): void {
48
+ try {
49
+ const cwd = ctx.currentCtx?.cwd ?? process.cwd();
50
+ if (loadConfig(cwd).config.reliability?.loopGuard === false) return;
51
+ } catch {
52
+ /* config read failure: keep the guard on (fail-safe) */
53
+ }
54
+ installToolLoopGuard(pi);
38
55
  }
39
56
 
40
57
  /**
@@ -48,7 +65,7 @@ function installResourcesDiscoverHook(pi: ExtensionAPI, ctx: RegistrationContext
48
65
  pi.on("resources_discover", () => {
49
66
  const sessionCwd = ctx.currentCtx?.cwd ?? process.cwd();
50
67
  const skillDir = path.resolve(sessionCwd, "skills");
51
- const extSkillDir = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..", "skills");
68
+ const extSkillDir = path.join(packageRoot(), "skills");
52
69
  const paths: string[] = [];
53
70
  if (fs.existsSync(extSkillDir)) paths.push(extSkillDir);
54
71
  if (skillDir !== extSkillDir && fs.existsSync(skillDir)) {
@@ -97,7 +114,7 @@ function installToolCallHook(_pi: ExtensionAPI, _ctx: RegistrationContext): void
97
114
  * tool result on failure — catches malformed config the moment it's
98
115
  * written, not at the next load. Latency-safe by construction: no process
99
116
  * spawn, one disk read ONLY for validated extensions, dedup'd by content.
100
- * Toggle via runtime.reliability.perWriteValidation (default true).
117
+ * Toggle via reliability.perWriteValidation (default true).
101
118
  * Process-spawning validators (.js/.sh/.py) are a future opt-in.
102
119
  */
103
120
  function installToolResultHook(_pi: ExtensionAPI, _ctx: RegistrationContext): void {
@@ -0,0 +1,243 @@
1
+ /**
2
+ * ARCH-1: Tool loop guard.
3
+ *
4
+ * Detects a session re-issuing the exact same tool call (same tool, same
5
+ * arguments) consecutively with identical results — how model-side infinite
6
+ * loops present (precedent: a pi-crew run where a worker re-verified the
7
+ * same completed files 14+ times; OMO-slim issue #1071).
8
+ *
9
+ * Ported from oh-my-opencode-slim src/hooks/tool-loop-guard/hook.ts, adapted
10
+ * to pi's extension hook surface:
11
+ * - `tool_result` advances the counter: identical args AND byte-identical
12
+ * output continues the run; new output resets it (a legitimate re-read
13
+ * after a file changed can never accumulate toward a block).
14
+ * - `tool_call` blocks when a hard-block tool's confirmed run count reached
15
+ * the block threshold. The before-hook never increments — overlapping
16
+ * parallel calls cannot inflate the count.
17
+ * - Hard-block set is read-only file tools only (read/grep/glob/find/ls);
18
+ * everything else warns. bash and edit/write stay warn-only: their
19
+ * identical repeats may be legitimate side-effect retries.
20
+ * - Wait-style tool (`ask` — its contract is "stop and wait"): per-turn
21
+ * counter keyed by tool name only, warn at 2, block the 3rd; any non-ask
22
+ * tool result resets the turn (OMO #1139 semantics).
23
+ *
24
+ * Scope is per-process: each child worker runs in its own process, and the
25
+ * main session is one more, so module-level state needs no session keying.
26
+ * Tracked fingerprints are FIFO-bounded to guard memory in long sessions.
27
+ *
28
+ * Toggle: reliability.loopGuard = false in config disables install
29
+ * (mirrors perWriteValidation).
30
+ */
31
+
32
+ import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
33
+
34
+ const LOOP_GUARD_WARN_AT = 3;
35
+ const LOOP_GUARD_BLOCK_AT = 5;
36
+
37
+ /** Tools exempt from the entire guard: identical repeated invocation is legitimate. */
38
+ const LOOP_GUARD_EXEMPT: Record<string, true> = {
39
+ team: true,
40
+ crew_agent: true,
41
+ Agent: true,
42
+ get_subagent_result: true,
43
+ };
44
+
45
+ /** Tools that may be hard-blocked: read-only file analysis only. */
46
+ const LOOP_GUARD_BLOCK_TOOLS: Record<string, true> = {
47
+ read: true,
48
+ grep: true,
49
+ glob: true,
50
+ find: true,
51
+ ls: true,
52
+ };
53
+
54
+ /** Wait-style tools: repeating within a turn is always degenerate. */
55
+ const WAIT_TOOL = "ask";
56
+ const WAIT_GUARD_WARN_AT = 2;
57
+ /** Once this many ask calls have COMPLETED, the next ask call is refused (OMO #1139: warn at 2, refuse the 3rd). */
58
+ const WAIT_GUARD_BLOCK_AT = 2;
59
+
60
+ /** Max tracked fingerprints before evicting the oldest (FIFO bound). */
61
+ const MAX_TRACKED_FINGERPRINTS = 512;
62
+
63
+ export const LOOP_GUARD_MARKER = "[REPEATED TOOL CALLS - STOP]";
64
+ export const WAIT_GUARD_MARKER = "[REPEATED WAIT TOOL - END TURN]";
65
+
66
+ export const LOOP_GUARD_WARNING = `
67
+ ${LOOP_GUARD_MARKER}
68
+
69
+ You have issued the exact same tool call with identical arguments ${LOOP_GUARD_WARN_AT} times in a row and received identical results. This is an infinite loop and you are making no progress.
70
+
71
+ STOP repeating this call. Instead:
72
+ 1. Reconsider what you are looking for — the result above already contains what this call can tell you.
73
+ 2. If you need different information, make a DIFFERENT call (different path, pattern, or tool).
74
+ 3. If the task is actually done, produce your final answer now instead of calling more tools.
75
+ `;
76
+
77
+ export const WAIT_GUARD_WARNING = `
78
+ ${WAIT_GUARD_MARKER}
79
+
80
+ You have called \`ask\` ${WAIT_GUARD_WARN_AT} times in this turn. Its contract is to stop and wait — do not call it again in the same turn.
81
+
82
+ STOP calling tools that wait. Continue with what you can do, or produce your final answer; the reply arrives as a separate message later.
83
+ `;
84
+
85
+ /** Deterministic JSON: object keys sorted recursively, insensitive to key order. */
86
+ export function stableStringify(value: unknown): string {
87
+ return JSON.stringify(sortValue(value));
88
+ }
89
+
90
+ function sortValue(value: unknown): unknown {
91
+ if (Array.isArray(value)) return value.map(sortValue);
92
+ if (value && typeof value === "object") {
93
+ const record = value as Record<string, unknown>;
94
+ const out: Record<string, unknown> = {};
95
+ for (const key of Object.keys(record).sort()) {
96
+ out[key] = sortValue(record[key]);
97
+ }
98
+ return out;
99
+ }
100
+ return value;
101
+ }
102
+
103
+ /** Deterministic fingerprint of tool + args, insensitive to key order. */
104
+ export function fingerprint(tool: string, args: unknown): string {
105
+ return `${tool.toLowerCase()}:${stableStringify(args ?? null)}`;
106
+ }
107
+
108
+ interface FingerprintState {
109
+ /** Confirmed consecutive identical-args + identical-output count. */
110
+ runCount: number;
111
+ /** Last output fingerprint seen for this tool+args (null until first result). */
112
+ lastOutput: string | null;
113
+ }
114
+
115
+ export interface LoopGuardResultAppend {
116
+ type: "text";
117
+ text: string;
118
+ }
119
+
120
+ export interface LoopGuardCallVerdict {
121
+ block?: boolean;
122
+ reason?: string;
123
+ }
124
+
125
+ /**
126
+ * Pure state machine — exported for tests and reused by the hook wiring.
127
+ * One instance per process.
128
+ */
129
+ export function createLoopGuardState() {
130
+ const runs = new Map<string, FingerprintState>();
131
+ let lastFingerprint: string | null = null;
132
+ let waitRunCount = 0;
133
+
134
+ function evictIfNeeded(): void {
135
+ while (runs.size > MAX_TRACKED_FINGERPRINTS) {
136
+ const oldest = runs.keys().next().value;
137
+ if (oldest === undefined) break;
138
+ runs.delete(oldest);
139
+ }
140
+ }
141
+
142
+ /** tool_result side: returns warning content to append, if any. */
143
+ function onToolResult(tool: string, args: unknown, output: unknown): LoopGuardResultAppend[] {
144
+ const toolLower = tool.toLowerCase();
145
+ if (toolLower === WAIT_TOOL) {
146
+ waitRunCount += 1;
147
+ if (waitRunCount === WAIT_GUARD_WARN_AT) return [{ type: "text", text: WAIT_GUARD_WARNING }];
148
+ return [];
149
+ }
150
+ // Any completed non-ask tool call ends the "turn" for the wait guard.
151
+ waitRunCount = 0;
152
+
153
+ if (LOOP_GUARD_EXEMPT[tool]) return [];
154
+
155
+ const fp = fingerprint(tool, args);
156
+ const outputKey = stableStringify(output);
157
+ const state = runs.get(fp) ?? { runCount: 0, lastOutput: null };
158
+
159
+ if (fp !== lastFingerprint) {
160
+ // A different call intervened: this starts a fresh run for this fp.
161
+ state.runCount = 0;
162
+ }
163
+ if (state.lastOutput === outputKey) {
164
+ state.runCount += 1;
165
+ } else {
166
+ // New information: never accumulates toward a block.
167
+ state.runCount = 1;
168
+ state.lastOutput = outputKey;
169
+ }
170
+ runs.set(fp, state);
171
+ evictIfNeeded();
172
+ lastFingerprint = fp;
173
+
174
+ if (state.runCount === LOOP_GUARD_WARN_AT) {
175
+ return [{ type: "text", text: LOOP_GUARD_WARNING }];
176
+ }
177
+ return [];
178
+ }
179
+
180
+ /** tool_call side: returns a block verdict for degenerate repeats. */
181
+ function onToolCall(tool: string, args: unknown): LoopGuardCallVerdict {
182
+ const toolLower = tool.toLowerCase();
183
+ if (toolLower === WAIT_TOOL) {
184
+ if (waitRunCount >= WAIT_GUARD_BLOCK_AT) {
185
+ return {
186
+ block: true,
187
+ reason: `pi-crew loop guard: \`ask\` called ${waitRunCount} times this turn — its contract is to wait. End your turn; the reply arrives as a separate message.`,
188
+ };
189
+ }
190
+ return {};
191
+ }
192
+ if (LOOP_GUARD_EXEMPT[tool]) return {};
193
+ if (!LOOP_GUARD_BLOCK_TOOLS[toolLower]) return {};
194
+
195
+ const fp = fingerprint(tool, args);
196
+ const state = runs.get(fp);
197
+ if (state && state.runCount >= LOOP_GUARD_BLOCK_AT) {
198
+ return {
199
+ block: true,
200
+ reason: `pi-crew loop guard: this exact ${tool} call (identical arguments) has returned identical results ${state.runCount} times in a row. Make a DIFFERENT call (different path, pattern, or tool), or produce your final answer.`,
201
+ };
202
+ }
203
+ return {};
204
+ }
205
+
206
+ function reset(): void {
207
+ runs.clear();
208
+ lastFingerprint = null;
209
+ waitRunCount = 0;
210
+ }
211
+
212
+ return { onToolResult, onToolCall, reset };
213
+ }
214
+
215
+ /**
216
+ * Install the loop guard on a Pi instance. Both hooks are best-effort:
217
+ * older Pi versions without these events are tolerated (same pattern as
218
+ * installResourcesDiscoverHook).
219
+ */
220
+ export function installToolLoopGuard(pi: ExtensionAPI): void {
221
+ const state = createLoopGuardState();
222
+ try {
223
+ pi.on("tool_call", async (event: { toolName: string; input?: unknown }) => {
224
+ const verdict = state.onToolCall(event.toolName, event.input);
225
+ if (verdict.block) {
226
+ return { block: true as const, reason: verdict.reason };
227
+ }
228
+ return undefined;
229
+ });
230
+ } catch {
231
+ /* older Pi without tool_call events */
232
+ }
233
+ try {
234
+ pi.on("tool_result", (event: { toolName: string; input?: unknown; content?: unknown }) => {
235
+ const appends = state.onToolResult(event.toolName, event.input, event.content);
236
+ if (appends.length === 0) return undefined;
237
+ const existing = Array.isArray(event.content) ? event.content : [];
238
+ return { content: [...existing, ...appends] };
239
+ });
240
+ } catch {
241
+ /* older Pi without tool_result events */
242
+ }
243
+ }
@@ -51,6 +51,11 @@ const EFFECTIVE_DEFAULTS: Record<string, unknown> = {
51
51
  "reliability.autoRetry": false,
52
52
  "reliability.autoRecover": false,
53
53
  "reliability.cleanupOrphanedTempDirs": true,
54
+ "reliability.loopGuard": true,
55
+ "reliability.perWriteValidation": true,
56
+ "reliability.ambientStatusInjection": true,
57
+ "reliability.forcePreflight": false,
58
+ "reliability.scopeModels": false,
54
59
  "telemetry.enabled": false,
55
60
  "notifications.enabled": false,
56
61
  };
@@ -216,6 +221,11 @@ const KNOWN_KEYS = new Set([
216
221
  "reliability.autoRetry",
217
222
  "reliability.autoRecover",
218
223
  "reliability.cleanupOrphanedTempDirs",
224
+ "reliability.loopGuard",
225
+ "reliability.perWriteValidation",
226
+ "reliability.ambientStatusInjection",
227
+ "reliability.forcePreflight",
228
+ "reliability.scopeModels",
219
229
  "reliability.deadletterThreshold",
220
230
  "reliability.retryPolicy.maxAttempts",
221
231
  "reliability.retryPolicy.backoffMs",
@@ -71,7 +71,7 @@ import * as path from "node:path";
71
71
  import { t } from "../../i18n.ts";
72
72
  import { hasAsyncStartMarker } from "../../runtime/async-marker.ts";
73
73
  import { checkProcessLiveness, isActiveRunStatus } from "../../runtime/process-status.ts";
74
- import { waitForRun } from "../../runtime/run-tracker.ts";
74
+ import { registerRunPromise, waitForRun } from "../../runtime/run-tracker.ts";
75
75
  import { collectRunMetrics } from "../../state/stores/run-metrics.ts";
76
76
  import type { PiTeamsToolResult } from "../tool-result.ts";
77
77
  import { effectiveRunConfig } from "./config-patch.ts";
@@ -710,6 +710,13 @@ export async function handleRun(params: TeamToolParamsValue, ctx: TeamContext):
710
710
  if (executeWorkers && ctx.startForegroundRun) {
711
711
  // CORE-8: unified deadline — resolves params > config > 1h default.
712
712
  const fgDeadline = resolveRunDeadline(ctx, params, executedConfig);
713
+ // F1 register/await race fix (2026-09-12): the waitForRun below runs
714
+ // immediately after startForegroundRun returns (void), while
715
+ // executeTeamRunCore registers its promise only after several awaits —
716
+ // the waiter could land on the polling path and MISS the broker's
717
+ // waiting-push. Pre-register here so the medium path is guaranteed;
718
+ // registerRunPromise is idempotent (the core's later call is a no-op).
719
+ registerRunPromise(updatedManifest.runId);
713
720
  ctx.onRunStarted?.(updatedManifest.runId);
714
721
  const fgSignal = fgDeadline.signal;
715
722
  let fgAbortListener: (() => void) | undefined;
@@ -770,6 +777,40 @@ export async function handleRun(params: TeamToolParamsValue, ctx: TeamContext):
770
777
  // Wait for the foreground run to complete and return actual results.
771
778
  try {
772
779
  const completed = await waitForRun(updatedManifest.runId, resolvedCtx.cwd, { timeoutMs: fgDeadline.deadlineMs });
780
+ if (completed.waiting) {
781
+ // F1 (2026-09-12 live battery): a task parked on `ask` released this
782
+ // waiter — return the QUESTION so the leader can answer inline
783
+ // instead of blocking until the response watchdog kills the
784
+ // parked worker. Run keeps executing in the foreground-run lane.
785
+ const w = completed.waiting;
786
+ const secondsLeft = Math.max(0, Math.round((w.deadline - Date.now()) / 1000));
787
+ const lines = [
788
+ `pi-crew run WAITING for your answer: ${updatedManifest.runId}`,
789
+ `Team: ${team.name} · Workflow: ${workflow.name}`,
790
+ `Task ${w.taskId} (${w.questionId.substring(0, 8)}) parked on ask — ${secondsLeft}s until the deadline (then the worker proceeds with best judgment):`,
791
+ "",
792
+ `Q: ${w.question}`,
793
+ ];
794
+ if (w.options?.length) {
795
+ lines.push("", "Options:", ...w.options.map((o, i) => ` ${i + 1}. ${o}`));
796
+ }
797
+ lines.push(
798
+ "",
799
+ "Answer now (run keeps executing):",
800
+ ` team action='respond' taskId='${w.taskId}' message='<your answer>'`,
801
+ "then re-block until the run finishes:",
802
+ ` team action='wait' runId='${updatedManifest.runId}'`,
803
+ );
804
+ return result(lines.join("\n"), {
805
+ action: "run",
806
+ status: "ok",
807
+ runId: updatedManifest.runId,
808
+ artifactsRoot: updatedManifest.artifactsRoot,
809
+ taskId: w.taskId,
810
+ questionId: w.questionId,
811
+ waiting: true,
812
+ });
813
+ }
773
814
  if (completed.detached) {
774
815
  // The waiter was released so this turn can settle (agent view
775
816
  // switch); the run keeps executing and reports on completion.
@@ -23,6 +23,12 @@ export interface TeamToolDetails {
23
23
  durationMs?: number;
24
24
  consistencyScore?: number;
25
25
  };
26
+ /** F1 (2026-09-12): set when a run returned early because a task parked on
27
+ * `ask` — the tool result carries the question; answer via respond then
28
+ * re-block via wait. */
29
+ taskId?: string;
30
+ questionId?: string;
31
+ waiting?: boolean;
26
32
  /** Structured data for programmatic consumption (e.g. TUI widgets). */
27
33
  data?: Record<string, unknown>;
28
34
  }
@@ -224,7 +224,7 @@ export function rewriteTeamWorkerPrompt(prompt: string, options: { inheritProjec
224
224
 
225
225
  // ── WP-2/R2 (ADR-0 2026-08-17-waiting-producer-ask): worker-side `ask` tool ──
226
226
  // Binding ADR items 1, 4, 5:
227
- // 1. `ask({ question, options?, timeoutSec? = 600 })` — the SERVER clamps
227
+ // 1. `ask({ question, options?, timeoutSec? = 480 })` — the SERVER clamps
228
228
  // timeoutSec ≤ 3600 (P2-7); the client mirrors the clamp defensively.
229
229
  // 4. Option-(b) delivery: poll the run mailbox stream
230
230
  // (<PI_CREW_STATE_ROOT>/mailbox via readAllMailboxMessages) every 500ms
@@ -282,8 +282,22 @@ export function effectiveSteeringInterval(realtimeActive: boolean): number {
282
282
  return realtimeActive ? STEER_POLL_ACTIVE_MS : STEER_POLL_IDLE_MS;
283
283
  }
284
284
 
285
- const ASK_TIMEOUT_SEC_DEFAULT = 600;
285
+ // F2 (2026-09-12 live battery): 480, NOT 600 — a parked worker emits no
286
+ // output, so the 600s response watchdog counts the whole park; at 600==600
287
+ // the kill raced the wake (team_20260912014448). 480s leaves 120s grace for
288
+ // the worker to wake, answer its fallback, and finish the turn. This client
289
+ // default must stay in lockstep with the server default
290
+ // (WAIT_REQUEST_TIMEOUT_SEC_DEFAULT) — an explicit value here overrides the
291
+ // server default, so fixing only the broker side changed nothing.
292
+ const ASK_TIMEOUT_SEC_DEFAULT = 480;
286
293
  const ASK_TIMEOUT_SEC_MAX = 3600;
294
+ /** F2 live-probe follow-up (2026-09-12, team_20260912053049): the model may
295
+ * pass an EXPLICIT timeoutSec (it passed 600, racing the watchdog again
296
+ * despite the 480 default). The EFFECTIVE deadline is clamped to this
297
+ * ceiling regardless of what the model asks for — a parked worker emits no
298
+ * output, so any deadline ≥ the 600s response watchdog is a guaranteed kill.
299
+ * Must stay strictly below RESPONSE_TIMEOUT_MS/1000 (child-pi-constants). */
300
+ const ASK_TIMEOUT_SEC_CEILING = 480;
287
301
  /** Client-side mirrors of the broker's parseWaitRequestParams bounds — the
288
302
  * typebox schema below enforces them at the tool-call boundary so an
289
303
  * out-of-bounds ask fails validation BEFORE a park is attempted. */
@@ -684,16 +698,21 @@ export function createAskTool(deps: AskToolDeps = {}): AskToolDefinition {
684
698
  "[ask] unavailable: no broker connection (PI_CREW_BROKER_SOCKET / PI_CREW_BROKER_TOKEN / PI_CREW_BROKER_RUN_ID / PI_CREW_STATE_ROOT absent — scaffold or mock mode) — proceed with best judgment; do not call ask again.",
685
699
  );
686
700
  }
687
- // Client-side mirror of the server clamp (P2-7): the broker clamps
688
- // again, so this only shortens the park window the model believes in.
689
- const timeoutSec = Math.min(Math.max(1, Math.floor(params.timeoutSec ?? ASK_TIMEOUT_SEC_DEFAULT)), ASK_TIMEOUT_SEC_MAX);
701
+ // Client-side mirror of the server clamp (P2-7) + the F2 ceiling: the
702
+ // broker clamps ≤3600 again, but the CEILING here is the one that keeps
703
+ // the park window strictly inside the 600s response watchdog — an
704
+ // explicit model value (observed: 600) must NOT override it.
705
+ const timeoutSec = Math.min(Math.max(1, Math.floor(params.timeoutSec ?? ASK_TIMEOUT_SEC_DEFAULT)), ASK_TIMEOUT_SEC_CEILING);
690
706
  const client = deps.makeBrokerClient
691
707
  ? deps.makeBrokerClient({ runId, taskId, socketPath, token })
692
708
  : new CrewBrokerClient({ runId, taskId, socketPath, token });
693
709
  try {
694
710
  const requestParams: Record<string, unknown> = { to: taskId, question: params.question, timeoutSec };
695
711
  if (params.options) requestParams.options = params.options;
696
- const parked = await client.request("wait.request", requestParams);
712
+ // F5: cap the RPC itself at the ask deadline + 5s grace — a response
713
+ // frame lost on a half-dead socket must not outlive the deadline the
714
+ // worker is prepared to wait anyway (fallback notice → proceed).
715
+ const parked = await client.request("wait.request", requestParams, { timeoutMs: timeoutSec * 1000 + 5_000 });
697
716
  if (!parked.ok) {
698
717
  // Policy rejection, auth failure, connect failure — all fast-fail.
699
718
  const code = parked.errorCode ?? "request-failed";