pi-crew 0.10.4 → 0.10.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +233 -0
- package/agents/analyst.md +37 -2
- package/agents/cold-verifier.md +10 -1
- package/agents/councillor-critic.md +39 -0
- package/agents/councillor-pragmatist.md +39 -0
- package/agents/councillor-skeptic.md +41 -0
- package/agents/critic.md +40 -2
- package/agents/designer.md +58 -0
- package/agents/executor.md +39 -2
- package/agents/explorer.md +38 -2
- package/agents/librarian.md +49 -0
- package/agents/oracle.md +54 -0
- package/agents/orchestrator.md +48 -0
- package/agents/planner.md +41 -2
- package/agents/reviewer.md +39 -2
- package/agents/security-reviewer.md +43 -2
- package/agents/test-engineer.md +48 -2
- package/agents/verifier.md +14 -1
- package/agents/writer.md +32 -2
- package/dist/index.mjs +1297 -853
- package/package.json +1 -1
- package/skills/async-worker-recovery/SKILL.md +4 -1
- package/skills/child-pi-spawning/SKILL.md +4 -1
- package/skills/context-artifact-hygiene/SKILL.md +4 -1
- package/skills/council/SKILL.md +24 -45
- package/skills/delegation-patterns/SKILL.md +18 -1
- package/skills/distill-persona/SKILL.md +4 -1
- package/skills/distill-software/SKILL.md +4 -1
- package/skills/event-log-tracing/SKILL.md +4 -1
- package/skills/git-master/SKILL.md +4 -1
- package/skills/iterative-audit/SKILL.md +4 -1
- package/skills/live-agent-lifecycle/SKILL.md +4 -1
- package/skills/mailbox-interactive/SKILL.md +4 -1
- package/skills/model-routing-context/SKILL.md +10 -1
- package/skills/multi-perspective-review/SKILL.md +18 -1
- package/skills/observability-reliability/SKILL.md +4 -1
- package/skills/orchestration/SKILL.md +18 -1
- package/skills/ownership-session-security/SKILL.md +4 -1
- package/skills/pi-extension-lifecycle/SKILL.md +4 -1
- package/skills/post-mortem/SKILL.md +4 -1
- package/skills/read-only-explorer/SKILL.md +4 -1
- package/skills/real-test-pi-crew/SKILL.md +165 -12
- package/skills/requirements-to-task-packet/SKILL.md +10 -1
- package/skills/research/SKILL.md +4 -1
- package/skills/resource-discovery-config/SKILL.md +10 -1
- package/skills/runtime-state-reader/SKILL.md +4 -1
- package/skills/safe-bash/SKILL.md +4 -1
- package/skills/scrutinize/SKILL.md +24 -1
- package/skills/secure-agent-orchestration-review/SKILL.md +4 -1
- package/skills/state-mutation-locking/SKILL.md +4 -1
- package/skills/systematic-debugging/SKILL.md +4 -1
- package/skills/verification-before-done/SKILL.md +18 -1
- package/skills/widget-rendering/SKILL.md +4 -1
- package/skills/workspace-isolation/SKILL.md +4 -1
- package/skills/worktree-isolation/SKILL.md +4 -1
- package/src/config/config-validation.ts +1 -0
- package/src/config/types.ts +8 -0
- package/src/errors.ts +1 -1
- package/src/extension/context-status-injection.ts +2 -2
- package/src/extension/knowledge-injection.ts +19 -7
- package/src/extension/post-init-skill-check.ts +32 -0
- package/src/extension/register.ts +9 -1
- package/src/extension/registration/hook-registration.ts +20 -3
- package/src/extension/registration/tool-loop-guard.ts +243 -0
- package/src/extension/team-tool/handle-settings.ts +10 -0
- package/src/extension/team-tool/run.ts +42 -1
- package/src/extension/team-tool-types.ts +6 -0
- package/src/prompt/prompt-runtime.ts +25 -6
- package/src/runtime/async-runner.ts +75 -11
- package/src/runtime/background-runner.ts +73 -7
- package/src/runtime/broker/crew-broker-client.ts +45 -2
- package/src/runtime/broker/crew-broker.ts +22 -27
- package/src/runtime/broker/protocol/request-parsers.ts +10 -2
- package/src/runtime/broker/stdin-handshake.ts +87 -0
- package/src/runtime/broker/wait-push.ts +45 -0
- package/src/runtime/detached-run-results.ts +25 -1
- package/src/runtime/foreground-watchdog.ts +24 -5
- package/src/runtime/live-session/live-session-runtime.ts +1 -1
- package/src/runtime/model/model-scope.ts +2 -2
- package/src/runtime/run-tracker.ts +74 -19
- package/src/runtime/skill-instructions.ts +20 -4
- package/src/runtime/task-runner/child-executor.ts +1 -1
- package/src/runtime/task-runner/prompt-builder.ts +22 -9
- package/src/schema/config-schema.ts +1 -0
- package/src/skills/discover-skills.ts +2 -2
- package/src/ui/settings-overlay.ts +40 -0
- package/src/utils/frontmatter.ts +7 -1
- package/src/utils/ndjson.ts +9 -1
|
@@ -28,7 +28,7 @@
|
|
|
28
28
|
* - `emitContext` already wraps handlers in try/catch and emits errors instead
|
|
29
29
|
* of crashing the loop (Pi `runner.ts:933`), so a throw here can't break the
|
|
30
30
|
* agent — but we also guard defensively.
|
|
31
|
-
* - Opt-out: `
|
|
31
|
+
* - Opt-out: `reliability.ambientStatusInjection: false` in config.
|
|
32
32
|
*/
|
|
33
33
|
|
|
34
34
|
import type { AgentMessage } from "@earendil-works/pi-agent-core";
|
|
@@ -163,7 +163,7 @@ export function handleContextEvent(event: ContextEvent, cwd: string, sessionId?:
|
|
|
163
163
|
* Register the ambient-status `context` event handler. Reads the project cwd
|
|
164
164
|
* from the session context on each call (crew state is per-project).
|
|
165
165
|
*
|
|
166
|
-
* Pass `enabled: false` (from `
|
|
166
|
+
* Pass `enabled: false` (from `reliability.ambientStatusInjection`) to
|
|
167
167
|
* disable the feature without unwiring the handler.
|
|
168
168
|
*/
|
|
169
169
|
export function registerContextStatusInjection(pi: ExtensionAPI, opts: { enabled?: boolean } = {}): void {
|
|
@@ -437,13 +437,21 @@ export function buildKnowledgeFragment(cwd: string, query?: KnowledgeQuery): str
|
|
|
437
437
|
|
|
438
438
|
/**
|
|
439
439
|
* Register the knowledge-injection hook. Appends project knowledge to the
|
|
440
|
-
* MAIN session's system prompt on `before_agent_start`.
|
|
441
|
-
*
|
|
442
|
-
*
|
|
443
|
-
*
|
|
444
|
-
*
|
|
445
|
-
*
|
|
446
|
-
*
|
|
440
|
+
* MAIN session's system prompt on `before_agent_start`.
|
|
441
|
+
*
|
|
442
|
+
* Worker knowledge path (ARCH-2 corrected): children are NOT spawned with
|
|
443
|
+
* `--no-extensions` anymore — `pi-args.ts` runs extension discovery like
|
|
444
|
+
* the main session, but a child only loads an extension when the agent's
|
|
445
|
+
* frontmatter declares it (`extensions:`; builtin pi-crew agents declare
|
|
446
|
+
* none, so in practice workers never load pi-crew). Workers instead receive
|
|
447
|
+
* knowledge via `buildKnowledgeFragment(task.cwd)` injected into their
|
|
448
|
+
* prompt stablePrefix by `prompt-builder.ts`.
|
|
449
|
+
*
|
|
450
|
+
* Defense-in-depth: if an agent DOES declare the pi-crew extension, this
|
|
451
|
+
* hook would fire in the worker process and double-inject knowledge (once
|
|
452
|
+
* here, once via prompt-builder). The `PI_CREW_KIND=subagent` early-return
|
|
453
|
+
* below keeps main-session hooks main-session-only regardless of how the
|
|
454
|
+
* child was spawned. Do NOT remove it.
|
|
447
455
|
*
|
|
448
456
|
* The hook calls buildKnowledgeFragment(cwd) with NO query — so the main
|
|
449
457
|
* session gets conventions-only (no session-log noise), which is the right
|
|
@@ -452,6 +460,10 @@ export function buildKnowledgeFragment(cwd: string, query?: KnowledgeQuery): str
|
|
|
452
460
|
*/
|
|
453
461
|
export function registerKnowledgeInjection(pi: ExtensionAPI): void {
|
|
454
462
|
pi.on("before_agent_start", (event: BeforeAgentStartEvent) => {
|
|
463
|
+
// ARCH-2: never fire in child worker processes — knowledge reaches
|
|
464
|
+
// workers via prompt-builder's stablePrefix fragment; firing here too
|
|
465
|
+
// would double-inject.
|
|
466
|
+
if (process.env.PI_CREW_KIND === "subagent") return;
|
|
455
467
|
const options =
|
|
456
468
|
(
|
|
457
469
|
event as BeforeAgentStartEvent & {
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
import { renderSkillInstructions } from "../runtime/skill-instructions.ts";
|
|
2
|
+
|
|
3
|
+
export type SkillCheckResult = {
|
|
4
|
+
total: number;
|
|
5
|
+
resolved: number;
|
|
6
|
+
missing: string[];
|
|
7
|
+
severity: "ok" | "warn" | "error";
|
|
8
|
+
message: string;
|
|
9
|
+
};
|
|
10
|
+
|
|
11
|
+
export async function runPostInitSkillCheck(cwd: string): Promise<SkillCheckResult> {
|
|
12
|
+
const result = renderSkillInstructions({ cwd, role: "executor" });
|
|
13
|
+
const total = result.names.length;
|
|
14
|
+
const missingMatches = result.block.match(/Skill '([^']+)' was selected but no SKILL\.md file was found/g);
|
|
15
|
+
const missing = missingMatches ? missingMatches.map((m) => m.match(/'([^']+)'/)![1]) : [];
|
|
16
|
+
const resolved = total - missing.length;
|
|
17
|
+
|
|
18
|
+
let severity: "ok" | "warn" | "error";
|
|
19
|
+
let message: string;
|
|
20
|
+
if (resolved === total) {
|
|
21
|
+
severity = "ok";
|
|
22
|
+
message = `All ${total} default skills resolved`;
|
|
23
|
+
} else if (resolved === 0) {
|
|
24
|
+
severity = "error";
|
|
25
|
+
message = `0/${total} default skills resolved — likely bundle stale. Run \`npm run build:bundle\`.`;
|
|
26
|
+
} else {
|
|
27
|
+
severity = "warn";
|
|
28
|
+
message = `${resolved}/${total} default skills resolved — degraded: ${missing.join(", ")}`;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
return { total, resolved, missing, severity, message };
|
|
32
|
+
}
|
|
@@ -32,6 +32,7 @@ import { registerCrewShortcuts } from "./crew-shortcuts.ts";
|
|
|
32
32
|
import { registerCrewVibes } from "./crew-vibes/index.ts";
|
|
33
33
|
import { registerKnowledgeInjection } from "./knowledge-injection.ts";
|
|
34
34
|
import { registerCrewMessageRenderers } from "./message-renderers.ts";
|
|
35
|
+
import { runPostInitSkillCheck } from "./post-init-skill-check.ts";
|
|
35
36
|
import { registerPiCommands } from "./registration/command-registration.ts";
|
|
36
37
|
import { buildRegistrationContext } from "./registration/context-builder.ts";
|
|
37
38
|
import { importCrashRecovery, purgeStaleActiveRunIndexSyncIfLoaded } from "./registration/crash-recovery-cache.ts";
|
|
@@ -50,7 +51,7 @@ export { __test__subagentSpawnParams };
|
|
|
50
51
|
/**
|
|
51
52
|
* Pi extension entry point. See module-level docstring for the full pipeline.
|
|
52
53
|
*/
|
|
53
|
-
export function registerPiTeams(pi: ExtensionAPI): void {
|
|
54
|
+
export async function registerPiTeams(pi: ExtensionAPI): Promise<void> {
|
|
54
55
|
resetTimings();
|
|
55
56
|
time("register:start");
|
|
56
57
|
|
|
@@ -127,6 +128,13 @@ export function registerPiTeams(pi: ExtensionAPI): void {
|
|
|
127
128
|
} catch (err) {
|
|
128
129
|
console.warn("[pi-crew] crew-vibes initialization failed:", err instanceof Error ? err.message : err);
|
|
129
130
|
}
|
|
131
|
+
|
|
132
|
+
const skillCheck = await runPostInitSkillCheck(process.cwd());
|
|
133
|
+
if (skillCheck.severity === "error") {
|
|
134
|
+
console.error(`[pi-crew] ${skillCheck.message}`);
|
|
135
|
+
} else if (skillCheck.severity === "warn") {
|
|
136
|
+
console.warn(`[pi-crew] ${skillCheck.message}`);
|
|
137
|
+
}
|
|
130
138
|
}
|
|
131
139
|
|
|
132
140
|
// Bottom-of-file import: keeps compaction-guard out of the top-level
|
|
@@ -16,13 +16,14 @@
|
|
|
16
16
|
*/
|
|
17
17
|
import * as fs from "node:fs";
|
|
18
18
|
import * as path from "node:path";
|
|
19
|
-
import { fileURLToPath } from "node:url";
|
|
20
19
|
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
21
20
|
import { asRecord, loadConfig } from "../../config/config.ts";
|
|
22
21
|
import { buildValidationBlocker, extractPathFromInput, validateWrittenFile } from "../../runtime/per-write-validator.ts";
|
|
22
|
+
import { packageRoot } from "../../utils/paths.ts";
|
|
23
23
|
import { resolveRealContainedPath } from "../../utils/safe-paths.ts";
|
|
24
24
|
import { shouldBlockDestructiveTeamAction } from "../team-tool/destructive-gate.ts";
|
|
25
25
|
import type { RegistrationContext } from "./registration-types.ts";
|
|
26
|
+
import { installToolLoopGuard } from "./tool-loop-guard.ts";
|
|
26
27
|
|
|
27
28
|
/**
|
|
28
29
|
* Register all non-lifecycle event hooks on the ExtensionAPI.
|
|
@@ -35,6 +36,22 @@ export function installPiHooks(pi: ExtensionAPI, ctx: RegistrationContext): void
|
|
|
35
36
|
installResourcesDiscoverHook(pi, ctx);
|
|
36
37
|
installToolCallHook(pi, ctx);
|
|
37
38
|
installToolResultHook(pi, ctx);
|
|
39
|
+
installToolLoopGuardIfEnabled(pi, ctx);
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* ARCH-1: dispatch loop guard — warn at 3 identical results, block read-only
|
|
44
|
+
* tools at 5. Toggle via reliability.loopGuard (default on), mirroring
|
|
45
|
+
* perWriteValidation.
|
|
46
|
+
*/
|
|
47
|
+
function installToolLoopGuardIfEnabled(pi: ExtensionAPI, ctx: RegistrationContext): void {
|
|
48
|
+
try {
|
|
49
|
+
const cwd = ctx.currentCtx?.cwd ?? process.cwd();
|
|
50
|
+
if (loadConfig(cwd).config.reliability?.loopGuard === false) return;
|
|
51
|
+
} catch {
|
|
52
|
+
/* config read failure: keep the guard on (fail-safe) */
|
|
53
|
+
}
|
|
54
|
+
installToolLoopGuard(pi);
|
|
38
55
|
}
|
|
39
56
|
|
|
40
57
|
/**
|
|
@@ -48,7 +65,7 @@ function installResourcesDiscoverHook(pi: ExtensionAPI, ctx: RegistrationContext
|
|
|
48
65
|
pi.on("resources_discover", () => {
|
|
49
66
|
const sessionCwd = ctx.currentCtx?.cwd ?? process.cwd();
|
|
50
67
|
const skillDir = path.resolve(sessionCwd, "skills");
|
|
51
|
-
const extSkillDir = path.
|
|
68
|
+
const extSkillDir = path.join(packageRoot(), "skills");
|
|
52
69
|
const paths: string[] = [];
|
|
53
70
|
if (fs.existsSync(extSkillDir)) paths.push(extSkillDir);
|
|
54
71
|
if (skillDir !== extSkillDir && fs.existsSync(skillDir)) {
|
|
@@ -97,7 +114,7 @@ function installToolCallHook(_pi: ExtensionAPI, _ctx: RegistrationContext): void
|
|
|
97
114
|
* tool result on failure — catches malformed config the moment it's
|
|
98
115
|
* written, not at the next load. Latency-safe by construction: no process
|
|
99
116
|
* spawn, one disk read ONLY for validated extensions, dedup'd by content.
|
|
100
|
-
* Toggle via
|
|
117
|
+
* Toggle via reliability.perWriteValidation (default true).
|
|
101
118
|
* Process-spawning validators (.js/.sh/.py) are a future opt-in.
|
|
102
119
|
*/
|
|
103
120
|
function installToolResultHook(_pi: ExtensionAPI, _ctx: RegistrationContext): void {
|
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* ARCH-1: Tool loop guard.
|
|
3
|
+
*
|
|
4
|
+
* Detects a session re-issuing the exact same tool call (same tool, same
|
|
5
|
+
* arguments) consecutively with identical results — how model-side infinite
|
|
6
|
+
* loops present (precedent: a pi-crew run where a worker re-verified the
|
|
7
|
+
* same completed files 14+ times; OMO-slim issue #1071).
|
|
8
|
+
*
|
|
9
|
+
* Ported from oh-my-opencode-slim src/hooks/tool-loop-guard/hook.ts, adapted
|
|
10
|
+
* to pi's extension hook surface:
|
|
11
|
+
* - `tool_result` advances the counter: identical args AND byte-identical
|
|
12
|
+
* output continues the run; new output resets it (a legitimate re-read
|
|
13
|
+
* after a file changed can never accumulate toward a block).
|
|
14
|
+
* - `tool_call` blocks when a hard-block tool's confirmed run count reached
|
|
15
|
+
* the block threshold. The before-hook never increments — overlapping
|
|
16
|
+
* parallel calls cannot inflate the count.
|
|
17
|
+
* - Hard-block set is read-only file tools only (read/grep/glob/find/ls);
|
|
18
|
+
* everything else warns. bash and edit/write stay warn-only: their
|
|
19
|
+
* identical repeats may be legitimate side-effect retries.
|
|
20
|
+
* - Wait-style tool (`ask` — its contract is "stop and wait"): per-turn
|
|
21
|
+
* counter keyed by tool name only, warn at 2, block the 3rd; any non-ask
|
|
22
|
+
* tool result resets the turn (OMO #1139 semantics).
|
|
23
|
+
*
|
|
24
|
+
* Scope is per-process: each child worker runs in its own process, and the
|
|
25
|
+
* main session is one more, so module-level state needs no session keying.
|
|
26
|
+
* Tracked fingerprints are FIFO-bounded to guard memory in long sessions.
|
|
27
|
+
*
|
|
28
|
+
* Toggle: reliability.loopGuard = false in config disables install
|
|
29
|
+
* (mirrors perWriteValidation).
|
|
30
|
+
*/
|
|
31
|
+
|
|
32
|
+
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
33
|
+
|
|
34
|
+
const LOOP_GUARD_WARN_AT = 3;
|
|
35
|
+
const LOOP_GUARD_BLOCK_AT = 5;
|
|
36
|
+
|
|
37
|
+
/** Tools exempt from the entire guard: identical repeated invocation is legitimate. */
|
|
38
|
+
const LOOP_GUARD_EXEMPT: Record<string, true> = {
|
|
39
|
+
team: true,
|
|
40
|
+
crew_agent: true,
|
|
41
|
+
Agent: true,
|
|
42
|
+
get_subagent_result: true,
|
|
43
|
+
};
|
|
44
|
+
|
|
45
|
+
/** Tools that may be hard-blocked: read-only file analysis only. */
|
|
46
|
+
const LOOP_GUARD_BLOCK_TOOLS: Record<string, true> = {
|
|
47
|
+
read: true,
|
|
48
|
+
grep: true,
|
|
49
|
+
glob: true,
|
|
50
|
+
find: true,
|
|
51
|
+
ls: true,
|
|
52
|
+
};
|
|
53
|
+
|
|
54
|
+
/** Wait-style tools: repeating within a turn is always degenerate. */
|
|
55
|
+
const WAIT_TOOL = "ask";
|
|
56
|
+
const WAIT_GUARD_WARN_AT = 2;
|
|
57
|
+
/** Once this many ask calls have COMPLETED, the next ask call is refused (OMO #1139: warn at 2, refuse the 3rd). */
|
|
58
|
+
const WAIT_GUARD_BLOCK_AT = 2;
|
|
59
|
+
|
|
60
|
+
/** Max tracked fingerprints before evicting the oldest (FIFO bound). */
|
|
61
|
+
const MAX_TRACKED_FINGERPRINTS = 512;
|
|
62
|
+
|
|
63
|
+
export const LOOP_GUARD_MARKER = "[REPEATED TOOL CALLS - STOP]";
|
|
64
|
+
export const WAIT_GUARD_MARKER = "[REPEATED WAIT TOOL - END TURN]";
|
|
65
|
+
|
|
66
|
+
export const LOOP_GUARD_WARNING = `
|
|
67
|
+
${LOOP_GUARD_MARKER}
|
|
68
|
+
|
|
69
|
+
You have issued the exact same tool call with identical arguments ${LOOP_GUARD_WARN_AT} times in a row and received identical results. This is an infinite loop and you are making no progress.
|
|
70
|
+
|
|
71
|
+
STOP repeating this call. Instead:
|
|
72
|
+
1. Reconsider what you are looking for — the result above already contains what this call can tell you.
|
|
73
|
+
2. If you need different information, make a DIFFERENT call (different path, pattern, or tool).
|
|
74
|
+
3. If the task is actually done, produce your final answer now instead of calling more tools.
|
|
75
|
+
`;
|
|
76
|
+
|
|
77
|
+
export const WAIT_GUARD_WARNING = `
|
|
78
|
+
${WAIT_GUARD_MARKER}
|
|
79
|
+
|
|
80
|
+
You have called \`ask\` ${WAIT_GUARD_WARN_AT} times in this turn. Its contract is to stop and wait — do not call it again in the same turn.
|
|
81
|
+
|
|
82
|
+
STOP calling tools that wait. Continue with what you can do, or produce your final answer; the reply arrives as a separate message later.
|
|
83
|
+
`;
|
|
84
|
+
|
|
85
|
+
/** Deterministic JSON: object keys sorted recursively, insensitive to key order. */
|
|
86
|
+
export function stableStringify(value: unknown): string {
|
|
87
|
+
return JSON.stringify(sortValue(value));
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
function sortValue(value: unknown): unknown {
|
|
91
|
+
if (Array.isArray(value)) return value.map(sortValue);
|
|
92
|
+
if (value && typeof value === "object") {
|
|
93
|
+
const record = value as Record<string, unknown>;
|
|
94
|
+
const out: Record<string, unknown> = {};
|
|
95
|
+
for (const key of Object.keys(record).sort()) {
|
|
96
|
+
out[key] = sortValue(record[key]);
|
|
97
|
+
}
|
|
98
|
+
return out;
|
|
99
|
+
}
|
|
100
|
+
return value;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/** Deterministic fingerprint of tool + args, insensitive to key order. */
|
|
104
|
+
export function fingerprint(tool: string, args: unknown): string {
|
|
105
|
+
return `${tool.toLowerCase()}:${stableStringify(args ?? null)}`;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
interface FingerprintState {
|
|
109
|
+
/** Confirmed consecutive identical-args + identical-output count. */
|
|
110
|
+
runCount: number;
|
|
111
|
+
/** Last output fingerprint seen for this tool+args (null until first result). */
|
|
112
|
+
lastOutput: string | null;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
export interface LoopGuardResultAppend {
|
|
116
|
+
type: "text";
|
|
117
|
+
text: string;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
export interface LoopGuardCallVerdict {
|
|
121
|
+
block?: boolean;
|
|
122
|
+
reason?: string;
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/**
|
|
126
|
+
* Pure state machine — exported for tests and reused by the hook wiring.
|
|
127
|
+
* One instance per process.
|
|
128
|
+
*/
|
|
129
|
+
export function createLoopGuardState() {
|
|
130
|
+
const runs = new Map<string, FingerprintState>();
|
|
131
|
+
let lastFingerprint: string | null = null;
|
|
132
|
+
let waitRunCount = 0;
|
|
133
|
+
|
|
134
|
+
function evictIfNeeded(): void {
|
|
135
|
+
while (runs.size > MAX_TRACKED_FINGERPRINTS) {
|
|
136
|
+
const oldest = runs.keys().next().value;
|
|
137
|
+
if (oldest === undefined) break;
|
|
138
|
+
runs.delete(oldest);
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
/** tool_result side: returns warning content to append, if any. */
|
|
143
|
+
function onToolResult(tool: string, args: unknown, output: unknown): LoopGuardResultAppend[] {
|
|
144
|
+
const toolLower = tool.toLowerCase();
|
|
145
|
+
if (toolLower === WAIT_TOOL) {
|
|
146
|
+
waitRunCount += 1;
|
|
147
|
+
if (waitRunCount === WAIT_GUARD_WARN_AT) return [{ type: "text", text: WAIT_GUARD_WARNING }];
|
|
148
|
+
return [];
|
|
149
|
+
}
|
|
150
|
+
// Any completed non-ask tool call ends the "turn" for the wait guard.
|
|
151
|
+
waitRunCount = 0;
|
|
152
|
+
|
|
153
|
+
if (LOOP_GUARD_EXEMPT[tool]) return [];
|
|
154
|
+
|
|
155
|
+
const fp = fingerprint(tool, args);
|
|
156
|
+
const outputKey = stableStringify(output);
|
|
157
|
+
const state = runs.get(fp) ?? { runCount: 0, lastOutput: null };
|
|
158
|
+
|
|
159
|
+
if (fp !== lastFingerprint) {
|
|
160
|
+
// A different call intervened: this starts a fresh run for this fp.
|
|
161
|
+
state.runCount = 0;
|
|
162
|
+
}
|
|
163
|
+
if (state.lastOutput === outputKey) {
|
|
164
|
+
state.runCount += 1;
|
|
165
|
+
} else {
|
|
166
|
+
// New information: never accumulates toward a block.
|
|
167
|
+
state.runCount = 1;
|
|
168
|
+
state.lastOutput = outputKey;
|
|
169
|
+
}
|
|
170
|
+
runs.set(fp, state);
|
|
171
|
+
evictIfNeeded();
|
|
172
|
+
lastFingerprint = fp;
|
|
173
|
+
|
|
174
|
+
if (state.runCount === LOOP_GUARD_WARN_AT) {
|
|
175
|
+
return [{ type: "text", text: LOOP_GUARD_WARNING }];
|
|
176
|
+
}
|
|
177
|
+
return [];
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
/** tool_call side: returns a block verdict for degenerate repeats. */
|
|
181
|
+
function onToolCall(tool: string, args: unknown): LoopGuardCallVerdict {
|
|
182
|
+
const toolLower = tool.toLowerCase();
|
|
183
|
+
if (toolLower === WAIT_TOOL) {
|
|
184
|
+
if (waitRunCount >= WAIT_GUARD_BLOCK_AT) {
|
|
185
|
+
return {
|
|
186
|
+
block: true,
|
|
187
|
+
reason: `pi-crew loop guard: \`ask\` called ${waitRunCount} times this turn — its contract is to wait. End your turn; the reply arrives as a separate message.`,
|
|
188
|
+
};
|
|
189
|
+
}
|
|
190
|
+
return {};
|
|
191
|
+
}
|
|
192
|
+
if (LOOP_GUARD_EXEMPT[tool]) return {};
|
|
193
|
+
if (!LOOP_GUARD_BLOCK_TOOLS[toolLower]) return {};
|
|
194
|
+
|
|
195
|
+
const fp = fingerprint(tool, args);
|
|
196
|
+
const state = runs.get(fp);
|
|
197
|
+
if (state && state.runCount >= LOOP_GUARD_BLOCK_AT) {
|
|
198
|
+
return {
|
|
199
|
+
block: true,
|
|
200
|
+
reason: `pi-crew loop guard: this exact ${tool} call (identical arguments) has returned identical results ${state.runCount} times in a row. Make a DIFFERENT call (different path, pattern, or tool), or produce your final answer.`,
|
|
201
|
+
};
|
|
202
|
+
}
|
|
203
|
+
return {};
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
function reset(): void {
|
|
207
|
+
runs.clear();
|
|
208
|
+
lastFingerprint = null;
|
|
209
|
+
waitRunCount = 0;
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
return { onToolResult, onToolCall, reset };
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
/**
|
|
216
|
+
* Install the loop guard on a Pi instance. Both hooks are best-effort:
|
|
217
|
+
* older Pi versions without these events are tolerated (same pattern as
|
|
218
|
+
* installResourcesDiscoverHook).
|
|
219
|
+
*/
|
|
220
|
+
export function installToolLoopGuard(pi: ExtensionAPI): void {
|
|
221
|
+
const state = createLoopGuardState();
|
|
222
|
+
try {
|
|
223
|
+
pi.on("tool_call", async (event: { toolName: string; input?: unknown }) => {
|
|
224
|
+
const verdict = state.onToolCall(event.toolName, event.input);
|
|
225
|
+
if (verdict.block) {
|
|
226
|
+
return { block: true as const, reason: verdict.reason };
|
|
227
|
+
}
|
|
228
|
+
return undefined;
|
|
229
|
+
});
|
|
230
|
+
} catch {
|
|
231
|
+
/* older Pi without tool_call events */
|
|
232
|
+
}
|
|
233
|
+
try {
|
|
234
|
+
pi.on("tool_result", (event: { toolName: string; input?: unknown; content?: unknown }) => {
|
|
235
|
+
const appends = state.onToolResult(event.toolName, event.input, event.content);
|
|
236
|
+
if (appends.length === 0) return undefined;
|
|
237
|
+
const existing = Array.isArray(event.content) ? event.content : [];
|
|
238
|
+
return { content: [...existing, ...appends] };
|
|
239
|
+
});
|
|
240
|
+
} catch {
|
|
241
|
+
/* older Pi without tool_result events */
|
|
242
|
+
}
|
|
243
|
+
}
|
|
@@ -51,6 +51,11 @@ const EFFECTIVE_DEFAULTS: Record<string, unknown> = {
|
|
|
51
51
|
"reliability.autoRetry": false,
|
|
52
52
|
"reliability.autoRecover": false,
|
|
53
53
|
"reliability.cleanupOrphanedTempDirs": true,
|
|
54
|
+
"reliability.loopGuard": true,
|
|
55
|
+
"reliability.perWriteValidation": true,
|
|
56
|
+
"reliability.ambientStatusInjection": true,
|
|
57
|
+
"reliability.forcePreflight": false,
|
|
58
|
+
"reliability.scopeModels": false,
|
|
54
59
|
"telemetry.enabled": false,
|
|
55
60
|
"notifications.enabled": false,
|
|
56
61
|
};
|
|
@@ -216,6 +221,11 @@ const KNOWN_KEYS = new Set([
|
|
|
216
221
|
"reliability.autoRetry",
|
|
217
222
|
"reliability.autoRecover",
|
|
218
223
|
"reliability.cleanupOrphanedTempDirs",
|
|
224
|
+
"reliability.loopGuard",
|
|
225
|
+
"reliability.perWriteValidation",
|
|
226
|
+
"reliability.ambientStatusInjection",
|
|
227
|
+
"reliability.forcePreflight",
|
|
228
|
+
"reliability.scopeModels",
|
|
219
229
|
"reliability.deadletterThreshold",
|
|
220
230
|
"reliability.retryPolicy.maxAttempts",
|
|
221
231
|
"reliability.retryPolicy.backoffMs",
|
|
@@ -71,7 +71,7 @@ import * as path from "node:path";
|
|
|
71
71
|
import { t } from "../../i18n.ts";
|
|
72
72
|
import { hasAsyncStartMarker } from "../../runtime/async-marker.ts";
|
|
73
73
|
import { checkProcessLiveness, isActiveRunStatus } from "../../runtime/process-status.ts";
|
|
74
|
-
import { waitForRun } from "../../runtime/run-tracker.ts";
|
|
74
|
+
import { registerRunPromise, waitForRun } from "../../runtime/run-tracker.ts";
|
|
75
75
|
import { collectRunMetrics } from "../../state/stores/run-metrics.ts";
|
|
76
76
|
import type { PiTeamsToolResult } from "../tool-result.ts";
|
|
77
77
|
import { effectiveRunConfig } from "./config-patch.ts";
|
|
@@ -710,6 +710,13 @@ export async function handleRun(params: TeamToolParamsValue, ctx: TeamContext):
|
|
|
710
710
|
if (executeWorkers && ctx.startForegroundRun) {
|
|
711
711
|
// CORE-8: unified deadline — resolves params > config > 1h default.
|
|
712
712
|
const fgDeadline = resolveRunDeadline(ctx, params, executedConfig);
|
|
713
|
+
// F1 register/await race fix (2026-09-12): the waitForRun below runs
|
|
714
|
+
// immediately after startForegroundRun returns (void), while
|
|
715
|
+
// executeTeamRunCore registers its promise only after several awaits —
|
|
716
|
+
// the waiter could land on the polling path and MISS the broker's
|
|
717
|
+
// waiting-push. Pre-register here so the medium path is guaranteed;
|
|
718
|
+
// registerRunPromise is idempotent (the core's later call is a no-op).
|
|
719
|
+
registerRunPromise(updatedManifest.runId);
|
|
713
720
|
ctx.onRunStarted?.(updatedManifest.runId);
|
|
714
721
|
const fgSignal = fgDeadline.signal;
|
|
715
722
|
let fgAbortListener: (() => void) | undefined;
|
|
@@ -770,6 +777,40 @@ export async function handleRun(params: TeamToolParamsValue, ctx: TeamContext):
|
|
|
770
777
|
// Wait for the foreground run to complete and return actual results.
|
|
771
778
|
try {
|
|
772
779
|
const completed = await waitForRun(updatedManifest.runId, resolvedCtx.cwd, { timeoutMs: fgDeadline.deadlineMs });
|
|
780
|
+
if (completed.waiting) {
|
|
781
|
+
// F1 (2026-09-12 live battery): a task parked on `ask` released this
|
|
782
|
+
// waiter — return the QUESTION so the leader can answer inline
|
|
783
|
+
// instead of blocking until the response watchdog kills the
|
|
784
|
+
// parked worker. Run keeps executing in the foreground-run lane.
|
|
785
|
+
const w = completed.waiting;
|
|
786
|
+
const secondsLeft = Math.max(0, Math.round((w.deadline - Date.now()) / 1000));
|
|
787
|
+
const lines = [
|
|
788
|
+
`pi-crew run WAITING for your answer: ${updatedManifest.runId}`,
|
|
789
|
+
`Team: ${team.name} · Workflow: ${workflow.name}`,
|
|
790
|
+
`Task ${w.taskId} (${w.questionId.substring(0, 8)}) parked on ask — ${secondsLeft}s until the deadline (then the worker proceeds with best judgment):`,
|
|
791
|
+
"",
|
|
792
|
+
`Q: ${w.question}`,
|
|
793
|
+
];
|
|
794
|
+
if (w.options?.length) {
|
|
795
|
+
lines.push("", "Options:", ...w.options.map((o, i) => ` ${i + 1}. ${o}`));
|
|
796
|
+
}
|
|
797
|
+
lines.push(
|
|
798
|
+
"",
|
|
799
|
+
"Answer now (run keeps executing):",
|
|
800
|
+
` team action='respond' taskId='${w.taskId}' message='<your answer>'`,
|
|
801
|
+
"then re-block until the run finishes:",
|
|
802
|
+
` team action='wait' runId='${updatedManifest.runId}'`,
|
|
803
|
+
);
|
|
804
|
+
return result(lines.join("\n"), {
|
|
805
|
+
action: "run",
|
|
806
|
+
status: "ok",
|
|
807
|
+
runId: updatedManifest.runId,
|
|
808
|
+
artifactsRoot: updatedManifest.artifactsRoot,
|
|
809
|
+
taskId: w.taskId,
|
|
810
|
+
questionId: w.questionId,
|
|
811
|
+
waiting: true,
|
|
812
|
+
});
|
|
813
|
+
}
|
|
773
814
|
if (completed.detached) {
|
|
774
815
|
// The waiter was released so this turn can settle (agent view
|
|
775
816
|
// switch); the run keeps executing and reports on completion.
|
|
@@ -23,6 +23,12 @@ export interface TeamToolDetails {
|
|
|
23
23
|
durationMs?: number;
|
|
24
24
|
consistencyScore?: number;
|
|
25
25
|
};
|
|
26
|
+
/** F1 (2026-09-12): set when a run returned early because a task parked on
|
|
27
|
+
* `ask` — the tool result carries the question; answer via respond then
|
|
28
|
+
* re-block via wait. */
|
|
29
|
+
taskId?: string;
|
|
30
|
+
questionId?: string;
|
|
31
|
+
waiting?: boolean;
|
|
26
32
|
/** Structured data for programmatic consumption (e.g. TUI widgets). */
|
|
27
33
|
data?: Record<string, unknown>;
|
|
28
34
|
}
|
|
@@ -224,7 +224,7 @@ export function rewriteTeamWorkerPrompt(prompt: string, options: { inheritProjec
|
|
|
224
224
|
|
|
225
225
|
// ── WP-2/R2 (ADR-0 2026-08-17-waiting-producer-ask): worker-side `ask` tool ──
|
|
226
226
|
// Binding ADR items 1, 4, 5:
|
|
227
|
-
// 1. `ask({ question, options?, timeoutSec? =
|
|
227
|
+
// 1. `ask({ question, options?, timeoutSec? = 480 })` — the SERVER clamps
|
|
228
228
|
// timeoutSec ≤ 3600 (P2-7); the client mirrors the clamp defensively.
|
|
229
229
|
// 4. Option-(b) delivery: poll the run mailbox stream
|
|
230
230
|
// (<PI_CREW_STATE_ROOT>/mailbox via readAllMailboxMessages) every 500ms
|
|
@@ -282,8 +282,22 @@ export function effectiveSteeringInterval(realtimeActive: boolean): number {
|
|
|
282
282
|
return realtimeActive ? STEER_POLL_ACTIVE_MS : STEER_POLL_IDLE_MS;
|
|
283
283
|
}
|
|
284
284
|
|
|
285
|
-
|
|
285
|
+
// F2 (2026-09-12 live battery): 480, NOT 600 — a parked worker emits no
|
|
286
|
+
// output, so the 600s response watchdog counts the whole park; at 600==600
|
|
287
|
+
// the kill raced the wake (team_20260912014448). 480s leaves 120s grace for
|
|
288
|
+
// the worker to wake, answer its fallback, and finish the turn. This client
|
|
289
|
+
// default must stay in lockstep with the server default
|
|
290
|
+
// (WAIT_REQUEST_TIMEOUT_SEC_DEFAULT) — an explicit value here overrides the
|
|
291
|
+
// server default, so fixing only the broker side changed nothing.
|
|
292
|
+
const ASK_TIMEOUT_SEC_DEFAULT = 480;
|
|
286
293
|
const ASK_TIMEOUT_SEC_MAX = 3600;
|
|
294
|
+
/** F2 live-probe follow-up (2026-09-12, team_20260912053049): the model may
|
|
295
|
+
* pass an EXPLICIT timeoutSec (it passed 600, racing the watchdog again
|
|
296
|
+
* despite the 480 default). The EFFECTIVE deadline is clamped to this
|
|
297
|
+
* ceiling regardless of what the model asks for — a parked worker emits no
|
|
298
|
+
* output, so any deadline ≥ the 600s response watchdog is a guaranteed kill.
|
|
299
|
+
* Must stay strictly below RESPONSE_TIMEOUT_MS/1000 (child-pi-constants). */
|
|
300
|
+
const ASK_TIMEOUT_SEC_CEILING = 480;
|
|
287
301
|
/** Client-side mirrors of the broker's parseWaitRequestParams bounds — the
|
|
288
302
|
* typebox schema below enforces them at the tool-call boundary so an
|
|
289
303
|
* out-of-bounds ask fails validation BEFORE a park is attempted. */
|
|
@@ -684,16 +698,21 @@ export function createAskTool(deps: AskToolDeps = {}): AskToolDefinition {
|
|
|
684
698
|
"[ask] unavailable: no broker connection (PI_CREW_BROKER_SOCKET / PI_CREW_BROKER_TOKEN / PI_CREW_BROKER_RUN_ID / PI_CREW_STATE_ROOT absent — scaffold or mock mode) — proceed with best judgment; do not call ask again.",
|
|
685
699
|
);
|
|
686
700
|
}
|
|
687
|
-
// Client-side mirror of the server clamp (P2-7)
|
|
688
|
-
//
|
|
689
|
-
|
|
701
|
+
// Client-side mirror of the server clamp (P2-7) + the F2 ceiling: the
|
|
702
|
+
// broker clamps ≤3600 again, but the CEILING here is the one that keeps
|
|
703
|
+
// the park window strictly inside the 600s response watchdog — an
|
|
704
|
+
// explicit model value (observed: 600) must NOT override it.
|
|
705
|
+
const timeoutSec = Math.min(Math.max(1, Math.floor(params.timeoutSec ?? ASK_TIMEOUT_SEC_DEFAULT)), ASK_TIMEOUT_SEC_CEILING);
|
|
690
706
|
const client = deps.makeBrokerClient
|
|
691
707
|
? deps.makeBrokerClient({ runId, taskId, socketPath, token })
|
|
692
708
|
: new CrewBrokerClient({ runId, taskId, socketPath, token });
|
|
693
709
|
try {
|
|
694
710
|
const requestParams: Record<string, unknown> = { to: taskId, question: params.question, timeoutSec };
|
|
695
711
|
if (params.options) requestParams.options = params.options;
|
|
696
|
-
|
|
712
|
+
// F5: cap the RPC itself at the ask deadline + 5s grace — a response
|
|
713
|
+
// frame lost on a half-dead socket must not outlive the deadline the
|
|
714
|
+
// worker is prepared to wait anyway (fallback notice → proceed).
|
|
715
|
+
const parked = await client.request("wait.request", requestParams, { timeoutMs: timeoutSec * 1000 + 5_000 });
|
|
697
716
|
if (!parked.ok) {
|
|
698
717
|
// Policy rejection, auth failure, connect failure — all fast-fail.
|
|
699
718
|
const code = parked.errorCode ?? "request-failed";
|