pi-ui-extend 1.0.38 → 1.0.40
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/app/extensions/extension-ui-controller.js +3 -0
- package/dist/app/rendering/extension-entry-renderer.js +2 -0
- package/dist/bundled-extensions/question/contract.d.ts +3 -0
- package/dist/bundled-extensions/question/contract.js +24 -0
- package/dist/bundled-extensions/question/desktop.d.ts +11 -0
- package/dist/bundled-extensions/question/desktop.js +143 -0
- package/dist/bundled-extensions/question/index.d.ts +1 -0
- package/dist/bundled-extensions/question/index.js +5 -1
- package/dist/bundled-extensions/question/render.js +6 -0
- package/dist/bundled-extensions/question/result.js +71 -9
- package/dist/bundled-extensions/question/tool-description.js +4 -3
- package/dist/bundled-extensions/question/tui.js +127 -12
- package/dist/bundled-extensions/question/types.d.ts +23 -2
- package/dist/tool-renderers/question.js +20 -1
- package/docs/desktop-markdown-media.md +77 -0
- package/docs/desktop-mvp.md +134 -0
- package/docs/desktop-task-manager.md +124 -0
- package/external/pi-tools-suite/README.md +207 -0
- package/external/pi-tools-suite/docs/evals.md +684 -0
- package/external/pi-tools-suite/package.json +7 -3
- package/external/pi-tools-suite/src/async-subagents/async-subagents.sample.jsonc +8 -0
- package/external/pi-tools-suite/src/async-subagents/core/agent-strategy.ts +41 -2
- package/external/pi-tools-suite/src/async-subagents/core/config.ts +12 -0
- package/external/pi-tools-suite/src/async-subagents/index.ts +6 -2
- package/external/pi-tools-suite/src/async-subagents/subagent-overlay.ts +1 -1
- package/external/pi-tools-suite/src/coding-discipline/index.ts +41 -142
- package/external/pi-tools-suite/src/config.ts +0 -21
- package/external/pi-tools-suite/src/dcp/auto-compress.ts +1 -1
- package/external/pi-tools-suite/src/dcp/compression-blocks.ts +1 -51
- package/external/pi-tools-suite/src/dcp/debug-log.ts +6 -0
- package/external/pi-tools-suite/src/dcp/index.ts +27 -125
- package/external/pi-tools-suite/src/dcp/prompts.ts +2 -2
- package/external/pi-tools-suite/src/dcp/provider-tool-results.ts +2 -1
- package/external/pi-tools-suite/src/dcp/pruner-candidates.ts +31 -10
- package/external/pi-tools-suite/src/dcp/pruner-compression-blocks.ts +6 -7
- package/external/pi-tools-suite/src/dcp/pruner-message-ids.ts +153 -111
- package/external/pi-tools-suite/src/dcp/pruner-metadata.ts +28 -17
- package/external/pi-tools-suite/src/dcp/pruner-nudge.ts +61 -27
- package/external/pi-tools-suite/src/dcp/pruner.ts +21 -13
- package/external/pi-tools-suite/src/dcp/state.ts +59 -0
- package/external/pi-tools-suite/src/default-pi-tools-suite-config.ts +27 -6
- package/external/pi-tools-suite/src/lib/rpc-session-state.ts +34 -0
- package/external/pi-tools-suite/src/todo/todo.ts +8 -3
- package/external/pi-tools-suite/src/tool-descriptions.ts +4 -3
- package/package.json +7 -4
|
@@ -234,6 +234,14 @@
|
|
|
234
234
|
"description": "Use when the sub-agent should make or plan code changes for a feature, bug fix, or refactor.",
|
|
235
235
|
"model": "openai-codex/gpt-5.6-sol",
|
|
236
236
|
"fallbackModels": ["zai/glm-5.3"],
|
|
237
|
+
// Avoid Sol recursively doing routine implementation work for a Sol parent.
|
|
238
|
+
// Luna also escalates substantial implementation to Terra. modelByParent
|
|
239
|
+
// wins over preset role models, so this prevents both Luna -> Sol and
|
|
240
|
+
// Sol -> Sol for implement tasks under the built-in `gpt` preset.
|
|
241
|
+
"modelByParent": {
|
|
242
|
+
"openai-codex/gpt-5.6-luna*": { "model": "openai-codex/gpt-5.6-terra", "fallbackModels": ["zai/glm-5.3"] },
|
|
243
|
+
"openai-codex/gpt-5.6-sol*": { "model": "openai-codex/gpt-5.6-terra", "fallbackModels": ["zai/glm-5.3"] }
|
|
244
|
+
},
|
|
237
245
|
"thinking": "high"
|
|
238
246
|
},
|
|
239
247
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { isGptLikeModel } from "./ultrawork-auto.js";
|
|
2
2
|
|
|
3
|
-
export type AgentStrategyName = "parallel-first" | "deep-work";
|
|
3
|
+
export type AgentStrategyName = "parallel-first" | "deep-work" | "escalation-aware" | "cost-aware-orchestrator";
|
|
4
4
|
|
|
5
5
|
export interface AgentStrategyOptions {
|
|
6
6
|
modelRef?: string;
|
|
@@ -27,13 +27,40 @@ Default: autonomous deep worker. Build context directly, make progress, edit, an
|
|
|
27
27
|
For broad work, keep delegation explicit and bounded: focused review/research/tests/frontend/deep tracks, plus one oracle only for high-stakes uncertainty or final plan checks. Read compact results, decide in the parent session, and report only what matters. If compressing unfinished work, preserve active objective + next step via todo/DCP rules.
|
|
28
28
|
</agent_strategy>`;
|
|
29
29
|
|
|
30
|
+
const ESCALATION_AWARE_STRATEGY_PROMPT = `<agent_strategy name="escalation-aware">
|
|
31
|
+
Execution hint for Pi, not a replacement for system/developer/user instructions.
|
|
32
|
+
|
|
33
|
+
Default: self-sufficient with tiered escalation. Solve narrow, well-bounded work directly, but do not grind through a large or uncertain task at the current model tier when a stronger focused subagent is appropriate. Use todo for the plan and async subagents for escalation; read compact results first and keep the parent context lean.
|
|
34
|
+
|
|
35
|
+
For a Luna parent, prefer Terra workers for substantial multi-file research, tests, or implementation, and escalate deep root-cause analysis, architecture/security review, high-risk decisions, or repeatedly failing complex work to Sol through the deep/review roles. For a Terra parent, handle routine research/tests/implementation directly; escalate deep root-cause analysis, architecture/security review, high-risk decisions, or stubborn complex failures to Sol through deep/review. Do not escalate merely because a plan has several steps, and do not delegate a tiny known-file edit or exact lookup.
|
|
36
|
+
|
|
37
|
+
Keep user questions, plan/todo changes, integration decisions, and the final report in the parent. Independent read-only escalations may run in parallel; serialize overlapping edits unless scopes are clearly disjoint. When work is delegated, synchronize its todo lifecycle: mark it in progress, collect and verify the worker result, then complete/update it before moving on.
|
|
38
|
+
</agent_strategy>`;
|
|
39
|
+
|
|
40
|
+
const COST_AWARE_ORCHESTRATOR_STRATEGY_PROMPT = `<agent_strategy name="cost-aware-orchestrator">
|
|
41
|
+
Execution hint for Pi, not a replacement for system/developer/user instructions.
|
|
42
|
+
|
|
43
|
+
Default: cost-aware orchestration. You are an expensive parent model, so keep the parent session focused on planning, decisions, integration, verification, and the final user-facing answer. For non-trivial todo work, prefer focused async subagents for repo scanning, multi-file research, documentation, tests, frontend work, and implementation steps that would otherwise require several repository tool calls. Read compact subagent results first; inspect raw artifacts or redo work in the parent only when verification or uncertainty requires it.
|
|
44
|
+
|
|
45
|
+
Keep user questions, todo/plan changes, architecture tradeoffs, cross-worker integration decisions, high-stakes review, and the final report in the parent. Do not delegate merely to avoid one cheap exact lookup or a tiny known-file edit. Independent read-only tracks may run in parallel; serialize overlapping edits unless scopes are clearly disjoint. When a todo item is delegated, keep its lifecycle synchronized: mark it in progress, collect and verify the worker result, then complete/update it before moving on.
|
|
46
|
+
</agent_strategy>`;
|
|
47
|
+
|
|
30
48
|
export function agentStrategyPrompt(options: AgentStrategyOptions = {}): string | undefined {
|
|
31
49
|
const env = options.env ?? process.env;
|
|
32
50
|
const override = strategyOverride(env);
|
|
33
51
|
if (override === "off") return undefined;
|
|
34
52
|
if (options.customPrompt && shouldSkipCustomPrompt(env)) return undefined;
|
|
35
53
|
|
|
36
|
-
const strategy = override
|
|
54
|
+
const strategy = override
|
|
55
|
+
?? (isExpensiveGptParent(options.modelRef)
|
|
56
|
+
? "cost-aware-orchestrator"
|
|
57
|
+
: isEscalationAwareGptParent(options.modelRef)
|
|
58
|
+
? "escalation-aware"
|
|
59
|
+
: isGptLikeModel(options.modelRef)
|
|
60
|
+
? "deep-work"
|
|
61
|
+
: "parallel-first");
|
|
62
|
+
if (strategy === "cost-aware-orchestrator") return COST_AWARE_ORCHESTRATOR_STRATEGY_PROMPT;
|
|
63
|
+
if (strategy === "escalation-aware") return ESCALATION_AWARE_STRATEGY_PROMPT;
|
|
37
64
|
return strategy === "deep-work" ? DEEP_WORK_STRATEGY_PROMPT : PARALLEL_FIRST_STRATEGY_PROMPT;
|
|
38
65
|
}
|
|
39
66
|
|
|
@@ -49,10 +76,22 @@ function strategyOverride(env: NodeJS.ProcessEnv): AgentStrategyName | "off" | u
|
|
|
49
76
|
if (FALSE_ENV_PATTERN.test(value)) return "off";
|
|
50
77
|
if (value === "parallel-first") return "parallel-first";
|
|
51
78
|
if (value === "deep-work") return "deep-work";
|
|
79
|
+
if (value === "escalation" || value === "escalation-aware") return "escalation-aware";
|
|
80
|
+
if (value === "cost-aware" || value === "cost-aware-orchestrator" || value === "orchestrator") return "cost-aware-orchestrator";
|
|
52
81
|
if (TRUE_ENV_PATTERN.test(value)) return undefined;
|
|
53
82
|
return undefined;
|
|
54
83
|
}
|
|
55
84
|
|
|
85
|
+
function isExpensiveGptParent(modelRef: string | undefined): boolean {
|
|
86
|
+
if (!modelRef) return false;
|
|
87
|
+
return /(?:^|\/)gpt-5\.6-sol(?:$|[-.:])/i.test(modelRef.trim());
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
function isEscalationAwareGptParent(modelRef: string | undefined): boolean {
|
|
91
|
+
if (!modelRef) return false;
|
|
92
|
+
return /(?:^|\/)gpt-5\.6-(?:luna|terra)(?:$|[-.:])/i.test(modelRef.trim());
|
|
93
|
+
}
|
|
94
|
+
|
|
56
95
|
function shouldSkipCustomPrompt(env: NodeJS.ProcessEnv): boolean {
|
|
57
96
|
const raw = firstEnv(env, "PI_AGENT_STRATEGY_WITH_CUSTOM_PROMPT", "ASYNC_SUBAGENTS_AGENT_STRATEGY_WITH_CUSTOM_PROMPT");
|
|
58
97
|
return raw ? !TRUE_ENV_PATTERN.test(raw.trim()) : true;
|
|
@@ -224,6 +224,10 @@ const BUILTIN_CONFIG: SubagentConfig = {
|
|
|
224
224
|
},
|
|
225
225
|
implement: {
|
|
226
226
|
description: "Use when the sub-agent should make or plan code changes for a feature, bug fix, or refactor.",
|
|
227
|
+
modelByParent: {
|
|
228
|
+
"openai-codex/gpt-5.6-luna*": { model: "openai-codex/gpt-5.6-terra", fallbackModels: ["zai/glm-5.3"] },
|
|
229
|
+
"openai-codex/gpt-5.6-sol*": { model: "openai-codex/gpt-5.6-terra", fallbackModels: ["zai/glm-5.3"] },
|
|
230
|
+
},
|
|
227
231
|
thinking: "high",
|
|
228
232
|
},
|
|
229
233
|
tests: {
|
|
@@ -234,10 +238,18 @@ const BUILTIN_CONFIG: SubagentConfig = {
|
|
|
234
238
|
},
|
|
235
239
|
review: {
|
|
236
240
|
description: "Use for review/audit of existing code or changes: correctness, security, performance, maintainability, API risks, quality. Do not implement new code.",
|
|
241
|
+
modelByParent: {
|
|
242
|
+
"openai-codex/gpt-5.6-luna*": { model: "openai-codex/gpt-5.6-sol", fallbackModels: ["zai/glm-5.3"] },
|
|
243
|
+
"openai-codex/gpt-5.6-terra*": { model: "openai-codex/gpt-5.6-sol", fallbackModels: ["zai/glm-5.3"] },
|
|
244
|
+
},
|
|
237
245
|
thinking: "high",
|
|
238
246
|
},
|
|
239
247
|
deep: {
|
|
240
248
|
description: "Use for broad hard reasoning: architecture, root-cause analysis, cross-module impact, complex debugging or tradeoffs.",
|
|
249
|
+
modelByParent: {
|
|
250
|
+
"openai-codex/gpt-5.6-luna*": { model: "openai-codex/gpt-5.6-sol", fallbackModels: ["zai/glm-5.3"] },
|
|
251
|
+
"openai-codex/gpt-5.6-terra*": { model: "openai-codex/gpt-5.6-sol", fallbackModels: ["zai/glm-5.3"] },
|
|
252
|
+
},
|
|
241
253
|
thinking: "high",
|
|
242
254
|
},
|
|
243
255
|
oracle: {
|
|
@@ -32,6 +32,7 @@ import { registerSubagentsTool } from "./tools/subagents.js";
|
|
|
32
32
|
import type { LiveAgent, SubagentsLiveStateEvent } from "./types.js";
|
|
33
33
|
import type { AgentState } from "./core/types.js";
|
|
34
34
|
import { publishStartupSection } from "../startup-section.js";
|
|
35
|
+
import { publishRpcSessionState } from "../lib/rpc-session-state.js";
|
|
35
36
|
|
|
36
37
|
function isTerminalAgentStatus(status: AgentState["status"]): boolean {
|
|
37
38
|
return status === "done" || status === "failed" || status === "stopped";
|
|
@@ -83,8 +84,8 @@ function createLiveStatePayload(
|
|
|
83
84
|
}
|
|
84
85
|
|
|
85
86
|
function agentMatchesSession(agent: LiveAgent, sessionFile: string | undefined): boolean {
|
|
86
|
-
if (!sessionFile
|
|
87
|
-
return pathsEqual(sessionFile, agent.parentSession);
|
|
87
|
+
if (!sessionFile) return true;
|
|
88
|
+
return agent.parentSession !== undefined && pathsEqual(sessionFile, agent.parentSession);
|
|
88
89
|
}
|
|
89
90
|
|
|
90
91
|
function isStaleExtensionContextError(error: unknown): boolean {
|
|
@@ -100,6 +101,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
100
101
|
const subagentOverlay = new SubagentOverlay(liveAgents);
|
|
101
102
|
let sawAutoUltraworkCandidate = false;
|
|
102
103
|
let currentSessionFile: string | undefined;
|
|
104
|
+
let currentSessionStateContext: Parameters<typeof publishRpcSessionState>[0];
|
|
103
105
|
let completionWatchTimer: ReturnType<typeof setInterval> | undefined;
|
|
104
106
|
publishSubagentPresetsStartupSection();
|
|
105
107
|
|
|
@@ -109,6 +111,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
109
111
|
const liveState = createLiveStatePayload(liveAgents, currentSessionFile);
|
|
110
112
|
pi.events?.emit?.(SUBAGENTS_LIVE_COUNT_EVENT, { count: liveState.count });
|
|
111
113
|
pi.events?.emit?.(SUBAGENTS_LIVE_STATE_EVENT, liveState);
|
|
114
|
+
publishRpcSessionState(currentSessionStateContext, SUBAGENTS_LIVE_STATE_EVENT, liveState);
|
|
112
115
|
updateCompletionWatcher();
|
|
113
116
|
} catch (error) {
|
|
114
117
|
ignoreStaleExtensionContextError(error);
|
|
@@ -168,6 +171,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
168
171
|
try {
|
|
169
172
|
sawAutoUltraworkCandidate = false;
|
|
170
173
|
currentSessionFile = sessionFileFromContext(ctx);
|
|
174
|
+
currentSessionStateContext = ctx;
|
|
171
175
|
subagentOverlay.restoreRunningAgents(ctx.cwd, currentSessionFile);
|
|
172
176
|
refreshSubagentOverlay();
|
|
173
177
|
} catch (error) {
|
|
@@ -23,7 +23,7 @@ export class SubagentOverlay {
|
|
|
23
23
|
if (liveRun.has(agent.id)) continue;
|
|
24
24
|
const agentDir = path.join(runDir, agent.id);
|
|
25
25
|
const agentParentSession = readParentSessionLink(agentDir);
|
|
26
|
-
if (parentSession && agentParentSession
|
|
26
|
+
if (parentSession && (!agentParentSession || !pathsEqual(parentSession, agentParentSession))) continue;
|
|
27
27
|
liveRun.set(agent.id, { runDir, agentId: agent.id, parentSession: agentParentSession, completed: Promise.resolve() });
|
|
28
28
|
}
|
|
29
29
|
if (liveRun.size === 0) this.liveAgents.delete(runDir);
|
|
@@ -3,7 +3,7 @@ import * as path from "node:path";
|
|
|
3
3
|
import type { Api, AssistantMessage, ImageContent, Model, ProviderHeaders, TextContent } from "@earendil-works/pi-ai";
|
|
4
4
|
import { Type } from "typebox";
|
|
5
5
|
|
|
6
|
-
import { loadPiToolsSuiteConfig
|
|
6
|
+
import { loadPiToolsSuiteConfig } from "../config.js";
|
|
7
7
|
import { ignoreStaleExtensionContextError } from "../context-usage.js";
|
|
8
8
|
import { completeWithModelRegistry, type ModelCompletionRegistry } from "../model-completion.js";
|
|
9
9
|
|
|
@@ -41,13 +41,6 @@ const DEFAULT_LOOKUP_MAX_IMAGES = 6;
|
|
|
41
41
|
const DEFAULT_LOOKUP_MAX_TOKENS = 1_600;
|
|
42
42
|
const DEFAULT_LOOKUP_TIMEOUT_MS = 120_000;
|
|
43
43
|
const MAX_IMAGE_BYTES = 16 * 1024 * 1024;
|
|
44
|
-
// Keep recovery nudges sparse: newer GLM coding models are substantially better at
|
|
45
|
-
// long-horizon tool use, and repeated developer reminders can become prompt noise.
|
|
46
|
-
const SILENCE_REMINDER_MIN_VIOLATION_GAP = 5;
|
|
47
|
-
const SILENCE_REMINDER_MIN_MESSAGE_GAP = 20;
|
|
48
|
-
// When the visible history shrinks by more than this many messages (compaction or
|
|
49
|
-
// truncation), the chatter baseline is reset so it isn't measured against a stale peak.
|
|
50
|
-
const SILENCE_REMINDER_COMPACTION_MARGIN = 8;
|
|
51
44
|
const LOOKUP_TOOL_NAME = "lookup";
|
|
52
45
|
|
|
53
46
|
const LOOKUP_TOOL_PARAMS = Type.Object(
|
|
@@ -65,48 +58,46 @@ const LOOKUP_TOOL_PARAMS = Type.Object(
|
|
|
65
58
|
{ additionalProperties: false },
|
|
66
59
|
);
|
|
67
60
|
|
|
68
|
-
type DisciplinePromptOptions = { lookupEnabled?: boolean
|
|
61
|
+
type DisciplinePromptOptions = { lookupEnabled?: boolean };
|
|
69
62
|
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
];
|
|
109
|
-
}
|
|
63
|
+
const QUALITY_DISCIPLINE_LINES = [
|
|
64
|
+
"GLM CODING QUALITY CONTRACT.",
|
|
65
|
+
"",
|
|
66
|
+
"Optimize for correct, evidence-backed changes rather than fast patching or conversational polish.",
|
|
67
|
+
"",
|
|
68
|
+
"Evidence and hypothesis discipline:",
|
|
69
|
+
"- inspect before editing; do not invent APIs, files, commands, behavior, or repository conventions;",
|
|
70
|
+
"- separate observed facts from hypotheses; do not commit to the first plausible root cause;",
|
|
71
|
+
"- for non-obvious bugs, keep at least two plausible hypotheses until evidence distinguishes them, and choose the next inspection/test for information gain;",
|
|
72
|
+
"- do not use edits as diagnostic probes when read/search/repro/test evidence can answer the question first;",
|
|
73
|
+
"- when behavior crosses modules, trace the relevant execution/data flow before editing instead of serially opening plausible files; use indexed architecture/search/dependency tools when available;",
|
|
74
|
+
"- before changing an API, state contract, persistence format, or async lifecycle, inspect relevant callers, consumers, tests, persistence boundaries, and cancellation/error paths.",
|
|
75
|
+
"",
|
|
76
|
+
"Implementation discipline:",
|
|
77
|
+
"- make the smallest change that fully fixes the issue and follow nearby conventions;",
|
|
78
|
+
"- for bugs, reproduce the failure or its defining condition when practical before changing code;",
|
|
79
|
+
"- high-risk changes (security, data/schema, public APIs, concurrency, irreversible behavior) need a short evidence-backed spec before implementation;",
|
|
80
|
+
"- handle relevant edge cases, errors, cancellation, stale state, and async behavior; do not block UI/event loops;",
|
|
81
|
+
"- avoid duplicate state, duplicate prompts, repeated side effects, and speculative refactors unrelated to the verified cause;",
|
|
82
|
+
"- write code, identifiers, comments, and commit messages in English.",
|
|
83
|
+
"",
|
|
84
|
+
"Failure and retry discipline:",
|
|
85
|
+
"- treat failed verification as new evidence; do not repeat the same approach with cosmetic variations;",
|
|
86
|
+
"- after two materially failed fix/verification attempts, stop patching: re-inspect evidence, revise the root-cause hypothesis, or escalate for an independent opinion.",
|
|
87
|
+
"",
|
|
88
|
+
"Verification discipline:",
|
|
89
|
+
"- verification must cover the behavioral claim that justified the change; typecheck/build/lint prove structural validity, not behavioral correctness;",
|
|
90
|
+
"- for bug fixes, verify the old failure or its defining condition no longer occurs whenever practical;",
|
|
91
|
+
"- for async/stateful changes, verify the success path plus the relevant cancellation, error, retry, duplicate-event, or stale-state path;",
|
|
92
|
+
"- before finalizing a non-trivial change, actively inspect or test one likely counterexample, regression path, or missed edge case;",
|
|
93
|
+
"- never claim a test/check passed unless it actually ran and its result was observed.",
|
|
94
|
+
"",
|
|
95
|
+
"Escalation discipline:",
|
|
96
|
+
"- when async subagents are available, use an independent `oracle` (preferably Sol through the configured oracle role) when focused inspection still leaves conflicting evidence, multiple incompatible root causes, a subtle cross-module invariant you cannot verify locally, or a high-risk security/schema/public-API/concurrency/irreversible decision;",
|
|
97
|
+
"- also escalate after two materially different fixes fail for the same unresolved cause;",
|
|
98
|
+
"- do not escalate routine implementation, mechanical edits, straightforward test failures, or facts that local tools can verify directly;",
|
|
99
|
+
"- treat oracle output as a second opinion, not authority: reconcile it with repository evidence before acting.",
|
|
100
|
+
];
|
|
110
101
|
|
|
111
102
|
const LOOKUP_DISCIPLINE_LINES = [
|
|
112
103
|
"",
|
|
@@ -120,17 +111,9 @@ const LOOKUP_DISCIPLINE_LINES = [
|
|
|
120
111
|
const FINAL_DISCIPLINE_LINES = [
|
|
121
112
|
"",
|
|
122
113
|
"When uncertain, test or inspect instead of assuming; otherwise proceed with grounded best effort.",
|
|
123
|
-
"Verify every non-trivial change. Never claim tests passed unless actually run.",
|
|
124
114
|
"Final report: what changed, what was verified, what was not verified, and any risks.",
|
|
125
115
|
];
|
|
126
116
|
|
|
127
|
-
const SILENCE_REMINDER_TEXT = [
|
|
128
|
-
"GLM silence reminder: remain in WORKING state.",
|
|
129
|
-
"Continue with tool-only discipline: inspect, verify, and act through tools only.",
|
|
130
|
-
"For the next step, emit tool calls with no accompanying assistant text.",
|
|
131
|
-
"Do not acknowledge this reminder.",
|
|
132
|
-
].join("\n");
|
|
133
|
-
|
|
134
117
|
const LEGACY_SILENT_PROMPT_BLOCK_PATTERN = new RegExp(
|
|
135
118
|
`${escapeRegExp(SILENT_PROMPT_MARKER_START)}[\\s\\S]*?${escapeRegExp(SILENT_PROMPT_MARKER_END)}\\s*`,
|
|
136
119
|
"g",
|
|
@@ -167,10 +150,6 @@ const LOOKUP_SYSTEM_PROMPT = [
|
|
|
167
150
|
export default function codingDiscipline(pi: ExtensionAPI) {
|
|
168
151
|
let selectedModelRef: string | undefined;
|
|
169
152
|
let lookupRegistered = false;
|
|
170
|
-
let silenceViolationCount = 0;
|
|
171
|
-
let lastReminderViolationCount = 0;
|
|
172
|
-
let lastReminderMessageCount = -SILENCE_REMINDER_MIN_MESSAGE_GAP;
|
|
173
|
-
let peakMessageCount = 0;
|
|
174
153
|
|
|
175
154
|
function maybeRegisterLookupTool(cwd?: string): void {
|
|
176
155
|
if (lookupRegistered) return;
|
|
@@ -220,7 +199,6 @@ export default function codingDiscipline(pi: ExtensionAPI) {
|
|
|
220
199
|
const cwd = contextCwd(ctx);
|
|
221
200
|
const injected = injectCodingDisciplineIntoPayload(event.payload, {
|
|
222
201
|
lookupEnabled: Boolean(lookupModelFromConfig(cwd)) && isBlindGlmModel(modelRef),
|
|
223
|
-
strictness: codingDisciplineStrictnessFromConfig(cwd),
|
|
224
202
|
});
|
|
225
203
|
if (process.env.PI_DEBUG_PROMPT === "1") {
|
|
226
204
|
logFinalPrompt(injected, modelRef, contextCwd(ctx) ?? process.cwd());
|
|
@@ -292,36 +270,6 @@ export default function codingDiscipline(pi: ExtensionAPI) {
|
|
|
292
270
|
return undefined;
|
|
293
271
|
});
|
|
294
272
|
|
|
295
|
-
pi.on("context", async (event: { messages?: unknown[] }, ctx: unknown) => {
|
|
296
|
-
const modelRef = selectedModelRef ?? modelRefFromContext(ctx);
|
|
297
|
-
if (!isGlmModel(modelRef) || !Array.isArray(event.messages)) return undefined;
|
|
298
|
-
|
|
299
|
-
const messageCount = event.messages.length;
|
|
300
|
-
// Compaction/truncation prunes the visible history. Reset the stale chatter
|
|
301
|
-
// baseline so the post-compaction turn isn't measured against a pre-compaction
|
|
302
|
-
// violation peak (otherwise the model could chatter freely until the count
|
|
303
|
-
// climbed back above the old peak, or get nagged on a freshly small count).
|
|
304
|
-
if (messageCount + SILENCE_REMINDER_COMPACTION_MARGIN < peakMessageCount) {
|
|
305
|
-
silenceViolationCount = 0;
|
|
306
|
-
lastReminderViolationCount = 0;
|
|
307
|
-
lastReminderMessageCount = messageCount - SILENCE_REMINDER_MIN_MESSAGE_GAP;
|
|
308
|
-
}
|
|
309
|
-
peakMessageCount = Math.max(peakMessageCount, messageCount);
|
|
310
|
-
|
|
311
|
-
const strictness = codingDisciplineStrictnessFromConfig(contextCwd(ctx));
|
|
312
|
-
const violationCount = countAssistantToolChatter(event.messages, strictness);
|
|
313
|
-
if (violationCount <= silenceViolationCount) return undefined;
|
|
314
|
-
|
|
315
|
-
const violationGap = violationCount - lastReminderViolationCount;
|
|
316
|
-
const messageGap = messageCount - lastReminderMessageCount;
|
|
317
|
-
silenceViolationCount = violationCount;
|
|
318
|
-
|
|
319
|
-
if (violationGap < SILENCE_REMINDER_MIN_VIOLATION_GAP && messageGap < SILENCE_REMINDER_MIN_MESSAGE_GAP) return undefined;
|
|
320
|
-
|
|
321
|
-
lastReminderViolationCount = violationCount;
|
|
322
|
-
lastReminderMessageCount = messageCount;
|
|
323
|
-
return { messages: [...event.messages, createSilenceReminderMessage()] };
|
|
324
|
-
});
|
|
325
273
|
}
|
|
326
274
|
|
|
327
275
|
export function prependCodingDisciplinePrompt(systemPrompt: string, options: DisciplinePromptOptions = {}): string {
|
|
@@ -335,10 +283,9 @@ export function prependCodingDisciplinePrompt(systemPrompt: string, options: Dis
|
|
|
335
283
|
}
|
|
336
284
|
|
|
337
285
|
export function buildCodingDisciplinePrompt(options: DisciplinePromptOptions = {}): string {
|
|
338
|
-
const strictness = options.strictness ?? DEFAULT_CODING_DISCIPLINE_STRICTNESS;
|
|
339
286
|
return [
|
|
340
287
|
DISCIPLINE_PROMPT_MARKER_START,
|
|
341
|
-
...
|
|
288
|
+
...QUALITY_DISCIPLINE_LINES,
|
|
342
289
|
...(options.lookupEnabled ? LOOKUP_DISCIPLINE_LINES : []),
|
|
343
290
|
...FINAL_DISCIPLINE_LINES,
|
|
344
291
|
DISCIPLINE_PROMPT_MARKER_END,
|
|
@@ -637,58 +584,10 @@ function isInstructionMessage(message: unknown): boolean {
|
|
|
637
584
|
return message.role === "system" || message.role === "developer";
|
|
638
585
|
}
|
|
639
586
|
|
|
640
|
-
function countAssistantToolChatter(messages: readonly unknown[], strictness: CodingDisciplineStrictness): number {
|
|
641
|
-
let count = 0;
|
|
642
|
-
for (const message of messages) {
|
|
643
|
-
if (!isAssistantToolChatter(message, strictness)) continue;
|
|
644
|
-
count++;
|
|
645
|
-
}
|
|
646
|
-
return count;
|
|
647
|
-
}
|
|
648
|
-
|
|
649
|
-
/**
|
|
650
|
-
* Detects assistant chatter (text alongside a tool call).
|
|
651
|
-
* - "strict" — any such text counts as chatter.
|
|
652
|
-
* - "lenient" — text counts only when a thinking/reasoning block already captured the
|
|
653
|
-
* reasoning; without a thinking block, the visible text is the model's only reasoning
|
|
654
|
-
* channel and must not be suppressed.
|
|
655
|
-
*/
|
|
656
|
-
function isAssistantToolChatter(message: unknown, strictness: CodingDisciplineStrictness): boolean {
|
|
657
|
-
if (!isRecord(message) || message.role !== "assistant") return false;
|
|
658
|
-
if (!Array.isArray(message.content)) return false;
|
|
659
|
-
const hasToolCall = message.content.some((part) => isRecord(part) && part.type === "toolCall");
|
|
660
|
-
if (!hasToolCall) return false;
|
|
661
|
-
const hasText = message.content.some((part) => isRecord(part) && part.type === "text" && hasNonEmptyText(part.text));
|
|
662
|
-
if (!hasText) return false;
|
|
663
|
-
if (strictness === "strict") return true;
|
|
664
|
-
const hasThinking = message.content.some(
|
|
665
|
-
(part) => isRecord(part) && (part.type === "thinking" || part.type === "reasoning"),
|
|
666
|
-
);
|
|
667
|
-
return Boolean(hasThinking);
|
|
668
|
-
}
|
|
669
|
-
|
|
670
|
-
function hasNonEmptyText(value: unknown): boolean {
|
|
671
|
-
return typeof value === "string" && value.trim().length > 0;
|
|
672
|
-
}
|
|
673
|
-
|
|
674
|
-
function createSilenceReminderMessage() {
|
|
675
|
-
return {
|
|
676
|
-
// Inject as a developer/system-level nudge rather than impersonating the user,
|
|
677
|
-
// so it reads as an automated reminder, not a user instruction.
|
|
678
|
-
role: "developer" as const,
|
|
679
|
-
content: [{ type: "text" as const, text: SILENCE_REMINDER_TEXT }],
|
|
680
|
-
timestamp: Date.now(),
|
|
681
|
-
};
|
|
682
|
-
}
|
|
683
|
-
|
|
684
587
|
function lookupModelFromConfig(cwd?: string): string | undefined {
|
|
685
588
|
return loadPiToolsSuiteConfig(["coding-discipline"], { cwd: cwd ?? process.cwd() }).lookupModel;
|
|
686
589
|
}
|
|
687
590
|
|
|
688
|
-
function codingDisciplineStrictnessFromConfig(cwd?: string): CodingDisciplineStrictness {
|
|
689
|
-
return loadPiToolsSuiteConfig(["coding-discipline"], { cwd: cwd ?? process.cwd() }).codingDisciplineStrictness ?? DEFAULT_CODING_DISCIPLINE_STRICTNESS;
|
|
690
|
-
}
|
|
691
|
-
|
|
692
591
|
function buildLookupPrompt(params: LookupParams, recentContext: string, imageCount: number, warnings: string[]): string {
|
|
693
592
|
return [
|
|
694
593
|
"Lookup request from a text-only GLM parent model.",
|
|
@@ -12,14 +12,6 @@ export interface PiToolsSuiteConfig {
|
|
|
12
12
|
todoThinkingOverrides: Record<string, TodoThinkingLevel>;
|
|
13
13
|
/** Vision-capable model used by the coding-discipline lookup tool; unset disables lookup. */
|
|
14
14
|
lookupModel?: string;
|
|
15
|
-
/**
|
|
16
|
-
* Chatter-detector strictness for the coding-discipline module:
|
|
17
|
-
* "strict" — any assistant text alongside a tool call is chatter (Opus-like);
|
|
18
|
-
* "lenient" — text is only chatter when a thinking block already captured the
|
|
19
|
-
* reasoning; without thinking, visible text is the reasoning channel.
|
|
20
|
-
* Default: "lenient".
|
|
21
|
-
*/
|
|
22
|
-
codingDisciplineStrictness?: CodingDisciplineStrictness;
|
|
23
15
|
}
|
|
24
16
|
|
|
25
17
|
type MutableConfig = {
|
|
@@ -28,12 +20,8 @@ type MutableConfig = {
|
|
|
28
20
|
todoThinking: boolean;
|
|
29
21
|
todoThinkingOverrides: Map<string, TodoThinkingLevel>;
|
|
30
22
|
lookupModel: string | undefined;
|
|
31
|
-
codingDisciplineStrictness: CodingDisciplineStrictness;
|
|
32
23
|
};
|
|
33
24
|
|
|
34
|
-
export const CODING_DISCIPLINE_STRICTNESS_VALUES = ["strict", "lenient"] as const;
|
|
35
|
-
export type CodingDisciplineStrictness = (typeof CODING_DISCIPLINE_STRICTNESS_VALUES)[number];
|
|
36
|
-
export const DEFAULT_CODING_DISCIPLINE_STRICTNESS: CodingDisciplineStrictness = "lenient";
|
|
37
25
|
const TODO_THINKING_OVERRIDE_LEVELS = ["off", "minimal", "low", "medium", "high", "xhigh", "max"] as const;
|
|
38
26
|
export type TodoThinkingLevel = (typeof TODO_THINKING_OVERRIDE_LEVELS)[number];
|
|
39
27
|
|
|
@@ -80,10 +68,6 @@ function normalizeLookupModel(raw: unknown): string | undefined {
|
|
|
80
68
|
return trimmed ? trimmed : undefined;
|
|
81
69
|
}
|
|
82
70
|
|
|
83
|
-
function normalizeCodingDisciplineStrictness(raw: unknown): CodingDisciplineStrictness {
|
|
84
|
-
return raw === "strict" ? "strict" : "lenient";
|
|
85
|
-
}
|
|
86
|
-
|
|
87
71
|
function isTodoThinkingLevel(raw: unknown): raw is TodoThinkingLevel {
|
|
88
72
|
return TODO_THINKING_OVERRIDE_LEVELS.includes(raw as TodoThinkingLevel);
|
|
89
73
|
}
|
|
@@ -161,9 +145,6 @@ function mergeConfigLayer(config: MutableConfig, raw: Record<string, unknown>, k
|
|
|
161
145
|
if (typeof raw.todoThinking === "boolean") config.todoThinking = raw.todoThinking;
|
|
162
146
|
mergeTodoThinkingOverrides(config, raw.todoThinkingOverrides);
|
|
163
147
|
if (Object.prototype.hasOwnProperty.call(raw, "lookupModel")) config.lookupModel = normalizeLookupModel(raw.lookupModel);
|
|
164
|
-
if (Object.prototype.hasOwnProperty.call(raw, "codingDisciplineStrictness")) {
|
|
165
|
-
config.codingDisciplineStrictness = normalizeCodingDisciplineStrictness(raw.codingDisciplineStrictness);
|
|
166
|
-
}
|
|
167
148
|
|
|
168
149
|
for (const key of DISABLED_LIST_KEYS) addDisabled(config, raw[key], knownModules);
|
|
169
150
|
for (const key of ENABLED_LIST_KEYS) removeDisabled(config, raw[key], knownModules);
|
|
@@ -222,7 +203,6 @@ export function loadPiToolsSuiteConfig(moduleNames: readonly string[], options:
|
|
|
222
203
|
todoThinking: false,
|
|
223
204
|
todoThinkingOverrides: new Map(DEFAULT_TODO_THINKING_OVERRIDES),
|
|
224
205
|
lookupModel: undefined,
|
|
225
|
-
codingDisciplineStrictness: DEFAULT_CODING_DISCIPLINE_STRICTNESS,
|
|
226
206
|
};
|
|
227
207
|
const userConfigPath = getPiToolsSuiteUserConfigPath(options.homeDir);
|
|
228
208
|
|
|
@@ -243,6 +223,5 @@ export function loadPiToolsSuiteConfig(moduleNames: readonly string[], options:
|
|
|
243
223
|
todoThinking: config.todoThinking,
|
|
244
224
|
todoThinkingOverrides: Object.fromEntries(config.todoThinkingOverrides),
|
|
245
225
|
...(config.lookupModel ? { lookupModel: config.lookupModel } : {}),
|
|
246
|
-
codingDisciplineStrictness: config.codingDisciplineStrictness,
|
|
247
226
|
};
|
|
248
227
|
}
|
|
@@ -116,7 +116,7 @@ export function buildProgrammaticSummary(
|
|
|
116
116
|
return lines.join("\n")
|
|
117
117
|
}
|
|
118
118
|
|
|
119
|
-
const SUMMARIZER_SYSTEM_PROMPT = `You summarize a slice of a coding agent's conversation so it can replace the raw messages in context. Produce a dense, continuation-focused summary: preserve user intent, decisions made, files/symbols changed or inspected, exact errors still actionable, verification status, and next steps. Do not infer, invent, or add facts absent from the source; preserve uncertainty instead of filling gaps. Drop full logs, repeated output, and incidental detail. Be concise (roughly 4-10 bullets). Output ONLY the summary text, no preamble.`
|
|
119
|
+
const SUMMARIZER_SYSTEM_PROMPT = `You summarize a slice of a coding agent's conversation so it can replace the raw messages in context. Produce a dense, continuation-focused summary: preserve user intent, decisions made, files/symbols changed or inspected, exact errors still actionable, verification status, and next steps. Preserve exact identifiers and explicit continuity markers verbatim, including uppercase labels before colons; never paraphrase or omit those labels. Do not infer, invent, or add facts absent from the source; preserve uncertainty instead of filling gaps. Drop full logs, repeated output, and incidental detail without quoting or naming the discarded log lines or their markers. Be concise (roughly 4-10 bullets). Output ONLY the summary text, no preamble.`
|
|
120
120
|
|
|
121
121
|
/** Outcome of one summarizer-model attempt, surfaced in DCP debug logs. */
|
|
122
122
|
export interface ModelSummaryAttempt {
|
|
@@ -388,55 +388,6 @@ export function resolveIdToBoundary(
|
|
|
388
388
|
const ts = state.messageIdSnapshot.get(id)
|
|
389
389
|
if (ts !== undefined) return { timestamp: ts }
|
|
390
390
|
|
|
391
|
-
// ── Stale mNNN fallback: direction-only clamp ────────────────────────
|
|
392
|
-
// When compression or pruning removes messages between context passes,
|
|
393
|
-
// positional mNNN IDs shift (e.g. an end boundary m145 becomes m123
|
|
394
|
-
// after 22 messages are removed). Clamp to the closest valid ID, but
|
|
395
|
-
// ONLY in a direction that preserves the range's semantics:
|
|
396
|
-
// - start boundary: clamp upward to the first available ID at or after
|
|
397
|
-
// the requested number. The start must never move backwards into
|
|
398
|
-
// older content; if no such ID exists, the requested start is gone
|
|
399
|
-
// and we cannot safely compress — throw.
|
|
400
|
-
// - end boundary: clamp downward to the last available ID at or before
|
|
401
|
-
// the requested number. The end must never move forwards into newer
|
|
402
|
-
// content; if no such ID exists, throw.
|
|
403
|
-
// The previous implementation fell back to the highest available ID in
|
|
404
|
-
// both "no match" cases, which could clamp e.g. m010..m010 over a
|
|
405
|
-
// snapshot of m001..m003 to a single-message block over m003 — silently
|
|
406
|
-
// compressing the wrong content.
|
|
407
|
-
const mMatch = id.match(/^m(\d+)$/i)
|
|
408
|
-
if (mMatch && state.messageIdSnapshot.size > 0) {
|
|
409
|
-
const requestedNum = parseInt(mMatch[1]!, 10)
|
|
410
|
-
const allIds = sortIds([...state.messageIdSnapshot.keys()])
|
|
411
|
-
const allNums = allIds
|
|
412
|
-
.map((mid) => {
|
|
413
|
-
const n = mid.match(/^m(\d+)$/i)
|
|
414
|
-
return n ? { id: mid, num: parseInt(n[1]!, 10) } : null
|
|
415
|
-
})
|
|
416
|
-
.filter((entry): entry is { id: string; num: number } => entry !== null)
|
|
417
|
-
.sort((a, b) => a.num - b.num)
|
|
418
|
-
|
|
419
|
-
if (allNums.length > 0) {
|
|
420
|
-
let clamped: { id: string; num: number } | undefined
|
|
421
|
-
if (field === "startTimestamp") {
|
|
422
|
-
clamped = allNums.find((entry) => entry.num >= requestedNum)
|
|
423
|
-
} else {
|
|
424
|
-
for (let i = allNums.length - 1; i >= 0; i--) {
|
|
425
|
-
if (allNums[i]!.num <= requestedNum) {
|
|
426
|
-
clamped = allNums[i]
|
|
427
|
-
break
|
|
428
|
-
}
|
|
429
|
-
}
|
|
430
|
-
}
|
|
431
|
-
if (clamped) {
|
|
432
|
-
const clampedMeta = state.messageMetaSnapshot.get(clamped.id)
|
|
433
|
-
if (clampedMeta) return resolveMetaBoundary(clampedMeta, field, state)
|
|
434
|
-
const clampedTs = state.messageIdSnapshot.get(clamped.id)
|
|
435
|
-
if (clampedTs !== undefined) return { timestamp: clampedTs }
|
|
436
|
-
}
|
|
437
|
-
}
|
|
438
|
-
}
|
|
439
|
-
|
|
440
391
|
throw unknownCompressionIdError(id, state)
|
|
441
392
|
}
|
|
442
393
|
|
|
@@ -445,8 +396,7 @@ export function resolveIdToBoundary(
|
|
|
445
396
|
* synthetic placeholder for an active compression block (`meta.blockId` is
|
|
446
397
|
* set), resolve to that block's stored boundary so the caller rolls the
|
|
447
398
|
* block up instead of nesting a new block on top of the placeholder. This
|
|
448
|
-
*
|
|
449
|
-
* may itself represent a previously compressed section.
|
|
399
|
+
* A model-visible mNNN may itself represent a previously compressed section.
|
|
450
400
|
*/
|
|
451
401
|
function resolveMetaBoundary(
|
|
452
402
|
meta: MessageIdMeta,
|
|
@@ -127,7 +127,13 @@ export function summarizeDcpState(state: DcpState): Record<string, unknown> {
|
|
|
127
127
|
nextBlockId: state.nextBlockId,
|
|
128
128
|
},
|
|
129
129
|
inactiveBlocksTail: inactiveBlocks,
|
|
130
|
+
persistentMessageIds: state.messageIdsByStableId.size,
|
|
131
|
+
nextMessageId: state.nextMessageId,
|
|
130
132
|
prunedTools: state.prunedToolIds.size,
|
|
133
|
+
automaticPruneCheckpoint: {
|
|
134
|
+
turn: state.lastAutomaticPruneTurn,
|
|
135
|
+
blockId: state.lastAutomaticPruneBlockId,
|
|
136
|
+
},
|
|
131
137
|
providerSeenTools: state.providerSeenToolIds.size,
|
|
132
138
|
consecutiveEmergencyPasses: state.consecutiveIgnoredStrongNudges,
|
|
133
139
|
nudgeAnchors: state.nudgeAnchors.map((anchor) => ({
|