@cydm/pie 1.0.48 → 1.0.50
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/builtin/extensions/ask-user/index.js +2 -9
- package/dist/builtin/extensions/ask-user/package.json +0 -7
- package/dist/builtin/extensions/changelog/index.js +15 -20
- package/dist/builtin/extensions/document-attachments/index.js +0 -1
- package/dist/builtin/extensions/document-attachments/package.json +1 -1
- package/dist/builtin/extensions/init/index.js +0 -13
- package/dist/builtin/extensions/plan-mode/index.js +29 -17
- package/dist/builtin/extensions/subagent/index.js +15 -31
- package/dist/builtin/extensions/todo/index.js +4 -8
- package/dist/builtin/skills/skill-creator/SKILL.md +15 -43
- package/dist/builtin/skills/skill-creator/eval-viewer/generate_review.mjs +1 -1
- package/dist/builtin/skills/skill-creator/eval-viewer/viewer.html +2 -2
- package/dist/builtin/skills/skill-creator/references/schemas.md +1 -1
- package/dist/builtin/skills/skill-creator/scripts/generate_report.mjs +1 -1
- package/dist/builtin/skills/skill-creator/scripts/improve_description.mjs +3 -3
- package/dist/builtin/skills/skill-creator/scripts/pie_runner.mjs +32 -4
- package/dist/builtin/skills/skill-creator/scripts/run_loop.mjs +1 -1
- package/dist/chunks/{chunk-IJMP3JTX.js → chunk-DZS34ACX.js} +230 -46
- package/dist/chunks/{chunk-U4B6RJ6Z.js → chunk-XBWFDIAU.js} +1 -1
- package/dist/cli.js +127 -80
- package/package.json +3 -3
- package/dist/builtin/extensions/deploy/index.js +0 -11
- package/dist/builtin/extensions/deploy/package.json +0 -11
- package/dist/builtin/extensions/files/index.js +0 -10
- package/dist/builtin/extensions/files/package.json +0 -14
- package/dist/builtin/extensions/kimi-attachments/index.js +0 -47
- package/dist/builtin/extensions/kimi-attachments/package.json +0 -11
- package/dist/builtin/skills/skill-creator/scripts/claude_cli.mjs +0 -115
package/README.md
CHANGED
|
@@ -274,7 +274,7 @@ That is part of what makes Pie embeddable: the CLI is only one interface on top
|
|
|
274
274
|
|
|
275
275
|
Extensions are discovered from four sources in ascending priority (later sources shadow earlier ones by stable extension identity):
|
|
276
276
|
|
|
277
|
-
1. `builtin` — shipped with the CLI (`ask-user`, `todo`, `plan-mode`, `subagent`, `init`, `changelog`,
|
|
277
|
+
1. `builtin` — shipped with the CLI (`ask-user`, `todo`, `plan-mode`, `subagent`, `init`, `changelog`, attachment handlers)
|
|
278
278
|
2. `global` — `~/.pie/extensions/`
|
|
279
279
|
3. `project` — `./.pie/extensions/`
|
|
280
280
|
4. `configured` — repeated `--extension-path <dir>` flags
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { createRequire as __createRequire } from "node:module"; const require = __createRequire(import.meta.url);
|
|
2
2
|
import {
|
|
3
3
|
createAskUserCapability
|
|
4
|
-
} from "../../../chunks/chunk-
|
|
4
|
+
} from "../../../chunks/chunk-DZS34ACX.js";
|
|
5
5
|
import "../../../chunks/chunk-FKAO25XH.js";
|
|
6
6
|
import "../../../chunks/chunk-K6ZOCJ4V.js";
|
|
7
7
|
import "../../../chunks/chunk-LZQXQOPJ.js";
|
|
@@ -10,14 +10,7 @@ import "../../../chunks/chunk-LZQXQOPJ.js";
|
|
|
10
10
|
function askUserExtension(ctx) {
|
|
11
11
|
ctx.log("Ask user extension loaded");
|
|
12
12
|
const capability = createAskUserCapability();
|
|
13
|
-
|
|
14
|
-
ctx.registerTool({
|
|
15
|
-
...tool,
|
|
16
|
-
async execute(args) {
|
|
17
|
-
ctx.log(`[${tool.name}] Called: ${JSON.stringify(args)}`);
|
|
18
|
-
return tool.execute(args);
|
|
19
|
-
}
|
|
20
|
-
});
|
|
13
|
+
ctx.registerTool(capability.tools[0]);
|
|
21
14
|
}
|
|
22
15
|
export {
|
|
23
16
|
askUserExtension as default
|
|
@@ -2,15 +2,8 @@
|
|
|
2
2
|
"name": "@pie/ask-user-extension",
|
|
3
3
|
"version": "1.0.0",
|
|
4
4
|
"type": "module",
|
|
5
|
-
"main": "dist/index.js",
|
|
6
5
|
"pie": {
|
|
7
6
|
"name": "ask-user",
|
|
8
7
|
"main": "index.ts"
|
|
9
|
-
},
|
|
10
|
-
"scripts": {
|
|
11
|
-
"build": "tsc"
|
|
12
|
-
},
|
|
13
|
-
"dependencies": {
|
|
14
|
-
"@sinclair/typebox": "^0.34.0"
|
|
15
8
|
}
|
|
16
9
|
}
|
|
@@ -1,12 +1,14 @@
|
|
|
1
1
|
import { createRequire as __createRequire } from "node:module"; const require = __createRequire(import.meta.url);
|
|
2
|
-
import
|
|
3
|
-
__require
|
|
4
|
-
} from "../../../chunks/chunk-LZQXQOPJ.js";
|
|
2
|
+
import "../../../chunks/chunk-LZQXQOPJ.js";
|
|
5
3
|
|
|
6
4
|
// builtin/extensions/changelog/index.ts
|
|
5
|
+
import * as fs from "node:fs";
|
|
6
|
+
import * as path from "node:path";
|
|
7
|
+
import { spawnSync } from "node:child_process";
|
|
8
|
+
var RECENT_COMMIT_COUNT = 15;
|
|
9
|
+
var RECENT_COMMIT_RANGE = "HEAD~5";
|
|
7
10
|
function getRecentCommits(cwd, count = 10) {
|
|
8
11
|
try {
|
|
9
|
-
const { spawnSync } = __require("child_process");
|
|
10
12
|
const result = spawnSync(
|
|
11
13
|
"git",
|
|
12
14
|
["log", `--max-count=${count}`, "--pretty=format:%h %s", "--no-merges"],
|
|
@@ -19,10 +21,9 @@ function getRecentCommits(cwd, count = 10) {
|
|
|
19
21
|
}
|
|
20
22
|
function getChangedFiles(cwd) {
|
|
21
23
|
try {
|
|
22
|
-
const { spawnSync } = __require("child_process");
|
|
23
24
|
const result = spawnSync(
|
|
24
25
|
"git",
|
|
25
|
-
["diff",
|
|
26
|
+
["diff", RECENT_COMMIT_RANGE, "--name-only", "--stat"],
|
|
26
27
|
{ cwd, encoding: "utf8" }
|
|
27
28
|
);
|
|
28
29
|
return result.status === 0 ? result.stdout : "";
|
|
@@ -32,8 +33,6 @@ function getChangedFiles(cwd) {
|
|
|
32
33
|
}
|
|
33
34
|
async function readChangelog(cwd) {
|
|
34
35
|
try {
|
|
35
|
-
const fs = await import("fs");
|
|
36
|
-
const path = await import("path");
|
|
37
36
|
const changelogPath = path.join(cwd, "CHANGELOG.md");
|
|
38
37
|
if (fs.existsSync(changelogPath)) {
|
|
39
38
|
return fs.readFileSync(changelogPath, "utf-8");
|
|
@@ -48,8 +47,6 @@ function changelogExtension(ctx) {
|
|
|
48
47
|
path: ["tools", "changelog"],
|
|
49
48
|
description: "Smart changelog management with gap detection",
|
|
50
49
|
handler: async (ctx2, args) => {
|
|
51
|
-
const fs = await import("fs");
|
|
52
|
-
const path = await import("path");
|
|
53
50
|
const changelogPath = path.join(ctx2.cwd, "CHANGELOG.md");
|
|
54
51
|
if (args?.includes("--show")) {
|
|
55
52
|
if (!fs.existsSync(changelogPath)) {
|
|
@@ -59,7 +56,7 @@ function changelogExtension(ctx) {
|
|
|
59
56
|
return "Current CHANGELOG:\n\n" + content.slice(0, 2e3) + (content.length > 2e3 ? "\n\n..." : "");
|
|
60
57
|
}
|
|
61
58
|
if (args?.trim() === "check") {
|
|
62
|
-
const commits2 = getRecentCommits(ctx2.cwd,
|
|
59
|
+
const commits2 = getRecentCommits(ctx2.cwd, RECENT_COMMIT_COUNT);
|
|
63
60
|
const changelog2 = await readChangelog(ctx2.cwd);
|
|
64
61
|
const changedFiles2 = getChangedFiles(ctx2.cwd);
|
|
65
62
|
if (!commits2) {
|
|
@@ -97,8 +94,7 @@ ${existingChangelog ? `
|
|
|
97
94
|
Use this existing style as reference:
|
|
98
95
|
\`\`\`
|
|
99
96
|
${existingChangelog.slice(0, 800)}
|
|
100
|
-
|
|
101
|
-
\`\`\`` : ""}
|
|
97
|
+
...\`\`\`` : ""}
|
|
102
98
|
|
|
103
99
|
Requirements:
|
|
104
100
|
1. Choose the right category: **Added** / **Changed** / **Fixed** / **Removed** / **Deprecated** / **Security**.
|
|
@@ -118,16 +114,15 @@ Use the write tool to update CHANGELOG.md:
|
|
|
118
114
|
const versionHint = args.slice(7).trim();
|
|
119
115
|
const existingChangelog = await readChangelog(ctx2.cwd);
|
|
120
116
|
if (!existingChangelog) {
|
|
121
|
-
return "No CHANGELOG.md found. Create one first with /changelog";
|
|
117
|
+
return "No CHANGELOG.md found. Create one first with /tools changelog";
|
|
122
118
|
}
|
|
123
|
-
ctx2.ui.notify(
|
|
119
|
+
ctx2.ui.notify(`Preparing release ${versionHint || "..."}`, "info");
|
|
124
120
|
return `Prepare the [Unreleased] section as a new release.
|
|
125
121
|
|
|
126
122
|
Current CHANGELOG:
|
|
127
123
|
\`\`\`
|
|
128
124
|
${existingChangelog.slice(0, 1500)}
|
|
129
|
-
|
|
130
|
-
\`\`\`
|
|
125
|
+
...\`\`\`
|
|
131
126
|
|
|
132
127
|
Version hint: ${versionHint || "infer from changes (patch/minor/major)"}
|
|
133
128
|
|
|
@@ -146,11 +141,11 @@ Tasks:
|
|
|
146
141
|
|
|
147
142
|
Use the write tool to update CHANGELOG.md.`;
|
|
148
143
|
}
|
|
149
|
-
const commits = getRecentCommits(ctx2.cwd,
|
|
144
|
+
const commits = getRecentCommits(ctx2.cwd, RECENT_COMMIT_COUNT);
|
|
150
145
|
const changelog = await readChangelog(ctx2.cwd);
|
|
151
146
|
const changedFiles = getChangedFiles(ctx2.cwd);
|
|
152
147
|
if (!commits && !changelog) {
|
|
153
|
-
return "No git history or CHANGELOG.md found. Initialize with: /changelog add <description>";
|
|
148
|
+
return "No git history or CHANGELOG.md found. Initialize with: /tools changelog add <description>";
|
|
154
149
|
}
|
|
155
150
|
ctx2.ui.notify("Analyzing changes and checking for missing changelog entries...", "info");
|
|
156
151
|
return `Smart changelog update
|
|
@@ -164,7 +159,7 @@ ${commits || "(git history unavailable)"}
|
|
|
164
159
|
${changedFiles || "(unavailable)"}
|
|
165
160
|
|
|
166
161
|
## Current CHANGELOG.md
|
|
167
|
-
${changelog ? "\n```\n" + changelog.slice(0, 2e3) + "\n
|
|
162
|
+
${changelog ? "\n```\n" + changelog.slice(0, 2e3) + "\n...```" : "(file does not exist; create it if needed)"}
|
|
168
163
|
|
|
169
164
|
## Tasks
|
|
170
165
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@pie/extension-document-attachments",
|
|
3
3
|
"version": "1.0.0",
|
|
4
|
-
"description": "
|
|
4
|
+
"description": "Local document attachment preprocessing (PDF, HTML, plain text)",
|
|
5
5
|
"main": "index.ts",
|
|
6
6
|
"type": "module",
|
|
7
7
|
"pie": {
|
|
@@ -96,24 +96,11 @@ CLI file tools use the project root as their access boundary; do not pass an ext
|
|
|
96
96
|
}
|
|
97
97
|
async function initExtension(ctx) {
|
|
98
98
|
ctx.log("Init extension loaded");
|
|
99
|
-
let cachedKnowledge = null;
|
|
100
|
-
try {
|
|
101
|
-
const existing = await loadExistingKnowledge(ctx.cwd);
|
|
102
|
-
cachedKnowledge = existing.content;
|
|
103
|
-
if (cachedKnowledge) {
|
|
104
|
-
ctx.log(`Pre-loaded root AGENTS.md (${cachedKnowledge.length} chars) from ${existing.target.agentsPath}`);
|
|
105
|
-
} else {
|
|
106
|
-
ctx.log(`No existing root AGENTS.md found at ${existing.target.agentsPath}`);
|
|
107
|
-
}
|
|
108
|
-
} catch (err) {
|
|
109
|
-
ctx.log(`Failed to pre-load root AGENTS.md: ${err}`);
|
|
110
|
-
}
|
|
111
99
|
const handler = async (ctx2, args) => {
|
|
112
100
|
const existing = await loadExistingKnowledge(ctx2.cwd);
|
|
113
101
|
const existingContent = existing.content;
|
|
114
102
|
const isUpdate = !!existing.content;
|
|
115
103
|
const userIntent = args?.trim();
|
|
116
|
-
cachedKnowledge = existingContent;
|
|
117
104
|
if (isUpdate) {
|
|
118
105
|
if (userIntent) {
|
|
119
106
|
ctx2.ui.notify(`Updating project knowledge (${userIntent})...`, "info");
|
|
@@ -7,7 +7,7 @@ import {
|
|
|
7
7
|
isPlanModeSafeCommand,
|
|
8
8
|
markCompletedPlanSteps,
|
|
9
9
|
restoreExecutionState
|
|
10
|
-
} from "../../../chunks/chunk-
|
|
10
|
+
} from "../../../chunks/chunk-DZS34ACX.js";
|
|
11
11
|
import "../../../chunks/chunk-FKAO25XH.js";
|
|
12
12
|
import {
|
|
13
13
|
Type
|
|
@@ -21,11 +21,10 @@ import os from "node:os";
|
|
|
21
21
|
import path from "node:path";
|
|
22
22
|
var SESSION_METADATA_KEY = "planModeEnabled";
|
|
23
23
|
var EXECUTION_METADATA_KEY = "executionState";
|
|
24
|
+
var EXECUTION_FLAG_KEY = "planExecution";
|
|
24
25
|
var SCRATCH_METADATA_KEY = "planScratchId";
|
|
25
26
|
var SCRATCH_NAME = /^[a-zA-Z0-9][a-zA-Z0-9._-]{0,63}$/;
|
|
26
27
|
var MAX_SCRATCH_CHARS = 64e3;
|
|
27
|
-
var planModeEnabled = false;
|
|
28
|
-
var executionMode = false;
|
|
29
28
|
function toPlanTodoItems(state) {
|
|
30
29
|
return state.steps.map((step) => ({
|
|
31
30
|
step: step.id,
|
|
@@ -35,6 +34,18 @@ function toPlanTodoItems(state) {
|
|
|
35
34
|
}
|
|
36
35
|
function planModeExtension(ctx) {
|
|
37
36
|
const capability = createPlanCapability();
|
|
37
|
+
let planModeEnabled = false;
|
|
38
|
+
let executingPlan = false;
|
|
39
|
+
function setExecutingPlan(value) {
|
|
40
|
+
executingPlan = value;
|
|
41
|
+
ctx.setSessionMetadata(EXECUTION_FLAG_KEY, value);
|
|
42
|
+
}
|
|
43
|
+
function readPersistedPlanMode() {
|
|
44
|
+
return ctx.getSessionMetadata()?.[SESSION_METADATA_KEY] === true;
|
|
45
|
+
}
|
|
46
|
+
function readPersistedExecuting() {
|
|
47
|
+
return ctx.getSessionMetadata()?.[EXECUTION_FLAG_KEY] === true;
|
|
48
|
+
}
|
|
38
49
|
function scratchDirectory() {
|
|
39
50
|
const metadata = ctx.getSessionMetadata();
|
|
40
51
|
let id = metadata?.[SCRATCH_METADATA_KEY];
|
|
@@ -91,9 +102,6 @@ function planModeExtension(ctx) {
|
|
|
91
102
|
}
|
|
92
103
|
}
|
|
93
104
|
});
|
|
94
|
-
function readPersistedPlanMode() {
|
|
95
|
-
return ctx.getSessionMetadata()?.[SESSION_METADATA_KEY] === true;
|
|
96
|
-
}
|
|
97
105
|
function readExecutionState() {
|
|
98
106
|
return restoreExecutionState(ctx.getSessionMetadata()?.[EXECUTION_METADATA_KEY] ?? ctx.getExecutionState());
|
|
99
107
|
}
|
|
@@ -106,7 +114,7 @@ function planModeExtension(ctx) {
|
|
|
106
114
|
}
|
|
107
115
|
function updateStatus() {
|
|
108
116
|
const state = readExecutionState();
|
|
109
|
-
if (
|
|
117
|
+
if (executingPlan && state.steps.length > 0) {
|
|
110
118
|
const completed = state.steps.filter((step) => step.status === "completed").length;
|
|
111
119
|
ctx.ui.notify(`\u{1F4CB} Plan: ${completed}/${state.steps.length} completed`, "info");
|
|
112
120
|
} else if (planModeEnabled) {
|
|
@@ -115,14 +123,14 @@ function planModeExtension(ctx) {
|
|
|
115
123
|
}
|
|
116
124
|
function applyPlanModeState(enabled, forcePromptInjection = false) {
|
|
117
125
|
planModeEnabled = enabled;
|
|
118
|
-
if (
|
|
119
|
-
|
|
126
|
+
if (enabled) {
|
|
127
|
+
setExecutingPlan(false);
|
|
120
128
|
}
|
|
121
129
|
ctx.setPlanMode(enabled, { forcePromptInjection });
|
|
122
130
|
}
|
|
123
131
|
function togglePlanMode() {
|
|
124
132
|
planModeEnabled = !planModeEnabled;
|
|
125
|
-
|
|
133
|
+
setExecutingPlan(false);
|
|
126
134
|
writeExecutionState(null);
|
|
127
135
|
ctx.setSessionMetadata(SESSION_METADATA_KEY, planModeEnabled);
|
|
128
136
|
applyPlanModeState(planModeEnabled, true);
|
|
@@ -146,8 +154,8 @@ function planModeExtension(ctx) {
|
|
|
146
154
|
});
|
|
147
155
|
ctx.on("tool:call", async (event) => {
|
|
148
156
|
if (!planModeEnabled || event.toolName !== "bash") return;
|
|
149
|
-
const command = event.args
|
|
150
|
-
if (command && !isPlanModeSafeCommand(command)) {
|
|
157
|
+
const command = typeof event.args === "object" && event.args !== null ? event.args.command : void 0;
|
|
158
|
+
if (typeof command === "string" && !isPlanModeSafeCommand(command)) {
|
|
151
159
|
ctx.ui.notify(
|
|
152
160
|
`Plan mode: Command blocked. Use read-only commands, or plan_scratch for temporary drafts. Use /plan to leave Plan mode.
|
|
153
161
|
Command: ${command}`,
|
|
@@ -159,16 +167,18 @@ Command: ${command}`,
|
|
|
159
167
|
ctx.on("agent:start", async () => {
|
|
160
168
|
const persisted = readPersistedPlanMode();
|
|
161
169
|
applyPlanModeState(persisted);
|
|
170
|
+
executingPlan = !persisted && readPersistedExecuting();
|
|
162
171
|
writeExecutionState(readExecutionState());
|
|
163
172
|
});
|
|
164
173
|
ctx.on("session:changed", async () => {
|
|
165
174
|
const enabled = readPersistedPlanMode();
|
|
166
175
|
applyPlanModeState(enabled, enabled);
|
|
176
|
+
executingPlan = !enabled && readPersistedExecuting();
|
|
167
177
|
writeExecutionState(readExecutionState());
|
|
168
178
|
});
|
|
169
179
|
ctx.on("turn:end", async (event) => {
|
|
170
180
|
const state = readExecutionState();
|
|
171
|
-
if (!
|
|
181
|
+
if (!executingPlan || state.steps.length === 0) return;
|
|
172
182
|
const message = event.message;
|
|
173
183
|
if (message?.role !== "assistant" || !Array.isArray(message.content)) return;
|
|
174
184
|
const text = message.content.filter((item) => item.type === "text").map((item) => item.text).join("\n");
|
|
@@ -180,13 +190,13 @@ Command: ${command}`,
|
|
|
180
190
|
});
|
|
181
191
|
ctx.on("agent:end", async (event) => {
|
|
182
192
|
const state = readExecutionState();
|
|
183
|
-
if (
|
|
193
|
+
if (executingPlan && state.steps.length > 0) {
|
|
184
194
|
if (state.steps.every((step) => step.status === "completed")) {
|
|
185
195
|
const completedList = state.steps.map((step) => `\u2713 ${step.title}`).join("\n");
|
|
186
196
|
ctx.ui.notify(`**Plan Complete!** \u2713
|
|
187
197
|
|
|
188
198
|
${completedList}`, "info");
|
|
189
|
-
|
|
199
|
+
setExecutingPlan(false);
|
|
190
200
|
writeExecutionState(null);
|
|
191
201
|
}
|
|
192
202
|
return;
|
|
@@ -197,7 +207,9 @@ ${completedList}`, "info");
|
|
|
197
207
|
(message) => message.role === "assistant" && Array.isArray(message.content)
|
|
198
208
|
);
|
|
199
209
|
if (!lastAssistant) return;
|
|
200
|
-
const
|
|
210
|
+
const content = lastAssistant.content;
|
|
211
|
+
if (!Array.isArray(content)) return;
|
|
212
|
+
const text = content.filter((item) => item.type === "text").map((item) => item.text).join("\n");
|
|
201
213
|
const extracted = extractPlanTodoItems(text);
|
|
202
214
|
if (extracted.length === 0) return;
|
|
203
215
|
const extractedState = writeExecutionState(createExecutionStateFromPlan(extracted));
|
|
@@ -211,7 +223,7 @@ ${todoListText}`, "info");
|
|
|
211
223
|
]);
|
|
212
224
|
if (choice?.startsWith("Execute")) {
|
|
213
225
|
planModeEnabled = false;
|
|
214
|
-
|
|
226
|
+
setExecutingPlan(extractedState.steps.length > 0);
|
|
215
227
|
ctx.setSessionMetadata(SESSION_METADATA_KEY, false);
|
|
216
228
|
ctx.setPlanMode(false);
|
|
217
229
|
updateStatus();
|
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
import { createRequire as __createRequire } from "node:module"; const require = __createRequire(import.meta.url);
|
|
2
2
|
import {
|
|
3
3
|
createCliHostCapabilities
|
|
4
|
-
} from "../../../chunks/chunk-
|
|
4
|
+
} from "../../../chunks/chunk-XBWFDIAU.js";
|
|
5
5
|
import {
|
|
6
6
|
createSharedFileSystemTools,
|
|
7
7
|
createSubagentCapability
|
|
8
|
-
} from "../../../chunks/chunk-
|
|
8
|
+
} from "../../../chunks/chunk-DZS34ACX.js";
|
|
9
9
|
import "../../../chunks/chunk-FKAO25XH.js";
|
|
10
10
|
import "../../../chunks/chunk-K6ZOCJ4V.js";
|
|
11
11
|
import "../../../chunks/chunk-LZQXQOPJ.js";
|
|
@@ -13,40 +13,24 @@ import "../../../chunks/chunk-LZQXQOPJ.js";
|
|
|
13
13
|
// builtin/extensions/subagent/index.ts
|
|
14
14
|
import * as os from "os";
|
|
15
15
|
import * as path from "path";
|
|
16
|
-
import * as fs from "fs";
|
|
17
|
-
function debugLog(module, message, data) {
|
|
18
|
-
try {
|
|
19
|
-
const homeDir = process.env.HOME || process.env.USERPROFILE || ".";
|
|
20
|
-
const logDir = path.join(homeDir, ".pie", "sessions");
|
|
21
|
-
const files = fs.readdirSync(logDir).filter((f) => f.endsWith(".log")).sort().reverse();
|
|
22
|
-
if (files.length > 0) {
|
|
23
|
-
const logPath = path.join(logDir, files[0]);
|
|
24
|
-
const entry = `[${(/* @__PURE__ */ new Date()).toISOString()}] [subagent:${module}] ${message}${data ? " " + JSON.stringify(data) : ""}
|
|
25
|
-
`;
|
|
26
|
-
fs.appendFileSync(logPath, entry);
|
|
27
|
-
}
|
|
28
|
-
} catch {
|
|
29
|
-
}
|
|
30
|
-
}
|
|
31
16
|
function subagentExtension(ctx) {
|
|
32
17
|
ctx.log("Subagent extension loaded (transparent parallel mode)");
|
|
33
|
-
const apiKey = ctx.apiKey || process.env.KIMI_API_KEY || process.env.OPENAI_API_KEY;
|
|
34
18
|
let model = ctx.model;
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
contextWindow: 2e5,
|
|
43
|
-
maxTokens: 8192
|
|
44
|
-
};
|
|
19
|
+
let apiKey = ctx.apiKey;
|
|
20
|
+
if (!model || !apiKey) {
|
|
21
|
+
const fallback = ctx.resolveModelClass?.("balanced");
|
|
22
|
+
if (fallback?.model && fallback?.apiKey) {
|
|
23
|
+
model = fallback.model;
|
|
24
|
+
apiKey = fallback.apiKey;
|
|
25
|
+
}
|
|
45
26
|
}
|
|
46
27
|
const skills = ctx.skills || [];
|
|
47
28
|
if (!apiKey) {
|
|
48
29
|
ctx.log("Warning: No API key available for subagent.");
|
|
49
30
|
}
|
|
31
|
+
if (!model) {
|
|
32
|
+
ctx.log("Warning: No model configured for subagent; tasks will fail until a model is available.");
|
|
33
|
+
}
|
|
50
34
|
const capability = createSubagentCapability({
|
|
51
35
|
apiKey,
|
|
52
36
|
model,
|
|
@@ -54,11 +38,11 @@ function subagentExtension(ctx) {
|
|
|
54
38
|
yoloMode: ctx.yoloMode,
|
|
55
39
|
getParentTools: () => ctx.getAllTools(),
|
|
56
40
|
log: (message) => ctx.log(message),
|
|
57
|
-
notify: ctx.hasUI ? (message, type) => ctx.ui.notify(message, type) : void 0,
|
|
58
|
-
debugLog,
|
|
41
|
+
notify: ctx.hasUI ? (message, type) => ctx.ui.notify(message, type === "success" ? "info" : type) : void 0,
|
|
42
|
+
debugLog: (module, message, data) => ctx.log(`[subagent:${module}] ${message}${data !== void 0 ? ` ${JSON.stringify(data)}` : ""}`),
|
|
59
43
|
resolveModelClass: (modelClass) => ctx.resolveModelClass?.(modelClass),
|
|
60
44
|
createRuntimeTools: () => {
|
|
61
|
-
const parentRuntimeTools = ctx.getRuntimeTools
|
|
45
|
+
const parentRuntimeTools = ctx.getRuntimeTools();
|
|
62
46
|
if (Array.isArray(parentRuntimeTools) && parentRuntimeTools.length > 0) {
|
|
63
47
|
return parentRuntimeTools.filter((tool) => tool?.name !== "spawn_subagents_parallel");
|
|
64
48
|
}
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { createRequire as __createRequire } from "node:module"; const require = __createRequire(import.meta.url);
|
|
2
2
|
import {
|
|
3
|
+
MANAGE_TODO_LIST_TOOL_DESCRIPTION,
|
|
3
4
|
ManageTodoListParamsSchema,
|
|
4
5
|
buildExecutionReminder,
|
|
5
6
|
createExecutionStateFromTodos,
|
|
@@ -7,18 +8,13 @@ import {
|
|
|
7
8
|
executeManageTodoList,
|
|
8
9
|
executionStateToTodos,
|
|
9
10
|
restoreExecutionState
|
|
10
|
-
} from "../../../chunks/chunk-
|
|
11
|
+
} from "../../../chunks/chunk-DZS34ACX.js";
|
|
11
12
|
import "../../../chunks/chunk-FKAO25XH.js";
|
|
12
13
|
import "../../../chunks/chunk-K6ZOCJ4V.js";
|
|
13
14
|
import "../../../chunks/chunk-LZQXQOPJ.js";
|
|
14
15
|
|
|
15
16
|
// builtin/extensions/todo/index.ts
|
|
16
17
|
var EXECUTION_METADATA_KEY = "executionState";
|
|
17
|
-
var TOOL_DESCRIPTION = `Manage a session-local todo list. Use when the user explicitly asks to use the todo tool, create/manage todos, track tasks, follow a todo list, show tasks, or mark a task done.
|
|
18
|
-
NEVER use for: internal planning, tracking your own progress, remembering subtasks, or creating checklists for yourself.
|
|
19
|
-
Use action=create with items to create a list; the tool starts item 1 automatically. Use action=complete_current after finishing the current item; the tool advances to the next item or clears the list after the final item.
|
|
20
|
-
When a todo list exists, keep executing from the current in-progress item until it is cleared. If the work cannot continue, explain the failure and use action=clear with a reason.
|
|
21
|
-
Actions: create, read, complete_current, clear.`;
|
|
22
18
|
function persistExecutionState(ctx, state) {
|
|
23
19
|
const nextState = state ?? restoreExecutionState(null);
|
|
24
20
|
ctx.setSessionMetadata(EXECUTION_METADATA_KEY, nextState);
|
|
@@ -27,7 +23,7 @@ function persistExecutionState(ctx, state) {
|
|
|
27
23
|
return nextState;
|
|
28
24
|
}
|
|
29
25
|
function notifyWidget(ctx, state) {
|
|
30
|
-
if (!ctx.hasUI
|
|
26
|
+
if (!ctx.hasUI) return;
|
|
31
27
|
const steps = createTodoWidgetView(state);
|
|
32
28
|
if (steps.length === 0) return;
|
|
33
29
|
const completed = steps.filter((step) => step.status === "completed").length;
|
|
@@ -37,7 +33,7 @@ function todoExtension(ctx) {
|
|
|
37
33
|
ctx.registerTool({
|
|
38
34
|
name: "manage_todo_list",
|
|
39
35
|
label: "manage_todo_list",
|
|
40
|
-
description:
|
|
36
|
+
description: MANAGE_TODO_LIST_TOOL_DESCRIPTION,
|
|
41
37
|
parameters: ManageTodoListParamsSchema,
|
|
42
38
|
async execute(args) {
|
|
43
39
|
const currentTodos = executionStateToTodos(ctx.getExecutionState());
|
|
@@ -11,7 +11,7 @@ At a high level, the process of creating a skill goes like this:
|
|
|
11
11
|
|
|
12
12
|
- Decide what you want the skill to do and roughly how it should do it
|
|
13
13
|
- Write a draft of the skill
|
|
14
|
-
- Create a few test prompts and run
|
|
14
|
+
- Create a few test prompts and run the agent with the skill loaded on them
|
|
15
15
|
- Help the user evaluate the results both qualitatively and quantitatively
|
|
16
16
|
- While the runs happen in the background, draft some quantitative evals if there aren't any (if there are some, you can either use as is or modify if you feel something needs to change about them). Then explain them to the user (or if they already existed, explain the ones that already exist)
|
|
17
17
|
- Use the `eval-viewer/generate_review.mjs` script to show the user the results for them to look at, and also let them look at the quantitative metrics
|
|
@@ -31,7 +31,7 @@ Cool? Cool.
|
|
|
31
31
|
|
|
32
32
|
## Communicating with the user
|
|
33
33
|
|
|
34
|
-
The skill creator is liable to be used by people across a wide range of familiarity with coding jargon. If you haven't heard (and how could you, it's only very recently that it started), there's a trend now where the power of
|
|
34
|
+
The skill creator is liable to be used by people across a wide range of familiarity with coding jargon. If you haven't heard (and how could you, it's only very recently that it started), there's a trend now where the power of AI coding agents is inspiring plumbers to open up their terminals, parents and grandparents to google "how to install npm". On the other hand, the bulk of users are probably fairly computer-literate.
|
|
35
35
|
|
|
36
36
|
So please pay attention to context cues to understand how to phrase your communication! In the default case, just to give you some idea:
|
|
37
37
|
|
|
@@ -48,7 +48,7 @@ It's OK to briefly explain terms if you're in doubt, and feel free to clarify te
|
|
|
48
48
|
|
|
49
49
|
Start by understanding the user's intent. The current conversation might already contain a workflow the user wants to capture (e.g., they say "turn this into a skill"). If so, extract answers from the conversation history first — the tools used, the sequence of steps, corrections the user made, input/output formats observed. The user may need to fill the gaps, and should confirm before proceeding to the next step.
|
|
50
50
|
|
|
51
|
-
1. What should this skill enable
|
|
51
|
+
1. What should this skill enable the agent to do?
|
|
52
52
|
2. When should this skill trigger? (what user phrases/contexts)
|
|
53
53
|
3. What's the expected output format?
|
|
54
54
|
4. Should we set up test cases to verify the skill works? Skills with objectively verifiable outputs (file transforms, data extraction, code generation, fixed workflow steps) benefit from test cases. Skills with subjective outputs (writing style, art) often don't need them. Suggest the appropriate default based on the skill type, but let the user decide.
|
|
@@ -64,7 +64,7 @@ Check available MCPs - if useful for research (searching docs, finding similar s
|
|
|
64
64
|
Based on the user interview, fill in these components:
|
|
65
65
|
|
|
66
66
|
- **name**: Skill identifier
|
|
67
|
-
- **description**: When to trigger, what it does. This is the primary triggering mechanism - include both what the skill does AND specific contexts for when to use it. All "when to use" info goes here, not in the body. Note: currently
|
|
67
|
+
- **description**: When to trigger, what it does. This is the primary triggering mechanism - include both what the skill does AND specific contexts for when to use it. All "when to use" info goes here, not in the body. Note: models currently tend to "undertrigger" skills -- to not use them when they'd be useful. To combat this, please make the skill descriptions a little bit "pushy". So for instance, instead of "How to build a simple fast dashboard to display internal company data.", you might write "How to build a simple fast dashboard to display internal company data. Make sure to use this skill whenever the user mentions dashboards, data visualization, internal metrics, or wants to display any kind of company data, even if they don't explicitly ask for a 'dashboard.'"
|
|
68
68
|
- **compatibility**: Required tools, dependencies (optional, rarely needed)
|
|
69
69
|
- **the rest of the skill :)**
|
|
70
70
|
|
|
@@ -106,7 +106,7 @@ cloud-deploy/
|
|
|
106
106
|
├── gcp.md
|
|
107
107
|
└── azure.md
|
|
108
108
|
```
|
|
109
|
-
|
|
109
|
+
The agent reads only the relevant reference file.
|
|
110
110
|
|
|
111
111
|
#### Principle of Lack of Surprise
|
|
112
112
|
|
|
@@ -244,7 +244,7 @@ Put each with_skill version before its baseline counterpart.
|
|
|
244
244
|
```
|
|
245
245
|
For iteration 2+, also pass `--previous-workspace <workspace>/iteration-<N-1>`.
|
|
246
246
|
|
|
247
|
-
**
|
|
247
|
+
**Headless environments:** If `webbrowser.open()` is not available or the environment has no display, use `--static <output_path>` to write a standalone HTML file instead of starting a server. Feedback will be downloaded as a `feedback.json` file when the user clicks "Submit All Reviews". After download, copy `feedback.json` into the workspace directory for the next iteration to pick up.
|
|
248
248
|
|
|
249
249
|
Note: please use `generate_review.mjs` to create the viewer; there's no need to write custom HTML.
|
|
250
250
|
|
|
@@ -345,7 +345,7 @@ Create 20 eval queries — a mix of should-trigger and should-not-trigger. Save
|
|
|
345
345
|
]
|
|
346
346
|
```
|
|
347
347
|
|
|
348
|
-
The queries must be realistic and something a
|
|
348
|
+
The queries must be realistic and something a real user of a coding agent would actually type. Not abstract requests, but requests that are concrete and specific and have a good amount of detail. For instance, file paths, personal context about the user's job or situation, column names and values, company names, URLs. A little bit of backstory. Some might be in lowercase or contain abbreviations or typos or casual speech. Use a mix of different lengths, and focus on edge cases rather than making them clear-cut (the user will get a chance to sign off on them).
|
|
349
349
|
|
|
350
350
|
Bad: `"Format this data"`, `"Extract text from PDF"`, `"Create a chart"`
|
|
351
351
|
|
|
@@ -397,7 +397,7 @@ This handles the full optimization loop automatically. It splits the eval set in
|
|
|
397
397
|
|
|
398
398
|
Understanding the triggering mechanism helps design better eval queries. Skills appear in Pie's `available_skills` list with their name + description, and Pie decides whether to consult a skill based on that description. The important thing to know is that Pie only consults skills for tasks it can't easily handle on its own — simple, one-step queries like "read this PDF" may not trigger a skill even if the description matches perfectly, because Pie can handle them directly with basic tools. Complex, multi-step, or specialized queries reliably trigger skills when the description matches.
|
|
399
399
|
|
|
400
|
-
This means your eval queries should be substantive enough that
|
|
400
|
+
This means your eval queries should be substantive enough that the model would actually benefit from consulting a skill. Simple queries like "read file X" are poor test cases — they won't trigger skills regardless of description quality.
|
|
401
401
|
|
|
402
402
|
### Step 4: Apply the result
|
|
403
403
|
|
|
@@ -405,9 +405,9 @@ Take `best_description` from the JSON output and update the skill's SKILL.md fro
|
|
|
405
405
|
|
|
406
406
|
---
|
|
407
407
|
|
|
408
|
-
### Package
|
|
408
|
+
### Package the skill
|
|
409
409
|
|
|
410
|
-
|
|
410
|
+
Package the skill and point the user to the resulting file:
|
|
411
411
|
|
|
412
412
|
```bash
|
|
413
413
|
node <skill-creator-path>/scripts/package_skill.mjs <path/to/skill-folder>
|
|
@@ -417,42 +417,14 @@ After packaging, direct the user to the resulting `.skill` file path so they can
|
|
|
417
417
|
|
|
418
418
|
---
|
|
419
419
|
|
|
420
|
-
##
|
|
420
|
+
## Updating an existing skill
|
|
421
421
|
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
**Running test cases**: No subagents means no parallel execution. For each test case, read the skill's SKILL.md, then follow its instructions to accomplish the test prompt yourself. Do them one at a time. This is less rigorous than independent subagents (you wrote the skill and you're also running it, so you have full context), but it's a useful sanity check — and the human review step compensates. Skip the baseline runs — just use the skill to complete the task as requested.
|
|
425
|
-
|
|
426
|
-
**Reviewing results**: If you can't open a browser (e.g., Claude.ai's VM has no display, or you're on a remote server), skip the browser reviewer entirely. Instead, present results directly in the conversation. For each test case, show the prompt and the output. If the output is a file the user needs to see (like a .docx or .xlsx), save it to the filesystem and tell them where it is so they can download and inspect it. Ask for feedback inline: "How does this look? Anything you'd change?"
|
|
427
|
-
|
|
428
|
-
**Benchmarking**: Skip the quantitative benchmarking — it relies on baseline comparisons which aren't meaningful without subagents. Focus on qualitative feedback from the user.
|
|
429
|
-
|
|
430
|
-
**The iteration loop**: Same as before — improve the skill, rerun the test cases, ask for feedback — just without the browser reviewer in the middle. You can still organize results into iteration directories on the filesystem if you have one.
|
|
431
|
-
|
|
432
|
-
**Description optimization**: This section now defaults to Pie's own non-interactive runner. It does not require `claude -p`.
|
|
433
|
-
|
|
434
|
-
**Blind comparison**: Requires subagents. Skip it.
|
|
435
|
-
|
|
436
|
-
**Packaging**: The `package_skill.mjs` script works anywhere with Node.js and a filesystem. On Claude.ai or Codex CLI, you can run it and the user can download the resulting `.skill` file.
|
|
437
|
-
|
|
438
|
-
**Updating an existing skill**: The user might be asking you to update an existing skill, not create a new one. In this case:
|
|
422
|
+
The user might be asking you to update an existing skill, not create a new one. In this case:
|
|
439
423
|
- **Preserve the original name.** Note the skill's directory name and `name` frontmatter field -- use them unchanged. E.g., if the installed skill is `research-helper`, output `research-helper.skill` (not `research-helper-v2`).
|
|
440
424
|
- **Copy to a writeable location before editing.** The installed skill path may be read-only. Copy to `/tmp/skill-name/`, edit there, and package from the copy.
|
|
441
425
|
- **If packaging manually, stage in `/tmp/` first**, then copy to the output directory -- direct writes may fail due to permissions.
|
|
442
426
|
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
## Cowork-Specific Instructions
|
|
446
|
-
|
|
447
|
-
If you're in Cowork, the main things to know are:
|
|
448
|
-
|
|
449
|
-
- You have subagents, so the main workflow (spawn test cases in parallel, run baselines, grade, etc.) all works. (However, if you run into severe problems with timeouts, it's OK to run the test prompts in series rather than parallel.)
|
|
450
|
-
- You don't have a browser or display, so when generating the eval viewer, use `--static <output_path>` to write a standalone HTML file instead of starting a server. Then proffer a link that the user can click to open the HTML in their browser.
|
|
451
|
-
- For whatever reason, the Cowork setup seems to disincline Claude from generating the eval viewer after running the tests, so just to reiterate: whether you're in Cowork or in Claude Code, after running tests, you should always generate the eval viewer for the human to look at examples before revising the skill yourself and trying to make corrections, using `generate_review.mjs` (not writing your own boutique html code). Sorry in advance but I'm gonna go all caps here: GENERATE THE EVAL VIEWER *BEFORE* evaluating inputs yourself. You want to get them in front of the human ASAP!
|
|
452
|
-
- Feedback works differently: since there's no running server, the viewer's "Submit All Reviews" button will download `feedback.json` as a file. You can then read it from there (you may have to request access first).
|
|
453
|
-
- Packaging works — `package_skill.mjs` just needs Node.js and a filesystem.
|
|
454
|
-
- Description optimization (`run_loop.mjs` / `run_eval.mjs`) should work in Cowork just fine since it uses Pie's non-interactive runner, not a browser, but please save it until you've fully finished making the skill and the user agrees it's in good shape.
|
|
455
|
-
- **Updating an existing skill**: The user might be asking you to update an existing skill, not create a new one. Follow the update guidance in the claude.ai section above.
|
|
427
|
+
Note: description optimization (`run_loop.mjs` / `run_eval.mjs`) uses Pie's own non-interactive runner and works headless, but save it until you've fully finished making the skill and the user agrees it's in good shape.
|
|
456
428
|
|
|
457
429
|
---
|
|
458
430
|
|
|
@@ -473,13 +445,13 @@ Repeating one more time the core loop here for emphasis:
|
|
|
473
445
|
|
|
474
446
|
- Figure out what the skill is about
|
|
475
447
|
- Draft or edit the skill
|
|
476
|
-
- Run
|
|
448
|
+
- Run test prompts with the skill loaded
|
|
477
449
|
- With the user, evaluate the outputs:
|
|
478
450
|
- Create benchmark.json and run `eval-viewer/generate_review.mjs` to help the user review them
|
|
479
451
|
- Run quantitative evals
|
|
480
452
|
- Repeat until you and the user are satisfied
|
|
481
453
|
- Package the final skill and return it to the user.
|
|
482
454
|
|
|
483
|
-
Please add steps to your TodoList, if you have such a thing, to make sure you don't forget.
|
|
455
|
+
Please add steps to your TodoList, if you have such a thing, to make sure you don't forget. Specifically put "Create evals JSON and run `eval-viewer/generate_review.mjs` so human can review test cases" in your TodoList to make sure it happens.
|
|
484
456
|
|
|
485
457
|
Good luck!
|
|
@@ -6,7 +6,7 @@ import { readFile, readdir, stat, writeFile } from "node:fs/promises";
|
|
|
6
6
|
import { fileURLToPath } from "node:url";
|
|
7
7
|
import { execFile } from "node:child_process";
|
|
8
8
|
import { promisify } from "node:util";
|
|
9
|
-
import { openInBrowser } from "../scripts/
|
|
9
|
+
import { openInBrowser } from "../scripts/pie_runner.mjs";
|
|
10
10
|
|
|
11
11
|
const execFileAsync = promisify(execFile);
|
|
12
12
|
const METADATA_FILES = new Set(["transcript.md", "user_notes.md", "metrics.json"]);
|