@cydm/pie 1.0.49 → 1.0.51

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. package/README.md +1 -1
  2. package/dist/builtin/extensions/ask-user/index.js +4 -11
  3. package/dist/builtin/extensions/ask-user/package.json +0 -7
  4. package/dist/builtin/extensions/changelog/index.js +15 -20
  5. package/dist/builtin/extensions/document-attachments/index.js +51 -1
  6. package/dist/builtin/extensions/document-attachments/package.json +1 -1
  7. package/dist/builtin/extensions/init/index.js +0 -13
  8. package/dist/builtin/extensions/plan-mode/index.js +31 -19
  9. package/dist/builtin/extensions/subagent/index.js +19 -40
  10. package/dist/builtin/extensions/todo/index.js +6 -10
  11. package/dist/builtin/skills/skill-creator/SKILL.md +15 -43
  12. package/dist/builtin/skills/skill-creator/eval-viewer/generate_review.mjs +1 -1
  13. package/dist/builtin/skills/skill-creator/eval-viewer/viewer.html +2 -2
  14. package/dist/builtin/skills/skill-creator/references/schemas.md +1 -1
  15. package/dist/builtin/skills/skill-creator/scripts/generate_report.mjs +1 -1
  16. package/dist/builtin/skills/skill-creator/scripts/improve_description.mjs +3 -3
  17. package/dist/builtin/skills/skill-creator/scripts/pie_runner.mjs +32 -4
  18. package/dist/builtin/skills/skill-creator/scripts/run_loop.mjs +1 -1
  19. package/dist/chunks/{chunk-7RYOIXA2.js → chunk-A644H2XX.js} +104 -20
  20. package/dist/chunks/{chunk-FKAO25XH.js → chunk-B3SMW2XW.js} +1 -1
  21. package/dist/chunks/{chunk-J6PEH6ST.js → chunk-KDPAGHYY.js} +2 -2
  22. package/dist/chunks/{chunk-K6ZOCJ4V.js → chunk-S2J3SZZX.js} +44 -1
  23. package/dist/chunks/{src-HYZFAW4V.js → src-UKCIAX6F.js} +2 -2
  24. package/dist/chunks/{test-stream-5T7E7C25.js → test-stream-NCBFPM2I.js} +1 -1
  25. package/dist/cli.js +49 -22
  26. package/package.json +3 -3
  27. package/dist/builtin/extensions/deploy/index.js +0 -11
  28. package/dist/builtin/extensions/deploy/package.json +0 -11
  29. package/dist/builtin/extensions/files/index.js +0 -10
  30. package/dist/builtin/extensions/files/package.json +0 -14
  31. package/dist/builtin/skills/skill-creator/scripts/claude_cli.mjs +0 -115
@@ -11,7 +11,7 @@ At a high level, the process of creating a skill goes like this:
11
11
 
12
12
  - Decide what you want the skill to do and roughly how it should do it
13
13
  - Write a draft of the skill
14
- - Create a few test prompts and run claude-with-access-to-the-skill on them
14
+ - Create a few test prompts and run the agent with the skill loaded on them
15
15
  - Help the user evaluate the results both qualitatively and quantitatively
16
16
  - While the runs happen in the background, draft some quantitative evals if there aren't any (if there are some, you can either use as is or modify if you feel something needs to change about them). Then explain them to the user (or if they already existed, explain the ones that already exist)
17
17
  - Use the `eval-viewer/generate_review.mjs` script to show the user the results for them to look at, and also let them look at the quantitative metrics
@@ -31,7 +31,7 @@ Cool? Cool.
31
31
 
32
32
  ## Communicating with the user
33
33
 
34
- The skill creator is liable to be used by people across a wide range of familiarity with coding jargon. If you haven't heard (and how could you, it's only very recently that it started), there's a trend now where the power of Claude is inspiring plumbers to open up their terminals, parents and grandparents to google "how to install npm". On the other hand, the bulk of users are probably fairly computer-literate.
34
+ The skill creator is liable to be used by people across a wide range of familiarity with coding jargon. If you haven't heard (and how could you, it's only very recently that it started), there's a trend now where the power of AI coding agents is inspiring plumbers to open up their terminals, parents and grandparents to google "how to install npm". On the other hand, the bulk of users are probably fairly computer-literate.
35
35
 
36
36
  So please pay attention to context cues to understand how to phrase your communication! In the default case, just to give you some idea:
37
37
 
@@ -48,7 +48,7 @@ It's OK to briefly explain terms if you're in doubt, and feel free to clarify te
48
48
 
49
49
  Start by understanding the user's intent. The current conversation might already contain a workflow the user wants to capture (e.g., they say "turn this into a skill"). If so, extract answers from the conversation history first — the tools used, the sequence of steps, corrections the user made, input/output formats observed. The user may need to fill the gaps, and should confirm before proceeding to the next step.
50
50
 
51
- 1. What should this skill enable Claude to do?
51
+ 1. What should this skill enable the agent to do?
52
52
  2. When should this skill trigger? (what user phrases/contexts)
53
53
  3. What's the expected output format?
54
54
  4. Should we set up test cases to verify the skill works? Skills with objectively verifiable outputs (file transforms, data extraction, code generation, fixed workflow steps) benefit from test cases. Skills with subjective outputs (writing style, art) often don't need them. Suggest the appropriate default based on the skill type, but let the user decide.
@@ -64,7 +64,7 @@ Check available MCPs - if useful for research (searching docs, finding similar s
64
64
  Based on the user interview, fill in these components:
65
65
 
66
66
  - **name**: Skill identifier
67
- - **description**: When to trigger, what it does. This is the primary triggering mechanism - include both what the skill does AND specific contexts for when to use it. All "when to use" info goes here, not in the body. Note: currently Claude has a tendency to "undertrigger" skills -- to not use them when they'd be useful. To combat this, please make the skill descriptions a little bit "pushy". So for instance, instead of "How to build a simple fast dashboard to display internal Anthropic data.", you might write "How to build a simple fast dashboard to display internal Anthropic data. Make sure to use this skill whenever the user mentions dashboards, data visualization, internal metrics, or wants to display any kind of company data, even if they don't explicitly ask for a 'dashboard.'"
67
+ - **description**: When to trigger, what it does. This is the primary triggering mechanism - include both what the skill does AND specific contexts for when to use it. All "when to use" info goes here, not in the body. Note: models currently tend to "undertrigger" skills -- to not use them when they'd be useful. To combat this, please make the skill descriptions a little bit "pushy". So for instance, instead of "How to build a simple fast dashboard to display internal company data.", you might write "How to build a simple fast dashboard to display internal company data. Make sure to use this skill whenever the user mentions dashboards, data visualization, internal metrics, or wants to display any kind of company data, even if they don't explicitly ask for a 'dashboard.'"
68
68
  - **compatibility**: Required tools, dependencies (optional, rarely needed)
69
69
  - **the rest of the skill :)**
70
70
 
@@ -106,7 +106,7 @@ cloud-deploy/
106
106
  ├── gcp.md
107
107
  └── azure.md
108
108
  ```
109
- Claude reads only the relevant reference file.
109
+ The agent reads only the relevant reference file.
110
110
 
111
111
  #### Principle of Lack of Surprise
112
112
 
@@ -244,7 +244,7 @@ Put each with_skill version before its baseline counterpart.
244
244
  ```
245
245
  For iteration 2+, also pass `--previous-workspace <workspace>/iteration-<N-1>`.
246
246
 
247
- **Cowork / headless environments:** If `webbrowser.open()` is not available or the environment has no display, use `--static <output_path>` to write a standalone HTML file instead of starting a server. Feedback will be downloaded as a `feedback.json` file when the user clicks "Submit All Reviews". After download, copy `feedback.json` into the workspace directory for the next iteration to pick up.
247
+ **Headless environments:** If `webbrowser.open()` is not available or the environment has no display, use `--static <output_path>` to write a standalone HTML file instead of starting a server. Feedback will be downloaded as a `feedback.json` file when the user clicks "Submit All Reviews". After download, copy `feedback.json` into the workspace directory for the next iteration to pick up.
248
248
 
249
249
  Note: please use `generate_review.mjs` to create the viewer; there's no need to write custom HTML.
250
250
 
@@ -345,7 +345,7 @@ Create 20 eval queries — a mix of should-trigger and should-not-trigger. Save
345
345
  ]
346
346
  ```
347
347
 
348
- The queries must be realistic and something a Claude Code or Claude.ai user would actually type. Not abstract requests, but requests that are concrete and specific and have a good amount of detail. For instance, file paths, personal context about the user's job or situation, column names and values, company names, URLs. A little bit of backstory. Some might be in lowercase or contain abbreviations or typos or casual speech. Use a mix of different lengths, and focus on edge cases rather than making them clear-cut (the user will get a chance to sign off on them).
348
+ The queries must be realistic and something a real user of a coding agent would actually type. Not abstract requests, but requests that are concrete and specific and have a good amount of detail. For instance, file paths, personal context about the user's job or situation, column names and values, company names, URLs. A little bit of backstory. Some might be in lowercase or contain abbreviations or typos or casual speech. Use a mix of different lengths, and focus on edge cases rather than making them clear-cut (the user will get a chance to sign off on them).
349
349
 
350
350
  Bad: `"Format this data"`, `"Extract text from PDF"`, `"Create a chart"`
351
351
 
@@ -397,7 +397,7 @@ This handles the full optimization loop automatically. It splits the eval set in
397
397
 
398
398
  Understanding the triggering mechanism helps design better eval queries. Skills appear in Pie's `available_skills` list with their name + description, and Pie decides whether to consult a skill based on that description. The important thing to know is that Pie only consults skills for tasks it can't easily handle on its own — simple, one-step queries like "read this PDF" may not trigger a skill even if the description matches perfectly, because Pie can handle them directly with basic tools. Complex, multi-step, or specialized queries reliably trigger skills when the description matches.
399
399
 
400
- This means your eval queries should be substantive enough that Claude would actually benefit from consulting a skill. Simple queries like "read file X" are poor test cases — they won't trigger skills regardless of description quality.
400
+ This means your eval queries should be substantive enough that the model would actually benefit from consulting a skill. Simple queries like "read file X" are poor test cases — they won't trigger skills regardless of description quality.
401
401
 
402
402
  ### Step 4: Apply the result
403
403
 
@@ -405,9 +405,9 @@ Take `best_description` from the JSON output and update the skill's SKILL.md fro
405
405
 
406
406
  ---
407
407
 
408
- ### Package and Present (only if `present_files` tool is available)
408
+ ### Package the skill
409
409
 
410
- Check whether you have access to the `present_files` tool. If you don't, skip this step. If you do, package the skill and present the .skill file to the user:
410
+ Package the skill and point the user to the resulting file:
411
411
 
412
412
  ```bash
413
413
  node <skill-creator-path>/scripts/package_skill.mjs <path/to/skill-folder>
@@ -417,42 +417,14 @@ After packaging, direct the user to the resulting `.skill` file path so they can
417
417
 
418
418
  ---
419
419
 
420
- ## Claude.ai-specific instructions
420
+ ## Updating an existing skill
421
421
 
422
- In Claude.ai, the core workflow is the same (draft → test → review → improve → repeat), but because Claude.ai doesn't have subagents, some mechanics change. Here's what to adapt:
423
-
424
- **Running test cases**: No subagents means no parallel execution. For each test case, read the skill's SKILL.md, then follow its instructions to accomplish the test prompt yourself. Do them one at a time. This is less rigorous than independent subagents (you wrote the skill and you're also running it, so you have full context), but it's a useful sanity check — and the human review step compensates. Skip the baseline runs — just use the skill to complete the task as requested.
425
-
426
- **Reviewing results**: If you can't open a browser (e.g., Claude.ai's VM has no display, or you're on a remote server), skip the browser reviewer entirely. Instead, present results directly in the conversation. For each test case, show the prompt and the output. If the output is a file the user needs to see (like a .docx or .xlsx), save it to the filesystem and tell them where it is so they can download and inspect it. Ask for feedback inline: "How does this look? Anything you'd change?"
427
-
428
- **Benchmarking**: Skip the quantitative benchmarking — it relies on baseline comparisons which aren't meaningful without subagents. Focus on qualitative feedback from the user.
429
-
430
- **The iteration loop**: Same as before — improve the skill, rerun the test cases, ask for feedback — just without the browser reviewer in the middle. You can still organize results into iteration directories on the filesystem if you have one.
431
-
432
- **Description optimization**: This section now defaults to Pie's own non-interactive runner. It does not require `claude -p`.
433
-
434
- **Blind comparison**: Requires subagents. Skip it.
435
-
436
- **Packaging**: The `package_skill.mjs` script works anywhere with Node.js and a filesystem. On Claude.ai or Codex CLI, you can run it and the user can download the resulting `.skill` file.
437
-
438
- **Updating an existing skill**: The user might be asking you to update an existing skill, not create a new one. In this case:
422
+ The user might be asking you to update an existing skill, not create a new one. In this case:
439
423
  - **Preserve the original name.** Note the skill's directory name and `name` frontmatter field -- use them unchanged. E.g., if the installed skill is `research-helper`, output `research-helper.skill` (not `research-helper-v2`).
440
424
  - **Copy to a writeable location before editing.** The installed skill path may be read-only. Copy to `/tmp/skill-name/`, edit there, and package from the copy.
441
425
  - **If packaging manually, stage in `/tmp/` first**, then copy to the output directory -- direct writes may fail due to permissions.
442
426
 
443
- ---
444
-
445
- ## Cowork-Specific Instructions
446
-
447
- If you're in Cowork, the main things to know are:
448
-
449
- - You have subagents, so the main workflow (spawn test cases in parallel, run baselines, grade, etc.) all works. (However, if you run into severe problems with timeouts, it's OK to run the test prompts in series rather than parallel.)
450
- - You don't have a browser or display, so when generating the eval viewer, use `--static <output_path>` to write a standalone HTML file instead of starting a server. Then proffer a link that the user can click to open the HTML in their browser.
451
- - For whatever reason, the Cowork setup seems to disincline Claude from generating the eval viewer after running the tests, so just to reiterate: whether you're in Cowork or in Claude Code, after running tests, you should always generate the eval viewer for the human to look at examples before revising the skill yourself and trying to make corrections, using `generate_review.mjs` (not writing your own boutique html code). Sorry in advance but I'm gonna go all caps here: GENERATE THE EVAL VIEWER *BEFORE* evaluating inputs yourself. You want to get them in front of the human ASAP!
452
- - Feedback works differently: since there's no running server, the viewer's "Submit All Reviews" button will download `feedback.json` as a file. You can then read it from there (you may have to request access first).
453
- - Packaging works — `package_skill.mjs` just needs Node.js and a filesystem.
454
- - Description optimization (`run_loop.mjs` / `run_eval.mjs`) should work in Cowork just fine since it uses Pie's non-interactive runner, not a browser, but please save it until you've fully finished making the skill and the user agrees it's in good shape.
455
- - **Updating an existing skill**: The user might be asking you to update an existing skill, not create a new one. Follow the update guidance in the claude.ai section above.
427
+ Note: description optimization (`run_loop.mjs` / `run_eval.mjs`) uses Pie's own non-interactive runner and works headless, but save it until you've fully finished making the skill and the user agrees it's in good shape.
456
428
 
457
429
  ---
458
430
 
@@ -473,13 +445,13 @@ Repeating one more time the core loop here for emphasis:
473
445
 
474
446
  - Figure out what the skill is about
475
447
  - Draft or edit the skill
476
- - Run claude-with-access-to-the-skill on test prompts
448
+ - Run test prompts with the skill loaded
477
449
  - With the user, evaluate the outputs:
478
450
  - Create benchmark.json and run `eval-viewer/generate_review.mjs` to help the user review them
479
451
  - Run quantitative evals
480
452
  - Repeat until you and the user are satisfied
481
453
  - Package the final skill and return it to the user.
482
454
 
483
- Please add steps to your TodoList, if you have such a thing, to make sure you don't forget. If you're in Cowork, please specifically put "Create evals JSON and run `eval-viewer/generate_review.mjs` so human can review test cases" in your TodoList to make sure it happens.
455
+ Please add steps to your TodoList, if you have such a thing, to make sure you don't forget. Specifically put "Create evals JSON and run `eval-viewer/generate_review.mjs` so human can review test cases" in your TodoList to make sure it happens.
484
456
 
485
457
  Good luck!
@@ -6,7 +6,7 @@ import { readFile, readdir, stat, writeFile } from "node:fs/promises";
6
6
  import { fileURLToPath } from "node:url";
7
7
  import { execFile } from "node:child_process";
8
8
  import { promisify } from "node:util";
9
- import { openInBrowser } from "../scripts/claude_cli.mjs";
9
+ import { openInBrowser } from "../scripts/pie_runner.mjs";
10
10
 
11
11
  const execFileAsync = promisify(execFile);
12
12
  const METADATA_FILES = new Set(["transcript.md", "user_notes.md", "metrics.json"]);
@@ -545,7 +545,7 @@
545
545
  <div class="header">
546
546
  <div>
547
547
  <h1>Eval Review: <span id="skill-name"></span></h1>
548
- <div class="instructions">Review each output and leave feedback below. Navigate with arrow keys or buttons. When done, copy feedback and paste into Claude Code.</div>
548
+ <div class="instructions">Review each output and leave feedback below. Navigate with arrow keys or buttons. When done, copy feedback and paste it back into your Pie session.</div>
549
549
  </div>
550
550
  <div class="progress" id="progress"></div>
551
551
  </div>
@@ -634,7 +634,7 @@
634
634
  <div class="done-overlay" id="done-overlay">
635
635
  <div class="done-card">
636
636
  <h2>Review Complete</h2>
637
- <p>Your feedback has been saved. Go back to your Claude Code session and tell Claude you're done reviewing.</p>
637
+ <p>Your feedback has been saved. Go back to your Pie session and say you're done reviewing.</p>
638
638
  <div class="btn-row">
639
639
  <button onclick="closeDoneDialog()">OK</button>
640
640
  </div>
@@ -225,7 +225,7 @@ Output from Benchmark mode. Located at `benchmarks/<timestamp>/benchmark.json`.
225
225
  "metadata": {
226
226
  "skill_name": "pdf",
227
227
  "skill_path": "/path/to/pdf",
228
- "executor_model": "claude-sonnet-4-20250514",
228
+ "executor_model": "pie-agent",
229
229
  "analyzer_model": "most-capable-model",
230
230
  "timestamp": "2026-01-15T10:30:00Z",
231
231
  "evals_run": [1, 2, 3],
@@ -75,7 +75,7 @@ ${refreshTag} <title>${titlePrefix}Skill Description Optimization</title>
75
75
  <body>
76
76
  <h1>${titlePrefix}Skill Description Optimization</h1>
77
77
  <div class="explainer">
78
- <strong>Optimizing your skill's description.</strong> This page updates automatically as Claude tests different versions of your skill's description. Each row is an iteration — a new description attempt. The columns show test queries: green checkmarks mean the skill triggered correctly (or correctly didn't trigger), red crosses mean it got it wrong.
78
+ <strong>Optimizing your skill's description.</strong> This page updates automatically as the model tests different versions of your skill's description. Each row is an iteration — a new description attempt. The columns show test queries: green checkmarks mean the skill triggered correctly (or correctly didn't trigger), red crosses mean it got it wrong.
79
79
  </div>
80
80
  <div class="summary">
81
81
  <p><strong>Original:</strong> ${escapeHtml(data?.original_description ?? "N/A")}</p>
@@ -25,9 +25,9 @@ export async function improveDescription({
25
25
  ? `Train: ${trainScore}, Test: ${testResults.summary.passed}/${testResults.summary.total}`
26
26
  : `Train: ${trainScore}`;
27
27
 
28
- let prompt = `You are optimizing a skill description for a Claude Code skill called "${skillName}". A "skill" is sort of like a prompt, but with progressive disclosure -- there's a title and description that Claude sees when deciding whether to use the skill, and then if it does use the skill, it reads the .md file which has lots more details and potentially links to other resources in the skill folder like helper files and scripts and additional documentation or examples.
28
+ let prompt = `You are optimizing a skill description for a Pie skill called "${skillName}". A "skill" is sort of like a prompt, but with progressive disclosure -- there's a title and description that the agent sees when deciding whether to use the skill, and then if it does use the skill, it reads the .md file which has lots more details and potentially links to other resources in the skill folder like helper files and scripts and additional documentation or examples.
29
29
 
30
- The description appears in Claude's "available_skills" list. When a user sends a query, Claude decides whether to invoke the skill based solely on the title and on this description. Your goal is to write a description that triggers for relevant queries, and doesn't trigger for irrelevant ones.
30
+ The description appears in the agent's "available_skills" list. When a user sends a query, the agent decides whether to invoke the skill based solely on the title and on this description. Your goal is to write a description that triggers for relevant queries, and doesn't trigger for irrelevant ones.
31
31
 
32
32
  Here's the current description:
33
33
  <current_description>
@@ -93,7 +93,7 @@ Concretely, your description should not be more than about 100-200 words, even i
93
93
  Here are some tips that we've found to work well in writing these descriptions:
94
94
  - The skill should be phrased in the imperative -- "Use this skill for" rather than "this skill does"
95
95
  - The skill description should focus on the user's intent, what they are trying to achieve, vs. the implementation details of how the skill works.
96
- - The description competes with other skills for Claude's attention — make it distinctive and immediately recognizable.
96
+ - The description competes with other skills for the model's attention — make it distinctive and immediately recognizable.
97
97
  - If you're getting lots of failures after repeated attempts, change things up. Try different sentence structures or wordings.
98
98
 
99
99
  I'd encourage you to be creative and mix up the style in different iterations since you'll have multiple opportunities to try different approaches and we'll just grab the highest-scoring one at the end.
@@ -2,11 +2,11 @@
2
2
 
3
3
  import crypto from "node:crypto";
4
4
  import { spawn } from "node:child_process";
5
+ import fs from "node:fs";
5
6
  import path from "node:path";
6
7
  import { fileURLToPath } from "node:url";
7
8
 
8
9
  const SCRIPT_DIR = path.dirname(fileURLToPath(import.meta.url));
9
- const REPO_ROOT = path.resolve(SCRIPT_DIR, "../../../../../../");
10
10
 
11
11
  export function findProjectRoot(start = process.cwd()) {
12
12
  return path.resolve(start);
@@ -29,13 +29,41 @@ export function openInBrowser(target) {
29
29
  spawn("xdg-open", [target], { detached: true, stdio: "ignore" }).unref();
30
30
  }
31
31
 
32
- function getPieCliPath(cwd = process.cwd()) {
33
- return path.join(REPO_ROOT, "products/cli/dist/cli.js");
32
+ let cachedCliPath;
33
+
34
+ /**
35
+ * Locate the Pie CLI entry point across layouts:
36
+ * - installed package: <pkg>/dist/builtin/skills/skill-creator/scripts -> <pkg>/dist/cli.js
37
+ * - dev repository: <repo>/products/cli/builtin/skills/skill-creator/scripts -> <repo>/products/cli/dist/cli.js
38
+ * - explicit override via PIE_CLI_PATH
39
+ *
40
+ * Candidate order is layout-aware: an installed build always prefers the
41
+ * package root and a dev checkout always prefers products/cli, so an
42
+ * unrelated sibling artifact cannot win over the correct entry.
43
+ */
44
+ export function getPieCliPath() {
45
+ if (cachedCliPath) return cachedCliPath;
46
+ const installed = path.resolve(SCRIPT_DIR, "../../../../../dist/cli.js"); // installed package root
47
+ const dev = path.resolve(SCRIPT_DIR, "../../../../dist/cli.js"); // dev repo: products/cli
48
+ const inInstalledLayout = SCRIPT_DIR.includes(`${path.sep}dist${path.sep}builtin${path.sep}`);
49
+ const candidates = [
50
+ process.env.PIE_CLI_PATH,
51
+ ...(inInstalledLayout ? [installed, dev] : [dev, installed]),
52
+ ].filter((candidate) => typeof candidate === "string" && candidate.length > 0);
53
+ for (const candidate of candidates) {
54
+ if (fs.existsSync(candidate)) {
55
+ cachedCliPath = candidate;
56
+ return cachedCliPath;
57
+ }
58
+ }
59
+ throw new Error(
60
+ `Pie CLI entry not found. Tried: ${candidates.join(", ")}. Set PIE_CLI_PATH to the cli.js entry point.`,
61
+ );
34
62
  }
35
63
 
36
64
  export async function runPieJsonPrompt(prompt, { cwd = process.cwd(), timeout = 300_000, sessionId = null } = {}) {
37
65
  return new Promise((resolve, reject) => {
38
- const args = [getPieCliPath(cwd), prompt, "--json-output"];
66
+ const args = [getPieCliPath(), prompt, "--json-output"];
39
67
  if (sessionId) {
40
68
  args.push("--session-id", sessionId);
41
69
  }
@@ -6,7 +6,7 @@ import { mkdir, readFile, writeFile } from "node:fs/promises";
6
6
  import { fileURLToPath } from "node:url";
7
7
  import { generateHtml } from "./generate_report.mjs";
8
8
  import { improveDescription } from "./improve_description.mjs";
9
- import { findProjectRoot, openInBrowser } from "./claude_cli.mjs";
9
+ import { findProjectRoot, openInBrowser } from "./pie_runner.mjs";
10
10
  import { runEval } from "./run_eval.mjs";
11
11
  import { parseSkillFile, resolveSkillPath } from "./skill_metadata.mjs";
12
12
 
@@ -1,7 +1,7 @@
1
1
  import { createRequire as __createRequire } from "node:module"; const require = __createRequire(import.meta.url);
2
2
  import {
3
3
  Agent
4
- } from "./chunk-FKAO25XH.js";
4
+ } from "./chunk-B3SMW2XW.js";
5
5
  import {
6
6
  FileSystemGateway,
7
7
  Type,
@@ -11,7 +11,7 @@ import {
11
11
  getPlatformConfig,
12
12
  hashText,
13
13
  streamSimple
14
- } from "./chunk-K6ZOCJ4V.js";
14
+ } from "./chunk-S2J3SZZX.js";
15
15
  import {
16
16
  __commonJS,
17
17
  __toESM
@@ -848,6 +848,10 @@ var BUILTIN_TOOL_CAPABILITY_METADATA = {
848
848
  spawn_subagents_parallel: { riskClass: "read_only", concurrencySafe: false, permissionScope: "host", readsHostResource: true, availableInPlanMode: true, maxOutputChars: 1e5 },
849
849
  ask_user_multi: { riskClass: "user_interaction", concurrencySafe: false, permissionScope: "user", asksUser: true, availableInPlanMode: true, allowedForSubagentByDefault: true },
850
850
  manage_todo_list: { riskClass: "session_mutation", concurrencySafe: false, permissionScope: "session", mutatesSession: true, availableInPlanMode: true },
851
+ // plan_scratch ships as the plan-mode extension's session-local draft tool;
852
+ // registration feeds /permissions display and the legacy plan-mode name list,
853
+ // while runtime decisions still honor the tool's inline risk metadata first.
854
+ plan_scratch: { riskClass: "session_mutation", concurrencySafe: false, permissionScope: "session", mutatesSession: true, availableInPlanMode: true },
851
855
  unity_project_inspect: { riskClass: "read_only", concurrencySafe: true, permissionScope: "unity", readsHostResource: true, availableInPlanMode: true, allowedForSubagentByDefault: true },
852
856
  unity_scene_query: { riskClass: "read_only", concurrencySafe: true, permissionScope: "unity", readsHostResource: true, availableInPlanMode: true, allowedForSubagentByDefault: true },
853
857
  unity_scene_object_inspect: { riskClass: "read_only", concurrencySafe: true, permissionScope: "unity", readsHostResource: true, availableInPlanMode: true, allowedForSubagentByDefault: true },
@@ -1409,8 +1413,9 @@ function buildSubagentPlan(deps, task, name, priority, timeoutSeconds) {
1409
1413
  const complexity = inferComplexity(task, priority);
1410
1414
  const modelClass = selectModelClass(complexity, priority);
1411
1415
  const modelResolution = deps.resolveModelClass?.(modelClass);
1412
- const model = modelResolution?.model ?? deps.model;
1413
- const apiKey = modelResolution?.apiKey ?? deps.apiKey;
1416
+ const resolvedPair = modelResolution?.model && modelResolution?.apiKey ? { model: modelResolution.model, apiKey: modelResolution.apiKey } : void 0;
1417
+ const model = resolvedPair?.model ?? deps.model;
1418
+ const apiKey = resolvedPair?.apiKey ?? deps.apiKey;
1414
1419
  const partialPlan = {
1415
1420
  task,
1416
1421
  name,
@@ -1508,6 +1513,9 @@ async function runSubagentWithProgress(deps, task, reportProgress, signal) {
1508
1513
  if (!apiKey) {
1509
1514
  throw new Error("No API key available for subagent execution.");
1510
1515
  }
1516
+ if (!model) {
1517
+ throw new Error("No model configured for subagent execution.");
1518
+ }
1511
1519
  if (signal?.aborted) {
1512
1520
  throw new Error("Subagent aborted by parent");
1513
1521
  }
@@ -1682,22 +1690,30 @@ function buildParallelResult(tasks, totalDuration) {
1682
1690
  `;
1683
1691
  return resultText;
1684
1692
  }
1693
+ var TASK_FIELD_ALIASES = ["task", "description", "prompt", "objective"];
1685
1694
  function normalizeTaskDefs(args) {
1686
1695
  const taskDefs = args?.tasks;
1687
1696
  if (!Array.isArray(taskDefs) || taskDefs.length === 0) {
1688
1697
  return { error: "Invalid spawn_subagents_parallel arguments: tasks must be a non-empty array." };
1689
1698
  }
1690
1699
  const normalized = [];
1700
+ const allowedFields = [...TASK_FIELD_ALIASES, "name", "priority", "timeout"];
1691
1701
  for (const [index, def] of taskDefs.entries()) {
1692
1702
  if (!def || typeof def !== "object") {
1693
- return { error: `Invalid task #${index + 1}: expected an object with task.` };
1703
+ return { error: `Invalid task #${index + 1}: expected an object with the objective in "task" (aliases "description"/"prompt"/"objective" accepted).` };
1694
1704
  }
1695
- if ("prompt" in def || "kind" in def || "max_turns" in def) {
1696
- return { error: "Invalid spawn_subagents_parallel arguments: use tasks[].task plus optional name/priority; prompt, kind, and max_turns are no longer supported." };
1705
+ const unknownFields = Object.keys(def).filter((key) => !allowedFields.includes(key));
1706
+ if (unknownFields.length > 0) {
1707
+ return { error: `Invalid task #${index + 1}: unknown field(s) ${unknownFields.map((key) => `"${key}"`).join(", ")}; allowed fields: ${allowedFields.map((key) => `"${key}"`).join(", ")}.` };
1697
1708
  }
1698
- const task = typeof def.task === "string" ? def.task.trim() : "";
1709
+ const taskValues = [
1710
+ ...new Set(
1711
+ TASK_FIELD_ALIASES.map((key) => typeof def[key] === "string" ? def[key].trim() : "").filter((value) => value.length > 0)
1712
+ )
1713
+ ];
1714
+ const task = taskValues.join("\n\n");
1699
1715
  if (!task) {
1700
- return { error: `Invalid task #${index + 1}: task cannot be empty.` };
1716
+ return { error: `Invalid task #${index + 1}: provide the objective in "task" (aliases "description"/"prompt"/"objective" are accepted).` };
1701
1717
  }
1702
1718
  const timeoutSeconds = "timeout" in def ? Number(def.timeout) : DEFAULT_SUBAGENT_TIMEOUT_SECONDS;
1703
1719
  if (!Number.isFinite(timeoutSeconds) || timeoutSeconds <= 0) {
@@ -1713,25 +1729,61 @@ function normalizeTaskDefs(args) {
1713
1729
  }
1714
1730
  return normalized;
1715
1731
  }
1732
+ var DEFAULT_MAX_CONCURRENT_SUBAGENTS = 4;
1733
+ var MAX_TRANSIENT_RETRIES = 2;
1734
+ var TRANSIENT_RETRY_BACKOFF_MS = 2e3;
1735
+ var TRANSIENT_ERROR_PATTERN = /econnreset|econnaborted|etimedout|timeout|enotfound|eai_again|econnrefused|socket hang up|sse stream aborted|fetch failed|status 5\d\d|bad gateway|service unavailable|overloaded/i;
1736
+ function isTransientNetworkError(message) {
1737
+ return TRANSIENT_ERROR_PATTERN.test(message);
1738
+ }
1739
+ function sleep(ms, signal) {
1740
+ return new Promise((resolve3) => {
1741
+ const timer = setTimeout(resolve3, ms);
1742
+ signal?.addEventListener(
1743
+ "abort",
1744
+ () => {
1745
+ clearTimeout(timer);
1746
+ resolve3();
1747
+ },
1748
+ { once: true }
1749
+ );
1750
+ });
1751
+ }
1752
+ async function runWithConcurrencyLimit(jobs, limit) {
1753
+ let next = 0;
1754
+ const workers = Array.from({ length: Math.max(1, Math.min(limit, jobs.length)) }, async () => {
1755
+ while (next < jobs.length) {
1756
+ const current = jobs[next++];
1757
+ await current();
1758
+ }
1759
+ });
1760
+ await Promise.all(workers);
1761
+ }
1716
1762
  function createSubagentCapability(deps) {
1717
1763
  const spawnParallelTool = {
1718
1764
  name: "spawn_subagents_parallel",
1719
1765
  label: "Spawn Subagents",
1720
1766
  description: `Spawn child agents to run bounded sub-tasks in parallel.
1721
1767
 
1722
- Use this for independent investigation, search, review, or triage work. Provide each task as a plain objective. Pie chooses the model class automatically. Use timeout only for unusually long tasks.`,
1768
+ Use this for independent investigation, search, review, or triage work. Provide each task as a plain objective. Pie chooses the model class automatically. Use timeout only for unusually long tasks.
1769
+
1770
+ Arguments shape (each tasks[] item puts the objective in "task" \u2014 "description"/"prompt"/"objective" are accepted as aliases):
1771
+ { "tasks": [{ "task": "<objective text>", "name": "<optional short label>", "priority": "fast" | "thorough", "timeout": 600 }] }`,
1723
1772
  parameters: Type.Object({
1724
1773
  tasks: Type.Array(
1725
1774
  Type.Object(
1726
1775
  {
1727
- task: Type.String({ description: "The bounded sub-task objective. Be specific and include expected output." }),
1776
+ task: Type.Optional(Type.String({ description: "The bounded sub-task objective. Be specific and include expected output. Preferred field name." })),
1777
+ description: Type.Optional(Type.String({ description: "Alias for task." })),
1778
+ prompt: Type.Optional(Type.String({ description: "Alias for task." })),
1779
+ objective: Type.Optional(Type.String({ description: "Alias for task." })),
1728
1780
  name: Type.Optional(Type.String({ description: "Short name for this subagent, shown in progress updates." })),
1729
1781
  priority: Type.Optional(Type.Union([Type.Literal("fast"), Type.Literal("thorough")], { description: "fast uses lower-cost defaults; thorough raises model class when needed." })),
1730
- timeout: Type.Optional(Type.Number({ description: "Timeout in seconds. Default: 1200s (20 minutes). For longer tasks, explicitly set a higher value, e.g. { timeout: 3600 }." }))
1782
+ timeout: Type.Optional(Type.Number({ description: "Timeout in seconds. Default: 1200s (20 minutes). For longer tasks, explicitly set a higher value, e.g. 3600." }))
1731
1783
  },
1732
- { additionalProperties: false }
1784
+ { additionalProperties: false, description: "Objective goes in 'task' (or an accepted alias); plus optional name/priority/timeout." }
1733
1785
  ),
1734
- { description: "Array of tasks to run in parallel", minItems: 1, maxItems: 10 }
1786
+ { description: "Array of tasks to run in parallel; at least one of task/description/prompt/objective must be a non-empty string per item.", minItems: 1, maxItems: 10 }
1735
1787
  )
1736
1788
  }, { additionalProperties: false }),
1737
1789
  async execute(first, second, third, fourth) {
@@ -1774,7 +1826,30 @@ Use this for independent investigation, search, review, or triage work. Provide
1774
1826
  toolContext.log?.(`Spawning ${tasks.length} parallel subagents: ${tasks.map((task) => task.name).join(", ")}`);
1775
1827
  toolContext.reportProgress?.({ message: `Launching ${tasks.length} parallel subagents...`, increment: 5 });
1776
1828
  const startTime = Date.now();
1777
- await Promise.all(tasks.map((task) => runSubagentWithProgress(deps, task, toolContext.reportProgress, signal)));
1829
+ const runTask = async (task) => {
1830
+ for (let attempt = 0; ; attempt++) {
1831
+ await runSubagentWithProgress(deps, task, toolContext.reportProgress, signal);
1832
+ const canRetry = task.status === "failed" && isTransientNetworkError(task.error ?? "") && attempt < MAX_TRANSIENT_RETRIES && !signal?.aborted;
1833
+ if (!canRetry) return;
1834
+ const failureReason = task.error ?? "unknown";
1835
+ task.evidence.length = 0;
1836
+ task.runtimeEvidence.length = 0;
1837
+ task.reportedEvidence.length = 0;
1838
+ task.confidenceSignals.length = 0;
1839
+ task.status = "pending";
1840
+ task.error = void 0;
1841
+ task.completedAt = void 0;
1842
+ const backoffMs = (deps.transientRetryBackoffMs ?? TRANSIENT_RETRY_BACKOFF_MS) * (attempt + 1);
1843
+ toolContext.reportProgress?.({ message: `[${task.name}] Transient error, retrying in ${Math.round(backoffMs / 1e3)}s (attempt ${attempt + 2}/${MAX_TRANSIENT_RETRIES + 1})...` });
1844
+ deps.log(`[subagent] ${task.name} failed with a transient error (${failureReason}); retrying`);
1845
+ await sleep(backoffMs, signal);
1846
+ if (signal?.aborted) return;
1847
+ }
1848
+ };
1849
+ await runWithConcurrencyLimit(
1850
+ tasks.map((task) => () => runTask(task)),
1851
+ Math.max(1, deps.maxConcurrentSubagents ?? DEFAULT_MAX_CONCURRENT_SUBAGENTS)
1852
+ );
1778
1853
  const totalDuration = Date.now() - startTime;
1779
1854
  const completed = tasks.filter((task) => task.status === "completed").length;
1780
1855
  const failed = tasks.filter((task) => task.status === "failed" || task.status === "timeout").length;
@@ -1845,6 +1920,12 @@ var ManageTodoListParamsSchema = Type.Object(
1845
1920
  },
1846
1921
  { additionalProperties: false }
1847
1922
  );
1923
+ var MANAGE_TODO_LIST_TOOL_DESCRIPTION = `Manage a session-local todo list. Use when the user explicitly asks to use the todo tool, create/manage todos, track tasks, follow a todo list, show tasks, or mark a task done.
1924
+ NEVER use for: internal planning, tracking your own progress, remembering subtasks, or creating checklists for yourself.
1925
+ Use action=create with items to create a list; provide todo titles and optional descriptions only. The tool assigns ids and starts item 1 automatically.
1926
+ Use action=complete_current after finishing the current item; the tool advances to the next item or clears the list after the final item. Do not call complete_current after a required install, build, test, or verification command just failed; fix/retry first or clear with a reason.
1927
+ When a todo list exists, keep executing from the current in-progress item until it is cleared. If the work cannot continue, explain the failure and use action=clear with a reason.
1928
+ Actions: create, read, complete_current, clear.`;
1848
1929
  function cloneTodos(todos) {
1849
1930
  return todos.map((todo) => ({ ...todo }));
1850
1931
  }
@@ -2568,14 +2649,14 @@ function isTransientFileError(error) {
2568
2649
  const message = error instanceof Error ? error.message : String(error);
2569
2650
  return /^(EPERM|EACCES|EBUSY|ENOTEMPTY)$/.test(code) || /\b(EPERM|EACCES|EBUSY|locked|busy)\b/i.test(message);
2570
2651
  }
2571
- function sleep(ms) {
2652
+ function sleep2(ms) {
2572
2653
  return new Promise((resolve3) => setTimeout(resolve3, ms));
2573
2654
  }
2574
2655
  async function retryFileOperation(operation, retryDelaysMs) {
2575
2656
  let lastError;
2576
2657
  for (const [index, delayMs] of retryDelaysMs.entries()) {
2577
2658
  if (delayMs > 0) {
2578
- await sleep(delayMs);
2659
+ await sleep2(delayMs);
2579
2660
  }
2580
2661
  try {
2581
2662
  operation();
@@ -3964,7 +4045,8 @@ var SessionManager = class {
3964
4045
  name: newName ?? `${this.activeSession.metadata.name} (Branch)`,
3965
4046
  createdAt: now,
3966
4047
  updatedAt: now,
3967
- tags: this.activeSession.metadata.tags
4048
+ tags: this.activeSession.metadata.tags,
4049
+ cwd: this.activeSession.metadata.cwd
3968
4050
  }
3969
4051
  );
3970
4052
  return newSessionId;
@@ -4001,7 +4083,8 @@ var SessionManager = class {
4001
4083
  createdAt: now,
4002
4084
  updatedAt: now,
4003
4085
  activeEntryId: newActiveEntryId,
4004
- tags: this.activeSession.metadata.tags
4086
+ tags: this.activeSession.metadata.tags,
4087
+ cwd: this.activeSession.metadata.cwd
4005
4088
  }
4006
4089
  );
4007
4090
  return newSessionId;
@@ -12532,7 +12615,7 @@ function sanitizeTitle(title) {
12532
12615
  }
12533
12616
  function isTextCompletionStepTitle(title) {
12534
12617
  const normalized = sanitizeTitle(title).toLowerCase();
12535
- return /反馈|总结|结论|汇报|说明|解释|收尾|验证|确认|位置|结果/u.test(title) || /\b(summarize|summary|report|explain|explanation|wrap[\s-]?up|verify|verification|confirm|result|results|location|findings)\b/i.test(normalized);
12618
+ return /反馈|总结|结论|汇报|说明|解释|收尾|验证|确认|位置|结果/u.test(normalized) || /\b(summarize|summary|report|explain|explanation|wrap[\s-]?up|verify|verification|confirm|result|results|location|findings)\b/i.test(normalized);
12536
12619
  }
12537
12620
  function maybeAdvanceTodoExecutionState(state, signal) {
12538
12621
  if (state.mode !== "todo" || state.source === "user_todo" || state.lifecycle !== "active" || !state.currentStepId) {
@@ -15166,6 +15249,7 @@ export {
15166
15249
  interpretShellExit,
15167
15250
  createSubagentCapability,
15168
15251
  ManageTodoListParamsSchema,
15252
+ MANAGE_TODO_LIST_TOOL_DESCRIPTION,
15169
15253
  executeManageTodoList,
15170
15254
  maybeAdvanceTodoExecutionState,
15171
15255
  createAskUserCapability,
@@ -3,7 +3,7 @@ import {
3
3
  EventStream,
4
4
  streamSimple,
5
5
  validateToolArguments
6
- } from "./chunk-K6ZOCJ4V.js";
6
+ } from "./chunk-S2J3SZZX.js";
7
7
 
8
8
  // ../../packages/agent-core/src/agent-loop.ts
9
9
  function createLoopErrorMessage(error, config, signal) {
@@ -8,10 +8,10 @@ import {
8
8
  createSharedWebSearchTool,
9
9
  interpretShellExit,
10
10
  requestInteraction
11
- } from "./chunk-7RYOIXA2.js";
11
+ } from "./chunk-A644H2XX.js";
12
12
  import {
13
13
  Type
14
- } from "./chunk-K6ZOCJ4V.js";
14
+ } from "./chunk-S2J3SZZX.js";
15
15
 
16
16
  // src/config.ts
17
17
  import { existsSync, mkdirSync, readFileSync, renameSync } from "fs";
@@ -9911,7 +9911,50 @@ function validateToolArguments(tool, toolCall) {
9911
9911
  throw new Error(`Invalid arguments for tool "${tool.name}"`);
9912
9912
  }
9913
9913
  const path = firstError.path || "/";
9914
- throw new Error(`Invalid arguments for tool "${tool.name}" at ${path}: ${firstError.message}`);
9914
+ const hint = buildArgumentHint(tool.parameters, firstError.path || "", args);
9915
+ throw new Error(
9916
+ `Invalid arguments for tool "${tool.name}" at ${path}: ${firstError.message}${hint ? `; ${hint}` : ""}`
9917
+ );
9918
+ }
9919
+ function buildArgumentHint(schema, errorPath, args) {
9920
+ const parentPath = errorPath.includes("/") ? errorPath.slice(0, errorPath.lastIndexOf("/")) : "";
9921
+ const node = resolveSchemaNode(schema, parentPath);
9922
+ if (!node || typeof node !== "object" || !node.properties) return "";
9923
+ const parts = [];
9924
+ const allowed = Object.keys(node.properties);
9925
+ if (allowed.length > 0) parts.push(`allowed keys: [${allowed.join(", ")}]`);
9926
+ if (Array.isArray(node.required) && node.required.length > 0) {
9927
+ parts.push(`required: [${node.required.map(String).join(", ")}]`);
9928
+ }
9929
+ const received = receivedKeysAt(args, parentPath);
9930
+ if (received) parts.push(`received keys: [${received.join(", ")}]`);
9931
+ return parts.join("; ");
9932
+ }
9933
+ function resolveSchemaNode(schema, path) {
9934
+ let node = schema;
9935
+ for (const segment of path.split("/").filter(Boolean)) {
9936
+ if (!node || typeof node !== "object") return void 0;
9937
+ if (/^\d+$/.test(segment)) {
9938
+ node = node.items;
9939
+ } else if (node.properties && segment in node.properties) {
9940
+ node = node.properties[segment];
9941
+ } else if (node.additionalProperties && typeof node.additionalProperties === "object") {
9942
+ node = node.additionalProperties;
9943
+ } else {
9944
+ return void 0;
9945
+ }
9946
+ }
9947
+ return node;
9948
+ }
9949
+ function receivedKeysAt(args, path) {
9950
+ let target = args;
9951
+ for (const segment of path.split("/").filter(Boolean)) {
9952
+ target = target?.[segment];
9953
+ }
9954
+ if (target && typeof target === "object" && !Array.isArray(target)) {
9955
+ return Object.keys(target);
9956
+ }
9957
+ return void 0;
9915
9958
  }
9916
9959
 
9917
9960
  // ../../packages/ai/src/index.ts
@@ -3,8 +3,8 @@ import {
3
3
  Agent,
4
4
  agentLoop,
5
5
  agentLoopContinue
6
- } from "./chunk-FKAO25XH.js";
7
- import "./chunk-K6ZOCJ4V.js";
6
+ } from "./chunk-B3SMW2XW.js";
7
+ import "./chunk-S2J3SZZX.js";
8
8
  import "./chunk-LZQXQOPJ.js";
9
9
  export {
10
10
  Agent,