@earendil-works/pi-coding-agent 0.80.10 → 0.81.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (144) hide show
  1. package/CHANGELOG.md +60 -0
  2. package/README.md +6 -3
  3. package/dist/cli/args.d.ts.map +1 -1
  4. package/dist/cli/args.js +2 -0
  5. package/dist/cli/args.js.map +1 -1
  6. package/dist/core/agent-session-services.d.ts.map +1 -1
  7. package/dist/core/agent-session-services.js +13 -0
  8. package/dist/core/agent-session-services.js.map +1 -1
  9. package/dist/core/agent-session.d.ts +22 -3
  10. package/dist/core/agent-session.d.ts.map +1 -1
  11. package/dist/core/agent-session.js +75 -56
  12. package/dist/core/agent-session.js.map +1 -1
  13. package/dist/core/compaction/branch-summarization.d.ts +7 -1
  14. package/dist/core/compaction/branch-summarization.d.ts.map +1 -1
  15. package/dist/core/compaction/branch-summarization.js +8 -11
  16. package/dist/core/compaction/branch-summarization.js.map +1 -1
  17. package/dist/core/compaction/compaction.d.ts +19 -3
  18. package/dist/core/compaction/compaction.d.ts.map +1 -1
  19. package/dist/core/compaction/compaction.js +63 -26
  20. package/dist/core/compaction/compaction.js.map +1 -1
  21. package/dist/core/compaction/utils.d.ts +1 -1
  22. package/dist/core/compaction/utils.d.ts.map +1 -1
  23. package/dist/core/compaction/utils.js +6 -17
  24. package/dist/core/compaction/utils.js.map +1 -1
  25. package/dist/core/extensions/loader.d.ts.map +1 -1
  26. package/dist/core/extensions/loader.js +13 -2
  27. package/dist/core/extensions/loader.js.map +1 -1
  28. package/dist/core/extensions/runner.d.ts +2 -1
  29. package/dist/core/extensions/runner.d.ts.map +1 -1
  30. package/dist/core/extensions/runner.js +31 -0
  31. package/dist/core/extensions/runner.js.map +1 -1
  32. package/dist/core/extensions/types.d.ts +16 -2
  33. package/dist/core/extensions/types.d.ts.map +1 -1
  34. package/dist/core/extensions/types.js.map +1 -1
  35. package/dist/core/model-registry.d.ts +5 -1
  36. package/dist/core/model-registry.d.ts.map +1 -1
  37. package/dist/core/model-registry.js +17 -2
  38. package/dist/core/model-registry.js.map +1 -1
  39. package/dist/core/model-resolver.d.ts.map +1 -1
  40. package/dist/core/model-resolver.js +2 -0
  41. package/dist/core/model-resolver.js.map +1 -1
  42. package/dist/core/model-runtime.d.ts +7 -2
  43. package/dist/core/model-runtime.d.ts.map +1 -1
  44. package/dist/core/model-runtime.js +41 -18
  45. package/dist/core/model-runtime.js.map +1 -1
  46. package/dist/core/prompt-templates.d.ts +1 -0
  47. package/dist/core/prompt-templates.d.ts.map +1 -1
  48. package/dist/core/prompt-templates.js +4 -4
  49. package/dist/core/prompt-templates.js.map +1 -1
  50. package/dist/core/remote-catalog-provider.d.ts +1 -1
  51. package/dist/core/remote-catalog-provider.d.ts.map +1 -1
  52. package/dist/core/remote-catalog-provider.js +26 -7
  53. package/dist/core/remote-catalog-provider.js.map +1 -1
  54. package/dist/core/resource-loader.d.ts.map +1 -1
  55. package/dist/core/resource-loader.js +1 -0
  56. package/dist/core/resource-loader.js.map +1 -1
  57. package/dist/core/sdk.d.ts.map +1 -1
  58. package/dist/core/sdk.js +6 -2
  59. package/dist/core/sdk.js.map +1 -1
  60. package/dist/core/session-manager.d.ts +9 -4
  61. package/dist/core/session-manager.d.ts.map +1 -1
  62. package/dist/core/session-manager.js +99 -23
  63. package/dist/core/session-manager.js.map +1 -1
  64. package/dist/core/tools/read.d.ts.map +1 -1
  65. package/dist/core/tools/read.js +1 -1
  66. package/dist/core/tools/read.js.map +1 -1
  67. package/dist/core/usage-totals.d.ts +19 -0
  68. package/dist/core/usage-totals.d.ts.map +1 -0
  69. package/dist/core/usage-totals.js +52 -0
  70. package/dist/core/usage-totals.js.map +1 -0
  71. package/dist/extensions/index.d.ts +3 -0
  72. package/dist/extensions/index.d.ts.map +1 -0
  73. package/dist/extensions/index.js +3 -0
  74. package/dist/extensions/index.js.map +1 -0
  75. package/dist/extensions/llama/client.d.ts +61 -0
  76. package/dist/extensions/llama/client.d.ts.map +1 -0
  77. package/dist/extensions/llama/client.js +302 -0
  78. package/dist/extensions/llama/client.js.map +1 -0
  79. package/dist/extensions/llama/huggingface.d.ts +23 -0
  80. package/dist/extensions/llama/huggingface.d.ts.map +1 -0
  81. package/dist/extensions/llama/huggingface.js +141 -0
  82. package/dist/extensions/llama/huggingface.js.map +1 -0
  83. package/dist/extensions/llama/index.d.ts +3 -0
  84. package/dist/extensions/llama/index.d.ts.map +1 -0
  85. package/dist/extensions/llama/index.js +208 -0
  86. package/dist/extensions/llama/index.js.map +1 -0
  87. package/dist/extensions/llama/provider.d.ts +10 -0
  88. package/dist/extensions/llama/provider.d.ts.map +1 -0
  89. package/dist/extensions/llama/provider.js +102 -0
  90. package/dist/extensions/llama/provider.js.map +1 -0
  91. package/dist/extensions/llama/ui.d.ts +42 -0
  92. package/dist/extensions/llama/ui.d.ts.map +1 -0
  93. package/dist/extensions/llama/ui.js +416 -0
  94. package/dist/extensions/llama/ui.js.map +1 -0
  95. package/dist/index.d.ts +2 -2
  96. package/dist/index.d.ts.map +1 -1
  97. package/dist/index.js +1 -1
  98. package/dist/index.js.map +1 -1
  99. package/dist/main.d.ts.map +1 -1
  100. package/dist/main.js +10 -4
  101. package/dist/main.js.map +1 -1
  102. package/dist/modes/interactive/components/footer.d.ts.map +1 -1
  103. package/dist/modes/interactive/components/footer.js +24 -23
  104. package/dist/modes/interactive/components/footer.js.map +1 -1
  105. package/dist/modes/interactive/interactive-mode.d.ts +1 -0
  106. package/dist/modes/interactive/interactive-mode.d.ts.map +1 -1
  107. package/dist/modes/interactive/interactive-mode.js +48 -26
  108. package/dist/modes/interactive/interactive-mode.js.map +1 -1
  109. package/dist/modes/rpc/rpc-client.d.ts +4 -0
  110. package/dist/modes/rpc/rpc-client.d.ts.map +1 -1
  111. package/dist/modes/rpc/rpc-client.js +7 -0
  112. package/dist/modes/rpc/rpc-client.js.map +1 -1
  113. package/dist/modes/rpc/rpc-mode.d.ts.map +1 -1
  114. package/dist/modes/rpc/rpc-mode.js +4 -0
  115. package/dist/modes/rpc/rpc-mode.js.map +1 -1
  116. package/dist/modes/rpc/rpc-types.d.ts +11 -0
  117. package/dist/modes/rpc/rpc-types.d.ts.map +1 -1
  118. package/dist/modes/rpc/rpc-types.js.map +1 -1
  119. package/docs/compaction.md +8 -3
  120. package/docs/custom-provider.md +28 -0
  121. package/docs/extensions.md +51 -9
  122. package/docs/index.md +1 -0
  123. package/docs/json.md +5 -1
  124. package/docs/llama-cpp.md +99 -0
  125. package/docs/prompt-templates.md +1 -0
  126. package/docs/providers.md +11 -0
  127. package/docs/rpc.md +81 -2
  128. package/docs/sdk.md +3 -0
  129. package/docs/session-format.md +17 -3
  130. package/docs/tui.md +41 -26
  131. package/docs/usage.md +2 -1
  132. package/examples/extensions/README.md +1 -1
  133. package/examples/extensions/custom-compaction.ts +1 -0
  134. package/examples/extensions/custom-provider-anthropic/package-lock.json +2 -2
  135. package/examples/extensions/custom-provider-anthropic/package.json +1 -1
  136. package/examples/extensions/custom-provider-gitlab-duo/package.json +1 -1
  137. package/examples/extensions/gondolin/package-lock.json +2 -2
  138. package/examples/extensions/gondolin/package.json +1 -1
  139. package/examples/extensions/sandbox/package-lock.json +2 -2
  140. package/examples/extensions/sandbox/package.json +1 -1
  141. package/examples/extensions/with-deps/package-lock.json +2 -2
  142. package/examples/extensions/with-deps/package.json +1 -1
  143. package/npm-shrinkwrap.json +16 -31
  144. package/package.json +4 -4
@@ -468,6 +468,7 @@ pi.on("session_before_compact", async (event, ctx) => {
468
468
  summary: "...",
469
469
  firstKeptEntryId: preparation.firstKeptEntryId,
470
470
  tokensBefore: preparation.tokensBefore,
471
+ // usage: summaryResponse.usage, // Optional; included in session totals
471
472
  }
472
473
  };
473
474
  });
@@ -489,7 +490,13 @@ pi.on("session_before_tree", async (event, ctx) => {
489
490
  const { preparation, signal } = event;
490
491
  return { cancel: true };
491
492
  // OR provide custom summary:
492
- return { summary: { summary: "...", details: {} } };
493
+ return {
494
+ summary: {
495
+ summary: "...",
496
+ // usage: summaryResponse.usage, // Optional; included in session totals
497
+ details: {},
498
+ },
499
+ };
493
500
  });
494
501
 
495
502
  pi.on("session_tree", async (event, ctx) => {
@@ -813,7 +820,7 @@ In parallel tool mode, `tool_result` and `tool_execution_end` may interleave in
813
820
  `tool_result` handlers chain like middleware:
814
821
  - Handlers run in extension load order
815
822
  - Each handler sees the latest result after previous handler changes
816
- - Handlers can return partial patches (`content`, `details`, or `isError`); omitted fields keep their current values
823
+ - Handlers can return partial patches (`content`, `details`, `isError`, or `usage`); omitted fields keep their current values
817
824
 
818
825
  Use `ctx.signal` for nested async work inside the handler. This lets Esc cancel model calls, `fetch()`, and other abort-aware operations started by the extension.
819
826
 
@@ -822,7 +829,7 @@ import { isBashToolResult } from "@earendil-works/pi-coding-agent";
822
829
 
823
830
  pi.on("tool_result", async (event, ctx) => {
824
831
  // event.toolName, event.toolCallId, event.input
825
- // event.content, event.details, event.isError
832
+ // event.content, event.details, event.isError, event.usage
826
833
 
827
834
  if (isBashToolResult(event)) {
828
835
  // event.details is typed as BashToolDetails
@@ -835,7 +842,7 @@ pi.on("tool_result", async (event, ctx) => {
835
842
  });
836
843
 
837
844
  // Modify result:
838
- return { content: [...], details: {...}, isError: false };
845
+ return { content: [...], details: {...}, isError: false, usage: nestedModelUsage };
839
846
  });
840
847
  ```
841
848
 
@@ -977,7 +984,7 @@ ctx.sessionManager.getLeafId() // Current leaf entry ID
977
984
 
978
985
  ### ctx.modelRegistry / ctx.model
979
986
 
980
- Access to models and API keys.
987
+ Access to models, providers, and resolved authentication. `ctx.modelRegistry.getProvider(id)` returns the effective pi-ai provider, while `getProviderAuth(id)` resolves its current API key, headers, base URL, and provider-scoped environment without requiring a loaded model. `ctx.model` is the active model.
981
988
 
982
989
  ### ctx.signal
983
990
 
@@ -1679,7 +1686,37 @@ Calls made during the extension factory function are queued and applied once the
1679
1686
 
1680
1687
  Dynamic providers can implement `refreshModels`. Pi calls it during model refresh, publishes the returned list synchronously through the provider, and passes the canonical credential/store/network/signal context. The extension decides whether to persist the catalog through `context.store`; live servers such as llama.cpp can ignore it.
1681
1688
 
1689
+ Extensions that need native provider auth, filtering, refresh, or stream behavior can register a complete `Provider` from `@earendil-works/pi-ai`. The provider becomes the composition base and `models.json` overrides still apply above it.
1690
+
1682
1691
  ```typescript
1692
+ import { createProvider, openAICompletionsApi } from "@earendil-works/pi-ai";
1693
+
1694
+ const provider = createProvider({
1695
+ id: "local-server",
1696
+ name: "Local Server",
1697
+ baseUrl: "http://localhost:8080/v1",
1698
+ auth: {
1699
+ apiKey: {
1700
+ name: "Local server setup",
1701
+ async login(interaction) {
1702
+ return {
1703
+ type: "api_key",
1704
+ key: await interaction.prompt({ type: "secret", message: "API key" }),
1705
+ };
1706
+ },
1707
+ async resolve({ credential }) {
1708
+ return credential?.key
1709
+ ? { auth: { apiKey: credential.key }, source: "stored API key" }
1710
+ : undefined;
1711
+ },
1712
+ },
1713
+ },
1714
+ models: [],
1715
+ api: openAICompletionsApi(),
1716
+ });
1717
+
1718
+ pi.registerProvider(provider);
1719
+
1683
1720
  // Register a new provider with custom models
1684
1721
  pi.registerProvider("my-proxy", {
1685
1722
  name: "My Proxy",
@@ -1748,7 +1785,9 @@ pi.registerProvider("corporate-ai", {
1748
1785
  });
1749
1786
  ```
1750
1787
 
1751
- **Config options:**
1788
+ The object form accepts a complete pi-ai `Provider`, including native `auth`, `getModels`, `refreshModels`, `filterModels`, `stream`, and `streamSimple` behavior.
1789
+
1790
+ **Legacy config options:**
1752
1791
  - `name` - Display name for the provider in UI such as `/login`.
1753
1792
  - `baseUrl` - API endpoint URL. Required when defining models.
1754
1793
  - `apiKey` - API key literal, environment interpolation (`$ENV_VAR` or `${ENV_VAR}`), or leading `!command`. Required when defining models (unless `oauth` provided). `$$` escapes `$`, and `$!` escapes a literal `!` without triggering command execution.
@@ -1900,6 +1939,7 @@ pi.registerTool({
1900
1939
  return {
1901
1940
  content: [{ type: "text", text: "Done" }], // Sent to LLM
1902
1941
  details: { data: result }, // For rendering & state
1942
+ // usage: nestedModelResponse.usage, // Optional nested LLM usage
1903
1943
  // Optional: stop after this tool batch when every finalized tool result
1904
1944
  // in the batch also returns terminate: true.
1905
1945
  terminate: true,
@@ -1912,6 +1952,8 @@ pi.registerTool({
1912
1952
  });
1913
1953
  ```
1914
1954
 
1955
+ **Usage accounting:** If a tool makes nested LLM calls, return their combined `Usage` as `usage`. Pi persists it on the tool result and includes it in footer, `/session`, and RPC session totals. `tool_result` handlers can inspect or replace this value.
1956
+
1915
1957
  **Signaling errors:** To mark a tool execution as failed (sets `isError: true` on the result and reports it to the LLM), throw an error from `execute`. Returning a value never sets the error flag regardless of what properties you include in the return object.
1916
1958
 
1917
1959
  **Early termination:** Return `terminate: true` from `execute()` to hint that the automatic follow-up LLM call should be skipped after the current tool batch. This only takes effect when every finalized tool result in that batch is terminating. See [examples/extensions/structured-output.ts](../examples/extensions/structured-output.ts) for a minimal example where the agent ends on a final structured-output tool call.
@@ -2708,8 +2750,8 @@ class VimEditor extends CustomEditor {
2708
2750
 
2709
2751
  export default function (pi: ExtensionAPI) {
2710
2752
  pi.on("session_start", (_event, ctx) => {
2711
- ctx.ui.setEditorComponent((_tui, theme, keybindings) =>
2712
- new VimEditor(theme, keybindings)
2753
+ ctx.ui.setEditorComponent((tui, theme, keybindings) =>
2754
+ new VimEditor(tui, theme, keybindings)
2713
2755
  );
2714
2756
  });
2715
2757
  }
@@ -2718,7 +2760,7 @@ export default function (pi: ExtensionAPI) {
2718
2760
  **Key points:**
2719
2761
  - Extend `CustomEditor` (not base `Editor`) to get app keybindings (escape to abort, ctrl+d, model switching)
2720
2762
  - Call `super.handleInput(data)` for keys you don't handle
2721
- - Factory receives `theme` and `keybindings` from the app
2763
+ - Factory receives `tui`, `theme`, and `keybindings` from the app
2722
2764
  - Use `ctx.ui.getEditorComponent()` before `setEditorComponent()` to wrap the previously configured custom editor
2723
2765
  - Pass `undefined` to restore default: `ctx.ui.setEditorComponent(undefined)`
2724
2766
 
package/docs/index.md CHANGED
@@ -41,6 +41,7 @@ For the full first-run flow, see [Quickstart](quickstart.md).
41
41
  - [Quickstart](quickstart.md) - install, authenticate, and run a first session.
42
42
  - [Using Pi](usage.md) - interactive mode, slash commands, context files, and CLI reference.
43
43
  - [Providers](providers.md) - subscription and API-key setup for built-in providers.
44
+ - [llama.cpp](llama-cpp.md) - run a local router and manage models with `/llama`.
44
45
  - [Security](security.md) - project trust, sandbox boundaries, and vulnerability reporting.
45
46
  - [Containerization](containerization.md) - sandbox pi with Gondolin, Docker, or OpenShell.
46
47
  - [Settings](settings.md) - global and project settings.
package/docs/json.md CHANGED
@@ -17,7 +17,11 @@ type AgentSessionEvent =
17
17
  | { type: "compaction_start"; reason: "manual" | "threshold" | "overflow" }
18
18
  | { type: "compaction_end"; reason: "manual" | "threshold" | "overflow"; result: CompactionResult | undefined; aborted: boolean; willRetry: boolean; errorMessage?: string }
19
19
  | { type: "auto_retry_start"; attempt: number; maxAttempts: number; delayMs: number; errorMessage: string }
20
- | { type: "auto_retry_end"; success: boolean; attempt: number; finalError?: string };
20
+ | { type: "auto_retry_end"; success: boolean; attempt: number; finalError?: string }
21
+ | { type: "summarization_retry_scheduled"; attempt: number; maxAttempts: number; delayMs: number; errorMessage: string }
22
+ | { type: "summarization_retry_attempt_start"; source: "branchSummary" }
23
+ | { type: "summarization_retry_attempt_start"; source: "compaction"; reason: "manual" | "threshold" | "overflow" }
24
+ | { type: "summarization_retry_finished" };
21
25
  ```
22
26
 
23
27
  `queue_update` emits the full pending steering and follow-up queues whenever they change. `compaction_start` and `compaction_end` cover both manual and automatic compaction.
@@ -0,0 +1,99 @@
1
+ # llama.cpp
2
+
3
+ Pi supports the [llama.cpp](https://github.com/ggml-org/llama.cpp) router server. The router discovers multiple GGUF models and loads or unloads them on demand.
4
+
5
+ Use a current llama.cpp build with router support. Follow the [build instructions](https://github.com/ggml-org/llama.cpp/blob/master/docs/build.md) or install a [prebuilt release](https://github.com/ggml-org/llama.cpp/releases) for your platform.
6
+
7
+ ## Start the router
8
+
9
+ Start `llama-server` without `--model` or `-m`. Passing a model starts single-model mode instead of router mode.
10
+
11
+ ```bash
12
+ llama-server \
13
+ --models-dir ~/models \
14
+ --no-models-autoload \
15
+ --jinja \
16
+ --host 127.0.0.1 \
17
+ --port 8080 \
18
+ -ngl 999 \
19
+ -c 32768
20
+ ```
21
+
22
+ Important options:
23
+
24
+ - `--models-dir ~/models` discovers local GGUF files.
25
+ - `--no-models-autoload` keeps loading explicit through `/llama`.
26
+ - `--jinja` enables compatible chat templates and tool calling.
27
+ - `-ngl 999` offloads as many layers as possible to the GPU.
28
+ - `-c 32768` sets the context window for each loaded model. Omit it to use the model's native context, which may require substantially more memory.
29
+
30
+ A single-file model can sit directly in the model directory. Put multimodal and multi-shard models in separate subdirectories:
31
+
32
+ ```text
33
+ ~/models/
34
+ ├── llama-3.2-1b-Q4_K_M.gguf
35
+ ├── gemma-3-4b-it-Q4_K_M/
36
+ │ ├── gemma-3-4b-it-Q4_K_M.gguf
37
+ │ └── mmproj-F16.gguf
38
+ └── large-model-Q4_K_M/
39
+ ├── large-model-Q4_K_M-00001-of-00003.gguf
40
+ ├── large-model-Q4_K_M-00002-of-00003.gguf
41
+ └── large-model-Q4_K_M-00003-of-00003.gguf
42
+ ```
43
+
44
+ Restart the router after manually adding files. For per-model context sizes and other options, use [llama.cpp model presets](https://github.com/ggml-org/llama.cpp/blob/master/tools/server/README.md#model-presets).
45
+
46
+ ## Configure Pi
47
+
48
+ Start Pi and configure the provider:
49
+
50
+ ```text
51
+ /login llama.cpp
52
+ ```
53
+
54
+ Enter the router URL and optional API key. The default URL is `http://127.0.0.1:8080`.
55
+
56
+ Environment variables can configure the same values without `/login`:
57
+
58
+ ```bash
59
+ export LLAMA_BASE_URL=http://127.0.0.1:8080
60
+ export LLAMA_API_KEY=optional-secret
61
+ pi
62
+ ```
63
+
64
+ If the server uses an API key, start `llama-server` with the matching `--api-key` value. Keep `--host 127.0.0.1` for local-only access.
65
+
66
+ ## Manage models
67
+
68
+ Run:
69
+
70
+ ```text
71
+ /llama
72
+ ```
73
+
74
+ - Select an unloaded model to load it.
75
+ - Select a loaded model to unload it.
76
+ - Select **Download model…**, search Hugging Face, then choose a repository and quantization. Exact `owner/repository[:quant]` values also work.
77
+ - Press Escape during a load or download to confirm cancellation.
78
+
79
+ Hugging Face search uses `HF_TOKEN` when set, then checks `$HF_TOKEN_PATH`, `$HF_HOME/token`, `$XDG_CACHE_HOME/huggingface/token`, and `~/.cache/huggingface/token`. Search also works without authentication, subject to lower rate limits. Pi warns before downloading gated repositories and links to their access page. The llama.cpp server performs the download, so its process must also have `HF_TOKEN` when the selected repository requires access.
80
+
81
+ If other models are loaded, Pi asks whether to unload them first or keep them loaded. Pi does not silently unload models and never deletes model files. The router may be shared with other clients, so `/llama` always displays the router's current state.
82
+
83
+ Only loaded models appear in `/model`. After loading a model, run `/model` to select it for the current Pi session.
84
+
85
+ If the router disconnects, `/llama` shows **Retry** and **Close**. Retry reconnects and refreshes model state without replaying the interrupted operation.
86
+
87
+ ## Troubleshooting
88
+
89
+ Check that the router is reachable:
90
+
91
+ ```bash
92
+ curl http://127.0.0.1:8080/health
93
+ curl http://127.0.0.1:8080/models
94
+ ```
95
+
96
+ - **No models in `/llama`:** Check `--models-dir`, the directory layout, and restart the router.
97
+ - **Model missing from `/model`:** Load it with `/llama` first.
98
+ - **Load fails or uses too much memory:** Lower `-c` or unload another model.
99
+ - **Server is not in router mode:** Start it without `--model`, `-m`, or `-hf`.
@@ -69,6 +69,7 @@ Templates support positional arguments, defaults, and simple slicing:
69
69
  - `$1`, `$2`, ... positional args
70
70
  - `$@` or `$ARGUMENTS` for all args joined
71
71
  - `${1:-default}` uses arg 1 when present/non-empty, otherwise `default`
72
+ - `${@:-default}` or `${ARGUMENTS:-default}` uses all arguments when present/non-empty, otherwise `default`
72
73
  - `${@:N}` for args from the Nth position (1-indexed)
73
74
  - `${@:N:L}` for `L` args starting at N
74
75
 
package/docs/providers.md CHANGED
@@ -8,6 +8,7 @@ Pi supports subscription-based providers via OAuth and API key providers via env
8
8
  - [API Keys](#api-keys)
9
9
  - [Auth File](#auth-file)
10
10
  - [Cloud Providers](#cloud-providers)
11
+ - [llama.cpp](#llamacpp)
11
12
  - [Custom Providers](#custom-providers)
12
13
  - [Resolution Order](#resolution-order)
13
14
 
@@ -86,6 +87,8 @@ pi
86
87
  | Kimi For Coding | `KIMI_API_KEY` | `kimi-coding` |
87
88
  | MiniMax | `MINIMAX_API_KEY` | `minimax` |
88
89
  | MiniMax (China) | `MINIMAX_CN_API_KEY` | `minimax-cn` |
90
+ | Qwen Token Plan | `QWEN_TOKEN_PLAN_API_KEY` | `qwen-token-plan` |
91
+ | Qwen Token Plan (China) | `QWEN_TOKEN_PLAN_CN_API_KEY` | `qwen-token-plan-cn` |
89
92
  | Xiaomi MiMo | `XIAOMI_API_KEY` | `xiaomi` |
90
93
  | Xiaomi MiMo Token Plan (China) | `XIAOMI_TOKEN_PLAN_CN_API_KEY` | `xiaomi-token-plan-cn` |
91
94
  | Xiaomi MiMo Token Plan (Amsterdam) | `XIAOMI_TOKEN_PLAN_AMS_API_KEY` | `xiaomi-token-plan-ams` |
@@ -108,6 +111,8 @@ Store credentials in `~/.pi/agent/auth.json`:
108
111
  "opencode": { "type": "api_key", "key": "..." },
109
112
  "opencode-go": { "type": "api_key", "key": "..." },
110
113
  "together": { "type": "api_key", "key": "..." },
114
+ "qwen-token-plan": { "type": "api_key", "key": "sk-sp-..." },
115
+ "qwen-token-plan-cn": { "type": "api_key", "key": "sk-sp-..." },
111
116
  "xiaomi": { "type": "api_key", "key": "..." },
112
117
  "xiaomi-token-plan-cn": { "type": "api_key", "key": "..." },
113
118
  "xiaomi-token-plan-ams": { "type": "api_key", "key": "..." },
@@ -274,6 +279,12 @@ export GOOGLE_CLOUD_LOCATION=us-central1
274
279
 
275
280
  Or set `GOOGLE_APPLICATION_CREDENTIALS` to a service account key file.
276
281
 
282
+ ## llama.cpp
283
+
284
+ Pi supports the llama.cpp router server. Configure it with `/login llama.cpp`, manage loaded models with `/llama`, and select a loaded model with `/model`.
285
+
286
+ See [llama.cpp](llama-cpp.md) for server setup, model directory layout, environment variables, and command usage.
287
+
277
288
  ## Custom Providers
278
289
 
279
290
  **Via models.json:** Add Ollama, LM Studio, vLLM, or any provider that speaks a supported API (OpenAI Completions, OpenAI Responses, Anthropic Messages, Google Generative AI). See [models.md](models.md).
package/docs/rpc.md CHANGED
@@ -313,6 +313,26 @@ Response:
313
313
  }
314
314
  ```
315
315
 
316
+ #### get_available_thinking_levels
317
+
318
+ List the thinking levels supported by the current model. Returns `["off"]` for a model without reasoning support.
319
+
320
+ ```json
321
+ {"type": "get_available_thinking_levels"}
322
+ ```
323
+
324
+ Response:
325
+ ```json
326
+ {
327
+ "type": "response",
328
+ "command": "get_available_thinking_levels",
329
+ "success": true,
330
+ "data": {
331
+ "levels": ["off", "minimal", "low", "medium", "high"]
332
+ }
333
+ }
334
+ ```
335
+
316
336
  ### Queue Modes
317
337
 
318
338
  #### set_steering_mode
@@ -375,12 +395,20 @@ Response:
375
395
  "firstKeptEntryId": "abc123",
376
396
  "tokensBefore": 150000,
377
397
  "estimatedTokensAfter": 32000,
398
+ "usage": {
399
+ "input": 32000,
400
+ "output": 1200,
401
+ "cacheRead": 0,
402
+ "cacheWrite": 0,
403
+ "totalTokens": 33200,
404
+ "cost": {"input": 0.01, "output": 0.02, "cacheRead": 0, "cacheWrite": 0, "total": 0.03}
405
+ },
378
406
  "details": {}
379
407
  }
380
408
  }
381
409
  ```
382
410
 
383
- `estimatedTokensAfter` is a heuristic estimate over the rebuilt message context immediately after compaction, not a provider-exact token count.
411
+ `estimatedTokensAfter` is a heuristic estimate over the rebuilt message context immediately after compaction, not a provider-exact token count. `usage` reports the LLM call or calls that generated the summary and may be omitted by custom compaction handlers.
384
412
 
385
413
  #### set_auto_compaction
386
414
 
@@ -537,7 +565,7 @@ Response:
537
565
  }
538
566
  ```
539
567
 
540
- `tokens` contains assistant usage totals for the current session state. `contextUsage` contains the actual current context-window estimate used for compaction and footer display.
568
+ `tokens` and `cost` include assistant messages, usage reported by tools, and compaction/branch-summary generation across the full session. `contextUsage` contains the actual current context-window estimate used for compaction and footer display.
541
569
 
542
570
  `contextUsage` is omitted when no model or context window is available. `contextUsage.tokens` and `contextUsage.percent` are `null` immediately after compaction until a fresh post-compaction assistant response provides valid usage data.
543
571
 
@@ -823,6 +851,9 @@ Events are streamed to stdout as JSON lines during agent operation. Events do NO
823
851
  | `compaction_end` | Compaction completes |
824
852
  | `auto_retry_start` | Auto-retry begins (after transient error) |
825
853
  | `auto_retry_end` | Auto-retry completes (success or final failure) |
854
+ | `summarization_retry_scheduled` | Retry scheduled for a transient compaction or branch-summary summarization error |
855
+ | `summarization_retry_attempt_start` | Retried summarization request starts |
856
+ | `summarization_retry_finished` | Summarization retry loop completes |
826
857
  | `extension_error` | Extension threw an error |
827
858
 
828
859
  ### agent_start
@@ -996,6 +1027,14 @@ The `reason` field is `"manual"`, `"threshold"`, or `"overflow"`.
996
1027
  "firstKeptEntryId": "abc123",
997
1028
  "tokensBefore": 150000,
998
1029
  "estimatedTokensAfter": 32000,
1030
+ "usage": {
1031
+ "input": 32000,
1032
+ "output": 1200,
1033
+ "cacheRead": 0,
1034
+ "cacheWrite": 0,
1035
+ "totalTokens": 33200,
1036
+ "cost": {"input": 0.01, "output": 0.02, "cacheRead": 0, "cacheWrite": 0, "total": 0.03}
1037
+ },
999
1038
  "details": {}
1000
1039
  },
1001
1040
  "aborted": false,
@@ -1041,6 +1080,36 @@ On final failure (max retries exceeded):
1041
1080
  }
1042
1081
  ```
1043
1082
 
1083
+ ### summarization_retry_scheduled / summarization_retry_attempt_start / summarization_retry_finished
1084
+
1085
+ Emitted when compaction or branch-summary summarization retries after a transient provider error. These events use the same retry settings as automatic assistant-turn retries.
1086
+
1087
+ ```json
1088
+ {
1089
+ "type": "summarization_retry_scheduled",
1090
+ "attempt": 1,
1091
+ "maxAttempts": 3,
1092
+ "delayMs": 2000,
1093
+ "errorMessage": "terminated"
1094
+ }
1095
+ ```
1096
+
1097
+ ```json
1098
+ {
1099
+ "type": "summarization_retry_attempt_start",
1100
+ "source": "compaction",
1101
+ "reason": "threshold"
1102
+ }
1103
+ ```
1104
+
1105
+ For branch summaries, `source` is `"branchSummary"` and no `reason` is present.
1106
+
1107
+ ```json
1108
+ {
1109
+ "type": "summarization_retry_finished"
1110
+ }
1111
+ ```
1112
+
1044
1113
  ### extension_error
1045
1114
 
1046
1115
  Emitted when an extension throws an error.
@@ -1348,11 +1417,21 @@ Stop reasons: `"stop"`, `"length"`, `"toolUse"`, `"error"`, `"aborted"`
1348
1417
  "toolCallId": "call_123",
1349
1418
  "toolName": "bash",
1350
1419
  "content": [{"type": "text", "text": "total 48\ndrwxr-xr-x ..."}],
1420
+ "usage": {
1421
+ "input": 100,
1422
+ "output": 50,
1423
+ "cacheRead": 0,
1424
+ "cacheWrite": 0,
1425
+ "totalTokens": 150,
1426
+ "cost": {"input": 0.0003, "output": 0.00075, "cacheRead": 0, "cacheWrite": 0, "total": 0.00105}
1427
+ },
1351
1428
  "isError": false,
1352
1429
  "timestamp": 1733234567890
1353
1430
  }
1354
1431
  ```
1355
1432
 
1433
+ `usage` is optional and reports nested LLM work performed by the tool. When present, it contributes to session token and cost totals.
1434
+
1356
1435
  ### BashExecutionMessage
1357
1436
 
1358
1437
  Created by the `bash` RPC command (not by LLM tool calls):
package/docs/sdk.md CHANGED
@@ -319,6 +319,9 @@ session.subscribe((event) => {
319
319
  case "compaction_end":
320
320
  case "auto_retry_start":
321
321
  case "auto_retry_end":
322
+ case "summarization_retry_scheduled":
323
+ case "summarization_retry_attempt_start":
324
+ case "summarization_retry_finished":
322
325
  break;
323
326
  }
324
327
  });
@@ -96,6 +96,7 @@ interface ToolResultMessage {
96
96
  toolName: string;
97
97
  content: (TextContent | ImageContent)[];
98
98
  details?: any; // Tool-specific metadata
99
+ usage?: Usage; // Nested LLM work performed by the tool
99
100
  isError: boolean;
100
101
  timestamp: number;
101
102
  }
@@ -231,9 +232,18 @@ Created when context is compacted. Stores a summary of earlier messages.
231
232
  {"type":"compaction","id":"f6g7h8i9","parentId":"e5f6g7h8","timestamp":"2024-12-03T14:10:00.000Z","summary":"User discussed X, Y, Z...","firstKeptEntryId":"c3d4e5f6","tokensBefore":50000}
232
233
  ```
233
234
 
235
+ Newer harness-generated compactions embed the retained post-compaction context directly on the entry, instead of `firstKeptEntryId`:
236
+
237
+ ```json
238
+ {"type":"compaction","id":"f6g7h8i9","parentId":"e5f6g7h8","timestamp":"2024-12-03T14:10:00.000Z","summary":"User discussed X, Y, Z...","tokensBefore":50000,"retainedTail":[{"role":"user","content":"latest request"},{"role":"assistant","content":[{"type":"text","text":"latest reply"}],"provider":"anthropic","model":"claude-sonnet-4-5","usage":{...},"stopReason":"stop"}]}
239
+ ```
240
+
234
241
  Optional fields:
242
+ - `usage`: LLM usage from generating the summary; included in session token and cost totals
243
+ - `retainedTail`: Materialized `AgentMessage[]` kept after compaction. This is optional only for backward compatibility with older sessions. Newer harness-generated compactions include it so we can rebuild context from this checkpoint without walking older entries before the compaction entry.
235
244
  - `details`: Implementation-specific data (e.g., `{ readFiles: string[], modifiedFiles: string[] }` for default, or custom data for extensions)
236
245
  - `fromHook`: `true` if generated by an extension, `false`/`undefined` if pi-generated (legacy field name)
246
+ - `firstKeptEntryId`: for compatibility with old entry format.
237
247
 
238
248
  ### BranchSummaryEntry
239
249
 
@@ -244,6 +254,7 @@ Created when switching branches via `/tree` with an LLM generated summary of the
244
254
  ```
245
255
 
246
256
  Optional fields:
257
+ - `usage`: LLM usage from generating the summary; included in session token and cost totals
247
258
  - `details`: File tracking data (`{ readFiles: string[], modifiedFiles: string[] }`) for default, or custom data for extensions
248
259
  - `fromHook`: `true` if generated by an extension, `false`/`undefined` if pi-generated (legacy field name)
249
260
 
@@ -311,8 +322,9 @@ Entries form a tree:
311
322
  1. Collects all entries on the path
312
323
  2. If a `CompactionEntry` is on the path:
313
324
  - Includes the compaction entry first
314
- - Then entries from `firstKeptEntryId` to compaction
315
- - Then entries after compaction
325
+ - If `retainedTail` is present, it acts as a self-contained checkpoint and entries after the compaction are included
326
+ - Otherwise entries from `firstKeptEntryId` to the compaction are included
327
+ - Then entries after compaction are included
316
328
  3. Preserves non-message entries in the selected range so interactive mode can render them
317
329
 
318
330
  `buildSessionContext()` builds on that entry list to produce the message list for the LLM:
@@ -320,11 +332,13 @@ Entries form a tree:
320
332
  1. Extracts current model and thinking level settings from the full path
321
333
  2. Converts selected entries to messages:
322
334
  - `message` -> stored `AgentMessage`
323
- - `compaction` -> `compactionSummary`
335
+ - `compaction` -> `compactionSummary` plus `retainedTail` when present
324
336
  - `branch_summary` -> `branchSummary`
325
337
  - `custom_message` -> `CustomMessage`
326
338
  - `custom` -> no context message
327
339
 
340
+ This makes newer compactions act like self-contained checkpoints. `retainedTail` is optional only so older sessions that only store `firstKeptEntryId` continue to load correctly.
341
+
328
342
  ## Parsing Example
329
343
 
330
344
  ```typescript
package/docs/tui.md CHANGED
@@ -90,19 +90,32 @@ Without this propagation, typing with an IME (Chinese, Japanese, Korean, etc.) w
90
90
 
91
91
  ```typescript
92
92
  pi.on("session_start", async (_event, ctx) => {
93
- const handle = ctx.ui.custom(myComponent);
94
- // handle.requestRender() - trigger re-render
95
- // handle.close() - restore normal UI
93
+ const result = await ctx.ui.custom<string | null>((tui, theme, keybindings, done) =>
94
+ new MyComponent({
95
+ theme,
96
+ keybindings,
97
+ onChange: () => tui.requestRender(),
98
+ onSelect: (value) => done(value),
99
+ onCancel: () => done(null),
100
+ })
101
+ );
96
102
  });
97
103
  ```
98
104
 
99
- **In custom tools** via `pi.ui.custom()`:
105
+ **In custom tools** via `ctx.ui.custom()`:
100
106
 
101
107
  ```typescript
102
- async execute(toolCallId, params, onUpdate, ctx, signal) {
103
- const handle = pi.ui.custom(myComponent);
104
- // ...
105
- handle.close();
108
+ async execute(toolCallId, params, signal, onUpdate, ctx) {
109
+ const result = await ctx.ui.custom<string | null>((tui, theme, keybindings, done) =>
110
+ new MyComponent({
111
+ theme,
112
+ keybindings,
113
+ onChange: () => tui.requestRender(),
114
+ onSelect: (value) => done(value),
115
+ onCancel: () => done(null),
116
+ })
117
+ );
118
+ // Use result...
106
119
  }
107
120
  ```
108
121
 
@@ -374,24 +387,26 @@ Usage in an extension:
374
387
  ```typescript
375
388
  pi.registerCommand("pick", {
376
389
  description: "Pick an item",
377
- handler: async (args, ctx) => {
390
+ handler: async (_args, ctx) => {
378
391
  const items = ["Option A", "Option B", "Option C"];
379
- const selector = new MySelector(items);
380
-
381
- let handle: { close: () => void; requestRender: () => void };
382
-
383
- await new Promise<void>((resolve) => {
384
- selector.onSelect = (item) => {
385
- ctx.ui.notify(`Selected: ${item}`, "info");
386
- handle.close();
387
- resolve();
388
- };
389
- selector.onCancel = () => {
390
- handle.close();
391
- resolve();
392
+ const selected = await ctx.ui.custom<string | null>((tui, _theme, _keybindings, done) => {
393
+ const selector = new MySelector(items);
394
+ selector.onSelect = done;
395
+ selector.onCancel = () => done(null);
396
+
397
+ return {
398
+ render: (width) => selector.render(width),
399
+ handleInput: (data) => {
400
+ selector.handleInput(data);
401
+ tui.requestRender();
402
+ },
403
+ invalidate: () => selector.invalidate(),
392
404
  };
393
- handle = ctx.ui.custom(selector);
394
405
  });
406
+
407
+ if (selected !== null) {
408
+ ctx.ui.notify(`Selected: ${selected}`, "info");
409
+ }
395
410
  }
396
411
  });
397
412
  ```
@@ -486,7 +501,7 @@ class CachedComponent {
486
501
  }
487
502
  ```
488
503
 
489
- Call `invalidate()` when state changes, then `handle.requestRender()` to trigger re-render.
504
+ Call `invalidate()` when state changes, then use the injected `tui.requestRender()` to trigger re-render.
490
505
 
491
506
  ## Invalidation and Theme Changes
492
507
 
@@ -885,9 +900,9 @@ class VimEditor extends CustomEditor {
885
900
 
886
901
  export default function (pi: ExtensionAPI) {
887
902
  pi.on("session_start", (_event, ctx) => {
888
- // Factory receives theme and keybindings from the app
903
+ // Factory receives the TUI, theme, and keybindings from the app
889
904
  ctx.ui.setEditorComponent((tui, theme, keybindings) =>
890
- new VimEditor(theme, keybindings)
905
+ new VimEditor(tui, theme, keybindings)
891
906
  );
892
907
  });
893
908
  }
package/docs/usage.md CHANGED
@@ -11,7 +11,7 @@ The interface has four main areas:
11
11
  - **Startup header** - shortcuts, loaded context files, prompt templates, skills, and extensions
12
12
  - **Messages** - user messages, assistant responses, tool calls, tool results, notifications, errors, and extension UI
13
13
  - **Editor** - where you type; border color indicates the current thinking level
14
- - **Footer** - working directory, session name, token/cache usage, cost, context usage, and current model
14
+ - **Footer** - working directory, session name, token/cache usage, cost, context usage, and current model. Totals include assistant responses, usage reported by tools, and summary generation.
15
15
 
16
16
  The editor can be replaced temporarily by built-in UI such as `/settings` or by custom extension UI.
17
17
 
@@ -37,6 +37,7 @@ Type `/` in the editor to open command completion. Extensions can register custo
37
37
  | Command | Description |
38
38
  |---------|-------------|
39
39
  | `/login`, `/logout` | Manage OAuth or API-key credentials |
40
+ | [`/llama`](llama-cpp.md) | Download, load, and unload llama.cpp router models |
40
41
  | `/model` | Switch models |
41
42
  | `/scoped-models` | Enable/disable models for Ctrl+P cycling |
42
43
  | `/settings` | Thinking level, theme, message delivery, transport |
@@ -162,7 +162,7 @@ export default function (pi: ExtensionAPI) {
162
162
  parameters: Type.Object({
163
163
  name: Type.String({ description: "Name to greet" }),
164
164
  }),
165
- async execute(toolCallId, params, onUpdate, ctx, signal) {
165
+ async execute(toolCallId, params, signal, onUpdate, ctx) {
166
166
  return {
167
167
  content: [{ type: "text", text: `Hello, ${params.name}!` }],
168
168
  details: {},
@@ -116,6 +116,7 @@ ${conversationText}
116
116
  summary,
117
117
  firstKeptEntryId,
118
118
  tokensBefore,
119
+ usage: response.usage,
119
120
  },
120
121
  };
121
122
  } catch (error) {