bermudis-pi-goodies 0.23.2 → 0.23.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (4) hide show
  1. package/README.md +22 -10
  2. package/clean-tui.ts +127 -33
  3. package/goodies.ts +10 -25
  4. package/package.json +1 -1
package/README.md CHANGED
@@ -25,7 +25,7 @@ extensions. One entry point, twelve independent features.
25
25
  After publishing the package to npm:
26
26
 
27
27
  ```bash
28
- pi install npm:bermudis-pi-goodies@0.23.2
28
+ pi install npm:bermudis-pi-goodies@0.23.4
29
29
  ```
30
30
 
31
31
  Remove any old `bermudis-pi-goodies.ts` symlink before reloading Pi. Each
@@ -60,7 +60,13 @@ widget line above the editor shows the cause and clears itself on the first
60
60
  success. Transient provider failures (upstream 5xx, stalls, network blips)
61
61
  are retried in place — up to three tries with a progressive pause of 1s,
62
62
  5s, then 10s — before any of that engages, so a single blip costs nothing;
63
- rate limits and other 4xx go straight to the pause. Summary log events carry
63
+ rate limits and other 4xx go straight to the pause. A response whose whole
64
+ budget went to reasoning (thinking blocks but no answer — how router models
65
+ like `kilo-auto/free` fail when their random upstream is a thinking model
66
+ that ignores effort hints) gets one in-place retry at a raised 4096-token
67
+ cap before the pause; the retry is logged as a `summary_reasoning_retry`
68
+ event carrying the upstream the router actually picked, so the log names
69
+ who ate the budget. Summary log events carry
64
70
  a `kind` field — `bash` or `thinking` — so `jq -r 'select(.type ==
65
71
  "summary_request") | [.kind, .outcome] | @tsv'` splits request counts by
66
72
  feature. The log at
@@ -74,10 +80,15 @@ there first; e.g. `jq -r 'select(.type == "summary_request") | .outcome'
74
80
 
75
81
  Two practical notes:
76
82
 
77
- - **Thinking models are handled, non-thinking ones are cheaper.** Requests
78
- pin the model's lowest reasoning effort and carry a 512-token budget that
79
- covers thinking plus the answer, so reasoning models work; a quick
80
- chat-class model still costs the least.
83
+ - **Thinking models are handled, non-thinking ones are cheaper.** The
84
+ reasoning effort comes from the model's own declared levels — the catalog
85
+ says what each model supports and the request is clamped to that, in the
86
+ model's own dialect. Models that declare nothing (routers like
87
+ `kilo-auto/free`) get the documented floor `low` — never an undocumented
88
+ word a gateway might silently drop — plus a 512-token budget covering
89
+ thinking plus the answer, with the raised-cap retry above for upstreams
90
+ that ignore it. A quick chat-class model still costs the least and can't
91
+ fail this way.
81
92
  - **Privacy:** qualifying commands (longer than 80 characters) are sent —
82
93
  first ~2000 characters — to whichever provider hosts the model you chose.
83
94
  That is the same trust decision as running an agent session against that
@@ -118,11 +129,12 @@ commands it is logged as a digest, never raw text.
118
129
 
119
130
  Two honest trade-offs:
120
131
 
121
- - **Separate opt-in, default off every launch.** A bash summary is one request per unique
132
+ - **Separate opt-in, off until you enable it.** A bash summary is one request per unique
122
133
  command; a thinking summary recurs for as long as the model reasons.
123
- Same provider, different volume — so `thinking-summaries on` asks for
124
- that explicitly, for this pi run only (off again next launch, never saved).
125
- It takes effect immediately, no `/reload`.
134
+ Same provider, different volume — so thinking summaries are a separate
135
+ toggle (`/goodies thinking-summaries on|off`), default off, persisted to
136
+ `goodies.json` like every other toggle. It takes effect immediately, no
137
+ `/reload`.
126
138
  - **It does not replace the `Thinking...` row itself.** Pi's only seam for
127
139
  that text is one global label applied to every assistant message at
128
140
  once — updating it mid-stream would rewrite every past thinking row, and
package/clean-tui.ts CHANGED
@@ -44,11 +44,12 @@ import { Box, Container, Text } from "@earendil-works/pi-tui";
44
44
  import { homedir } from "node:os";
45
45
  import { readFileSync } from "node:fs";
46
46
  import { completeSimple } from "@earendil-works/pi-ai/compat";
47
- import type {
48
- Api,
49
- AssistantMessage,
50
- Model,
51
- ThinkingLevel,
47
+ import {
48
+ clampThinkingLevel,
49
+ type Api,
50
+ type AssistantMessage,
51
+ type Model,
52
+ type ThinkingLevel,
52
53
  } from "@earendil-works/pi-ai";
53
54
  import { logGoodiesEvent, setGoodiesLogPathForTesting } from "./goodies-log.ts";
54
55
  import { describeError } from "./json-file.ts";
@@ -424,7 +425,12 @@ const SUMMARY_THRESHOLD_CHARS = 80;
424
425
  // old 30-token cap, gpt-oss running its API-default effort returned an empty
425
426
  // summary every time. 512 gives reasoning headroom; non-reasoning models
426
427
  // stop at the answer's natural end, so the raised cap costs them nothing.
428
+ // When even 512 is out-thought — router models (kilo-auto/free, openrouter
429
+ // routers) hop between upstreams per request and some ignore effort hints
430
+ // entirely — completeSummaryTurn escalates once to the cap below before
431
+ // declaring the empty-summary failure.
427
432
  const SUMMARY_MAX_TOKENS = 512;
433
+ const SUMMARY_REASONING_RETRY_MAX_TOKENS = 4096;
428
434
  const SUMMARY_PROMPT =
429
435
  "Summarize this shell command in less than 13 words, plain English, no quotes, no formatting. " +
430
436
  'Examples: "cat >> file << \'EOF\' with 20 lines of log" -> "Appends reboot log to migration file". ' +
@@ -692,6 +698,41 @@ async function resolveSummaryTransport(): Promise<{
692
698
  return { model: found, label, apiKey: auth.apiKey, headers };
693
699
  }
694
700
 
701
+ /** Whether any content block carries non-empty answer text. */
702
+ function hasAnswerText(response: {
703
+ content: Array<{ type: string; text?: string }>;
704
+ }): boolean {
705
+ return response.content.some(
706
+ (c) => c.type === "text" && (c.text ?? "").trim() !== "",
707
+ );
708
+ }
709
+
710
+ /**
711
+ * Whether a textless response looks like "reasoning consumed the shared
712
+ * completion budget": thinking blocks arrived (their text lives in the
713
+ * `thinking` field, not `text`), or the provider hit the token ceiling
714
+ * before any answer existed. Routers (kilo-auto/free, openrouter/*) make
715
+ * this nondeterministic — every request can land on a different upstream,
716
+ * and thinking upstreams may ignore reasoning-effort hints outright.
717
+ */
718
+ function reasoningAteBudget(response: {
719
+ stopReason: string;
720
+ content: Array<{
721
+ type: string;
722
+ text?: string;
723
+ thinking?: string;
724
+ redacted?: boolean;
725
+ }>;
726
+ }): boolean {
727
+ if (hasAnswerText(response)) return false;
728
+ if (response.stopReason === "length") return true;
729
+ return response.content.some(
730
+ (c) =>
731
+ c.type === "thinking" &&
732
+ ((c.thinking ?? c.text ?? "").trim() !== "" || c.redacted === true),
733
+ );
734
+ }
735
+
695
736
  /**
696
737
  * Convert a pi-ai completion response into a summary string or throw the
697
738
  * error shape the shared backoff/log/widget path expects. Extracted from
@@ -702,14 +743,17 @@ async function resolveSummaryTransport(): Promise<{
702
743
  *
703
744
  * - "aborted" → AbortError (per-request abort, not a provider failure)
704
745
  * - "error" → Error with the provider's errorMessage (truncated) + label
705
- * - empty text content → Error diagnosing a thinking-model that ate the budget
746
+ * - empty text after reasoning activity → Error naming the eaten budget
747
+ * (completeSummaryTurn has already retried at a raised cap by this point)
748
+ * - empty text without reasoning activity → the generic thinking-model
749
+ * diagnosis
706
750
  * - anything else → the joined text content
707
751
  */
708
752
  export function convertSummaryResponse(
709
753
  response: {
710
754
  stopReason: string;
711
755
  errorMessage?: string;
712
- content: Array<{ type: string; text?: string }>;
756
+ content: Array<{ type: string; text?: string; thinking?: string }>;
713
757
  },
714
758
  label: string,
715
759
  ): string {
@@ -726,10 +770,16 @@ export function convertSummaryResponse(
726
770
  .filter((c): c is { type: "text"; text: string } => c.type === "text")
727
771
  .map((c) => c.text)
728
772
  .join("\n");
729
- if (!text.trim())
773
+ if (!text.trim()) {
774
+ if (reasoningAteBudget(response)) {
775
+ throw new Error(
776
+ `empty summary — ${label} spent the raised token budget on reasoning; try a summary model that can disable thinking`,
777
+ );
778
+ }
730
779
  throw new Error(
731
780
  `empty summary — is ${label} a thinking model that cannot disable thinking?`,
732
781
  );
782
+ }
733
783
  return text;
734
784
  }
735
785
 
@@ -742,60 +792,104 @@ function activeBackend(): SummaryBackend {
742
792
  );
743
793
  }
744
794
 
745
- // Lowest-effort reasoning, but only where silence is broken: on OpenAI-
746
- // compatible endpoints a reasoning model with no mapped "off" (Groq's
747
- // gpt-oss maps off and minimal to null) runs its API-default effort when no
748
- // reasoning parameter is sent — medium for gpt-oss — which burns the shared
749
- // completion budget and returns an empty answer. "minimal" clamps to the
750
- // model's lowest supported level ("low" for gpt-oss). Everything else keeps
751
- // the absent option: non-reasoning models clamp to "off" (no parameter),
752
- // models whose catalog maps "off" to a concrete value already disable
753
- // thinking when nothing is sent, and non-OpenAI adapters enable thinking on
754
- // truthy values (see the retired no-reasoning note in git history).
795
+ // Lowest-effort reasoning, but only where silence is broken — and always the
796
+ // model's OWN declared capability, never a hardcoded guess:
797
+ //
798
+ // - Non-reasoning models, non-OpenAI adapters (a truthy effort ENABLES
799
+ // thinking there), and models whose map gives "off" a concrete wire
800
+ // value: send nothing. The adapter emits the declared off value itself.
801
+ // - Models with a capability map: request "minimal". completeSimple clamps
802
+ // against the map before anything reaches the wire (pi-ai streamSimple),
803
+ // so the model only ever sees a level it declares, translated to its own
804
+ // dialect word — whether the catalog says low..max, only off+high, or
805
+ // maps minimal to something else entirely.
806
+ // - Map-less models (routers: kilo-auto/free, openrouter/*): there is no
807
+ // translation layer, so the raw word must be one gateways actually
808
+ // document. "minimal" is not — OpenRouter-style wires silently drop it
809
+ // and the random upstream runs its DEFAULT effort, which is exactly the
810
+ // reasoning-eats-the-budget failure. Send the documented floor "low";
811
+ // the raised-cap retry absorbs upstreams that ignore even that.
755
812
  function summaryReasoning(model: Model<Api>): ThinkingLevel | undefined {
756
813
  if (model.api !== "openai-completions" && model.api !== "openai-responses")
757
814
  return undefined;
758
815
  if (!model.reasoning) return undefined;
759
816
  if (typeof model.thinkingLevelMap?.off === "string") return undefined;
817
+ if (model.thinkingLevelMap === undefined) return "low";
760
818
  return "minimal";
761
819
  }
762
820
 
763
821
  /**
764
822
  * Run one summary completion through pi-ai, shared by both summary kinds so
765
- * the reasoning_effort compatibility retry lives in exactly one place.
823
+ * the reasoning-effort compatibility retry and the reasoning-budget retry
824
+ * live in exactly one place.
766
825
  *
767
- * Some OpenAI-compatible endpoints validate reasoning_effort against their
768
- * own enum and reject "minimal" outright — observed on Command Code
769
- * (commandcode/poolside/*): 400 invalid_request_error, param:"reasoning_effort",
770
- * accepted values low|medium|high|xhigh|max. "low" is the floor of every known
771
- * enum, so retry once there before surfacing the failure; an endpoint that
772
- * rejects "low" too would pause as before.
826
+ * Retry 1 — enum compatibility: some OpenAI-compatible endpoints validate
827
+ * reasoning_effort against their own enum and reject "minimal" outright —
828
+ * observed on Command Code (commandcode/poolside/*): 400 invalid_request_error,
829
+ * param:"reasoning_effort", accepted values low|medium|high|xhigh|max. "low"
830
+ * is the floor of every known enum, so retry once there before surfacing the
831
+ * failure; an endpoint that rejects "low" too would pause as before.
832
+ *
833
+ * Retry 2 — reasoning budget: a response with no answer text whose budget
834
+ * went to thinking (thinking blocks present, or a length cutoff) gets one
835
+ * second chance at a raised cap. This is the router case (kilo-auto/free):
836
+ * each request lands on a random pool upstream (DeepSeek, Nemotron, Qwen,
837
+ * ...) that may ignore the effort hint entirely — no dialect word controls
838
+ * another vendor's thinking. The retry clamps the effort to the model's
839
+ * declared floor (never an undeclared word) and raises the cap — the cap is
840
+ * the lever that works regardless of dialect. Models where summaryReasoning
841
+ * sent no effort keep it absent: for non-OpenAI adapters a truthy effort
842
+ * ENABLES thinking. Still empty after this lands as the raised-budget
843
+ * diagnosis in convertSummaryResponse; the outer timeout still bounds the
844
+ * whole turn, so a slow retry degrades to the ordinary retryable-timeout
845
+ * path.
773
846
  */
774
847
  async function completeSummaryTurn(
775
848
  t: Awaited<ReturnType<typeof resolveSummaryTransport>>,
776
849
  context: Parameters<typeof completeSimple>[1],
777
850
  signal: AbortSignal,
778
851
  ): Promise<AssistantMessage> {
779
- const options = (reasoning: ThinkingLevel | undefined) => ({
852
+ const firstEffort = summaryReasoning(t.model);
853
+ const options = (
854
+ reasoning: ThinkingLevel | undefined,
855
+ maxTokens: number = SUMMARY_MAX_TOKENS,
856
+ ) => ({
780
857
  apiKey: t.apiKey,
781
858
  headers: t.headers,
782
- maxTokens: SUMMARY_MAX_TOKENS,
859
+ maxTokens,
783
860
  signal,
784
861
  reasoning,
785
862
  });
786
863
  // completeSimple does NOT throw for HTTP errors — it returns an
787
864
  // AssistantMessage with stopReason:"error" + errorMessage, so the
788
865
  // compatibility check inspects the response, not a catch block.
789
- const response = await completeSimple(
790
- t.model,
791
- context,
792
- options(summaryReasoning(t.model)),
793
- );
866
+ let response = await completeSimple(t.model, context, options(firstEffort));
794
867
  if (
795
868
  response.stopReason === "error" &&
796
869
  (response.errorMessage ?? "").includes("reasoning_effort")
797
870
  ) {
798
- return await completeSimple(t.model, context, options("low"));
871
+ response = await completeSimple(t.model, context, options("low"));
872
+ }
873
+ if (reasoningAteBudget(response)) {
874
+ logGoodiesEvent({
875
+ type: "summary_reasoning_retry",
876
+ model: t.label,
877
+ // The upstream a router actually picked — the response's own model
878
+ // field is the only way to know who ate the budget.
879
+ ...(response.responseModel
880
+ ? { responseModel: response.responseModel }
881
+ : {}),
882
+ });
883
+ response = await completeSimple(
884
+ t.model,
885
+ context,
886
+ options(
887
+ firstEffort === undefined
888
+ ? undefined
889
+ : (clampThinkingLevel(t.model, "low") as ThinkingLevel),
890
+ SUMMARY_REASONING_RETRY_MAX_TOKENS,
891
+ ),
892
+ );
799
893
  }
800
894
  return response;
801
895
  }
package/goodies.ts CHANGED
@@ -8,8 +8,7 @@
8
8
  * State persists to ~/.pi/agent/goodies.json. Toggling a feature writes the
9
9
  * config but does NOT unload/reload the extension — every feature is registered
10
10
  * at load time, so a toggle needs `/reload` or a new session to take effect.
11
- * Exceptions read at request time: `summary-model`, and `thinking-summaries`
12
- * which is session-only (off on every pi launch, never persisted — token spend).
11
+ * Exceptions read at request time: `summary-model` and `thinking-summaries`.
13
12
  */
14
13
  import type { Api, Model } from "@earendil-works/pi-ai";
15
14
  import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
@@ -26,16 +25,9 @@ import { rankCandidates } from "./vision-core.ts";
26
25
 
27
26
  let CONFIG_PATH = join(homedir(), ".pi", "agent", "goodies.json");
28
27
 
29
- /** Session-only flag: thinking summaries default off every pi launch. */
30
- let sessionThinkingEnabled = false;
31
-
32
28
  export function __setConfigPathForTesting(path: string): void {
33
29
  CONFIG_PATH = path;
34
30
  config = loadConfig();
35
- // Thinking summaries are session-only (default off every pi launch), so a
36
- // fresh config scope in tests also means a fresh flag — otherwise an
37
- // earlier test's `on` would leak into later ones sharing the process.
38
- sessionThinkingEnabled = false;
39
31
  }
40
32
 
41
33
  type FeatureName =
@@ -77,10 +69,10 @@ type Config = Partial<Record<FeatureName, boolean>> & {
77
69
  */
78
70
  "summary-model"?: string;
79
71
  /**
80
- * Legacy persisted key. Thinking summaries are now session-only (off on
81
- * every pi launch, on only when `/goodies thinking-summaries on` runs in
82
- * that pi run) because they can add meaningful token spend. The file key
83
- * is ignored; it is deleted on write for migration from older versions.
72
+ * Whether live thinking summaries are on. Persisted like every other
73
+ * toggle; read at request time so `/goodies thinking-summaries on|off`
74
+ * takes effect immediately. Unset means off — the volume (a request per
75
+ * reasoning run, not per command) is why the default stays off.
84
76
  */
85
77
  "thinking-summaries"?: boolean;
86
78
  };
@@ -185,20 +177,13 @@ export function setSummaryModel(model: string | undefined): void {
185
177
  }
186
178
 
187
179
  export function getThinkingSummariesEnabled(): boolean {
188
- return sessionThinkingEnabled; // default off every pi launch
180
+ return config["thinking-summaries"] === true; // default off when unset
189
181
  }
190
182
 
191
183
  export function setThinkingSummariesEnabled(enabled: boolean): void {
192
- sessionThinkingEnabled = enabled;
193
- // Migration: drop any stale persisted key from older versions so it can
194
- // never revive. Best-effort — a failure must not break the toggle.
195
- try {
196
- updateConfig((next) => {
197
- delete next["thinking-summaries"];
198
- });
199
- } catch {
200
- // ignore: the in-memory flag above already took effect
201
- }
184
+ updateConfig((next) => {
185
+ next["thinking-summaries"] = enabled;
186
+ });
202
187
  }
203
188
 
204
189
  // ── Summary-model resolution against pi's model registry ────────────────────
@@ -563,7 +548,7 @@ export default function goodies(pi: ExtensionAPI): void {
563
548
  setThinkingSummariesEnabled(value === "on");
564
549
  const model = getSummaryModel();
565
550
  ctx.ui.notify(
566
- `thinking summaries ${value} for this pi run only (off again next launch). ` +
551
+ `thinking summaries ${value} (persisted). ` +
567
552
  (value === "on" && !model
568
553
  ? `No summary model set yet — run /goodies summary-model <provider/model> or nothing will happen. `
569
554
  : "") +
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "bermudis-pi-goodies",
3
- "version": "0.23.2",
3
+ "version": "0.23.4",
4
4
  "repository": {
5
5
  "type": "git",
6
6
  "url": "git+https://github.com/bermudi/agent-extensions.git",