@matthewfl/pi-contemplator 0.1.15 → 0.1.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@matthewfl/pi-contemplator",
3
- "version": "0.1.15",
3
+ "version": "0.1.16",
4
4
  "description": "A Pi extension that keeps long-running agentic sessions on track with background memory, contemplation, and structural review.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -50,11 +50,11 @@
50
50
  "@earendil-works/pi-tui": "*"
51
51
  },
52
52
  "devDependencies": {
53
- "@earendil-works/pi-agent-core": "^0.85.0",
54
- "@earendil-works/pi-ai": "^0.85.0",
55
- "@earendil-works/pi-coding-agent": "^0.85.0",
56
- "@earendil-works/pi-server": "^0.85.0",
57
- "@earendil-works/pi-tui": "^0.85.0",
53
+ "@earendil-works/pi-agent-core": "^0.85.1",
54
+ "@earendil-works/pi-ai": "^0.85.1",
55
+ "@earendil-works/pi-coding-agent": "^0.85.1",
56
+ "@earendil-works/pi-server": "^0.85.1",
57
+ "@earendil-works/pi-tui": "^0.85.1",
58
58
  "@types/node": "^22.0.0",
59
59
  "typebox": "^1.1.38",
60
60
  "typescript": "^5.6.0",
@@ -4,6 +4,7 @@ import { Type } from "@earendil-works/pi-ai";
4
4
  import { streamSimple } from "@earendil-works/pi-ai/compat";
5
5
  import type { Static } from "typebox";
6
6
  import { hashId } from "../../ids.js";
7
+ import { replayTruncatedThinkingAsText } from "../replay-truncated-thinking.js";
7
8
  import { logAgentStreamError } from "../stream-errors.js";
8
9
  import { OBSERVER_AGENT_LOOP_MAX_TOKENS, boundedMaxTokens } from "../../model-budget.js";
9
10
  import { OBSERVER_SYSTEM } from "./prompts.js";
@@ -205,7 +206,7 @@ IMPORTANT: Now call record_observations to record the useful new observations fr
205
206
  apiKey,
206
207
  headers,
207
208
  maxTokens: boundedMaxTokens(model, OBSERVER_AGENT_LOOP_MAX_TOKENS),
208
- convertToLlm: (msgs) => msgs as Message[],
209
+ convertToLlm: replayTruncatedThinkingAsText,
209
210
  toolExecution: "sequential",
210
211
  shouldStopAfterTurn: () => {
211
212
  turnCount++;
@@ -270,9 +271,11 @@ IMPORTANT: Now call record_observations to record the useful new observations fr
270
271
  // maximum. agentLoop stops on `length` when no tool call was completed; it
271
272
  // does not automatically send a continuation request. Preserve the partial
272
273
  // response so the model can continue from work it already performed rather
273
- // than paying to reproduce it, then append a short tool-focused instruction
274
- // and reduce reasoning to minimal. A second length stop fails forward at the
275
- // bounded-chunk level.
274
+ // than paying to reproduce it. Plaintext thinking is replayed as ordinary
275
+ // assistant text at the LLM boundary because some provider templates strip
276
+ // historical reasoning; encrypted thinking retains its opaque structure.
277
+ // Then append a short tool-focused instruction and reduce reasoning to
278
+ // minimal. A second length stop fails forward at the bounded-chunk level.
276
279
  terminalFailure = undefined;
277
280
  const retryPrompt: Message = {
278
281
  role: "user",
@@ -0,0 +1,21 @@
1
+ import type { AgentMessage } from "@earendil-works/pi-agent-core";
2
+ import type { Message } from "@earendil-works/pi-ai";
3
+
4
+ /**
5
+ * Many provider chat templates discard historical reasoning blocks. Preserve
6
+ * unfinished plaintext work after an output-length stop by replaying it as
7
+ * ordinary assistant text. Redacted/encrypted blocks remain structured so
8
+ * their opaque provider payload stays replayable. The original transcript is
9
+ * never mutated; this transformation is only applied at the LLM boundary.
10
+ */
11
+ export function replayTruncatedThinkingAsText(messages: readonly AgentMessage[]): Message[] {
12
+ return messages.map((message) => {
13
+ if (message.role !== "assistant" || message.stopReason !== "length" || !message.content.some((part) => part.type === "thinking" && !part.redacted)) return message as Message;
14
+ return {
15
+ ...message,
16
+ content: message.content.map((part) => part.type === "thinking" && !part.redacted
17
+ ? { type: "text" as const, text: `[Incomplete analysis from the preceding truncated response]\n${part.thinking}` }
18
+ : part),
19
+ } as Message;
20
+ });
21
+ }
@@ -26,6 +26,7 @@ import {
26
26
  import { estimateStringTokens } from "../../tokens.js";
27
27
  import { createRecallAgentTool } from "../../tools/recall-observation.js";
28
28
  import { createSearchMemoriesAgentTool } from "../../tools/search-memories.js";
29
+ import { replayTruncatedThinkingAsText } from "../replay-truncated-thinking.js";
29
30
  import { logAgentStreamError } from "../stream-errors.js";
30
31
  import { summarizerContinue, SUMMARIZER_SYSTEM } from "./prompts.js";
31
32
  import {
@@ -106,27 +107,6 @@ function textResult(text: string, details: Record<string, unknown> = {}, termina
106
107
  return { content: [{ type: "text" as const, text }], details, ...(terminate ? { terminate: true } : {}) };
107
108
  }
108
109
 
109
- /**
110
- * Many provider chat templates discard historical reasoning blocks. Preserve
111
- * unfinished plaintext work after an output-length stop by replaying it as
112
- * ordinary assistant text; unlike provider-specific thinking metadata, text
113
- * survives every supported conversation serializer. Redacted/encrypted blocks
114
- * must remain structured so their opaque provider payload stays replayable.
115
- * The durable/in-memory transcript remains unchanged—this transformation is
116
- * only applied at the LLM boundary.
117
- */
118
- export function replayTruncatedThinkingAsText(messages: readonly AgentMessage[]): Message[] {
119
- return messages.map((message) => {
120
- if (message.role !== "assistant" || message.stopReason !== "length" || !message.content.some((part) => part.type === "thinking" && !part.redacted)) return message as Message;
121
- return {
122
- ...message,
123
- content: message.content.map((part) => part.type === "thinking" && !part.redacted
124
- ? { type: "text" as const, text: `[Incomplete analysis from the preceding truncated response]\n${part.thinking}` }
125
- : part),
126
- } as Message;
127
- });
128
- }
129
-
130
110
  function preview(content: string): string {
131
111
  const compact = content.replace(/\s+/g, " ").trim();
132
112
  return compact.length <= 100 ? compact : `${compact.slice(0, 100)}…`;
@@ -62,10 +62,6 @@ function truncateStatusText(value: string, limit = 1_000): string {
62
62
  return value.length <= limit ? value : `${value.slice(0, limit - 1)}…`;
63
63
  }
64
64
 
65
- function tokenSum(items: { tokenCount: number }[]): number {
66
- return items.reduce((sum, item) => sum + item.tokenCount, 0);
67
- }
68
-
69
65
  function addedSuffix(count: number): string | undefined {
70
66
  return count > 0 ? `+${count.toLocaleString()}` : undefined;
71
67
  }
@@ -90,8 +86,6 @@ export function registerStatusCommand(pi: ExtensionAPI, runtime: Runtime): void
90
86
  const full = fullProjection(entries);
91
87
  const drift = diffProjection(visible, full);
92
88
 
93
- const visibleObservationTokens = tokenSum(visible.observations);
94
- const visibleSummaryTokens = tokenSum(visible.summaries);
95
89
  const pools = partitionMemoryPools(folded.activeObservations, folded.activeSummaries, runtime.config.newMemoryPoolMaxTokens);
96
90
  const observationLine = appendSuffixes(
97
91
  `Observations: ${folded.observations.length} recorded / ${folded.activeObservations.length} active / ${visible.observations.length} visible`,
@@ -126,10 +120,9 @@ export function registerStatusCommand(pi: ExtensionAPI, runtime: Runtime): void
126
120
  `Observer source backlog: ~${obsProgress.toLocaleString()} / ${runtime.config.observeAfterTokens.toLocaleString()} tokens (${pct(obsProgress, runtime.config.observeAfterTokens)}%)`,
127
121
  `Summarizer trigger: old pool ~${pools.oldTokens.toLocaleString()} / ${summarizerTrigger.toLocaleString()} tokens (${pct(pools.oldTokens, summarizerTrigger)}%)`,
128
122
  `Automatic compaction source backlog: ~${compactionProgress.toLocaleString()} / ${compactThreshold.toLocaleString()} tokens (${pct(compactionProgress, compactThreshold)}%; injected memory excluded)`,
129
- `Visible observation pool: ~${visibleObservationTokens.toLocaleString()} tokens`,
123
+ `Active memory total: ~${pools.totalTokens.toLocaleString()} tokens (observations + summaries; split below)`,
130
124
  `New memory pool: ~${pools.newTokens.toLocaleString()} / ${runtime.config.newMemoryPoolMaxTokens.toLocaleString()} protection-budget tokens (${pct(pools.newTokens, runtime.config.newMemoryPoolMaxTokens)}%; newest memory always protected whole)`,
131
- `Old memory pool: ~${pools.oldTokens.toLocaleString()} / ${runtime.config.oldMemoryPoolTargetTokens.toLocaleString()} advisory target tokens (${pct(pools.oldTokens, runtime.config.oldMemoryPoolTargetTokens)}%)`,
132
- `Summary pool: ~${visibleSummaryTokens.toLocaleString()} visible tokens`,
125
+ `Old memory pool: ~${pools.oldTokens.toLocaleString()} / ${runtime.config.oldMemoryPoolTargetTokens.toLocaleString()} advisory target tokens (${pct(pools.oldTokens, runtime.config.oldMemoryPoolTargetTokens)}%; observations + summaries)`,
133
126
  `Summarizer: ${runtime.config.summarizerEnabled === false ? "disabled" : "enabled"}; retrigger after +${runtime.config.summarizerRetriggerTokens.toLocaleString()} old-pool tokens / sample above ~${summarizerSamplingTokens.toLocaleString()} tokens`,
134
127
  `Summarizer model: ${configuredModelLabel(runtime.configuredMemoryWorkerModel("summarizer"))}`,
135
128
  `Observer model: ${configuredModelLabel(runtime.configuredMemoryWorkerModel("observer"))}`,