@oh-my-pi/pi-agent-core 18.3.0 → 18.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/README.md +3 -2
- package/THIRD-PARTY-NOTICES.txt +2 -2
- package/dist/types/agent.d.ts +14 -0
- package/dist/types/compaction/messages.d.ts +9 -0
- package/dist/types/compaction/transcript-tokens.d.ts +14 -1
- package/dist/types/index.d.ts +2 -0
- package/dist/types/live-steering.d.ts +34 -0
- package/dist/types/output-budget.d.ts +43 -0
- package/dist/types/tool-context.d.ts +31 -0
- package/dist/types/types.d.ts +42 -1
- package/package.json +8 -8
- package/src/agent-loop.ts +164 -19
- package/src/agent.ts +79 -4
- package/src/compaction/messages.ts +12 -2
- package/src/compaction/transcript-tokens.ts +31 -1
- package/src/index.ts +4 -0
- package/src/live-steering.ts +89 -0
- package/src/output-budget.ts +130 -0
- package/src/tool-context.ts +49 -0
- package/src/types.ts +42 -1
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Agent side of provider live steering ({@link LiveSteering}).
|
|
3
|
+
*
|
|
4
|
+
* A provider that can put user input into the response it is streaming (OpenAI
|
|
5
|
+
* Responses `response.steer`) pulls queued steering through a
|
|
6
|
+
* {@link LiveSteeringChannel}. The loop records what the provider accepted right
|
|
7
|
+
* after that response, so the transcript matches what the model saw; anything
|
|
8
|
+
* it declined is injected at the next boundary like ordinary steering.
|
|
9
|
+
*/
|
|
10
|
+
import type { LiveSteerClaim, LiveSteering, UserMessage } from "@oh-my-pi/pi-ai";
|
|
11
|
+
import { logger } from "@oh-my-pi/pi-utils";
|
|
12
|
+
import type { AgentMessage } from "./types";
|
|
13
|
+
|
|
14
|
+
/** Steering-queue access for one provider call, supplied by the agent loop. */
|
|
15
|
+
export interface LiveSteeringQueue {
|
|
16
|
+
/** Resolves once steering is queued or `signal` aborts; never consumes. */
|
|
17
|
+
wait(signal: AbortSignal): Promise<void>;
|
|
18
|
+
/** Dequeues the next steering batch. */
|
|
19
|
+
take(signal: AbortSignal): Promise<AgentMessage[]>;
|
|
20
|
+
/**
|
|
21
|
+
* Provider view of `messages` appended to the in-flight call's context, or
|
|
22
|
+
* `undefined` when that view is not purely user messages.
|
|
23
|
+
*/
|
|
24
|
+
toProvider(messages: AgentMessage[], signal: AbortSignal): Promise<UserMessage[] | undefined>;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/** One provider call's {@link LiveSteering} source. */
|
|
28
|
+
export class LiveSteeringChannel implements LiveSteering {
|
|
29
|
+
/** Steering the provider delivered into the in-flight response, in queue order. */
|
|
30
|
+
readonly accepted: AgentMessage[] = [];
|
|
31
|
+
/** Steering taken from the queue but not delivered; always queued after {@link accepted}. */
|
|
32
|
+
readonly deferred: AgentMessage[] = [];
|
|
33
|
+
readonly #queue: LiveSteeringQueue;
|
|
34
|
+
|
|
35
|
+
constructor(queue: LiveSteeringQueue) {
|
|
36
|
+
this.#queue = queue;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
wait(signal: AbortSignal): Promise<void> {
|
|
40
|
+
// Once input is deferred, later input must follow it at the boundary;
|
|
41
|
+
// delivering it live would reorder the user's messages.
|
|
42
|
+
if (this.deferred.length === 0) return this.#queue.wait(signal);
|
|
43
|
+
if (signal.aborted) return Promise.resolve();
|
|
44
|
+
const { promise, resolve } = Promise.withResolvers<void>();
|
|
45
|
+
signal.addEventListener("abort", () => resolve(), { once: true });
|
|
46
|
+
return promise;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
async claim(signal: AbortSignal): Promise<LiveSteerClaim | undefined> {
|
|
50
|
+
if (this.deferred.length > 0 || signal.aborted) return undefined;
|
|
51
|
+
let messages: AgentMessage[];
|
|
52
|
+
try {
|
|
53
|
+
messages = await this.#queue.take(signal);
|
|
54
|
+
} catch (error) {
|
|
55
|
+
// The queue restores what it could not hand over.
|
|
56
|
+
logger.debug("Live steering dequeue failed", {
|
|
57
|
+
error: error instanceof Error ? error.message : String(error),
|
|
58
|
+
});
|
|
59
|
+
return undefined;
|
|
60
|
+
}
|
|
61
|
+
if (messages.length === 0) return undefined;
|
|
62
|
+
let providerMessages: UserMessage[] | undefined;
|
|
63
|
+
try {
|
|
64
|
+
providerMessages = await this.#queue.toProvider(messages, signal);
|
|
65
|
+
} catch (error) {
|
|
66
|
+
logger.debug("Live steering conversion failed", {
|
|
67
|
+
error: error instanceof Error ? error.message : String(error),
|
|
68
|
+
});
|
|
69
|
+
}
|
|
70
|
+
if (!providerMessages) {
|
|
71
|
+
this.deferred.push(...messages);
|
|
72
|
+
return undefined;
|
|
73
|
+
}
|
|
74
|
+
let settled = false;
|
|
75
|
+
return {
|
|
76
|
+
messages: providerMessages,
|
|
77
|
+
accept: () => {
|
|
78
|
+
if (settled) return;
|
|
79
|
+
settled = true;
|
|
80
|
+
this.accepted.push(...messages);
|
|
81
|
+
},
|
|
82
|
+
reject: () => {
|
|
83
|
+
if (settled) return;
|
|
84
|
+
settled = true;
|
|
85
|
+
this.deferred.push(...messages);
|
|
86
|
+
},
|
|
87
|
+
};
|
|
88
|
+
}
|
|
89
|
+
}
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
import type { Context, Model, Tool } from "@oh-my-pi/pi-ai";
|
|
2
|
+
import { stopsOutputAtContextWindow } from "@oh-my-pi/pi-catalog/compat/output-limits";
|
|
3
|
+
import { stringifyJson } from "@oh-my-pi/pi-utils";
|
|
4
|
+
import { findRequestUsageAnchor } from "./compaction/transcript-tokens";
|
|
5
|
+
import type { Tokenizer } from "./tokenizer";
|
|
6
|
+
|
|
7
|
+
/** Smallest output cap {@link fitOutputTokensToContextWindow} will request. */
|
|
8
|
+
export const MIN_FITTED_OUTPUT_TOKENS = 1024;
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Local counts are padded by 1/this before sizing the output cap: the
|
|
12
|
+
* provider's tokenizer can disagree with ours by a few percent, and
|
|
13
|
+
* undercounting reproduces the overflow this guards against. This is a
|
|
14
|
+
* tokenizer-error margin on locally counted text only (provider-reported
|
|
15
|
+
* usage is exact), not a context reserve; compaction's own reserve
|
|
16
|
+
* (`resolveBudgetReserveTokens`) still decides when to compact.
|
|
17
|
+
*/
|
|
18
|
+
const PROMPT_ESTIMATE_MARGIN_DIVISOR = 10;
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* Output cap for a request, so prompt plus output stays inside the model's
|
|
22
|
+
* context window.
|
|
23
|
+
*
|
|
24
|
+
* Chat Completions-style providers (DeepSeek, OpenAI, vLLM, ...) reject a
|
|
25
|
+
* request whose prompt tokens plus `max_tokens` exceed the window. Every
|
|
26
|
+
* request asks for `model.maxTokens` of output by default, so without this a
|
|
27
|
+
* large model output cap (DeepSeek V4: ~384k of a ~1M window) makes every
|
|
28
|
+
* request fail once the prompt passes window minus output cap, long before
|
|
29
|
+
* compaction triggers, and side turns (`/btw`, recaps) have no overflow
|
|
30
|
+
* recovery at all.
|
|
31
|
+
*
|
|
32
|
+
* The prompt size is the provider's own report from the newest trustworthy
|
|
33
|
+
* assistant turn (see {@link findRequestUsageAnchor}) plus a local count of
|
|
34
|
+
* only the messages appended after it; the whole context is counted locally
|
|
35
|
+
* only when no turn can anchor (fresh or freshly rewritten context).
|
|
36
|
+
*
|
|
37
|
+
* Returns `maxTokens` unchanged when the requested cap already fits, the
|
|
38
|
+
* model declares no window, the host ends generation at the window itself
|
|
39
|
+
* instead of rejecting the request (`stops-output-at-context-window`, e.g.
|
|
40
|
+
* Claude 4.5+ on the Claude API), or nothing would be requested (including an
|
|
41
|
+
* OpenRouter-hosted model with no caller cap: the transport omits the catalog
|
|
42
|
+
* default there so each upstream self-caps, and a fitted value would turn into
|
|
43
|
+
* an explicit cap that filters upstreams). Otherwise returns
|
|
44
|
+
* the remaining room (never below {@link MIN_FITTED_OUTPUT_TOKENS}); a
|
|
45
|
+
* prompt that fills the whole window still overflows and is left to the
|
|
46
|
+
* caller's compaction. Near a full window the floor means a turn can stop on
|
|
47
|
+
* `length` instead of failing with a 400.
|
|
48
|
+
*
|
|
49
|
+
* Lives here, not next to the default in pi-ai's `mapOptionsForApi`, because
|
|
50
|
+
* pi-ai has no tokenizer; callers apply it in their `streamFn` (coding-agent
|
|
51
|
+
* does so in its shared settings-aware wrapper).
|
|
52
|
+
*
|
|
53
|
+
* Not fixed: Anthropic budget-thinking transports raise `max_tokens` back to
|
|
54
|
+
* at least the thinking budget plus a fallback buffer downstream
|
|
55
|
+
* (`ensureMaxTokensForThinking`), so a fitted cap below that is overridden
|
|
56
|
+
* and the request can still exceed the window as before.
|
|
57
|
+
*/
|
|
58
|
+
export function fitOutputTokensToContextWindow(
|
|
59
|
+
model: Model,
|
|
60
|
+
context: Context,
|
|
61
|
+
maxTokens: number | undefined,
|
|
62
|
+
tokenizer: Tokenizer,
|
|
63
|
+
): number | undefined {
|
|
64
|
+
if (maxTokens === undefined && omitsDefaultOutputCap(model.compat)) return undefined;
|
|
65
|
+
const requested = maxTokens ?? model.maxTokens;
|
|
66
|
+
const contextWindow = model.contextWindow;
|
|
67
|
+
if (!requested || !contextWindow || contextWindow <= 0) return maxTokens;
|
|
68
|
+
if (stopsOutputAtContextWindow(model)) return maxTokens;
|
|
69
|
+
|
|
70
|
+
const room = contextWindow - countPromptTokens(context, tokenizer);
|
|
71
|
+
if (room >= requested) return maxTokens;
|
|
72
|
+
return Math.max(MIN_FITTED_OUTPUT_TOKENS, room);
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/** OpenRouter hosts drop the catalog default cap unless the endpoint always needs one. */
|
|
76
|
+
function omitsDefaultOutputCap(compat: Model["compat"] | undefined): boolean {
|
|
77
|
+
return (
|
|
78
|
+
compat !== undefined && "isOpenRouterHost" in compat && compat.isOpenRouterHost && !compat.alwaysSendMaxTokens
|
|
79
|
+
);
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* Framing tokens (system prompt, tool definitions) memoized per array: these
|
|
84
|
+
* are stable identities for the life of a turn, so side turns and repeat
|
|
85
|
+
* requests do not re-stringify and re-tokenize them. Length is part of the key
|
|
86
|
+
* to catch in-place growth.
|
|
87
|
+
*/
|
|
88
|
+
const framingCounts = new WeakMap<readonly unknown[], { tokenizer: Tokenizer; length: number; tokens: number }>();
|
|
89
|
+
|
|
90
|
+
function countFraming(items: readonly unknown[] | undefined, tokenizer: Tokenizer, fragments: () => string[]): number {
|
|
91
|
+
if (!items || items.length === 0) return 0;
|
|
92
|
+
const cached = framingCounts.get(items);
|
|
93
|
+
if (cached && cached.tokenizer === tokenizer && cached.length === items.length) return cached.tokens;
|
|
94
|
+
const tokens = tokenizer.countTokens(fragments());
|
|
95
|
+
framingCounts.set(items, { tokenizer, length: items.length, tokens });
|
|
96
|
+
return tokens;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
function toolFragments(tools: readonly Tool[]): string[] {
|
|
100
|
+
const fragments: string[] = [];
|
|
101
|
+
for (const tool of tools) fragments.push(tool.name, tool.description, stringifyJson(tool.parameters) ?? "");
|
|
102
|
+
return fragments;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
function withMargin(localTokens: number): number {
|
|
106
|
+
return localTokens + Math.ceil(localTokens / PROMPT_ESTIMATE_MARGIN_DIVISOR);
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
/** Provider-anchored prompt size; falls back to a full local count when nothing anchors. */
|
|
110
|
+
function countPromptTokens(context: Context, tokenizer: Tokenizer): number {
|
|
111
|
+
const { messages } = context;
|
|
112
|
+
const anchor = findRequestUsageAnchor(messages);
|
|
113
|
+
if (!anchor) return withMargin(countContextTokens(context, tokenizer));
|
|
114
|
+
let tail = 0;
|
|
115
|
+
for (let index = anchor.index + 1; index < messages.length; index++) {
|
|
116
|
+
tail += tokenizer.countMessage(messages[index]);
|
|
117
|
+
}
|
|
118
|
+
return anchor.tokens + withMargin(tail);
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
function countContextTokens(context: Context, tokenizer: Tokenizer): number {
|
|
122
|
+
const { systemPrompt, tools, inactiveTools } = context;
|
|
123
|
+
return (
|
|
124
|
+
countFraming(systemPrompt, tokenizer, () => [...(systemPrompt ?? [])]) +
|
|
125
|
+
countFraming(tools, tokenizer, () => toolFragments(tools ?? [])) +
|
|
126
|
+
// Anthropic replays retired tool definitions, so they are prompt too.
|
|
127
|
+
countFraming(inactiveTools, tokenizer, () => toolFragments(inactiveTools ?? [])) +
|
|
128
|
+
tokenizer.countMessages(context.messages)
|
|
129
|
+
);
|
|
130
|
+
}
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
import type { ToolResultMessage } from "@oh-my-pi/pi-ai";
|
|
2
|
+
import type { AgentMessage } from "./types";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Symbol-keyed carrier for passive context reported by a tool executed outside
|
|
6
|
+
* the agent loop (Cursor exec-channel dispatch). The executor attaches the
|
|
7
|
+
* joined context to the {@link ToolResultMessage} it returns; `Agent` reads it
|
|
8
|
+
* when the provider hands the result back and injects it after the buffered
|
|
9
|
+
* results. Symbol keys never serialize, so the context cannot leak into the
|
|
10
|
+
* persisted tool result.
|
|
11
|
+
*/
|
|
12
|
+
export const TOOL_RESULT_ADDITIONAL_CONTEXT = Symbol("tool-result-additional-context");
|
|
13
|
+
|
|
14
|
+
/** A tool result optionally carrying {@link TOOL_RESULT_ADDITIONAL_CONTEXT}. */
|
|
15
|
+
export type ToolResultWithAdditionalContext = ToolResultMessage & { [TOOL_RESULT_ADDITIONAL_CONTEXT]?: string };
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* True for a passive-context value worth delivering: a string with at least
|
|
19
|
+
* one non-whitespace character. Shared by every producer and aggregation site
|
|
20
|
+
* so blank values never produce a developer message.
|
|
21
|
+
*/
|
|
22
|
+
export function isNonBlankContext(value: unknown): value is string {
|
|
23
|
+
return typeof value === "string" && value.trim().length > 0;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* Join passive context values in order, dropping blanks. Returns undefined
|
|
28
|
+
* when nothing remains.
|
|
29
|
+
*/
|
|
30
|
+
export function joinAdditionalContext(values: Iterable<string | undefined>): string | undefined {
|
|
31
|
+
const kept: string[] = [];
|
|
32
|
+
for (const value of values) {
|
|
33
|
+
if (isNonBlankContext(value)) kept.push(value);
|
|
34
|
+
}
|
|
35
|
+
return kept.length > 0 ? kept.join("\n\n") : undefined;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Build the developer message that carries passive tool context to the next
|
|
40
|
+
* provider request. Emitted after the tool results it belongs to.
|
|
41
|
+
*/
|
|
42
|
+
export function createAdditionalContextMessage(text: string): AgentMessage {
|
|
43
|
+
return {
|
|
44
|
+
role: "developer",
|
|
45
|
+
content: [{ type: "text", text }],
|
|
46
|
+
attribution: "agent",
|
|
47
|
+
timestamp: Date.now(),
|
|
48
|
+
};
|
|
49
|
+
}
|
package/src/types.ts
CHANGED
|
@@ -69,6 +69,12 @@ export interface AgentTurnEndContext {
|
|
|
69
69
|
message: AgentMessage;
|
|
70
70
|
/** Tool results produced by this turn, already paired with `message` in the live context. */
|
|
71
71
|
toolResults: ToolResultMessage[];
|
|
72
|
+
/**
|
|
73
|
+
* Passive model-visible messages appended after the tool results at this
|
|
74
|
+
* boundary. The agent loop always sends an array (possibly empty);
|
|
75
|
+
* absent is equivalent to empty for hosts that construct the context.
|
|
76
|
+
*/
|
|
77
|
+
additionalMessages?: AgentMessage[];
|
|
72
78
|
/** True when the current tool-loop batch is continuing without yielding to post-turn steering. */
|
|
73
79
|
willContinue: boolean;
|
|
74
80
|
}
|
|
@@ -344,7 +350,11 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
|
|
|
344
350
|
|
|
345
351
|
/**
|
|
346
352
|
* Provides tool execution context, resolved per tool call.
|
|
347
|
-
* Use for late-bound UI or session state access.
|
|
353
|
+
* Use for late-bound UI or session state access. The loop passes the tool
|
|
354
|
+
* call's {@link ToolCallContext}; hosts that support passive tool context
|
|
355
|
+
* surface its `addAdditionalContext` sink as
|
|
356
|
+
* {@link AgentToolContext.addAdditionalContext}. The returned object is
|
|
357
|
+
* handed to the tool as-is.
|
|
348
358
|
*/
|
|
349
359
|
getToolContext?: (toolCall?: ToolCallContext) => AgentToolContext | undefined;
|
|
350
360
|
|
|
@@ -422,6 +432,12 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
|
|
|
422
432
|
* model's text output back into canonical `toolCall` blocks.
|
|
423
433
|
*/
|
|
424
434
|
dialect?: Dialect;
|
|
435
|
+
/**
|
|
436
|
+
* Per-call owned-dialect resolver, read once per LLM call with the model
|
|
437
|
+
* being requested. Authoritative when set: its return value (including
|
|
438
|
+
* `undefined` = native tool calling) replaces the static {@link dialect}.
|
|
439
|
+
*/
|
|
440
|
+
getDialect?: (model: Model) => Dialect | undefined;
|
|
425
441
|
/**
|
|
426
442
|
* When owned (in-band) tool calling is active and the model starts
|
|
427
443
|
* fabricating a tool result inside its own turn, control how the loop reacts:
|
|
@@ -615,6 +631,13 @@ export interface ToolCallContext {
|
|
|
615
631
|
* always safe (the message injects at the next batch boundary).
|
|
616
632
|
*/
|
|
617
633
|
steeringSignal?: AbortSignal;
|
|
634
|
+
/**
|
|
635
|
+
* Loop-owned sink for passive context reported while this call executes.
|
|
636
|
+
* Values join the call's context at the batch boundary and are injected
|
|
637
|
+
* after the batch's tool results, in assistant tool-call order, before the
|
|
638
|
+
* next provider request. Blank values are ignored.
|
|
639
|
+
*/
|
|
640
|
+
addAdditionalContext?: (context: string) => void;
|
|
618
641
|
}
|
|
619
642
|
|
|
620
643
|
/** A single tool-call content block emitted by an assistant message. */
|
|
@@ -817,11 +840,19 @@ export interface SpeculativeToolExecutionConfig {
|
|
|
817
840
|
* written back to the tool-call block on the assistant message, and seen by
|
|
818
841
|
* history, scheduling, execution events, and `tool.execute` alike. It is
|
|
819
842
|
* ignored when `block` is true.
|
|
843
|
+
*
|
|
844
|
+
* Set `additionalContext` to attach passive model-visible context to this call.
|
|
845
|
+
* Non-empty values from a tool batch are injected in assistant tool-call order
|
|
846
|
+
* after every result settles and before the next provider request. It is
|
|
847
|
+
* dropped when the call is blocked or skipped, or when its final result is an
|
|
848
|
+
* error (including an approval denial raised by the tool's own gate). Within a
|
|
849
|
+
* call it follows any context the tool reported during execution.
|
|
820
850
|
*/
|
|
821
851
|
export interface BeforeToolCallResult {
|
|
822
852
|
block?: boolean;
|
|
823
853
|
reason?: string;
|
|
824
854
|
args?: Record<string, unknown>;
|
|
855
|
+
additionalContext?: string;
|
|
825
856
|
}
|
|
826
857
|
|
|
827
858
|
/**
|
|
@@ -989,6 +1020,16 @@ export type ToolApproval = ToolApprovalDecision | ((args: unknown) => ToolApprov
|
|
|
989
1020
|
* Apps can extend via declaration merging.
|
|
990
1021
|
*/
|
|
991
1022
|
export interface AgentToolContext {
|
|
1023
|
+
/**
|
|
1024
|
+
* Attach trusted, agent-authored instructions to the next provider request.
|
|
1025
|
+
* The host emits them after tool results with developer/system priority where
|
|
1026
|
+
* the selected transport supports it. Do not use this channel for raw tool
|
|
1027
|
+
* output, retrieved documents, web content, or other untrusted data; return
|
|
1028
|
+
* those through the ordinary tool result instead. Hosts populate it from
|
|
1029
|
+
* {@link ToolCallContext.addAdditionalContext} (or their own collector for
|
|
1030
|
+
* calls dispatched outside the loop); absent when the host has no sink.
|
|
1031
|
+
*/
|
|
1032
|
+
addAdditionalContext?(context: string): void;
|
|
992
1033
|
/** Present only while the matching outer tool owns its finalized stream session. */
|
|
993
1034
|
[SPECULATIVE_STREAM_SESSION]?: ToolSpeculationStreamSession;
|
|
994
1035
|
}
|