@oh-my-pi/pi-agent-core 18.1.16 → 18.1.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +13 -0
- package/dist/types/compaction/compaction-v2-streaming.d.ts +2 -2
- package/dist/types/compaction/openai.d.ts +7 -7
- package/package.json +8 -8
- package/src/agent-loop.ts +64 -2
- package/src/compaction/compaction-v2-streaming.ts +2 -2
- package/src/compaction/compaction.ts +44 -63
- package/src/compaction/openai.ts +7 -7
- package/src/proxy.ts +1 -5
- package/src/tokenizer.ts +10 -2
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,19 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [18.1.17] - 2026-09-10
|
|
6
|
+
|
|
7
|
+
### Changed
|
|
8
|
+
|
|
9
|
+
- `Tool <name> not found` now names a plausible intended target when the advertised set contains one, e.g. `Tool mcp__abc123__xyz789_read not found. Did you mean read?`. A model that mis-transcribes a long opaque tool name reliably keeps the trailing segment, which is the only part carrying meaning, so the miss becomes recoverable in the same turn instead of costing a round trip. Purely advisory — the suggestion is only ever a string in the error, never a dispatch target, so an unrecognized name still fails ([#10109](https://github.com/can1357/oh-my-pi/issues/10109) by [@oldschoola](https://github.com/oldschoola)).
|
|
10
|
+
|
|
11
|
+
### Fixed
|
|
12
|
+
|
|
13
|
+
- Fixed the token estimator counting developer messages as free and ignoring images in user content, which let context budgeting, pruning and the compaction trigger read a transcript as far smaller than the one sent to the provider.
|
|
14
|
+
- Fixed repeated local compaction omitting messages retained before the previous compaction record, while preserving original entry IDs and `/clear` boundaries.
|
|
15
|
+
- Raised remote compaction request timeout from 3 minutes to 5 minutes so long Codex/gpt-6-astra compact streams can finish before the watchdog aborts them.
|
|
16
|
+
- Fixed proxy responses dropping the cost the server reported; recorded costs are kept instead of being recomputed.
|
|
17
|
+
|
|
5
18
|
## [18.1.10] - 2026-09-04
|
|
6
19
|
|
|
7
20
|
### Fixed
|
|
@@ -12,8 +12,8 @@ import { type OpenAICodexCompactionBody } from "@oh-my-pi/pi-ai/providers/openai
|
|
|
12
12
|
export declare const V2_RETAINED_MESSAGE_TOKEN_BUDGET = 64000;
|
|
13
13
|
/** Max retries for V2 streaming compaction on transient stream errors. */
|
|
14
14
|
export declare const V2_COMPACTION_MAX_RETRIES = 2;
|
|
15
|
-
/** Timeout for V2 streaming compaction (
|
|
16
|
-
export declare const V2_COMPACTION_TIMEOUT_MS =
|
|
15
|
+
/** Timeout for V2 streaming compaction (5 minutes, same as V1). */
|
|
16
|
+
export declare const V2_COMPACTION_TIMEOUT_MS = 300000;
|
|
17
17
|
/** Token usage reported by the streamed V2 Responses completion. */
|
|
18
18
|
export interface CompactionV2Usage {
|
|
19
19
|
inputTokens: number;
|
|
@@ -19,14 +19,14 @@ import { Tokenizer } from "../tokenizer.js";
|
|
|
19
19
|
export * from "./compaction-v2-streaming.js";
|
|
20
20
|
export declare const OPENAI_REMOTE_COMPACTION_PRESERVE_KEY = "openaiRemoteCompaction";
|
|
21
21
|
/**
|
|
22
|
-
* Hard ceiling on remote compaction HTTP requests. Unlike every
|
|
23
|
-
* stream (guarded by first-event/idle watchdogs in pi-ai), these are
|
|
24
|
-
* fetches awaiting one non-streamed JSON body — a connection silently
|
|
25
|
-
* by a middlebox would otherwise hang the whole compaction pipeline
|
|
26
|
-
* (frozen "Auto context-full maintenance…" spinner, manual /compact
|
|
27
|
-
* behind it). On timeout the caller falls back to local summarization.
|
|
22
|
+
* Hard ceiling on remote compaction HTTP requests (5 minutes). Unlike every
|
|
23
|
+
* provider stream (guarded by first-event/idle watchdogs in pi-ai), these are
|
|
24
|
+
* raw fetches awaiting one non-streamed JSON body — a connection silently
|
|
25
|
+
* dropped by a middlebox would otherwise hang the whole compaction pipeline
|
|
26
|
+
* forever (frozen "Auto context-full maintenance…" spinner, manual /compact
|
|
27
|
+
* queueing behind it). On timeout the caller falls back to local summarization.
|
|
28
28
|
*/
|
|
29
|
-
export declare const REMOTE_COMPACTION_TIMEOUT_MS =
|
|
29
|
+
export declare const REMOTE_COMPACTION_TIMEOUT_MS = 300000;
|
|
30
30
|
export declare const CONTEXT_WINDOW_TRUNCATED_OUTPUT_MESSAGE: string;
|
|
31
31
|
export interface TrimRemoteCompactionInputResult {
|
|
32
32
|
input: Array<Record<string, unknown>>;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@oh-my-pi/pi-agent-core",
|
|
4
|
-
"version": "18.1.
|
|
4
|
+
"version": "18.1.17",
|
|
5
5
|
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
|
|
6
6
|
"homepage": "https://omp.sh",
|
|
7
7
|
"author": "Stencil Labs, Inc.",
|
|
@@ -35,16 +35,16 @@
|
|
|
35
35
|
"fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
|
|
36
36
|
},
|
|
37
37
|
"dependencies": {
|
|
38
|
-
"@oh-my-pi/pi-ai": "18.1.
|
|
39
|
-
"@oh-my-pi/pi-catalog": "18.1.
|
|
40
|
-
"@oh-my-pi/pi-natives": "18.1.
|
|
41
|
-
"@oh-my-pi/pi-utils": "18.1.
|
|
42
|
-
"@oh-my-pi/pi-wire": "18.1.
|
|
43
|
-
"@oh-my-pi/snapcompact": "18.1.
|
|
38
|
+
"@oh-my-pi/pi-ai": "18.1.17",
|
|
39
|
+
"@oh-my-pi/pi-catalog": "18.1.17",
|
|
40
|
+
"@oh-my-pi/pi-natives": "18.1.17",
|
|
41
|
+
"@oh-my-pi/pi-utils": "18.1.17",
|
|
42
|
+
"@oh-my-pi/pi-wire": "18.1.17",
|
|
43
|
+
"@oh-my-pi/snapcompact": "18.1.17",
|
|
44
44
|
"@opentelemetry/api": "^1.9.1"
|
|
45
45
|
},
|
|
46
46
|
"devDependencies": {
|
|
47
|
-
"@oh-my-pi/omptype": "18.1.
|
|
47
|
+
"@oh-my-pi/omptype": "18.1.17",
|
|
48
48
|
"@opentelemetry/context-async-hooks": "^2.9.0",
|
|
49
49
|
"@opentelemetry/sdk-trace-base": "^2.9.0",
|
|
50
50
|
"@types/bun": "^1.3.14"
|
package/src/agent-loop.ts
CHANGED
|
@@ -2269,6 +2269,68 @@ function resolveToolForCall(
|
|
|
2269
2269
|
);
|
|
2270
2270
|
}
|
|
2271
2271
|
|
|
2272
|
+
/** Shortest suggestable segment; below this the match is noise (`id`, `to`). */
|
|
2273
|
+
const MIN_TOOL_NAME_SUGGESTION_SEGMENT = 3;
|
|
2274
|
+
/** Cap on names listed for an ambiguous miss, so the error stays readable. */
|
|
2275
|
+
const MAX_TOOL_NAME_SUGGESTIONS = 3;
|
|
2276
|
+
|
|
2277
|
+
/**
|
|
2278
|
+
* Advertised tool names sharing a trailing `_`-delimited segment with `name`.
|
|
2279
|
+
*
|
|
2280
|
+
* A model that mis-transcribes a long opaque tool name reliably keeps the
|
|
2281
|
+
* trailing verb — that segment is the only part carrying meaning, while any
|
|
2282
|
+
* leading id segments are high-entropy and mnemonic-free. Matching on it turns
|
|
2283
|
+
* an otherwise dead `not found` into a self-correcting one.
|
|
2284
|
+
*
|
|
2285
|
+
* Both the last `__` and last `_` boundary are tried, so a name that lost only
|
|
2286
|
+
* its separator (`…__resolve_library_id`) and one that lost a whole id segment
|
|
2287
|
+
* (`…__read`) both recover. Purely advisory: this only builds an error string
|
|
2288
|
+
* and never selects a tool, so dispatch semantics are unchanged.
|
|
2289
|
+
*/
|
|
2290
|
+
function suggestToolNames(
|
|
2291
|
+
name: string,
|
|
2292
|
+
tools: ReadonlyArray<Pick<AgentTool, "name" | "customWireName">> | undefined,
|
|
2293
|
+
): string[] {
|
|
2294
|
+
if (!tools || tools.length === 0) return [];
|
|
2295
|
+
const segments: string[] = [];
|
|
2296
|
+
for (const boundary of ["__", "_"]) {
|
|
2297
|
+
const idx = name.lastIndexOf(boundary);
|
|
2298
|
+
if (idx < 0) continue;
|
|
2299
|
+
const segment = name.slice(idx + boundary.length);
|
|
2300
|
+
if (segment.length >= MIN_TOOL_NAME_SUGGESTION_SEGMENT && !segments.includes(segment)) segments.push(segment);
|
|
2301
|
+
}
|
|
2302
|
+
if (segments.length === 0) return [];
|
|
2303
|
+
// Longest tail first. A distinctive `__` tail (`resolve_library_get`) is a
|
|
2304
|
+
// far stronger signal than the generic `_` tail it contains (`get`), and the
|
|
2305
|
+
// caller truncates the list — so the strongest match has to sort ahead of
|
|
2306
|
+
// however many tools happen to share the weak one.
|
|
2307
|
+
segments.sort((a, b) => b.length - a.length);
|
|
2308
|
+
const matches: string[] = [];
|
|
2309
|
+
for (const segment of segments) {
|
|
2310
|
+
for (const tool of tools) {
|
|
2311
|
+
for (const candidate of [tool.name, tool.customWireName]) {
|
|
2312
|
+
if (candidate === undefined || candidate === name || matches.includes(candidate)) continue;
|
|
2313
|
+
if (candidate === segment || candidate.endsWith(`_${segment}`)) matches.push(candidate);
|
|
2314
|
+
}
|
|
2315
|
+
}
|
|
2316
|
+
}
|
|
2317
|
+
return matches;
|
|
2318
|
+
}
|
|
2319
|
+
|
|
2320
|
+
/**
|
|
2321
|
+
* `Tool <name> not found`, plus a suggestion when the advertised set contains a
|
|
2322
|
+
* plausible intended target. Exact wording is not a contract; the model reads it.
|
|
2323
|
+
*/
|
|
2324
|
+
function formatToolNotFoundMessage(
|
|
2325
|
+
name: string,
|
|
2326
|
+
tools: ReadonlyArray<Pick<AgentTool, "name" | "customWireName">> | undefined,
|
|
2327
|
+
): string {
|
|
2328
|
+
const suggestions = suggestToolNames(name, tools);
|
|
2329
|
+
if (suggestions.length === 0) return `Tool ${name} not found`;
|
|
2330
|
+
if (suggestions.length === 1) return `Tool ${name} not found. Did you mean ${suggestions[0]}?`;
|
|
2331
|
+
return `Tool ${name} not found. Closest available: ${suggestions.slice(0, MAX_TOOL_NAME_SUGGESTIONS).join(", ")}`;
|
|
2332
|
+
}
|
|
2333
|
+
|
|
2272
2334
|
/**
|
|
2273
2335
|
* Pre-dispatch phase for every pending tool call on `assistantMessage`, run in
|
|
2274
2336
|
* call order: intent extraction, argument validation, and the `beforeToolCall`
|
|
@@ -2312,7 +2374,7 @@ async function prepareToolCallDispatch(
|
|
|
2312
2374
|
}
|
|
2313
2375
|
const validate = (args: Record<string, unknown>): Record<string, unknown> | undefined => {
|
|
2314
2376
|
try {
|
|
2315
|
-
if (!tool) throw new Error(
|
|
2377
|
+
if (!tool) throw new Error(formatToolNotFoundMessage(toolCall.name, context.tools));
|
|
2316
2378
|
return validateToolArguments(tool, { ...toolCall, arguments: args });
|
|
2317
2379
|
} catch (validationError) {
|
|
2318
2380
|
if (tool?.lenientArgValidation) {
|
|
@@ -2637,7 +2699,7 @@ async function executeToolCalls(
|
|
|
2637
2699
|
|
|
2638
2700
|
await runInActiveSpan(toolSpan, async () => {
|
|
2639
2701
|
try {
|
|
2640
|
-
if (!tool) throw new Error(
|
|
2702
|
+
if (!tool) throw new Error(formatToolNotFoundMessage(toolCall.name, tools));
|
|
2641
2703
|
if (record.signal.aborted) {
|
|
2642
2704
|
result = createToolSignalAbortedResult(record.signal);
|
|
2643
2705
|
isError = true;
|
|
@@ -44,8 +44,8 @@ export const V2_RETAINED_MESSAGE_TOKEN_BUDGET = 64_000;
|
|
|
44
44
|
/** Max retries for V2 streaming compaction on transient stream errors. */
|
|
45
45
|
export const V2_COMPACTION_MAX_RETRIES = 2;
|
|
46
46
|
|
|
47
|
-
/** Timeout for V2 streaming compaction (
|
|
48
|
-
export const V2_COMPACTION_TIMEOUT_MS =
|
|
47
|
+
/** Timeout for V2 streaming compaction (5 minutes, same as V1). */
|
|
48
|
+
export const V2_COMPACTION_TIMEOUT_MS = 300_000;
|
|
49
49
|
|
|
50
50
|
const DEFAULT_AZURE_API_VERSION = "v1";
|
|
51
51
|
const OPENAI_REMOTE_COMPACTION_PRESERVE_KEY = "openaiRemoteCompaction";
|
|
@@ -395,22 +395,6 @@ export function resolveThresholdTokens(contextWindow: number, settings: Compacti
|
|
|
395
395
|
// Cut point detection
|
|
396
396
|
// ============================================================================
|
|
397
397
|
|
|
398
|
-
function estimateEntriesTokens(
|
|
399
|
-
entries: SessionEntry[],
|
|
400
|
-
tokenizer: Tokenizer,
|
|
401
|
-
startIndex: number,
|
|
402
|
-
endIndex: number,
|
|
403
|
-
): number {
|
|
404
|
-
let total = 0;
|
|
405
|
-
for (let i = startIndex; i < endIndex; i++) {
|
|
406
|
-
const msg = getMessageFromEntry(entries[i]);
|
|
407
|
-
if (msg) {
|
|
408
|
-
total += tokenizer.countMessage(msg);
|
|
409
|
-
}
|
|
410
|
-
}
|
|
411
|
-
return total;
|
|
412
|
-
}
|
|
413
|
-
|
|
414
398
|
/**
|
|
415
399
|
* Find valid cut points: indices of user, assistant, custom, or bashExecution messages.
|
|
416
400
|
* Never cut at tool results (they must follow their tool call).
|
|
@@ -1331,16 +1315,10 @@ export function prepareCompaction(
|
|
|
1331
1315
|
|
|
1332
1316
|
let prevCompactionIndex = findReadableCompactionIndex(pathEntries, settings, activeModel);
|
|
1333
1317
|
|
|
1334
|
-
//
|
|
1335
|
-
//
|
|
1336
|
-
// must not resurrect the dropped pre-clear turns into its summary — matching
|
|
1337
|
-
// how buildSessionContext starts the model-context rebuild after the boundary.
|
|
1338
|
-
// A boundary after the last reusable compaction supersedes it: the pre-reset
|
|
1339
|
-
// summary was cleared too, so drop the previous-compaction reuse and start
|
|
1340
|
-
// fresh after the boundary. A boundary at or before that compaction is already
|
|
1341
|
-
// superseded by it, so only scan newer entries.
|
|
1318
|
+
// A reset after the reusable compaction clears its summary too. An older
|
|
1319
|
+
// reset still bounds how far we may recover that compaction's kept messages.
|
|
1342
1320
|
let resetBoundaryIndex = -1;
|
|
1343
|
-
for (let i = pathEntries.length - 1; i
|
|
1321
|
+
for (let i = pathEntries.length - 1; i >= 0; i--) {
|
|
1344
1322
|
if (pathEntries[i].type === "reset_boundary") {
|
|
1345
1323
|
resetBoundaryIndex = i;
|
|
1346
1324
|
break;
|
|
@@ -1349,14 +1327,43 @@ export function prepareCompaction(
|
|
|
1349
1327
|
if (resetBoundaryIndex > prevCompactionIndex) {
|
|
1350
1328
|
prevCompactionIndex = -1;
|
|
1351
1329
|
}
|
|
1352
|
-
const
|
|
1353
|
-
|
|
1330
|
+
const previousCompaction =
|
|
1331
|
+
prevCompactionIndex >= 0 ? (pathEntries[prevCompactionIndex] as CompactionEntry) : undefined;
|
|
1332
|
+
let boundaryStart = Math.max(prevCompactionIndex, resetBoundaryIndex) + 1;
|
|
1333
|
+
if (
|
|
1334
|
+
previousCompaction &&
|
|
1335
|
+
!getCompactionV2PreserveData(previousCompaction.preserveData) &&
|
|
1336
|
+
!getPreservedOpenAiRemoteCompactionData(previousCompaction.preserveData)
|
|
1337
|
+
) {
|
|
1338
|
+
// Local summaries exclude the retained tail, whose original entries precede
|
|
1339
|
+
// the compaction record. Native replay already carries that tail. Only look
|
|
1340
|
+
// backwards: advisor snapshots put all retained messages after the summary
|
|
1341
|
+
// and may carry a keep ID from their previous, differently indexed snapshot.
|
|
1342
|
+
for (let i = resetBoundaryIndex + 1; i < prevCompactionIndex; i++) {
|
|
1343
|
+
if (pathEntries[i].id === previousCompaction.firstKeptEntryId) {
|
|
1344
|
+
boundaryStart = i;
|
|
1345
|
+
break;
|
|
1346
|
+
}
|
|
1347
|
+
}
|
|
1348
|
+
}
|
|
1349
|
+
|
|
1350
|
+
// Keep original IDs beside the converted messages so estimation, cutting,
|
|
1351
|
+
// and all three output regions share one sequence without journal metadata.
|
|
1352
|
+
const compactionEntries: SessionEntry[] = [];
|
|
1353
|
+
const compactionMessages: AgentMessage[] = [];
|
|
1354
|
+
for (let i = boundaryStart; i < pathEntries.length; i++) {
|
|
1355
|
+
const entry = pathEntries[i];
|
|
1356
|
+
const message = getMessageFromEntry(entry);
|
|
1357
|
+
if (!message) continue;
|
|
1358
|
+
compactionEntries.push(entry);
|
|
1359
|
+
compactionMessages.push(message);
|
|
1360
|
+
}
|
|
1354
1361
|
|
|
1355
1362
|
const lastUsage = getLastAssistantUsage(pathEntries);
|
|
1356
1363
|
const tokensBefore = lastUsage ? calculateContextTokens(lastUsage) : 0;
|
|
1357
1364
|
let keepRecentTokens = settings.keepRecentTokens;
|
|
1358
1365
|
if (lastUsage) {
|
|
1359
|
-
const estimatedTokens =
|
|
1366
|
+
const estimatedTokens = tokenizer.countMessages(compactionMessages);
|
|
1360
1367
|
const promptTokens = calculatePromptTokens(lastUsage);
|
|
1361
1368
|
const ratio = estimatedTokens > 0 ? promptTokens / estimatedTokens : 0;
|
|
1362
1369
|
if (Number.isFinite(ratio) && ratio > 1) {
|
|
@@ -1364,10 +1371,10 @@ export function prepareCompaction(
|
|
|
1364
1371
|
}
|
|
1365
1372
|
}
|
|
1366
1373
|
|
|
1367
|
-
const cutPoint = findCutPoint(
|
|
1374
|
+
const cutPoint = findCutPoint(compactionEntries, tokenizer, 0, compactionEntries.length, keepRecentTokens);
|
|
1368
1375
|
|
|
1369
1376
|
// Get ID of first kept entry
|
|
1370
|
-
const firstKeptEntry =
|
|
1377
|
+
const firstKeptEntry = compactionEntries[cutPoint.firstKeptEntryIndex];
|
|
1371
1378
|
if (!firstKeptEntry?.id) {
|
|
1372
1379
|
return undefined; // Session needs migration
|
|
1373
1380
|
}
|
|
@@ -1375,42 +1382,16 @@ export function prepareCompaction(
|
|
|
1375
1382
|
|
|
1376
1383
|
const historyEnd = cutPoint.isSplitTurn ? cutPoint.turnStartIndex : cutPoint.firstKeptEntryIndex;
|
|
1377
1384
|
|
|
1378
|
-
|
|
1379
|
-
const
|
|
1380
|
-
|
|
1381
|
-
|
|
1382
|
-
|
|
1383
|
-
}
|
|
1384
|
-
|
|
1385
|
-
// Messages for turn prefix summary (if splitting a turn)
|
|
1386
|
-
const turnPrefixMessages: AgentMessage[] = [];
|
|
1387
|
-
if (cutPoint.isSplitTurn) {
|
|
1388
|
-
for (let i = cutPoint.turnStartIndex; i < cutPoint.firstKeptEntryIndex; i++) {
|
|
1389
|
-
const msg = getMessageFromEntry(pathEntries[i]);
|
|
1390
|
-
if (msg) turnPrefixMessages.push(msg);
|
|
1391
|
-
}
|
|
1392
|
-
}
|
|
1393
|
-
|
|
1394
|
-
// Messages kept after compaction (recent history)
|
|
1395
|
-
const recentMessages: AgentMessage[] = [];
|
|
1396
|
-
for (let i = cutPoint.firstKeptEntryIndex; i < boundaryEnd; i++) {
|
|
1397
|
-
const msg = getMessageFromEntry(pathEntries[i]);
|
|
1398
|
-
if (msg) recentMessages.push(msg);
|
|
1399
|
-
}
|
|
1385
|
+
const messagesToSummarize = compactionMessages.slice(0, historyEnd);
|
|
1386
|
+
const turnPrefixMessages = cutPoint.isSplitTurn
|
|
1387
|
+
? compactionMessages.slice(cutPoint.turnStartIndex, cutPoint.firstKeptEntryIndex)
|
|
1388
|
+
: [];
|
|
1389
|
+
const recentMessages = compactionMessages.slice(cutPoint.firstKeptEntryIndex);
|
|
1400
1390
|
// Nothing to summarize means compaction would be a no-op.
|
|
1401
1391
|
if (messagesToSummarize.length === 0 && turnPrefixMessages.length === 0) {
|
|
1402
1392
|
return undefined;
|
|
1403
1393
|
}
|
|
1404
1394
|
|
|
1405
|
-
// Get previous summary and preserved data for iterative updates
|
|
1406
|
-
let previousSummary: string | undefined;
|
|
1407
|
-
let previousPreserveData: Record<string, unknown> | undefined;
|
|
1408
|
-
if (prevCompactionIndex >= 0) {
|
|
1409
|
-
const prevCompaction = pathEntries[prevCompactionIndex] as CompactionEntry;
|
|
1410
|
-
previousSummary = prevCompaction.summary;
|
|
1411
|
-
previousPreserveData = prevCompaction.preserveData;
|
|
1412
|
-
}
|
|
1413
|
-
|
|
1414
1395
|
// Extract file operations from messages and previous compaction
|
|
1415
1396
|
const fileOps = extractFileOperations(messagesToSummarize, pathEntries, prevCompactionIndex);
|
|
1416
1397
|
|
|
@@ -1428,8 +1409,8 @@ export function prepareCompaction(
|
|
|
1428
1409
|
recentMessages,
|
|
1429
1410
|
isSplitTurn: cutPoint.isSplitTurn,
|
|
1430
1411
|
tokensBefore,
|
|
1431
|
-
previousSummary,
|
|
1432
|
-
previousPreserveData,
|
|
1412
|
+
previousSummary: previousCompaction?.summary,
|
|
1413
|
+
previousPreserveData: previousCompaction?.preserveData,
|
|
1433
1414
|
fileOps,
|
|
1434
1415
|
settings,
|
|
1435
1416
|
};
|
package/src/compaction/openai.ts
CHANGED
|
@@ -66,14 +66,14 @@ export * from "./compaction-v2-streaming";
|
|
|
66
66
|
export const OPENAI_REMOTE_COMPACTION_PRESERVE_KEY = "openaiRemoteCompaction";
|
|
67
67
|
|
|
68
68
|
/**
|
|
69
|
-
* Hard ceiling on remote compaction HTTP requests. Unlike every
|
|
70
|
-
* stream (guarded by first-event/idle watchdogs in pi-ai), these are
|
|
71
|
-
* fetches awaiting one non-streamed JSON body — a connection silently
|
|
72
|
-
* by a middlebox would otherwise hang the whole compaction pipeline
|
|
73
|
-
* (frozen "Auto context-full maintenance…" spinner, manual /compact
|
|
74
|
-
* behind it). On timeout the caller falls back to local summarization.
|
|
69
|
+
* Hard ceiling on remote compaction HTTP requests (5 minutes). Unlike every
|
|
70
|
+
* provider stream (guarded by first-event/idle watchdogs in pi-ai), these are
|
|
71
|
+
* raw fetches awaiting one non-streamed JSON body — a connection silently
|
|
72
|
+
* dropped by a middlebox would otherwise hang the whole compaction pipeline
|
|
73
|
+
* forever (frozen "Auto context-full maintenance…" spinner, manual /compact
|
|
74
|
+
* queueing behind it). On timeout the caller falls back to local summarization.
|
|
75
75
|
*/
|
|
76
|
-
export const REMOTE_COMPACTION_TIMEOUT_MS =
|
|
76
|
+
export const REMOTE_COMPACTION_TIMEOUT_MS = 300_000;
|
|
77
77
|
|
|
78
78
|
const DEFAULT_AZURE_API_VERSION = "v1";
|
|
79
79
|
|
package/src/proxy.ts
CHANGED
|
@@ -20,7 +20,6 @@ import {
|
|
|
20
20
|
type StreamingPartialJsonCarrier,
|
|
21
21
|
setStreamingPartialJson,
|
|
22
22
|
} from "@oh-my-pi/pi-ai/utils/block-symbols";
|
|
23
|
-
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
|
|
24
23
|
import { parseStreamingJson, readSseJson } from "@oh-my-pi/pi-utils";
|
|
25
24
|
|
|
26
25
|
// Event stream adapter for proxy SSE events
|
|
@@ -171,7 +170,7 @@ export function streamProxy(model: Model, context: Context, options: ProxyStream
|
|
|
171
170
|
response.body as ReadableStream<Uint8Array>,
|
|
172
171
|
options.signal,
|
|
173
172
|
)) {
|
|
174
|
-
const parsedEvent = processProxyEvent(
|
|
173
|
+
const parsedEvent = processProxyEvent(event, partial, partialJsonByIndex);
|
|
175
174
|
if (parsedEvent) {
|
|
176
175
|
if (parsedEvent.type === "done" || parsedEvent.type === "error") {
|
|
177
176
|
sawTerminalEvent = true;
|
|
@@ -233,7 +232,6 @@ function scrubPartialJson(partial: AssistantMessage): void {
|
|
|
233
232
|
* reads as still-streaming.
|
|
234
233
|
*/
|
|
235
234
|
function processProxyEvent(
|
|
236
|
-
model: Model,
|
|
237
235
|
proxyEvent: ProxyAssistantMessageEvent,
|
|
238
236
|
partial: AssistantMessage,
|
|
239
237
|
partialJsonByIndex: Map<number, string>,
|
|
@@ -375,7 +373,6 @@ function processProxyEvent(
|
|
|
375
373
|
partial.stopReason = proxyEvent.reason;
|
|
376
374
|
partial.usage = proxyEvent.usage;
|
|
377
375
|
if (proxyEvent.content !== undefined) partial.content = proxyEvent.content;
|
|
378
|
-
calculateCost(model, partial.usage);
|
|
379
376
|
scrubPartialJson(partial);
|
|
380
377
|
return { type: "done", reason: proxyEvent.reason, message: partial };
|
|
381
378
|
|
|
@@ -384,7 +381,6 @@ function processProxyEvent(
|
|
|
384
381
|
partial.errorMessage = proxyEvent.errorMessage;
|
|
385
382
|
partial.usage = proxyEvent.usage;
|
|
386
383
|
if (proxyEvent.content !== undefined) partial.content = proxyEvent.content;
|
|
387
|
-
calculateCost(model, partial.usage);
|
|
388
384
|
scrubPartialJson(partial);
|
|
389
385
|
return { type: "error", reason: proxyEvent.reason, error: partial };
|
|
390
386
|
}
|
package/src/tokenizer.ts
CHANGED
|
@@ -219,14 +219,22 @@ export class Tokenizer {
|
|
|
219
219
|
}
|
|
220
220
|
|
|
221
221
|
switch (message.role) {
|
|
222
|
-
case "user":
|
|
223
|
-
|
|
222
|
+
case "user":
|
|
223
|
+
case "developer": {
|
|
224
|
+
// Both roles carry `string | (TextContent | ImageContent)[]` and both are
|
|
225
|
+
// sent to the provider -- convertMessageToLlm handles developer alongside
|
|
226
|
+
// user -- so they are counted alike. Without the developer case the switch
|
|
227
|
+
// fell through to `default: return 0`, and the old annotation narrowed the
|
|
228
|
+
// blocks to text-only, hiding the image charge the toolResult arm applies.
|
|
229
|
+
const content = message.content;
|
|
224
230
|
if (typeof content === "string") {
|
|
225
231
|
fragments.push(content);
|
|
226
232
|
} else if (Array.isArray(content)) {
|
|
227
233
|
for (const block of content) {
|
|
228
234
|
if (block.type === "text" && block.text) {
|
|
229
235
|
fragments.push(block.text);
|
|
236
|
+
} else if (block.type === "image") {
|
|
237
|
+
extra += IMAGE_TOKEN_ESTIMATE;
|
|
230
238
|
}
|
|
231
239
|
}
|
|
232
240
|
}
|