@yeaft/webchat-agent 1.0.509 → 1.0.510
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/local-runtime/version.json +1 -1
- package/package.json +1 -1
- package/yeaft/engine.js +233 -44
- package/yeaft/history-window.js +111 -49
- package/yeaft/post-compact.js +76 -0
- package/yeaft/tool-folding/index.js +9 -15
- package/yeaft/tool-folding/t1-reflector.js +3 -2
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":"1.0.
|
|
1
|
+
{"version":"1.0.510"}
|
package/package.json
CHANGED
package/yeaft/engine.js
CHANGED
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
* 5. If tool_calls → execute tools → append results → goto 3
|
|
10
10
|
* 6. Persist each completed message at its durability boundary
|
|
11
11
|
* 7. If max_tokens → auto-continue (up to maxContinueTurns)
|
|
12
|
-
* 8. On LLMContextError →
|
|
12
|
+
* 8. On LLMContextError → shrink the request copy and retry without replaying tools
|
|
13
13
|
* 9. On retryable error with fallbackModel → switch model → retry
|
|
14
14
|
*
|
|
15
15
|
* Pattern derived from Claude Code's query loop (src/query.ts).
|
|
@@ -22,7 +22,7 @@ import { promises as fsp } from 'fs';
|
|
|
22
22
|
import { join, resolve as resolvePath } from 'path';
|
|
23
23
|
import { buildSystemPrompt, buildWorkerPrompt } from './prompts.js';
|
|
24
24
|
import { getRuntimePlatformInfo } from './runtime-platform.js';
|
|
25
|
-
import { LLMAbortError, LLMAuthError, LLMPolicyError, LLMRateLimitError, LLMServerError, LLMStreamIdleTimeoutError } from './llm/adapter.js';
|
|
25
|
+
import { LLMAbortError, LLMAuthError, LLMContextError, LLMPolicyError, LLMRateLimitError, LLMServerError, LLMStreamIdleTimeoutError } from './llm/adapter.js';
|
|
26
26
|
import { runMemoryPreflow, buildRelevantScopes, memoryScopeLabel } from './sessions/pre-flow.js';
|
|
27
27
|
import {
|
|
28
28
|
readProjectDoc,
|
|
@@ -33,7 +33,7 @@ import {
|
|
|
33
33
|
DEFAULT_PROJECT_DOC_MAX_BYTES,
|
|
34
34
|
} from './sessions/project-doc.js';
|
|
35
35
|
import { archiveToolResults } from './archive/tool-results.js';
|
|
36
|
-
import { trimSnapshotForBudget, estimateMessageTokens, buildHistoryBuckets } from './history-window.js';
|
|
36
|
+
import { trimSnapshotForBudget, estimateMessageTokens, buildHistoryBuckets, fitProviderRequestToContext } from './history-window.js';
|
|
37
37
|
import { recallConversationTurns } from './conversation/history-index.js';
|
|
38
38
|
import { parseSeqFromId } from './conversation/persist.js';
|
|
39
39
|
import { isVpForeign, readContent as readScopeContent } from './memory/store.js';
|
|
@@ -46,6 +46,14 @@ import { perfNowMs, recordAgentPerfTrace } from './perf-trace.js';
|
|
|
46
46
|
const MAIN_THREAD_ID = 'main';
|
|
47
47
|
import { pickEffort, parseEffortPrefix, snapshotEffortDecision } from './effort.js';
|
|
48
48
|
import { bindProviderState } from './llm/provider-state.js';
|
|
49
|
+
import {
|
|
50
|
+
POST_COMPACT_CONTEXT_RATIO,
|
|
51
|
+
generatePostCompact,
|
|
52
|
+
loadPostCompact,
|
|
53
|
+
postCompactPath,
|
|
54
|
+
removePostCompactIfSource,
|
|
55
|
+
savePostCompact,
|
|
56
|
+
} from './post-compact.js';
|
|
49
57
|
import { DEFAULT_CONTEXT_WINDOW, getModelInfo, normalizeEffort, parseModelRef, resolveContextWindow, resolveMaxOutputTokens, resolveModel } from './models.js';
|
|
50
58
|
import { lookupModelLimitSync } from './llm/models-dev.js';
|
|
51
59
|
import { attachRouterPlan, extractPriorPlan, stripMetaForWire } from './router/continuity.js';
|
|
@@ -59,7 +67,7 @@ import { createPluginSkillManager } from './plugins.js';
|
|
|
59
67
|
import { extractDisplayImages, stripDisplayImageData } from './image-assets.js';
|
|
60
68
|
import { acknowledgePendingNotifications, formatNotificationsForPrompt, peekPendingNotifications } from './sub-agent/notifications.js';
|
|
61
69
|
import {
|
|
62
|
-
|
|
70
|
+
TOOL_LOOP_REFLECTION_INTERVAL,
|
|
63
71
|
TURN_SUMMARY_THRESHOLD,
|
|
64
72
|
DUP_TOOL_THRESHOLD,
|
|
65
73
|
ExecLog,
|
|
@@ -80,7 +88,7 @@ import {
|
|
|
80
88
|
* conversations (user report: Yeaft loop errored at the cap). The engine
|
|
81
89
|
* now runs until the LLM itself returns stopReason='end_turn' or a
|
|
82
90
|
* non-retryable error surfaces. Real runaway loops are still bounded by:
|
|
83
|
-
* • provider rate limits / context
|
|
91
|
+
* • provider rate limits / context recovery exhaustion
|
|
84
92
|
* • user-initiated abort (AbortController / cancel)
|
|
85
93
|
* • MAX_CONTINUE_TURNS for the max_tokens auto-continue path
|
|
86
94
|
*/
|
|
@@ -722,13 +730,18 @@ export class Engine {
|
|
|
722
730
|
* prior turn's history is rewritten with the reflection. If still
|
|
723
731
|
* pending, the engine falls back to the exec-log stub.
|
|
724
732
|
* • `#reflectedTurns` — Set<turnNumber>; ensures T1 fires at most
|
|
725
|
-
* once per
|
|
733
|
+
* once per reflection interval measured in tool loops.
|
|
726
734
|
*/
|
|
727
735
|
#execLog = null;
|
|
728
736
|
#pendingT2 = new Map();
|
|
729
737
|
#reflectedTurns = new Set();
|
|
730
738
|
#__queryCounter = 0;
|
|
731
739
|
|
|
740
|
+
/** Derived post-response summaries, keyed by Session/VP/thread scope. */
|
|
741
|
+
#postCompactSummaries = new Map();
|
|
742
|
+
#postCompactLoaded = new Set();
|
|
743
|
+
#postCompactRevisions = new Map();
|
|
744
|
+
|
|
732
745
|
/** @type {string} */
|
|
733
746
|
#currentThreadId = MAIN_THREAD_ID;
|
|
734
747
|
|
|
@@ -2069,6 +2082,12 @@ export class Engine {
|
|
|
2069
2082
|
&& typeof vpPersona.vpId === 'string'
|
|
2070
2083
|
? vpPersona.vpId
|
|
2071
2084
|
: (typeof senderVpId === 'string' ? senderVpId : null);
|
|
2085
|
+
const postCompactScope = this.#postCompactScope({
|
|
2086
|
+
sessionId: runtimeSessionId,
|
|
2087
|
+
vpId: queryVpId,
|
|
2088
|
+
threadId: runtimeThreadId,
|
|
2089
|
+
});
|
|
2090
|
+
const postCompactState = await this.#beginPostCompactScope(postCompactScope);
|
|
2072
2091
|
// Exact read-only tool results are safe to reuse within one query only
|
|
2073
2092
|
// when no intervening mutation can have changed the workspace. The map is
|
|
2074
2093
|
// intentionally local to this query; cross-turn reuse belongs to the
|
|
@@ -2249,6 +2268,10 @@ export class Engine {
|
|
|
2249
2268
|
if (amsContext && amsContext.snapshotBlock) {
|
|
2250
2269
|
memoryInjection = amsContext.snapshotBlock;
|
|
2251
2270
|
}
|
|
2271
|
+
if (postCompactState.summary) {
|
|
2272
|
+
const block = `## Prior conversation compact\n${postCompactState.summary}`;
|
|
2273
|
+
memoryInjection = memoryInjection ? `${memoryInjection}\n\n${block}` : block;
|
|
2274
|
+
}
|
|
2252
2275
|
const loadedMemoryForDebug = loadedMemoryDebugEntries(amsContext?.snapshot);
|
|
2253
2276
|
const loadedMemoryMetaForDebug = {
|
|
2254
2277
|
...(useMessageHistory ? historyRecallMeta : {}),
|
|
@@ -2475,18 +2498,18 @@ export class Engine {
|
|
|
2475
2498
|
// `turnStartIdx` is where the current user message lives; the arc
|
|
2476
2499
|
// we may collapse spans (arcStartIdx .. last assistant/tool).
|
|
2477
2500
|
//
|
|
2478
|
-
// Periodic
|
|
2479
|
-
//
|
|
2501
|
+
// Periodic T1 fires every 30 tool loops, not every 30 calls. A single
|
|
2502
|
+
// provider batch can contain many parallel calls but is still one loop.
|
|
2480
2503
|
// track:
|
|
2481
|
-
// • `
|
|
2504
|
+
// • `lastT1AtLoopCount` — tool-loop snapshot at the last T1
|
|
2482
2505
|
// ATTEMPT (success OR error). Trigger when
|
|
2483
|
-
// `
|
|
2506
|
+
// `completedToolLoops - lastT1AtLoopCount >= interval`.
|
|
2484
2507
|
// • `arcStartIdx` — first index of the current (uncollapsed)
|
|
2485
2508
|
// tool arc. Initialised to turnStartIdx + 1; reset after each
|
|
2486
2509
|
// successful T1 collapse to `conversationMessages.length`
|
|
2487
2510
|
// (i.e. the slot the next assistant message will land in).
|
|
2488
2511
|
// • `t1CollapsesDone` — count of T1 firings that ACTUALLY
|
|
2489
|
-
// rewrote history. Distinct from `
|
|
2512
|
+
// rewrote history. Distinct from `lastT1AtLoopCount` because
|
|
2490
2513
|
// the catch block bumps the latter to back off after a
|
|
2491
2514
|
// transient reflector error WITHOUT having collapsed
|
|
2492
2515
|
// anything. The T2 schedule check below is gated on this
|
|
@@ -2494,7 +2517,7 @@ export class Engine {
|
|
|
2494
2517
|
// fall back at end_turn").
|
|
2495
2518
|
const turnStartIdx = conversationMessages.length - 1;
|
|
2496
2519
|
let queryToolCount = 0;
|
|
2497
|
-
let
|
|
2520
|
+
let lastT1AtLoopCount = 0;
|
|
2498
2521
|
let arcStartIdx = turnStartIdx + 1;
|
|
2499
2522
|
let t1CollapsesDone = 0;
|
|
2500
2523
|
// Duplicate policy is scoped to one user query. Only successful, real
|
|
@@ -2561,6 +2584,10 @@ export class Engine {
|
|
|
2561
2584
|
let primaryModelAtLastBoundary = currentModel;
|
|
2562
2585
|
let cumulativeInputTokens = 0;
|
|
2563
2586
|
let cumulativeOutputTokens = 0;
|
|
2587
|
+
let maxContextOccupancyRatio = 0;
|
|
2588
|
+
let peakContextTokens = 0;
|
|
2589
|
+
let peakContextWindow = 0;
|
|
2590
|
+
let postCompactCandidate = null;
|
|
2564
2591
|
let activeProviderRequest = null;
|
|
2565
2592
|
// Skill events describe the selection injected into each provider request.
|
|
2566
2593
|
// The first request must report its initial selection; later loops report
|
|
@@ -2592,6 +2619,11 @@ export class Engine {
|
|
|
2592
2619
|
let consecutiveRetryableErrors = 0;
|
|
2593
2620
|
let consecutiveForbiddenErrors = 0;
|
|
2594
2621
|
let contentPolicyRecoveryAttempts = 0;
|
|
2622
|
+
let contextOverflowRecoveryAttempts = 0;
|
|
2623
|
+
// A provider can know about framing/tokenizer overhead that our monotonic
|
|
2624
|
+
// estimator cannot. Each real overflow retries the same unexecuted request
|
|
2625
|
+
// with a smaller provider-only window; no transcript row is rewritten.
|
|
2626
|
+
let providerContextScale = 1;
|
|
2595
2627
|
|
|
2596
2628
|
while (true) {
|
|
2597
2629
|
turnNumber++;
|
|
@@ -2877,12 +2909,9 @@ export class Engine {
|
|
|
2877
2909
|
// query tape remain complete; no summary is generated and no history
|
|
2878
2910
|
// rows are rewritten. This also protects later tool-loop requests,
|
|
2879
2911
|
// not just the initial snapshot assembled by the bridge.
|
|
2880
|
-
|
|
2881
|
-
|
|
2882
|
-
const historyBudget = Math.max(
|
|
2883
|
-
requestConfig.messageTokenBudget || 32768,
|
|
2884
|
-
Math.floor(currentContextWindow * 0.75) - estimateMessagesTokens(systemPrompt, []),
|
|
2885
|
-
) - continuationCost);
|
|
2912
|
+
// Retry continuation is part of the active turn. Like the opening user
|
|
2913
|
+
// row and tool-loop traffic, it must not consume the 32K history budget.
|
|
2914
|
+
const historyBudget = Math.max(0, requestConfig.messageTokenBudget || 32768);
|
|
2886
2915
|
const buckets = useMessageHistory ? buildHistoryBuckets(conversationMessages, {
|
|
2887
2916
|
prompt,
|
|
2888
2917
|
relatedTurns: relatedHistoryTurns,
|
|
@@ -2892,10 +2921,18 @@ export class Engine {
|
|
|
2892
2921
|
currentTurnStartIndex: turnStartIdx,
|
|
2893
2922
|
language: requestConfig.language,
|
|
2894
2923
|
}) : null;
|
|
2895
|
-
|
|
2896
|
-
|
|
2897
|
-
|
|
2898
|
-
|
|
2924
|
+
let historyMessageCount;
|
|
2925
|
+
const requestHistory = buckets?.messages || (() => {
|
|
2926
|
+
const historical = trimSnapshotForBudget(conversationMessages.slice(0, turnStartIdx), {
|
|
2927
|
+
messageTokenBudget: historyBudget,
|
|
2928
|
+
language: requestConfig.language,
|
|
2929
|
+
});
|
|
2930
|
+
historyMessageCount = historical.length;
|
|
2931
|
+
return [...historical, ...conversationMessages.slice(turnStartIdx)];
|
|
2932
|
+
})();
|
|
2933
|
+
if (buckets) {
|
|
2934
|
+
historyMessageCount = requestHistory.length - (buckets.meta?.current?.messageCount || 0);
|
|
2935
|
+
}
|
|
2899
2936
|
if (buckets) this.#trace.log?.('history_buckets', {
|
|
2900
2937
|
sessionId: runtimeSessionId, turnId: queryTurnId, ...historyRecallMeta, ...buckets.meta,
|
|
2901
2938
|
});
|
|
@@ -2919,6 +2956,39 @@ export class Engine {
|
|
|
2919
2956
|
} catch { /* best-effort */ }
|
|
2920
2957
|
}
|
|
2921
2958
|
|
|
2959
|
+
// Final request boundary: account for every component against the
|
|
2960
|
+
// actual model window. The 32K budget above applies only to historical
|
|
2961
|
+
// rows; current-turn rows are paid here together with system, schemas,
|
|
2962
|
+
// and the model-specific output reserve. This runs on every tool-loop
|
|
2963
|
+
// request and again with tighter headroom after a provider overflow.
|
|
2964
|
+
const requestMaxOutputTokens = Math.max(1, Math.min(
|
|
2965
|
+
requestConfig.maxOutputTokens || resolveMaxOutputTokens(currentModel, requestConfig),
|
|
2966
|
+
resolveMaxOutputTokens(currentModel, requestConfig),
|
|
2967
|
+
));
|
|
2968
|
+
const toolSchemaTokens = toolDefs.length > 0
|
|
2969
|
+
? approxTokens(JSON.stringify(toolDefs)) : 0;
|
|
2970
|
+
const fittedRequest = fitProviderRequestToContext(wireMessages, {
|
|
2971
|
+
contextWindow: Math.max(1, Math.floor(currentContextWindow * providerContextScale)),
|
|
2972
|
+
systemTokens: estimateMessagesTokens(systemPrompt, []),
|
|
2973
|
+
toolSchemaTokens,
|
|
2974
|
+
outputReserve: requestMaxOutputTokens,
|
|
2975
|
+
historyMessageCount,
|
|
2976
|
+
historyTokenBudget: historyBudget,
|
|
2977
|
+
language: requestConfig.language,
|
|
2978
|
+
});
|
|
2979
|
+
wireMessages = fittedRequest.messages;
|
|
2980
|
+
if (fittedRequest.meta.droppedHistoryMessages > 0
|
|
2981
|
+
|| fittedRequest.meta.droppedCurrentMessages > 0
|
|
2982
|
+
|| providerContextScale < 1) {
|
|
2983
|
+
this.#trace.log?.('request_context_trim', {
|
|
2984
|
+
sessionId: runtimeSessionId,
|
|
2985
|
+
turnId: queryTurnId,
|
|
2986
|
+
model: currentModel,
|
|
2987
|
+
recoveryAttempt: contextOverflowRecoveryAttempts,
|
|
2988
|
+
...fittedRequest.meta,
|
|
2989
|
+
});
|
|
2990
|
+
}
|
|
2991
|
+
|
|
2922
2992
|
// task-704b: pre-flight total-token guard. Even with the per-tool
|
|
2923
2993
|
// cap (registry.js: 10% of contextWindow per result), N tool
|
|
2924
2994
|
// results plus history can still breach the wire limit before we
|
|
@@ -3005,7 +3075,7 @@ export class Engine {
|
|
|
3005
3075
|
system: systemPrompt,
|
|
3006
3076
|
messages: wireMessages,
|
|
3007
3077
|
tools: toolDefs.length > 0 ? toolDefs : undefined,
|
|
3008
|
-
maxTokens:
|
|
3078
|
+
maxTokens: requestMaxOutputTokens,
|
|
3009
3079
|
effort: resolvedEffort,
|
|
3010
3080
|
effortConstraint,
|
|
3011
3081
|
requestIdentity,
|
|
@@ -3183,6 +3253,35 @@ export class Engine {
|
|
|
3183
3253
|
responseTextBytes: Buffer.byteLength(responseText, 'utf8'),
|
|
3184
3254
|
},
|
|
3185
3255
|
});
|
|
3256
|
+
const requestContextOccupancy = (totalUsage.inputTokens || 0)
|
|
3257
|
+
+ (totalUsage.cacheInputDeltaTokens || 0)
|
|
3258
|
+
+ (totalUsage.outputTokens || 0);
|
|
3259
|
+
const requestContextOccupancyRatio = requestContextOccupancy / currentContextWindow;
|
|
3260
|
+
if (requestContextOccupancyRatio >= maxContextOccupancyRatio) {
|
|
3261
|
+
maxContextOccupancyRatio = requestContextOccupancyRatio;
|
|
3262
|
+
peakContextTokens = requestContextOccupancy;
|
|
3263
|
+
peakContextWindow = currentContextWindow;
|
|
3264
|
+
}
|
|
3265
|
+
// Trigger from the peak request, but summarize the latest provider
|
|
3266
|
+
// state so an earlier, fuller tool loop cannot omit later reflection,
|
|
3267
|
+
// tool results, or the final answer from the derived artifact.
|
|
3268
|
+
postCompactCandidate = {
|
|
3269
|
+
scope: postCompactScope,
|
|
3270
|
+
revision: postCompactState.revision,
|
|
3271
|
+
sessionId: runtimeSessionId,
|
|
3272
|
+
turnId: queryTurnId,
|
|
3273
|
+
model: currentModel,
|
|
3274
|
+
config: requestConfig,
|
|
3275
|
+
adapter: requestAdapter,
|
|
3276
|
+
messages: [
|
|
3277
|
+
...wireMessages.map(message => ({ ...message })),
|
|
3278
|
+
...(toolCalls.length === 0 && responseText
|
|
3279
|
+
? [{ role: 'assistant', content: responseText }]
|
|
3280
|
+
: []),
|
|
3281
|
+
],
|
|
3282
|
+
contextTokens: peakContextTokens,
|
|
3283
|
+
contextWindow: peakContextWindow,
|
|
3284
|
+
};
|
|
3186
3285
|
// Stream completed without throwing — reset the retry counter so
|
|
3187
3286
|
// the next turn starts with a clean budget. In-band adapter errors
|
|
3188
3287
|
// are converted to throws above so they share the real error path.
|
|
@@ -3298,11 +3397,32 @@ export class Engine {
|
|
|
3298
3397
|
const earlyIsRateLimit = err instanceof LLMRateLimitError;
|
|
3299
3398
|
const earlyIsTransient = err instanceof LLMServerError;
|
|
3300
3399
|
const earlyIsContentPolicy = err instanceof LLMPolicyError;
|
|
3400
|
+
const earlyIsContextOverflow = err instanceof LLMContextError;
|
|
3301
3401
|
// A completed tool_call has already crossed the streaming boundary to
|
|
3302
3402
|
// the caller. Replaying that request would publish a duplicate call and
|
|
3303
3403
|
// leave ambiguous execution ownership, so only pre-tool failures are
|
|
3304
3404
|
// eligible for transparent retry or model fallback.
|
|
3305
3405
|
const canReplayProviderRequest = toolCalls.length === 0;
|
|
3406
|
+
if (earlyIsContextOverflow && canReplayProviderRequest
|
|
3407
|
+
&& contextOverflowRecoveryAttempts < 3) {
|
|
3408
|
+
contextOverflowRecoveryAttempts += 1;
|
|
3409
|
+
providerContextScale *= 0.75;
|
|
3410
|
+
endAttemptTrace('context_overflow_retry');
|
|
3411
|
+
if (responseText) prepareRetryContinuation();
|
|
3412
|
+
yield {
|
|
3413
|
+
type: 'llm_retry',
|
|
3414
|
+
attempt: contextOverflowRecoveryAttempts,
|
|
3415
|
+
maxRetries: 3,
|
|
3416
|
+
delayMs: 0,
|
|
3417
|
+
reason: 'context_overflow_recovery',
|
|
3418
|
+
recoveryMode: responseText ? 'continue' : 'restart',
|
|
3419
|
+
errorName: err.name,
|
|
3420
|
+
statusCode: err.statusCode ?? null,
|
|
3421
|
+
message: 'Provider rejected the context; retrying with a smaller request copy.',
|
|
3422
|
+
};
|
|
3423
|
+
yield { type: 'turn_end', turnNumber, stopReason: 'llm_retry', threadId };
|
|
3424
|
+
continue;
|
|
3425
|
+
}
|
|
3306
3426
|
if (earlyIsContentPolicy && canReplayProviderRequest && contentPolicyRecoveryAttempts === 0) {
|
|
3307
3427
|
contentPolicyRecoveryAttempts = 1;
|
|
3308
3428
|
endAttemptTrace('llm_retry');
|
|
@@ -3896,8 +4016,8 @@ export class Engine {
|
|
|
3896
4016
|
// `#applyPendingT2Reflections` carries the result forward.
|
|
3897
4017
|
//
|
|
3898
4018
|
// Periodic-T1 fix: gate on `t1CollapsesDone === 0`, NOT
|
|
3899
|
-
// `
|
|
3900
|
-
// `
|
|
4019
|
+
// `lastT1AtLoopCount === 0`. The catch block of T1 advances
|
|
4020
|
+
// `lastT1AtLoopCount` after a reflector error to avoid
|
|
3901
4021
|
// tight-loop retries — but no collapse happened, so T2 should
|
|
3902
4022
|
// still be allowed to fall back at end_turn. Fowler-review
|
|
3903
4023
|
// critical finding.
|
|
@@ -4697,29 +4817,28 @@ export class Engine {
|
|
|
4697
4817
|
break;
|
|
4698
4818
|
}
|
|
4699
4819
|
|
|
4700
|
-
// PR-L: T1 in-turn (synchronous) reflection. Fires
|
|
4701
|
-
//
|
|
4702
|
-
//
|
|
4703
|
-
// the first batch of the query(). Generates a markdown reflection
|
|
4820
|
+
// PR-L: T1 in-turn (synchronous) reflection. Fires every 30 completed
|
|
4821
|
+
// tool loops. A provider response containing many parallel tool calls is
|
|
4822
|
+
// one loop, not many. Generates a markdown reflection
|
|
4704
4823
|
// over the assistant+tool arc since the last T1 firing (or
|
|
4705
4824
|
// since the user prompt for the first batch) and rewrites that
|
|
4706
4825
|
// range to a SINGLE synthetic user message before the next
|
|
4707
4826
|
// adapter.stream() runs.
|
|
4708
4827
|
//
|
|
4709
4828
|
// Loop semantics:
|
|
4710
|
-
// - First
|
|
4711
|
-
//
|
|
4712
|
-
// - Each subsequent batch: arcStartIdx is updated to the slot
|
|
4829
|
+
// - First interval: arcStartIdx = turnStartIdx + 1.
|
|
4830
|
+
// - Each subsequent interval: arcStartIdx is updated to the slot
|
|
4713
4831
|
// right after the just-inserted reflection message; fires
|
|
4714
|
-
// again whenever
|
|
4715
|
-
//
|
|
4716
|
-
//
|
|
4717
|
-
// batch within the same query gets a distinct entry — without
|
|
4832
|
+
// again whenever 30 more tool loops have completed.
|
|
4833
|
+
// - The dedup Set key includes the loop count so each interval
|
|
4834
|
+
// within the same query gets a distinct entry — without
|
|
4718
4835
|
// this the second batch would be silently skipped.
|
|
4719
|
-
const
|
|
4720
|
-
|
|
4836
|
+
const completedToolLoops = toolLoopTurns + 1;
|
|
4837
|
+
const t1BatchDue = completedToolLoops - lastT1AtLoopCount
|
|
4838
|
+
>= TOOL_LOOP_REFLECTION_INTERVAL;
|
|
4839
|
+
if (t1BatchDue && !toolBatchBarrier
|
|
4721
4840
|
&& !abortedDuringTools && !signal?.aborted) {
|
|
4722
|
-
const t1DedupKey = `${queryNumber}:t1:${
|
|
4841
|
+
const t1DedupKey = `${queryNumber}:t1-loop:${completedToolLoops}`;
|
|
4723
4842
|
if (this.#reflectedTurns.has(t1DedupKey)) {
|
|
4724
4843
|
// Defensive: should never hit since t1BatchDue gates re-entry
|
|
4725
4844
|
// and queryNumber namespaces queries. Kept as belt-and-
|
|
@@ -4787,10 +4906,10 @@ export class Engine {
|
|
|
4787
4906
|
// immediately after it, i.e. at conversationMessages.length
|
|
4788
4907
|
// (the next assistant message will land here).
|
|
4789
4908
|
arcStartIdx = conversationMessages.length;
|
|
4790
|
-
|
|
4909
|
+
lastT1AtLoopCount = completedToolLoops;
|
|
4791
4910
|
// Bump the success counter — used by the T2 schedule check
|
|
4792
4911
|
// to decide whether T2 still has work to do at end_turn.
|
|
4793
|
-
// Distinct from
|
|
4912
|
+
// Distinct from lastT1AtLoopCount which the catch block
|
|
4794
4913
|
// also bumps (but without rewriting history).
|
|
4795
4914
|
t1CollapsesDone += 1;
|
|
4796
4915
|
yield {
|
|
@@ -4820,9 +4939,9 @@ export class Engine {
|
|
|
4820
4939
|
status: 'error',
|
|
4821
4940
|
error: err && err.message || String(err),
|
|
4822
4941
|
};
|
|
4823
|
-
// Advance
|
|
4942
|
+
// Advance lastT1AtLoopCount past this interval so we don't
|
|
4824
4943
|
// tight-loop on a hiccuping reflector. The next attempt is
|
|
4825
|
-
//
|
|
4944
|
+
// another 30 tool loops from now, not immediately. arcStartIdx is
|
|
4826
4945
|
// left alone because history wasn't rewritten — the tail still
|
|
4827
4946
|
// begins where it did. The trade-off: the next batch's
|
|
4828
4947
|
// reflection will cover the tools that just failed too,
|
|
@@ -4831,7 +4950,7 @@ export class Engine {
|
|
|
4831
4950
|
// We do NOT bump t1CollapsesDone — see the variable's
|
|
4832
4951
|
// declaration comment. This keeps the T2 fallback path live
|
|
4833
4952
|
// when every T1 attempt has errored.
|
|
4834
|
-
|
|
4953
|
+
lastT1AtLoopCount = completedToolLoops;
|
|
4835
4954
|
}
|
|
4836
4955
|
}
|
|
4837
4956
|
}
|
|
@@ -4879,6 +4998,76 @@ export class Engine {
|
|
|
4879
4998
|
totalTokens: cumulativeInputTokens + cumulativeOutputTokens,
|
|
4880
4999
|
loopCount: turnNumber,
|
|
4881
5000
|
};
|
|
5001
|
+
|
|
5002
|
+
// The visible response is complete at the yield above. Only when the
|
|
5003
|
+
// consumer resumes past that boundary do we inspect pressure and launch
|
|
5004
|
+
// best-effort post compact. It never blocks this turn or the next one.
|
|
5005
|
+
if (postCompactCandidate
|
|
5006
|
+
&& maxContextOccupancyRatio >= POST_COMPACT_CONTEXT_RATIO) {
|
|
5007
|
+
this.#schedulePostCompact(postCompactCandidate);
|
|
5008
|
+
}
|
|
5009
|
+
}
|
|
5010
|
+
|
|
5011
|
+
#postCompactScope({ sessionId, vpId, threadId }) {
|
|
5012
|
+
if (!this.#yeaftDir || !sessionId) return null;
|
|
5013
|
+
const path = postCompactPath(this.#yeaftDir, { sessionId, vpId, threadId });
|
|
5014
|
+
return { key: path, path };
|
|
5015
|
+
}
|
|
5016
|
+
|
|
5017
|
+
async #beginPostCompactScope(scope) {
|
|
5018
|
+
if (!scope) return { revision: 0, summary: '' };
|
|
5019
|
+
const revision = (this.#postCompactRevisions.get(scope.key) || 0) + 1;
|
|
5020
|
+
this.#postCompactRevisions.set(scope.key, revision);
|
|
5021
|
+
if (!this.#postCompactLoaded.has(scope.key)) {
|
|
5022
|
+
this.#postCompactLoaded.add(scope.key);
|
|
5023
|
+
const artifact = await loadPostCompact(scope.path);
|
|
5024
|
+
if (artifact) this.#postCompactSummaries.set(scope.key, artifact);
|
|
5025
|
+
}
|
|
5026
|
+
return {
|
|
5027
|
+
revision,
|
|
5028
|
+
summary: this.#postCompactSummaries.get(scope.key)?.summary || '',
|
|
5029
|
+
};
|
|
5030
|
+
}
|
|
5031
|
+
|
|
5032
|
+
#schedulePostCompact(candidate) {
|
|
5033
|
+
const { scope, revision } = candidate;
|
|
5034
|
+
if (!scope || this.#postCompactRevisions.get(scope.key) !== revision) return;
|
|
5035
|
+
const compactMaxTokens = Math.max(512, Math.min(4096,
|
|
5036
|
+
resolveMaxOutputTokens(candidate.model, candidate.config)));
|
|
5037
|
+
const task = async () => {
|
|
5038
|
+
try {
|
|
5039
|
+
const summary = await generatePostCompact({
|
|
5040
|
+
adapter: candidate.adapter,
|
|
5041
|
+
model: candidate.model,
|
|
5042
|
+
messages: candidate.messages,
|
|
5043
|
+
maxTokens: compactMaxTokens,
|
|
5044
|
+
});
|
|
5045
|
+
const current = () => this.#postCompactRevisions.get(scope.key) === revision;
|
|
5046
|
+
if (!current()) return;
|
|
5047
|
+
const artifact = {
|
|
5048
|
+
summary,
|
|
5049
|
+
model: candidate.model,
|
|
5050
|
+
sourceTurnId: candidate.turnId,
|
|
5051
|
+
sourceRevision: revision,
|
|
5052
|
+
sourceContextTokens: candidate.contextTokens,
|
|
5053
|
+
contextWindow: candidate.contextWindow,
|
|
5054
|
+
createdAt: new Date().toISOString(),
|
|
5055
|
+
};
|
|
5056
|
+
if (await savePostCompact(scope.path, artifact, current)) {
|
|
5057
|
+
if (current()) this.#postCompactSummaries.set(scope.key, artifact);
|
|
5058
|
+
else await removePostCompactIfSource(scope.path, candidate.turnId);
|
|
5059
|
+
}
|
|
5060
|
+
} catch (error) {
|
|
5061
|
+
// A response already reached the user. Compact failure is diagnostic
|
|
5062
|
+
// only and must not create a late error event.
|
|
5063
|
+
this.#trace.log?.('post_compact_failed', {
|
|
5064
|
+
sessionId: candidate.sessionId,
|
|
5065
|
+
turnId: candidate.turnId,
|
|
5066
|
+
message: String(error?.message || error).slice(0, 200),
|
|
5067
|
+
});
|
|
5068
|
+
}
|
|
5069
|
+
};
|
|
5070
|
+
void task();
|
|
4882
5071
|
}
|
|
4883
5072
|
|
|
4884
5073
|
/**
|
package/yeaft/history-window.js
CHANGED
|
@@ -27,7 +27,6 @@ export const DEFAULT_RUNTIME_CACHE_TURN_CAP = 25;
|
|
|
27
27
|
export const DEFAULT_RUNTIME_CACHE_TOKEN_BUDGET = 32768;
|
|
28
28
|
export const DEFAULT_RUNTIME_CACHE_MESSAGE_CAP = 256;
|
|
29
29
|
|
|
30
|
-
const MINIMUM_RECENT_PROVIDER_TURNS = 5;
|
|
31
30
|
const IMAGE_PART_TOKEN_COST = 1024;
|
|
32
31
|
const DOCUMENT_PART_TOKEN_COST = 2048;
|
|
33
32
|
const CONTENT_PART_FRAME_TOKENS = 2;
|
|
@@ -899,11 +898,11 @@ function describeBucket(turns, messages = turns.flatMap(turn => turn.text)) {
|
|
|
899
898
|
* result in the transcript. Past human turn boundaries are retained when the
|
|
900
899
|
* configured recent window fits; tools are optional enrichment, newest first.
|
|
901
900
|
*
|
|
902
|
-
* The active turn is outside both buckets and
|
|
903
|
-
*
|
|
904
|
-
*
|
|
905
|
-
*
|
|
906
|
-
*
|
|
901
|
+
* The active turn is outside both buckets and outside the history budget. Its
|
|
902
|
+
* opening user row is protected by the later whole-request fitter. Recent text
|
|
903
|
+
* stays complete when it fits, but a configured turn count is never a hard
|
|
904
|
+
* request-success floor. Related recall only uses remaining history budget and
|
|
905
|
+
* stays optional and complete. External recall must have
|
|
907
906
|
* comparable userSeq/source identities to establish
|
|
908
907
|
* that it predates recent/current history; unknown chronology fails closed.
|
|
909
908
|
*
|
|
@@ -927,41 +926,16 @@ export function buildHistoryBuckets(snapshot, options = {}) {
|
|
|
927
926
|
: (allTurns.at(-1)?.index ?? source.length);
|
|
928
927
|
const currentSource = source.slice(currentStart);
|
|
929
928
|
const currentIdentity = bucketTurn(currentSource, currentStart);
|
|
930
|
-
|
|
931
|
-
|
|
932
|
-
|
|
933
|
-
|
|
934
|
-
|
|
935
|
-
|
|
936
|
-
|
|
937
|
-
|
|
938
|
-
|
|
939
|
-
|
|
940
|
-
// directly, without legacy human-turn filtering or text projection.
|
|
941
|
-
const units = providerUnits(pairSanitize(truncateToolResultsForModel(
|
|
942
|
-
currentSource.slice(1), { language: options.language },
|
|
943
|
-
)));
|
|
944
|
-
const fitted = [];
|
|
945
|
-
let tokens = remainingTokens;
|
|
946
|
-
let rows = remainingRows;
|
|
947
|
-
for (let index = units.length - 1; index >= 0; index -= 1) {
|
|
948
|
-
const unit = fitProviderUnit(units[index], tokens);
|
|
949
|
-
const cost = estimateMessagesTokens(unit);
|
|
950
|
-
if (unit.length > rows || cost > tokens) continue;
|
|
951
|
-
fitted.unshift(unit);
|
|
952
|
-
tokens -= cost;
|
|
953
|
-
rows -= unit.length;
|
|
954
|
-
}
|
|
955
|
-
current.push(...fitted.flat());
|
|
956
|
-
}
|
|
957
|
-
// The legacy fitter assumes a normal positive budget; at tiny allowances
|
|
958
|
-
// even an empty row's framing can exceed it. Remove complete tail units.
|
|
959
|
-
while (estimateMessagesTokens(current) > tokenBudget || current.length > messageCap) {
|
|
960
|
-
current = pairSanitize(current.slice(0, -1));
|
|
961
|
-
}
|
|
962
|
-
}
|
|
963
|
-
const availableTokens = Math.max(0, tokenBudget - estimateMessagesTokens(current));
|
|
964
|
-
const availableRows = Math.max(0, messageCap - current.length);
|
|
929
|
+
// The 32K/default budget owns only rows before currentStart. Keep the active
|
|
930
|
+
// turn intact here; whole-request fitting runs at every provider boundary and
|
|
931
|
+
// uses the actual model context window. Tool bodies may still receive their
|
|
932
|
+
// normal deterministic per-result truncation, without touching the durable
|
|
933
|
+
// transcript or charging that copy against history.
|
|
934
|
+
const current = pairSanitize(truncateToolResultsForModel(
|
|
935
|
+
currentSource.map(message => ({ ...message })), { language: options.language },
|
|
936
|
+
));
|
|
937
|
+
const availableTokens = tokenBudget;
|
|
938
|
+
const availableRows = messageCap;
|
|
965
939
|
let duplicateCount = 0;
|
|
966
940
|
const past = [];
|
|
967
941
|
for (const turn of splitBucketTurns(source.slice(0, currentStart))) {
|
|
@@ -1037,12 +1011,6 @@ export function buildHistoryBuckets(snapshot, options = {}) {
|
|
|
1037
1011
|
recentTokens += turn.tokens;
|
|
1038
1012
|
recentRows += turn.text.length;
|
|
1039
1013
|
}
|
|
1040
|
-
const minimumRecent = Math.min(MINIMUM_RECENT_PROVIDER_TURNS, recentCandidates.length);
|
|
1041
|
-
if (recent.length < minimumRecent) {
|
|
1042
|
-
const error = new Error(`Context budget cannot retain ${minimumRecent} complete recent history turns`);
|
|
1043
|
-
error.code = 'HISTORY_RECENT_BUDGET_EXCEEDED';
|
|
1044
|
-
throw error;
|
|
1045
|
-
}
|
|
1046
1014
|
let remainingTokens = availableTokens - recent.reduce((total, turn) => total + turn.tokens, 0);
|
|
1047
1015
|
let remainingRows = availableRows - recent.reduce((total, turn) => total + turn.text.length, 0);
|
|
1048
1016
|
const related = [];
|
|
@@ -1097,9 +1065,12 @@ export function buildHistoryBuckets(snapshot, options = {}) {
|
|
|
1097
1065
|
budget: {
|
|
1098
1066
|
messageTokenBudget: tokenBudget, maxMessageCount: messageCap,
|
|
1099
1067
|
recentTurnCap: recentCap, relatedTurnCap: relatedCap,
|
|
1100
|
-
minimumRecentTurns:
|
|
1068
|
+
minimumRecentTurns: 0,
|
|
1101
1069
|
relatedReservedTokens: 0, availableHistoryTokens: availableTokens,
|
|
1102
|
-
usedTokens: estimateMessagesTokens(
|
|
1070
|
+
usedTokens: estimateMessagesTokens([...relatedMessages, ...recentMessages]),
|
|
1071
|
+
usedMessages: relatedMessages.length + recentMessages.length,
|
|
1072
|
+
requestTokensBeforeWholeRequestFit: estimateMessagesTokens(messages),
|
|
1073
|
+
requestMessagesBeforeWholeRequestFit: messages.length,
|
|
1103
1074
|
},
|
|
1104
1075
|
dropped: {
|
|
1105
1076
|
pastTurnCount: droppedTurns.length,
|
|
@@ -1113,6 +1084,97 @@ export function buildHistoryBuckets(snapshot, options = {}) {
|
|
|
1113
1084
|
};
|
|
1114
1085
|
}
|
|
1115
1086
|
|
|
1087
|
+
/**
|
|
1088
|
+
* Fit one provider-request copy to the actual model window. The caller tells
|
|
1089
|
+
* us where current-turn rows begin; only the prefix is subject to the history
|
|
1090
|
+
* budget. If the complete request is still too large, old history disappears
|
|
1091
|
+
* first, followed by the oldest disposable current-turn protocol units. The
|
|
1092
|
+
* source array and durable transcript are never mutated.
|
|
1093
|
+
*
|
|
1094
|
+
* @param {Array<object>} messages
|
|
1095
|
+
* @param {{ contextWindow:number, systemTokens?:number, toolSchemaTokens?:number,
|
|
1096
|
+
* outputReserve?:number, historyMessageCount?:number, historyTokenBudget?:number,
|
|
1097
|
+
* maxMessageCount?:number, language?:string }} options
|
|
1098
|
+
* @returns {{messages:Array<object>, meta:object}}
|
|
1099
|
+
*/
|
|
1100
|
+
export function fitProviderRequestToContext(messages, options = {}) {
|
|
1101
|
+
const source = Array.isArray(messages) ? messages : [];
|
|
1102
|
+
const contextWindow = bucketCap(options.contextWindow, 0);
|
|
1103
|
+
const staticTokens = bucketCap(options.systemTokens, 0)
|
|
1104
|
+
+ bucketCap(options.toolSchemaTokens, 0)
|
|
1105
|
+
+ bucketCap(options.outputReserve, 0);
|
|
1106
|
+
const messageBudget = Math.max(0, contextWindow - staticTokens);
|
|
1107
|
+
const split = Math.max(0, Math.min(source.length,
|
|
1108
|
+
Number.isInteger(options.historyMessageCount) ? options.historyMessageCount : 0));
|
|
1109
|
+
// The runtime cache's 256-row cap is a history-storage concern, not a model
|
|
1110
|
+
// request limit. Current-turn tool loops may legitimately exceed it while
|
|
1111
|
+
// remaining inside the model window. Only enforce a cap when the caller
|
|
1112
|
+
// explicitly supplies one.
|
|
1113
|
+
const messageCap = options.maxMessageCount === undefined
|
|
1114
|
+
? Number.MAX_SAFE_INTEGER
|
|
1115
|
+
: bucketCap(options.maxMessageCount, Number.MAX_SAFE_INTEGER);
|
|
1116
|
+
const historySource = source.slice(0, split);
|
|
1117
|
+
const currentSource = source.slice(split);
|
|
1118
|
+
|
|
1119
|
+
let current = pairSanitize(truncateToolResultsForModel(
|
|
1120
|
+
currentSource.map(message => ({ ...message })), { language: options.language },
|
|
1121
|
+
));
|
|
1122
|
+
if (estimateMessagesTokens(current) > messageBudget || current.length > messageCap) {
|
|
1123
|
+
const fitted = [];
|
|
1124
|
+
if (current.length > 0 && messageBudget >= 2 && messageCap > 0) {
|
|
1125
|
+
const first = shrinkMessageToBudget(stripAllToolNoise([current[0]])[0], messageBudget);
|
|
1126
|
+
if (first && estimateMessageTokens(first) <= messageBudget) fitted.push(first);
|
|
1127
|
+
let tokens = messageBudget - estimateMessagesTokens(fitted);
|
|
1128
|
+
let rows = messageCap - fitted.length;
|
|
1129
|
+
const units = providerUnits(pairSanitize(current.slice(1)));
|
|
1130
|
+
const tail = [];
|
|
1131
|
+
for (let index = units.length - 1; index >= 0; index -= 1) {
|
|
1132
|
+
const unit = fitProviderUnit(units[index], tokens);
|
|
1133
|
+
const cost = estimateMessagesTokens(unit);
|
|
1134
|
+
if (unit.length > rows || cost > tokens) continue;
|
|
1135
|
+
tail.unshift(unit);
|
|
1136
|
+
tokens -= cost;
|
|
1137
|
+
rows -= unit.length;
|
|
1138
|
+
}
|
|
1139
|
+
fitted.push(...tail.flat());
|
|
1140
|
+
}
|
|
1141
|
+
current = pairSanitize(fitted);
|
|
1142
|
+
}
|
|
1143
|
+
|
|
1144
|
+
const configuredHistoryBudget = bucketCap(
|
|
1145
|
+
options.historyTokenBudget, DEFAULT_MESSAGE_TOKEN_BUDGET,
|
|
1146
|
+
);
|
|
1147
|
+
const remainingTokens = Math.max(0, Math.min(
|
|
1148
|
+
configuredHistoryBudget,
|
|
1149
|
+
messageBudget - estimateMessagesTokens(current),
|
|
1150
|
+
));
|
|
1151
|
+
const remainingRows = Math.max(0, messageCap - current.length);
|
|
1152
|
+
const history = remainingTokens >= 2 && remainingRows > 0
|
|
1153
|
+
? trimSnapshotForBudget(historySource, {
|
|
1154
|
+
messageTokenBudget: remainingTokens,
|
|
1155
|
+
maxMessageCount: remainingRows,
|
|
1156
|
+
recentTurnCap: Number.MAX_SAFE_INTEGER,
|
|
1157
|
+
language: options.language,
|
|
1158
|
+
})
|
|
1159
|
+
: [];
|
|
1160
|
+
const fittedMessages = [...history, ...current];
|
|
1161
|
+
return {
|
|
1162
|
+
messages: fittedMessages,
|
|
1163
|
+
meta: {
|
|
1164
|
+
contextWindow,
|
|
1165
|
+
staticTokens,
|
|
1166
|
+
messageBudget,
|
|
1167
|
+
estimatedTokens: staticTokens + estimateMessagesTokens(fittedMessages),
|
|
1168
|
+
historyMessagesBefore: historySource.length,
|
|
1169
|
+
historyMessagesAfter: history.length,
|
|
1170
|
+
currentMessagesBefore: currentSource.length,
|
|
1171
|
+
currentMessagesAfter: current.length,
|
|
1172
|
+
droppedHistoryMessages: historySource.length - history.length,
|
|
1173
|
+
droppedCurrentMessages: currentSource.length - current.length,
|
|
1174
|
+
},
|
|
1175
|
+
};
|
|
1176
|
+
}
|
|
1177
|
+
|
|
1116
1178
|
/**
|
|
1117
1179
|
* Bound the Session-level runtime history cache. This is deliberately stricter
|
|
1118
1180
|
* than the provider configuration: the cache is only a disposable source
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Non-blocking, post-response conversation compaction.
|
|
3
|
+
*
|
|
4
|
+
* The compact artifact is a derived provider-context cache. It never replaces
|
|
5
|
+
* or tombstones ConversationStore rows. A generation fence in Engine decides
|
|
6
|
+
* whether a completed artifact is still current before this module writes it.
|
|
7
|
+
*/
|
|
8
|
+
import { promises as fs } from 'fs';
|
|
9
|
+
import { dirname, join } from 'path';
|
|
10
|
+
|
|
11
|
+
export const POST_COMPACT_CONTEXT_RATIO = 0.8;
|
|
12
|
+
|
|
13
|
+
function safePart(value, fallback) {
|
|
14
|
+
const text = typeof value === 'string' && value.trim() ? value.trim() : fallback;
|
|
15
|
+
return encodeURIComponent(text).replace(/%/g, '_');
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
export function postCompactPath(yeaftDir, { sessionId, vpId, threadId } = {}) {
|
|
19
|
+
if (!yeaftDir || !sessionId) return null;
|
|
20
|
+
const file = `${safePart(vpId, 'default')}--${safePart(threadId, 'main')}.json`;
|
|
21
|
+
return join(yeaftDir, 'sessions', safePart(sessionId, 'session'), 'conversation', 'post-compact', file);
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
export async function loadPostCompact(path) {
|
|
25
|
+
if (!path) return null;
|
|
26
|
+
try {
|
|
27
|
+
const parsed = JSON.parse(await fs.readFile(path, 'utf8'));
|
|
28
|
+
return parsed && parsed.version === 1 && typeof parsed.summary === 'string'
|
|
29
|
+
? parsed : null;
|
|
30
|
+
} catch {
|
|
31
|
+
return null;
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
export async function savePostCompact(path, artifact, isCurrent = null) {
|
|
36
|
+
if (!path) return false;
|
|
37
|
+
await fs.mkdir(dirname(path), { recursive: true });
|
|
38
|
+
const temp = `${path}.${process.pid}.${Date.now()}.tmp`;
|
|
39
|
+
await fs.writeFile(temp, `${JSON.stringify({ version: 1, ...artifact }, null, 2)}\n`, 'utf8');
|
|
40
|
+
if (typeof isCurrent === 'function' && !isCurrent()) {
|
|
41
|
+
await fs.unlink(temp).catch(() => {});
|
|
42
|
+
return false;
|
|
43
|
+
}
|
|
44
|
+
await fs.rename(temp, path);
|
|
45
|
+
return true;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export async function removePostCompactIfSource(path, sourceTurnId) {
|
|
49
|
+
if (!path || !sourceTurnId) return false;
|
|
50
|
+
try {
|
|
51
|
+
const parsed = JSON.parse(await fs.readFile(path, 'utf8'));
|
|
52
|
+
if (parsed?.sourceTurnId !== sourceTurnId) return false;
|
|
53
|
+
await fs.unlink(path);
|
|
54
|
+
return true;
|
|
55
|
+
} catch {
|
|
56
|
+
return false;
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export async function generatePostCompact({ adapter, model, messages, maxTokens = 4096 }) {
|
|
61
|
+
const transcript = JSON.stringify((Array.isArray(messages) ? messages : []).map(message => ({
|
|
62
|
+
role: message?.role,
|
|
63
|
+
content: message?.content,
|
|
64
|
+
...(Array.isArray(message?.toolCalls) ? { toolCalls: message.toolCalls } : {}),
|
|
65
|
+
...(message?.toolCallId ? { toolCallId: message.toolCallId } : {}),
|
|
66
|
+
})));
|
|
67
|
+
const result = await adapter.call({
|
|
68
|
+
model,
|
|
69
|
+
system: 'Summarize the earlier conversation for use as context in a later turn. Preserve user goals, decisions, constraints, unresolved work, and important results. Omit raw tool payloads and do not invent facts. Return only the compact summary.',
|
|
70
|
+
messages: [{ role: 'user', content: `Compact this transcript:\n${transcript}` }],
|
|
71
|
+
maxTokens,
|
|
72
|
+
});
|
|
73
|
+
const summary = typeof result?.text === 'string' ? result.text.trim() : '';
|
|
74
|
+
if (!summary) throw new Error('post compact returned empty content');
|
|
75
|
+
return summary;
|
|
76
|
+
}
|
|
@@ -2,7 +2,8 @@
|
|
|
2
2
|
* tool-folding/index.js — V7 reflection subsystem entry (PR-L).
|
|
3
3
|
*
|
|
4
4
|
* Exposes:
|
|
5
|
-
* - Constants
|
|
5
|
+
* - Constants TOOL_LOOP_REFLECTION_INTERVAL, TURN_SUMMARY_THRESHOLD,
|
|
6
|
+
* DUP_TOOL_THRESHOLD
|
|
6
7
|
* - Reflector helpers (T1 sync, T2 async, fallback stub)
|
|
7
8
|
* - Helpers for collapsing message ranges into a single assistant
|
|
8
9
|
* reflection message
|
|
@@ -10,19 +11,10 @@
|
|
|
10
11
|
*
|
|
11
12
|
* The constants are NOT config-driven — V7 design freezes them in code.
|
|
12
13
|
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
* end-of-turn reflection path. Keep a usefully wide gap between the two
|
|
18
|
-
* so the (T2, T1) band where T2-alone applies stays meaningful.
|
|
19
|
-
*
|
|
20
|
-
* TOOL_BATCH_SIZE history: was 13 originally; raised to 30 (2026-05-15)
|
|
21
|
-
* after user feedback that 13 fired too often inside a single task and
|
|
22
|
-
* fragmented otherwise-coherent tool arcs into multiple reflections. 30
|
|
23
|
-
* keeps the periodic-reflection contract (it still fires every N tools,
|
|
24
|
-
* not just once) but gives a single task arc room to breathe before the
|
|
25
|
-
* arc gets collapsed.
|
|
14
|
+
* T1 runs inside the turn and collapses history in place. Its cadence is
|
|
15
|
+
* measured in provider tool loops (assistant tool_use batch → execution →
|
|
16
|
+
* next provider boundary), not in the number of calls inside a batch. A model
|
|
17
|
+
* returning 30 parallel tools has completed one loop, not thirty.
|
|
26
18
|
*
|
|
27
19
|
* TURN_SUMMARY_THRESHOLD history: was 5 originally; raised to 8
|
|
28
20
|
* (2026-05-18). 5 was too aggressive — small "read a few files, edit one,
|
|
@@ -32,7 +24,9 @@
|
|
|
32
24
|
* before the next turn's history grows.
|
|
33
25
|
*/
|
|
34
26
|
|
|
35
|
-
export const
|
|
27
|
+
export const TOOL_LOOP_REFLECTION_INTERVAL = 30;
|
|
28
|
+
// Compatibility for external imports; the engine uses the loop-specific name.
|
|
29
|
+
export const TOOL_BATCH_SIZE = TOOL_LOOP_REFLECTION_INTERVAL;
|
|
36
30
|
export const TURN_SUMMARY_THRESHOLD = 8;
|
|
37
31
|
export const DUP_TOOL_THRESHOLD = 3;
|
|
38
32
|
|
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* t1-reflector.js — V7 in-turn (synchronous) reflection (PR-L).
|
|
3
3
|
*
|
|
4
|
-
* Triggered
|
|
5
|
-
*
|
|
4
|
+
* Triggered after each interval of 30 completed tool loops, immediately before
|
|
5
|
+
* the engine loops back into adapter.stream(). Parallel calls returned in one
|
|
6
|
+
* assistant tool-use batch count as one loop. Calls
|
|
6
7
|
* the PRIMARY model — never the fast model — to generate a markdown
|
|
7
8
|
* reflection over the batch.
|
|
8
9
|
*
|