@yeaft/webchat-agent 1.0.508 → 1.0.510
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/connection/message-router.js +1 -1
- package/local-runtime/server/handlers/agent-sync.js +1 -0
- package/local-runtime/server/handlers/client-conversation.js +38 -2
- package/local-runtime/server/handlers/client-misc.js +1 -1
- package/local-runtime/version.json +1 -1
- package/local-runtime/web/app.bundle.js +212 -99
- package/local-runtime/web/app.bundle.js.gz +0 -0
- package/local-runtime/web/index.html +2 -2
- package/local-runtime/web/style.bundle.css +1 -1
- package/local-runtime/web/style.bundle.css.gz +0 -0
- package/package.json +1 -1
- package/yeaft/config-api.js +59 -6
- package/yeaft/config.js +3 -3
- package/yeaft/conversation/history-index-worker.js +3 -2
- package/yeaft/conversation/history-index.js +6 -4
- package/yeaft/conversation/persist.js +37 -0
- package/yeaft/conversation/recall-relevance.js +44 -11
- package/yeaft/engine.js +262 -56
- package/yeaft/history-window.js +128 -122
- package/yeaft/post-compact.js +76 -0
- package/yeaft/tool-folding/index.js +9 -15
- package/yeaft/tool-folding/t1-reflector.js +3 -2
- package/yeaft/web-bridge.js +130 -39
package/yeaft/engine.js
CHANGED
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
* 5. If tool_calls → execute tools → append results → goto 3
|
|
10
10
|
* 6. Persist each completed message at its durability boundary
|
|
11
11
|
* 7. If max_tokens → auto-continue (up to maxContinueTurns)
|
|
12
|
-
* 8. On LLMContextError →
|
|
12
|
+
* 8. On LLMContextError → shrink the request copy and retry without replaying tools
|
|
13
13
|
* 9. On retryable error with fallbackModel → switch model → retry
|
|
14
14
|
*
|
|
15
15
|
* Pattern derived from Claude Code's query loop (src/query.ts).
|
|
@@ -22,7 +22,7 @@ import { promises as fsp } from 'fs';
|
|
|
22
22
|
import { join, resolve as resolvePath } from 'path';
|
|
23
23
|
import { buildSystemPrompt, buildWorkerPrompt } from './prompts.js';
|
|
24
24
|
import { getRuntimePlatformInfo } from './runtime-platform.js';
|
|
25
|
-
import { LLMAbortError, LLMAuthError, LLMPolicyError, LLMRateLimitError, LLMServerError, LLMStreamIdleTimeoutError } from './llm/adapter.js';
|
|
25
|
+
import { LLMAbortError, LLMAuthError, LLMContextError, LLMPolicyError, LLMRateLimitError, LLMServerError, LLMStreamIdleTimeoutError } from './llm/adapter.js';
|
|
26
26
|
import { runMemoryPreflow, buildRelevantScopes, memoryScopeLabel } from './sessions/pre-flow.js';
|
|
27
27
|
import {
|
|
28
28
|
readProjectDoc,
|
|
@@ -33,7 +33,7 @@ import {
|
|
|
33
33
|
DEFAULT_PROJECT_DOC_MAX_BYTES,
|
|
34
34
|
} from './sessions/project-doc.js';
|
|
35
35
|
import { archiveToolResults } from './archive/tool-results.js';
|
|
36
|
-
import { trimSnapshotForBudget, estimateMessageTokens, buildHistoryBuckets } from './history-window.js';
|
|
36
|
+
import { trimSnapshotForBudget, estimateMessageTokens, buildHistoryBuckets, fitProviderRequestToContext } from './history-window.js';
|
|
37
37
|
import { recallConversationTurns } from './conversation/history-index.js';
|
|
38
38
|
import { parseSeqFromId } from './conversation/persist.js';
|
|
39
39
|
import { isVpForeign, readContent as readScopeContent } from './memory/store.js';
|
|
@@ -46,7 +46,15 @@ import { perfNowMs, recordAgentPerfTrace } from './perf-trace.js';
|
|
|
46
46
|
const MAIN_THREAD_ID = 'main';
|
|
47
47
|
import { pickEffort, parseEffortPrefix, snapshotEffortDecision } from './effort.js';
|
|
48
48
|
import { bindProviderState } from './llm/provider-state.js';
|
|
49
|
-
import {
|
|
49
|
+
import {
|
|
50
|
+
POST_COMPACT_CONTEXT_RATIO,
|
|
51
|
+
generatePostCompact,
|
|
52
|
+
loadPostCompact,
|
|
53
|
+
postCompactPath,
|
|
54
|
+
removePostCompactIfSource,
|
|
55
|
+
savePostCompact,
|
|
56
|
+
} from './post-compact.js';
|
|
57
|
+
import { DEFAULT_CONTEXT_WINDOW, getModelInfo, normalizeEffort, parseModelRef, resolveContextWindow, resolveMaxOutputTokens, resolveModel } from './models.js';
|
|
50
58
|
import { lookupModelLimitSync } from './llm/models-dev.js';
|
|
51
59
|
import { attachRouterPlan, extractPriorPlan, stripMetaForWire } from './router/continuity.js';
|
|
52
60
|
import { resolveThinking } from './router/thinking.js';
|
|
@@ -59,7 +67,7 @@ import { createPluginSkillManager } from './plugins.js';
|
|
|
59
67
|
import { extractDisplayImages, stripDisplayImageData } from './image-assets.js';
|
|
60
68
|
import { acknowledgePendingNotifications, formatNotificationsForPrompt, peekPendingNotifications } from './sub-agent/notifications.js';
|
|
61
69
|
import {
|
|
62
|
-
|
|
70
|
+
TOOL_LOOP_REFLECTION_INTERVAL,
|
|
63
71
|
TURN_SUMMARY_THRESHOLD,
|
|
64
72
|
DUP_TOOL_THRESHOLD,
|
|
65
73
|
ExecLog,
|
|
@@ -80,7 +88,7 @@ import {
|
|
|
80
88
|
* conversations (user report: Yeaft loop errored at the cap). The engine
|
|
81
89
|
* now runs until the LLM itself returns stopReason='end_turn' or a
|
|
82
90
|
* non-retryable error surfaces. Real runaway loops are still bounded by:
|
|
83
|
-
* • provider rate limits / context
|
|
91
|
+
* • provider rate limits / context recovery exhaustion
|
|
84
92
|
* • user-initiated abort (AbortController / cancel)
|
|
85
93
|
* • MAX_CONTINUE_TURNS for the max_tokens auto-continue path
|
|
86
94
|
*/
|
|
@@ -722,13 +730,18 @@ export class Engine {
|
|
|
722
730
|
* prior turn's history is rewritten with the reflection. If still
|
|
723
731
|
* pending, the engine falls back to the exec-log stub.
|
|
724
732
|
* • `#reflectedTurns` — Set<turnNumber>; ensures T1 fires at most
|
|
725
|
-
* once per
|
|
733
|
+
* once per reflection interval measured in tool loops.
|
|
726
734
|
*/
|
|
727
735
|
#execLog = null;
|
|
728
736
|
#pendingT2 = new Map();
|
|
729
737
|
#reflectedTurns = new Set();
|
|
730
738
|
#__queryCounter = 0;
|
|
731
739
|
|
|
740
|
+
/** Derived post-response summaries, keyed by Session/VP/thread scope. */
|
|
741
|
+
#postCompactSummaries = new Map();
|
|
742
|
+
#postCompactLoaded = new Set();
|
|
743
|
+
#postCompactRevisions = new Map();
|
|
744
|
+
|
|
732
745
|
/** @type {string} */
|
|
733
746
|
#currentThreadId = MAIN_THREAD_ID;
|
|
734
747
|
|
|
@@ -1882,7 +1895,7 @@ export class Engine {
|
|
|
1882
1895
|
}
|
|
1883
1896
|
}
|
|
1884
1897
|
|
|
1885
|
-
async *#queryLifecycle({ prompt, promptParts = null, messages = [], signal, userEffort = null, scenario = 'chat', isSubAgent = false, parentEffortDecision = null, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, currentUserMessage = null, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null } = {}) {
|
|
1898
|
+
async *#queryLifecycle({ prompt, promptParts = null, messages = [], signal, turnConfig = null, userEffort = null, scenario = 'chat', isSubAgent = false, parentEffortDecision = null, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, currentUserMessage = null, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null } = {}) {
|
|
1886
1899
|
if (!prompt || typeof prompt !== 'string' || !prompt.trim()) {
|
|
1887
1900
|
const error = new Error('prompt is required and must be a non-empty string');
|
|
1888
1901
|
yield {
|
|
@@ -1974,7 +1987,7 @@ export class Engine {
|
|
|
1974
1987
|
try {
|
|
1975
1988
|
this.#currentThreadId = threadId || MAIN_THREAD_ID;
|
|
1976
1989
|
this.#currentCausalRootId = effectiveCausalRootId;
|
|
1977
|
-
yield* this.#runQuery({ prompt: effectivePrompt, promptParts: effectivePromptParts, messages, signal: runSignal, userEffort: explicitUserEffort, scenario, isSubAgent, parentEffortDecision, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds, projectInstruction, projectLabel, vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted, currentUserMessage, causalRootId: effectiveCausalRootId, getCurrentTodos, setCurrentTodos, askUser, threadId: this.#currentThreadId, vpTurnId, drainPendingUserMessages, prepareProviderRequest, startProviderRequest, finishProviderRequest, failProviderRequest, closePendingUserInput, collabToolPolicy: effectiveCollabToolPolicy, explicitSkillName: parsedSkill.skillName, retryLifecycle });
|
|
1990
|
+
yield* this.#runQuery({ prompt: effectivePrompt, promptParts: effectivePromptParts, messages, signal: runSignal, turnConfig: turnConfig ? { model: turnConfig.model, effort: turnConfig.effort, maxOutputTokens: turnConfig.maxOutputTokens } : null, userEffort: explicitUserEffort, scenario, isSubAgent, parentEffortDecision, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds, projectInstruction, projectLabel, vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted, currentUserMessage, causalRootId: effectiveCausalRootId, getCurrentTodos, setCurrentTodos, askUser, threadId: this.#currentThreadId, vpTurnId, drainPendingUserMessages, prepareProviderRequest, startProviderRequest, finishProviderRequest, failProviderRequest, closePendingUserInput, collabToolPolicy: effectiveCollabToolPolicy, explicitSkillName: parsedSkill.skillName, retryLifecycle });
|
|
1978
1991
|
} finally {
|
|
1979
1992
|
// Closing the async generator at a visible retry boundary means the
|
|
1980
1993
|
// continuation never reached a provider. Keep it out of history and
|
|
@@ -2029,7 +2042,7 @@ export class Engine {
|
|
|
2029
2042
|
* in a try/finally without indenting the whole loop.
|
|
2030
2043
|
* @private
|
|
2031
2044
|
*/
|
|
2032
|
-
async *#runQuery({ prompt, promptParts = null, messages, signal, userEffort = null, scenario = 'chat', isSubAgent = false, parentEffortDecision = null, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, currentUserMessage = null, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null, explicitSkillName = null, retryLifecycle }) {
|
|
2045
|
+
async *#runQuery({ prompt, promptParts = null, messages, signal, turnConfig = null, userEffort = null, scenario = 'chat', isSubAgent = false, parentEffortDecision = null, vpPersona, router, senderVpId, inboundEnvelope, taskId, taskMembers, sessionId, sessionMembers, projectSessionIds = null, projectInstruction = '', projectLabel = '', vpPlan, sessionAnnouncement, workCenterInstructions, workDir, userAlreadyPersisted = false, currentUserMessage = null, causalRootId = null, getCurrentTodos = null, setCurrentTodos = null, askUser = null, threadId = MAIN_THREAD_ID, vpTurnId = null, drainPendingUserMessages = null, prepareProviderRequest = null, startProviderRequest = null, finishProviderRequest = null, failProviderRequest = null, closePendingUserInput = null, collabToolPolicy = null, explicitSkillName = null, retryLifecycle }) {
|
|
2033
2046
|
|
|
2034
2047
|
const effectiveCollabToolPolicy = collabToolPolicy === COLLAB_TOOL_POLICY.SINGLE_VP || collabToolPolicy === COLLAB_TOOL_POLICY.MULTI_VP
|
|
2035
2048
|
? collabToolPolicy
|
|
@@ -2069,6 +2082,12 @@ export class Engine {
|
|
|
2069
2082
|
&& typeof vpPersona.vpId === 'string'
|
|
2070
2083
|
? vpPersona.vpId
|
|
2071
2084
|
: (typeof senderVpId === 'string' ? senderVpId : null);
|
|
2085
|
+
const postCompactScope = this.#postCompactScope({
|
|
2086
|
+
sessionId: runtimeSessionId,
|
|
2087
|
+
vpId: queryVpId,
|
|
2088
|
+
threadId: runtimeThreadId,
|
|
2089
|
+
});
|
|
2090
|
+
const postCompactState = await this.#beginPostCompactScope(postCompactScope);
|
|
2072
2091
|
// Exact read-only tool results are safe to reuse within one query only
|
|
2073
2092
|
// when no intervening mutation can have changed the workspace. The map is
|
|
2074
2093
|
// intentionally local to this query; cross-turn reuse belongs to the
|
|
@@ -2113,8 +2132,8 @@ export class Engine {
|
|
|
2113
2132
|
// const useDreamMemory = scenario === 'work-item' || !!vpPersona?.subAgent
|
|
2114
2133
|
// || (!runtimeSessionId && !internalTrigger);
|
|
2115
2134
|
const useDreamMemory = false;
|
|
2116
|
-
const recentTurnCap = this.#config.yeaft?.recentTurnsLimit ?? 20;
|
|
2117
|
-
const relatedTurnCap = this.#config.yeaft?.relatedTurnsLimit ??
|
|
2135
|
+
const recentTurnCap = Math.max(20, this.#config.yeaft?.recentTurnsLimit ?? 20);
|
|
2136
|
+
const relatedTurnCap = Math.min(5, this.#config.yeaft?.relatedTurnsLimit ?? 5);
|
|
2118
2137
|
let relatedHistoryTurns = [];
|
|
2119
2138
|
let historyRecallMeta = { source: 'messages', status: 'disabled' };
|
|
2120
2139
|
if (useMessageHistory && !internalTrigger && this.#conversationStore?.loadRecentBySession) {
|
|
@@ -2130,7 +2149,9 @@ export class Engine {
|
|
|
2130
2149
|
const beforeSeq = Number.isFinite(persistedQueryUser?.seq)
|
|
2131
2150
|
? persistedQueryUser.seq : parseSeqFromId(persistedQueryUser?.id);
|
|
2132
2151
|
if (Number.isFinite(beforeSeq)) {
|
|
2133
|
-
const
|
|
2152
|
+
const loadHistory = this.#conversationStore.loadProviderHistoryBySession
|
|
2153
|
+
|| this.#conversationStore.loadRecentBySession;
|
|
2154
|
+
const tail = await loadHistory.call(this.#conversationStore, runtimeSessionId, recentTurnCap, { beforeSeq });
|
|
2134
2155
|
messages = tail.filter(m => parseSeqFromId(m.id) < beforeSeq
|
|
2135
2156
|
&& (m.role !== 'tool' || !queryVpId || m.speakerVpId === queryVpId))
|
|
2136
2157
|
.map(m => {
|
|
@@ -2247,6 +2268,10 @@ export class Engine {
|
|
|
2247
2268
|
if (amsContext && amsContext.snapshotBlock) {
|
|
2248
2269
|
memoryInjection = amsContext.snapshotBlock;
|
|
2249
2270
|
}
|
|
2271
|
+
if (postCompactState.summary) {
|
|
2272
|
+
const block = `## Prior conversation compact\n${postCompactState.summary}`;
|
|
2273
|
+
memoryInjection = memoryInjection ? `${memoryInjection}\n\n${block}` : block;
|
|
2274
|
+
}
|
|
2250
2275
|
const loadedMemoryForDebug = loadedMemoryDebugEntries(amsContext?.snapshot);
|
|
2251
2276
|
const loadedMemoryMetaForDebug = {
|
|
2252
2277
|
...(useMessageHistory ? historyRecallMeta : {}),
|
|
@@ -2473,18 +2498,18 @@ export class Engine {
|
|
|
2473
2498
|
// `turnStartIdx` is where the current user message lives; the arc
|
|
2474
2499
|
// we may collapse spans (arcStartIdx .. last assistant/tool).
|
|
2475
2500
|
//
|
|
2476
|
-
// Periodic
|
|
2477
|
-
//
|
|
2501
|
+
// Periodic T1 fires every 30 tool loops, not every 30 calls. A single
|
|
2502
|
+
// provider batch can contain many parallel calls but is still one loop.
|
|
2478
2503
|
// track:
|
|
2479
|
-
// • `
|
|
2504
|
+
// • `lastT1AtLoopCount` — tool-loop snapshot at the last T1
|
|
2480
2505
|
// ATTEMPT (success OR error). Trigger when
|
|
2481
|
-
// `
|
|
2506
|
+
// `completedToolLoops - lastT1AtLoopCount >= interval`.
|
|
2482
2507
|
// • `arcStartIdx` — first index of the current (uncollapsed)
|
|
2483
2508
|
// tool arc. Initialised to turnStartIdx + 1; reset after each
|
|
2484
2509
|
// successful T1 collapse to `conversationMessages.length`
|
|
2485
2510
|
// (i.e. the slot the next assistant message will land in).
|
|
2486
2511
|
// • `t1CollapsesDone` — count of T1 firings that ACTUALLY
|
|
2487
|
-
// rewrote history. Distinct from `
|
|
2512
|
+
// rewrote history. Distinct from `lastT1AtLoopCount` because
|
|
2488
2513
|
// the catch block bumps the latter to back off after a
|
|
2489
2514
|
// transient reflector error WITHOUT having collapsed
|
|
2490
2515
|
// anything. The T2 schedule check below is gated on this
|
|
@@ -2492,7 +2517,7 @@ export class Engine {
|
|
|
2492
2517
|
// fall back at end_turn").
|
|
2493
2518
|
const turnStartIdx = conversationMessages.length - 1;
|
|
2494
2519
|
let queryToolCount = 0;
|
|
2495
|
-
let
|
|
2520
|
+
let lastT1AtLoopCount = 0;
|
|
2496
2521
|
let arcStartIdx = turnStartIdx + 1;
|
|
2497
2522
|
let t1CollapsesDone = 0;
|
|
2498
2523
|
// Duplicate policy is scoped to one user query. Only successful, real
|
|
@@ -2555,10 +2580,14 @@ export class Engine {
|
|
|
2555
2580
|
// `refreshConfig()` may publish a new Session model while a stream or a
|
|
2556
2581
|
// tool is running. Apply it only before the next provider request; the
|
|
2557
2582
|
// current request keeps the snapshot captured below.
|
|
2558
|
-
let currentModel = this.#config.model;
|
|
2583
|
+
let currentModel = turnConfig?.model || this.#config.model;
|
|
2559
2584
|
let primaryModelAtLastBoundary = currentModel;
|
|
2560
2585
|
let cumulativeInputTokens = 0;
|
|
2561
2586
|
let cumulativeOutputTokens = 0;
|
|
2587
|
+
let maxContextOccupancyRatio = 0;
|
|
2588
|
+
let peakContextTokens = 0;
|
|
2589
|
+
let peakContextWindow = 0;
|
|
2590
|
+
let postCompactCandidate = null;
|
|
2562
2591
|
let activeProviderRequest = null;
|
|
2563
2592
|
// Skill events describe the selection injected into each provider request.
|
|
2564
2593
|
// The first request must report its initial selection; later loops report
|
|
@@ -2590,6 +2619,11 @@ export class Engine {
|
|
|
2590
2619
|
let consecutiveRetryableErrors = 0;
|
|
2591
2620
|
let consecutiveForbiddenErrors = 0;
|
|
2592
2621
|
let contentPolicyRecoveryAttempts = 0;
|
|
2622
|
+
let contextOverflowRecoveryAttempts = 0;
|
|
2623
|
+
// A provider can know about framing/tokenizer overhead that our monotonic
|
|
2624
|
+
// estimator cannot. Each real overflow retries the same unexecuted request
|
|
2625
|
+
// with a smaller provider-only window; no transcript row is rewritten.
|
|
2626
|
+
let providerContextScale = 1;
|
|
2593
2627
|
|
|
2594
2628
|
while (true) {
|
|
2595
2629
|
turnNumber++;
|
|
@@ -2600,7 +2634,7 @@ export class Engine {
|
|
|
2600
2634
|
// Keep a retry fallback selected by this query; replacing it here would
|
|
2601
2635
|
// turn an exhausted primary into an endless retry loop.
|
|
2602
2636
|
if (currentModel === primaryModelAtLastBoundary) {
|
|
2603
|
-
const refreshedPrimaryModel = this.#config.model;
|
|
2637
|
+
const refreshedPrimaryModel = turnConfig?.model || this.#config.model;
|
|
2604
2638
|
if (refreshedPrimaryModel !== primaryModelAtLastBoundary) {
|
|
2605
2639
|
currentModel = refreshedPrimaryModel;
|
|
2606
2640
|
primaryModelAtLastBoundary = refreshedPrimaryModel;
|
|
@@ -2612,6 +2646,21 @@ export class Engine {
|
|
|
2612
2646
|
// this request. Fallback retries intentionally retain their selected
|
|
2613
2647
|
// model, but still use the current policy and configured effort.
|
|
2614
2648
|
const requestConfig = { ...this.#config };
|
|
2649
|
+
// Overlay only the request snapshot. Never publish temporary settings via
|
|
2650
|
+
// refreshConfig or mutate the shared config / AdapterRouter catalog.
|
|
2651
|
+
if (turnConfig) {
|
|
2652
|
+
requestConfig.model = currentModel;
|
|
2653
|
+
const entry = requestConfig.availableModels?.find(model => model.ref === currentModel);
|
|
2654
|
+
requestConfig.modelInfo = getModelInfo(parseModelRef(currentModel).modelId, entry) || null;
|
|
2655
|
+
if (turnConfig.effort != null) requestConfig.modelEffort = turnConfig.effort;
|
|
2656
|
+
// Blank means the selected model's default cap, not the Session's
|
|
2657
|
+
// previous model budget. Re-resolve at every boundary (including fallback
|
|
2658
|
+
// retries / catalog refresh), without leaking the old global ceiling.
|
|
2659
|
+
const outputLimit = resolveMaxOutputTokens(parseModelRef(currentModel).modelId, {
|
|
2660
|
+
modelInfo: requestConfig.modelInfo,
|
|
2661
|
+
});
|
|
2662
|
+
requestConfig.maxOutputTokens = Math.min(turnConfig.maxOutputTokens ?? outputLimit, outputLimit);
|
|
2663
|
+
}
|
|
2615
2664
|
// Capture the matching provider catalog in the same synchronous boundary
|
|
2616
2665
|
// as config/model. Preflight may yield user/task events before the stream
|
|
2617
2666
|
// is built, but one request must never mix two refresh revisions.
|
|
@@ -2808,7 +2857,7 @@ export class Engine {
|
|
|
2808
2857
|
// effect at the next loop. A caller override or `/effort` prefix stays
|
|
2809
2858
|
// fixed for this query and still wins over live Session config.
|
|
2810
2859
|
const configuredEffort = normalizeEffort(requestConfig.modelEffort);
|
|
2811
|
-
const requestUserEffort = userEffort || configuredEffort || null;
|
|
2860
|
+
const requestUserEffort = normalizeEffort(turnConfig?.effort) || userEffort || configuredEffort || null;
|
|
2812
2861
|
let resolvedEffort = pickEffort({ scenario, toolLoopTurns, userEffort: requestUserEffort });
|
|
2813
2862
|
|
|
2814
2863
|
// DESIGN.md §9.16: thinking-mode precedence chain. When a VP
|
|
@@ -2860,25 +2909,30 @@ export class Engine {
|
|
|
2860
2909
|
// query tape remain complete; no summary is generated and no history
|
|
2861
2910
|
// rows are rewritten. This also protects later tool-loop requests,
|
|
2862
2911
|
// not just the initial snapshot assembled by the bridge.
|
|
2863
|
-
|
|
2864
|
-
|
|
2865
|
-
const historyBudget = Math.max(
|
|
2866
|
-
requestConfig.messageTokenBudget || 32768,
|
|
2867
|
-
Math.floor(currentContextWindow * 0.75) - estimateMessagesTokens(systemPrompt, []),
|
|
2868
|
-
) - continuationCost);
|
|
2912
|
+
// Retry continuation is part of the active turn. Like the opening user
|
|
2913
|
+
// row and tool-loop traffic, it must not consume the 32K history budget.
|
|
2914
|
+
const historyBudget = Math.max(0, requestConfig.messageTokenBudget || 32768);
|
|
2869
2915
|
const buckets = useMessageHistory ? buildHistoryBuckets(conversationMessages, {
|
|
2870
2916
|
prompt,
|
|
2871
2917
|
relatedTurns: relatedHistoryTurns,
|
|
2872
|
-
recentTurnCap: requestConfig.yeaft?.recentTurnsLimit ?? 20,
|
|
2873
|
-
relatedTurnCap: requestConfig.yeaft?.relatedTurnsLimit ??
|
|
2918
|
+
recentTurnCap: Math.max(20, requestConfig.yeaft?.recentTurnsLimit ?? 20),
|
|
2919
|
+
relatedTurnCap: Math.min(5, requestConfig.yeaft?.relatedTurnsLimit ?? 5),
|
|
2874
2920
|
messageTokenBudget: historyBudget,
|
|
2875
2921
|
currentTurnStartIndex: turnStartIdx,
|
|
2876
2922
|
language: requestConfig.language,
|
|
2877
2923
|
}) : null;
|
|
2878
|
-
|
|
2879
|
-
|
|
2880
|
-
|
|
2881
|
-
|
|
2924
|
+
let historyMessageCount;
|
|
2925
|
+
const requestHistory = buckets?.messages || (() => {
|
|
2926
|
+
const historical = trimSnapshotForBudget(conversationMessages.slice(0, turnStartIdx), {
|
|
2927
|
+
messageTokenBudget: historyBudget,
|
|
2928
|
+
language: requestConfig.language,
|
|
2929
|
+
});
|
|
2930
|
+
historyMessageCount = historical.length;
|
|
2931
|
+
return [...historical, ...conversationMessages.slice(turnStartIdx)];
|
|
2932
|
+
})();
|
|
2933
|
+
if (buckets) {
|
|
2934
|
+
historyMessageCount = requestHistory.length - (buckets.meta?.current?.messageCount || 0);
|
|
2935
|
+
}
|
|
2882
2936
|
if (buckets) this.#trace.log?.('history_buckets', {
|
|
2883
2937
|
sessionId: runtimeSessionId, turnId: queryTurnId, ...historyRecallMeta, ...buckets.meta,
|
|
2884
2938
|
});
|
|
@@ -2902,6 +2956,39 @@ export class Engine {
|
|
|
2902
2956
|
} catch { /* best-effort */ }
|
|
2903
2957
|
}
|
|
2904
2958
|
|
|
2959
|
+
// Final request boundary: account for every component against the
|
|
2960
|
+
// actual model window. The 32K budget above applies only to historical
|
|
2961
|
+
// rows; current-turn rows are paid here together with system, schemas,
|
|
2962
|
+
// and the model-specific output reserve. This runs on every tool-loop
|
|
2963
|
+
// request and again with tighter headroom after a provider overflow.
|
|
2964
|
+
const requestMaxOutputTokens = Math.max(1, Math.min(
|
|
2965
|
+
requestConfig.maxOutputTokens || resolveMaxOutputTokens(currentModel, requestConfig),
|
|
2966
|
+
resolveMaxOutputTokens(currentModel, requestConfig),
|
|
2967
|
+
));
|
|
2968
|
+
const toolSchemaTokens = toolDefs.length > 0
|
|
2969
|
+
? approxTokens(JSON.stringify(toolDefs)) : 0;
|
|
2970
|
+
const fittedRequest = fitProviderRequestToContext(wireMessages, {
|
|
2971
|
+
contextWindow: Math.max(1, Math.floor(currentContextWindow * providerContextScale)),
|
|
2972
|
+
systemTokens: estimateMessagesTokens(systemPrompt, []),
|
|
2973
|
+
toolSchemaTokens,
|
|
2974
|
+
outputReserve: requestMaxOutputTokens,
|
|
2975
|
+
historyMessageCount,
|
|
2976
|
+
historyTokenBudget: historyBudget,
|
|
2977
|
+
language: requestConfig.language,
|
|
2978
|
+
});
|
|
2979
|
+
wireMessages = fittedRequest.messages;
|
|
2980
|
+
if (fittedRequest.meta.droppedHistoryMessages > 0
|
|
2981
|
+
|| fittedRequest.meta.droppedCurrentMessages > 0
|
|
2982
|
+
|| providerContextScale < 1) {
|
|
2983
|
+
this.#trace.log?.('request_context_trim', {
|
|
2984
|
+
sessionId: runtimeSessionId,
|
|
2985
|
+
turnId: queryTurnId,
|
|
2986
|
+
model: currentModel,
|
|
2987
|
+
recoveryAttempt: contextOverflowRecoveryAttempts,
|
|
2988
|
+
...fittedRequest.meta,
|
|
2989
|
+
});
|
|
2990
|
+
}
|
|
2991
|
+
|
|
2905
2992
|
// task-704b: pre-flight total-token guard. Even with the per-tool
|
|
2906
2993
|
// cap (registry.js: 10% of contextWindow per result), N tool
|
|
2907
2994
|
// results plus history can still breach the wire limit before we
|
|
@@ -2988,7 +3075,7 @@ export class Engine {
|
|
|
2988
3075
|
system: systemPrompt,
|
|
2989
3076
|
messages: wireMessages,
|
|
2990
3077
|
tools: toolDefs.length > 0 ? toolDefs : undefined,
|
|
2991
|
-
maxTokens:
|
|
3078
|
+
maxTokens: requestMaxOutputTokens,
|
|
2992
3079
|
effort: resolvedEffort,
|
|
2993
3080
|
effortConstraint,
|
|
2994
3081
|
requestIdentity,
|
|
@@ -3166,6 +3253,35 @@ export class Engine {
|
|
|
3166
3253
|
responseTextBytes: Buffer.byteLength(responseText, 'utf8'),
|
|
3167
3254
|
},
|
|
3168
3255
|
});
|
|
3256
|
+
const requestContextOccupancy = (totalUsage.inputTokens || 0)
|
|
3257
|
+
+ (totalUsage.cacheInputDeltaTokens || 0)
|
|
3258
|
+
+ (totalUsage.outputTokens || 0);
|
|
3259
|
+
const requestContextOccupancyRatio = requestContextOccupancy / currentContextWindow;
|
|
3260
|
+
if (requestContextOccupancyRatio >= maxContextOccupancyRatio) {
|
|
3261
|
+
maxContextOccupancyRatio = requestContextOccupancyRatio;
|
|
3262
|
+
peakContextTokens = requestContextOccupancy;
|
|
3263
|
+
peakContextWindow = currentContextWindow;
|
|
3264
|
+
}
|
|
3265
|
+
// Trigger from the peak request, but summarize the latest provider
|
|
3266
|
+
// state so an earlier, fuller tool loop cannot omit later reflection,
|
|
3267
|
+
// tool results, or the final answer from the derived artifact.
|
|
3268
|
+
postCompactCandidate = {
|
|
3269
|
+
scope: postCompactScope,
|
|
3270
|
+
revision: postCompactState.revision,
|
|
3271
|
+
sessionId: runtimeSessionId,
|
|
3272
|
+
turnId: queryTurnId,
|
|
3273
|
+
model: currentModel,
|
|
3274
|
+
config: requestConfig,
|
|
3275
|
+
adapter: requestAdapter,
|
|
3276
|
+
messages: [
|
|
3277
|
+
...wireMessages.map(message => ({ ...message })),
|
|
3278
|
+
...(toolCalls.length === 0 && responseText
|
|
3279
|
+
? [{ role: 'assistant', content: responseText }]
|
|
3280
|
+
: []),
|
|
3281
|
+
],
|
|
3282
|
+
contextTokens: peakContextTokens,
|
|
3283
|
+
contextWindow: peakContextWindow,
|
|
3284
|
+
};
|
|
3169
3285
|
// Stream completed without throwing — reset the retry counter so
|
|
3170
3286
|
// the next turn starts with a clean budget. In-band adapter errors
|
|
3171
3287
|
// are converted to throws above so they share the real error path.
|
|
@@ -3281,11 +3397,32 @@ export class Engine {
|
|
|
3281
3397
|
const earlyIsRateLimit = err instanceof LLMRateLimitError;
|
|
3282
3398
|
const earlyIsTransient = err instanceof LLMServerError;
|
|
3283
3399
|
const earlyIsContentPolicy = err instanceof LLMPolicyError;
|
|
3400
|
+
const earlyIsContextOverflow = err instanceof LLMContextError;
|
|
3284
3401
|
// A completed tool_call has already crossed the streaming boundary to
|
|
3285
3402
|
// the caller. Replaying that request would publish a duplicate call and
|
|
3286
3403
|
// leave ambiguous execution ownership, so only pre-tool failures are
|
|
3287
3404
|
// eligible for transparent retry or model fallback.
|
|
3288
3405
|
const canReplayProviderRequest = toolCalls.length === 0;
|
|
3406
|
+
if (earlyIsContextOverflow && canReplayProviderRequest
|
|
3407
|
+
&& contextOverflowRecoveryAttempts < 3) {
|
|
3408
|
+
contextOverflowRecoveryAttempts += 1;
|
|
3409
|
+
providerContextScale *= 0.75;
|
|
3410
|
+
endAttemptTrace('context_overflow_retry');
|
|
3411
|
+
if (responseText) prepareRetryContinuation();
|
|
3412
|
+
yield {
|
|
3413
|
+
type: 'llm_retry',
|
|
3414
|
+
attempt: contextOverflowRecoveryAttempts,
|
|
3415
|
+
maxRetries: 3,
|
|
3416
|
+
delayMs: 0,
|
|
3417
|
+
reason: 'context_overflow_recovery',
|
|
3418
|
+
recoveryMode: responseText ? 'continue' : 'restart',
|
|
3419
|
+
errorName: err.name,
|
|
3420
|
+
statusCode: err.statusCode ?? null,
|
|
3421
|
+
message: 'Provider rejected the context; retrying with a smaller request copy.',
|
|
3422
|
+
};
|
|
3423
|
+
yield { type: 'turn_end', turnNumber, stopReason: 'llm_retry', threadId };
|
|
3424
|
+
continue;
|
|
3425
|
+
}
|
|
3289
3426
|
if (earlyIsContentPolicy && canReplayProviderRequest && contentPolicyRecoveryAttempts === 0) {
|
|
3290
3427
|
contentPolicyRecoveryAttempts = 1;
|
|
3291
3428
|
endAttemptTrace('llm_retry');
|
|
@@ -3879,8 +4016,8 @@ export class Engine {
|
|
|
3879
4016
|
// `#applyPendingT2Reflections` carries the result forward.
|
|
3880
4017
|
//
|
|
3881
4018
|
// Periodic-T1 fix: gate on `t1CollapsesDone === 0`, NOT
|
|
3882
|
-
// `
|
|
3883
|
-
// `
|
|
4019
|
+
// `lastT1AtLoopCount === 0`. The catch block of T1 advances
|
|
4020
|
+
// `lastT1AtLoopCount` after a reflector error to avoid
|
|
3884
4021
|
// tight-loop retries — but no collapse happened, so T2 should
|
|
3885
4022
|
// still be allowed to fall back at end_turn. Fowler-review
|
|
3886
4023
|
// critical finding.
|
|
@@ -4680,29 +4817,28 @@ export class Engine {
|
|
|
4680
4817
|
break;
|
|
4681
4818
|
}
|
|
4682
4819
|
|
|
4683
|
-
// PR-L: T1 in-turn (synchronous) reflection. Fires
|
|
4684
|
-
//
|
|
4685
|
-
//
|
|
4686
|
-
// the first batch of the query(). Generates a markdown reflection
|
|
4820
|
+
// PR-L: T1 in-turn (synchronous) reflection. Fires every 30 completed
|
|
4821
|
+
// tool loops. A provider response containing many parallel tool calls is
|
|
4822
|
+
// one loop, not many. Generates a markdown reflection
|
|
4687
4823
|
// over the assistant+tool arc since the last T1 firing (or
|
|
4688
4824
|
// since the user prompt for the first batch) and rewrites that
|
|
4689
4825
|
// range to a SINGLE synthetic user message before the next
|
|
4690
4826
|
// adapter.stream() runs.
|
|
4691
4827
|
//
|
|
4692
4828
|
// Loop semantics:
|
|
4693
|
-
// - First
|
|
4694
|
-
//
|
|
4695
|
-
// - Each subsequent batch: arcStartIdx is updated to the slot
|
|
4829
|
+
// - First interval: arcStartIdx = turnStartIdx + 1.
|
|
4830
|
+
// - Each subsequent interval: arcStartIdx is updated to the slot
|
|
4696
4831
|
// right after the just-inserted reflection message; fires
|
|
4697
|
-
// again whenever
|
|
4698
|
-
//
|
|
4699
|
-
//
|
|
4700
|
-
// batch within the same query gets a distinct entry — without
|
|
4832
|
+
// again whenever 30 more tool loops have completed.
|
|
4833
|
+
// - The dedup Set key includes the loop count so each interval
|
|
4834
|
+
// within the same query gets a distinct entry — without
|
|
4701
4835
|
// this the second batch would be silently skipped.
|
|
4702
|
-
const
|
|
4703
|
-
|
|
4836
|
+
const completedToolLoops = toolLoopTurns + 1;
|
|
4837
|
+
const t1BatchDue = completedToolLoops - lastT1AtLoopCount
|
|
4838
|
+
>= TOOL_LOOP_REFLECTION_INTERVAL;
|
|
4839
|
+
if (t1BatchDue && !toolBatchBarrier
|
|
4704
4840
|
&& !abortedDuringTools && !signal?.aborted) {
|
|
4705
|
-
const t1DedupKey = `${queryNumber}:t1:${
|
|
4841
|
+
const t1DedupKey = `${queryNumber}:t1-loop:${completedToolLoops}`;
|
|
4706
4842
|
if (this.#reflectedTurns.has(t1DedupKey)) {
|
|
4707
4843
|
// Defensive: should never hit since t1BatchDue gates re-entry
|
|
4708
4844
|
// and queryNumber namespaces queries. Kept as belt-and-
|
|
@@ -4770,10 +4906,10 @@ export class Engine {
|
|
|
4770
4906
|
// immediately after it, i.e. at conversationMessages.length
|
|
4771
4907
|
// (the next assistant message will land here).
|
|
4772
4908
|
arcStartIdx = conversationMessages.length;
|
|
4773
|
-
|
|
4909
|
+
lastT1AtLoopCount = completedToolLoops;
|
|
4774
4910
|
// Bump the success counter — used by the T2 schedule check
|
|
4775
4911
|
// to decide whether T2 still has work to do at end_turn.
|
|
4776
|
-
// Distinct from
|
|
4912
|
+
// Distinct from lastT1AtLoopCount which the catch block
|
|
4777
4913
|
// also bumps (but without rewriting history).
|
|
4778
4914
|
t1CollapsesDone += 1;
|
|
4779
4915
|
yield {
|
|
@@ -4803,9 +4939,9 @@ export class Engine {
|
|
|
4803
4939
|
status: 'error',
|
|
4804
4940
|
error: err && err.message || String(err),
|
|
4805
4941
|
};
|
|
4806
|
-
// Advance
|
|
4942
|
+
// Advance lastT1AtLoopCount past this interval so we don't
|
|
4807
4943
|
// tight-loop on a hiccuping reflector. The next attempt is
|
|
4808
|
-
//
|
|
4944
|
+
// another 30 tool loops from now, not immediately. arcStartIdx is
|
|
4809
4945
|
// left alone because history wasn't rewritten — the tail still
|
|
4810
4946
|
// begins where it did. The trade-off: the next batch's
|
|
4811
4947
|
// reflection will cover the tools that just failed too,
|
|
@@ -4814,7 +4950,7 @@ export class Engine {
|
|
|
4814
4950
|
// We do NOT bump t1CollapsesDone — see the variable's
|
|
4815
4951
|
// declaration comment. This keeps the T2 fallback path live
|
|
4816
4952
|
// when every T1 attempt has errored.
|
|
4817
|
-
|
|
4953
|
+
lastT1AtLoopCount = completedToolLoops;
|
|
4818
4954
|
}
|
|
4819
4955
|
}
|
|
4820
4956
|
}
|
|
@@ -4862,6 +4998,76 @@ export class Engine {
|
|
|
4862
4998
|
totalTokens: cumulativeInputTokens + cumulativeOutputTokens,
|
|
4863
4999
|
loopCount: turnNumber,
|
|
4864
5000
|
};
|
|
5001
|
+
|
|
5002
|
+
// The visible response is complete at the yield above. Only when the
|
|
5003
|
+
// consumer resumes past that boundary do we inspect pressure and launch
|
|
5004
|
+
// best-effort post compact. It never blocks this turn or the next one.
|
|
5005
|
+
if (postCompactCandidate
|
|
5006
|
+
&& maxContextOccupancyRatio >= POST_COMPACT_CONTEXT_RATIO) {
|
|
5007
|
+
this.#schedulePostCompact(postCompactCandidate);
|
|
5008
|
+
}
|
|
5009
|
+
}
|
|
5010
|
+
|
|
5011
|
+
#postCompactScope({ sessionId, vpId, threadId }) {
|
|
5012
|
+
if (!this.#yeaftDir || !sessionId) return null;
|
|
5013
|
+
const path = postCompactPath(this.#yeaftDir, { sessionId, vpId, threadId });
|
|
5014
|
+
return { key: path, path };
|
|
5015
|
+
}
|
|
5016
|
+
|
|
5017
|
+
async #beginPostCompactScope(scope) {
|
|
5018
|
+
if (!scope) return { revision: 0, summary: '' };
|
|
5019
|
+
const revision = (this.#postCompactRevisions.get(scope.key) || 0) + 1;
|
|
5020
|
+
this.#postCompactRevisions.set(scope.key, revision);
|
|
5021
|
+
if (!this.#postCompactLoaded.has(scope.key)) {
|
|
5022
|
+
this.#postCompactLoaded.add(scope.key);
|
|
5023
|
+
const artifact = await loadPostCompact(scope.path);
|
|
5024
|
+
if (artifact) this.#postCompactSummaries.set(scope.key, artifact);
|
|
5025
|
+
}
|
|
5026
|
+
return {
|
|
5027
|
+
revision,
|
|
5028
|
+
summary: this.#postCompactSummaries.get(scope.key)?.summary || '',
|
|
5029
|
+
};
|
|
5030
|
+
}
|
|
5031
|
+
|
|
5032
|
+
#schedulePostCompact(candidate) {
|
|
5033
|
+
const { scope, revision } = candidate;
|
|
5034
|
+
if (!scope || this.#postCompactRevisions.get(scope.key) !== revision) return;
|
|
5035
|
+
const compactMaxTokens = Math.max(512, Math.min(4096,
|
|
5036
|
+
resolveMaxOutputTokens(candidate.model, candidate.config)));
|
|
5037
|
+
const task = async () => {
|
|
5038
|
+
try {
|
|
5039
|
+
const summary = await generatePostCompact({
|
|
5040
|
+
adapter: candidate.adapter,
|
|
5041
|
+
model: candidate.model,
|
|
5042
|
+
messages: candidate.messages,
|
|
5043
|
+
maxTokens: compactMaxTokens,
|
|
5044
|
+
});
|
|
5045
|
+
const current = () => this.#postCompactRevisions.get(scope.key) === revision;
|
|
5046
|
+
if (!current()) return;
|
|
5047
|
+
const artifact = {
|
|
5048
|
+
summary,
|
|
5049
|
+
model: candidate.model,
|
|
5050
|
+
sourceTurnId: candidate.turnId,
|
|
5051
|
+
sourceRevision: revision,
|
|
5052
|
+
sourceContextTokens: candidate.contextTokens,
|
|
5053
|
+
contextWindow: candidate.contextWindow,
|
|
5054
|
+
createdAt: new Date().toISOString(),
|
|
5055
|
+
};
|
|
5056
|
+
if (await savePostCompact(scope.path, artifact, current)) {
|
|
5057
|
+
if (current()) this.#postCompactSummaries.set(scope.key, artifact);
|
|
5058
|
+
else await removePostCompactIfSource(scope.path, candidate.turnId);
|
|
5059
|
+
}
|
|
5060
|
+
} catch (error) {
|
|
5061
|
+
// A response already reached the user. Compact failure is diagnostic
|
|
5062
|
+
// only and must not create a late error event.
|
|
5063
|
+
this.#trace.log?.('post_compact_failed', {
|
|
5064
|
+
sessionId: candidate.sessionId,
|
|
5065
|
+
turnId: candidate.turnId,
|
|
5066
|
+
message: String(error?.message || error).slice(0, 200),
|
|
5067
|
+
});
|
|
5068
|
+
}
|
|
5069
|
+
};
|
|
5070
|
+
void task();
|
|
4865
5071
|
}
|
|
4866
5072
|
|
|
4867
5073
|
/**
|