@yeaft/webchat-agent 1.0.509 → 1.0.511

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@yeaft/webchat-agent",
3
- "version": "1.0.509",
3
+ "version": "1.0.511",
4
4
  "description": "Remote worker agent for Yeaft Web Code Agent — connects the native Yeaft engine, CLI providers, and workbench tools",
5
5
  "main": "index.js",
6
6
  "type": "module",
package/yeaft/cli.js CHANGED
@@ -380,7 +380,7 @@ async function runREPL(config, args) {
380
380
 
381
381
  // Load persisted conversation as initial messages. `loadRecent` is now
382
382
  // turn-based (one user round-trip = one turn; multi-VP fan-out collapses
383
- // into one turn). 20 turns is the bootstrap window. Provider requests
383
+ // into one turn). 10 turns is the bootstrap window. Provider requests
384
384
  // apply deterministic history-window trimming; persisted history stays
385
385
  // authoritative and no LLM conversation summary is generated.
386
386
  let conversationMessages = conversationStore.loadRecent().map(m => ({
package/yeaft/config.js CHANGED
@@ -53,7 +53,7 @@ const DEFAULTS = {
53
53
  // ConversationStore.loadRecentBySession / loadSessionHistoryForVp
54
54
  // bring back after boot or reconnect. Older transcript remains available
55
55
  // through history pagination/search. Range: 1–500.
56
- yeaftRecentTurnsLimit: 20,
56
+ yeaftRecentTurnsLimit: 10,
57
57
  // Same-Session related Q&A turns, selected by deterministic full-text rules.
58
58
  // Range: 0–5; 0 disables related recall without changing recent history.
59
59
  yeaftRelatedTurnsLimit: 5,
@@ -42,7 +42,7 @@ import { markConversationDirty } from './history-index-state.js';
42
42
  * Chat-Completions adapter. Turn-based slicing always cuts at a user-
43
43
  * message boundary, which is pair-safe by construction.
44
44
  *
45
- * 20 turns is the bootstrap window the user signed off on (2026-05-01).
45
+ * 10 turns is the default bootstrap window.
46
46
  * It is the cold-start replay window after a fresh boot or reconnect. Runtime
47
47
  * provider requests apply a separate deterministic history-window transform;
48
48
  * no LLM summary is required for recovery.
@@ -57,7 +57,7 @@ import { markConversationDirty } from './history-index-state.js';
57
57
  * at module load (`const cap = DEFAULT_RECENT_TURNS`) would not see
58
58
  * runtime overrides. The reader function makes that always-correct.
59
59
  */
60
- let DEFAULT_RECENT_TURNS = 20;
60
+ let DEFAULT_RECENT_TURNS = 10;
61
61
 
62
62
  // Circuit breaker for newest-to-oldest session scans. This bounds event-loop
63
63
  // starvation when the newest transcript tail is dense with hidden/internal or
package/yeaft/engine.js CHANGED
@@ -9,7 +9,7 @@
9
9
  * 5. If tool_calls → execute tools → append results → goto 3
10
10
  * 6. Persist each completed message at its durability boundary
11
11
  * 7. If max_tokens → auto-continue (up to maxContinueTurns)
12
- * 8. On LLMContextError → fail the turn; no summary or hidden maintenance call
12
+ * 8. On LLMContextError → shrink the request copy and retry without replaying tools
13
13
  * 9. On retryable error with fallbackModel → switch model → retry
14
14
  *
15
15
  * Pattern derived from Claude Code's query loop (src/query.ts).
@@ -22,7 +22,7 @@ import { promises as fsp } from 'fs';
22
22
  import { join, resolve as resolvePath } from 'path';
23
23
  import { buildSystemPrompt, buildWorkerPrompt } from './prompts.js';
24
24
  import { getRuntimePlatformInfo } from './runtime-platform.js';
25
- import { LLMAbortError, LLMAuthError, LLMPolicyError, LLMRateLimitError, LLMServerError, LLMStreamIdleTimeoutError } from './llm/adapter.js';
25
+ import { LLMAbortError, LLMAuthError, LLMContextError, LLMPolicyError, LLMRateLimitError, LLMServerError, LLMStreamIdleTimeoutError } from './llm/adapter.js';
26
26
  import { runMemoryPreflow, buildRelevantScopes, memoryScopeLabel } from './sessions/pre-flow.js';
27
27
  import {
28
28
  readProjectDoc,
@@ -33,7 +33,7 @@ import {
33
33
  DEFAULT_PROJECT_DOC_MAX_BYTES,
34
34
  } from './sessions/project-doc.js';
35
35
  import { archiveToolResults } from './archive/tool-results.js';
36
- import { trimSnapshotForBudget, estimateMessageTokens, buildHistoryBuckets } from './history-window.js';
36
+ import { trimSnapshotForBudget, estimateMessageTokens, buildHistoryBuckets, fitProviderRequestToContext } from './history-window.js';
37
37
  import { recallConversationTurns } from './conversation/history-index.js';
38
38
  import { parseSeqFromId } from './conversation/persist.js';
39
39
  import { isVpForeign, readContent as readScopeContent } from './memory/store.js';
@@ -46,6 +46,14 @@ import { perfNowMs, recordAgentPerfTrace } from './perf-trace.js';
46
46
  const MAIN_THREAD_ID = 'main';
47
47
  import { pickEffort, parseEffortPrefix, snapshotEffortDecision } from './effort.js';
48
48
  import { bindProviderState } from './llm/provider-state.js';
49
+ import {
50
+ POST_COMPACT_CONTEXT_RATIO,
51
+ generatePostCompact,
52
+ loadPostCompact,
53
+ postCompactPath,
54
+ removePostCompactIfSource,
55
+ savePostCompact,
56
+ } from './post-compact.js';
49
57
  import { DEFAULT_CONTEXT_WINDOW, getModelInfo, normalizeEffort, parseModelRef, resolveContextWindow, resolveMaxOutputTokens, resolveModel } from './models.js';
50
58
  import { lookupModelLimitSync } from './llm/models-dev.js';
51
59
  import { attachRouterPlan, extractPriorPlan, stripMetaForWire } from './router/continuity.js';
@@ -59,7 +67,7 @@ import { createPluginSkillManager } from './plugins.js';
59
67
  import { extractDisplayImages, stripDisplayImageData } from './image-assets.js';
60
68
  import { acknowledgePendingNotifications, formatNotificationsForPrompt, peekPendingNotifications } from './sub-agent/notifications.js';
61
69
  import {
62
- TOOL_BATCH_SIZE,
70
+ TOOL_LOOP_REFLECTION_INTERVAL,
63
71
  TURN_SUMMARY_THRESHOLD,
64
72
  DUP_TOOL_THRESHOLD,
65
73
  ExecLog,
@@ -80,7 +88,7 @@ import {
80
88
  * conversations (user report: Yeaft loop errored at the cap). The engine
81
89
  * now runs until the LLM itself returns stopReason='end_turn' or a
82
90
  * non-retryable error surfaces. Real runaway loops are still bounded by:
83
- * • provider rate limits / context window (LLMContextError is surfaced)
91
+ * • provider rate limits / context recovery exhaustion
84
92
  * • user-initiated abort (AbortController / cancel)
85
93
  * • MAX_CONTINUE_TURNS for the max_tokens auto-continue path
86
94
  */
@@ -722,13 +730,18 @@ export class Engine {
722
730
  * prior turn's history is rewritten with the reflection. If still
723
731
  * pending, the engine falls back to the exec-log stub.
724
732
  * • `#reflectedTurns` — Set<turnNumber>; ensures T1 fires at most
725
- * once per turn (when toolCount crosses TOOL_BATCH_SIZE).
733
+ * once per reflection interval measured in tool loops.
726
734
  */
727
735
  #execLog = null;
728
736
  #pendingT2 = new Map();
729
737
  #reflectedTurns = new Set();
730
738
  #__queryCounter = 0;
731
739
 
740
+ /** Derived post-response summaries, keyed by Session/VP/thread scope. */
741
+ #postCompactSummaries = new Map();
742
+ #postCompactLoaded = new Set();
743
+ #postCompactRevisions = new Map();
744
+
732
745
  /** @type {string} */
733
746
  #currentThreadId = MAIN_THREAD_ID;
734
747
 
@@ -2069,6 +2082,12 @@ export class Engine {
2069
2082
  && typeof vpPersona.vpId === 'string'
2070
2083
  ? vpPersona.vpId
2071
2084
  : (typeof senderVpId === 'string' ? senderVpId : null);
2085
+ const postCompactScope = this.#postCompactScope({
2086
+ sessionId: runtimeSessionId,
2087
+ vpId: queryVpId,
2088
+ threadId: runtimeThreadId,
2089
+ });
2090
+ const postCompactState = await this.#beginPostCompactScope(postCompactScope);
2072
2091
  // Exact read-only tool results are safe to reuse within one query only
2073
2092
  // when no intervening mutation can have changed the workspace. The map is
2074
2093
  // intentionally local to this query; cross-turn reuse belongs to the
@@ -2113,7 +2132,7 @@ export class Engine {
2113
2132
  // const useDreamMemory = scenario === 'work-item' || !!vpPersona?.subAgent
2114
2133
  // || (!runtimeSessionId && !internalTrigger);
2115
2134
  const useDreamMemory = false;
2116
- const recentTurnCap = Math.max(20, this.#config.yeaft?.recentTurnsLimit ?? 20);
2135
+ const recentTurnCap = Math.max(10, this.#config.yeaft?.recentTurnsLimit ?? 10);
2117
2136
  const relatedTurnCap = Math.min(5, this.#config.yeaft?.relatedTurnsLimit ?? 5);
2118
2137
  let relatedHistoryTurns = [];
2119
2138
  let historyRecallMeta = { source: 'messages', status: 'disabled' };
@@ -2249,6 +2268,10 @@ export class Engine {
2249
2268
  if (amsContext && amsContext.snapshotBlock) {
2250
2269
  memoryInjection = amsContext.snapshotBlock;
2251
2270
  }
2271
+ if (postCompactState.summary) {
2272
+ const block = `## Prior conversation compact\n${postCompactState.summary}`;
2273
+ memoryInjection = memoryInjection ? `${memoryInjection}\n\n${block}` : block;
2274
+ }
2252
2275
  const loadedMemoryForDebug = loadedMemoryDebugEntries(amsContext?.snapshot);
2253
2276
  const loadedMemoryMetaForDebug = {
2254
2277
  ...(useMessageHistory ? historyRecallMeta : {}),
@@ -2475,18 +2498,18 @@ export class Engine {
2475
2498
  // `turnStartIdx` is where the current user message lives; the arc
2476
2499
  // we may collapse spans (arcStartIdx .. last assistant/tool).
2477
2500
  //
2478
- // Periodic-T1 fix: T1 must fire EVERY TOOL_BATCH_SIZE (30) tool
2479
- // calls, not just the first batch. So instead of a one-shot boolean,
2501
+ // Periodic T1 fires every 30 tool loops, not every 30 calls. A single
2502
+ // provider batch can contain many parallel calls but is still one loop.
2480
2503
  // track:
2481
- // • `lastT1AtToolCount` — toolCount snapshot at the last T1
2504
+ // • `lastT1AtLoopCount` — tool-loop snapshot at the last T1
2482
2505
  // ATTEMPT (success OR error). Trigger when
2483
- // `queryToolCount - lastT1AtToolCount >= TOOL_BATCH_SIZE`.
2506
+ // `completedToolLoops - lastT1AtLoopCount >= interval`.
2484
2507
  // • `arcStartIdx` — first index of the current (uncollapsed)
2485
2508
  // tool arc. Initialised to turnStartIdx + 1; reset after each
2486
2509
  // successful T1 collapse to `conversationMessages.length`
2487
2510
  // (i.e. the slot the next assistant message will land in).
2488
2511
  // • `t1CollapsesDone` — count of T1 firings that ACTUALLY
2489
- // rewrote history. Distinct from `lastT1AtToolCount` because
2512
+ // rewrote history. Distinct from `lastT1AtLoopCount` because
2490
2513
  // the catch block bumps the latter to back off after a
2491
2514
  // transient reflector error WITHOUT having collapsed
2492
2515
  // anything. The T2 schedule check below is gated on this
@@ -2494,7 +2517,7 @@ export class Engine {
2494
2517
  // fall back at end_turn").
2495
2518
  const turnStartIdx = conversationMessages.length - 1;
2496
2519
  let queryToolCount = 0;
2497
- let lastT1AtToolCount = 0;
2520
+ let lastT1AtLoopCount = 0;
2498
2521
  let arcStartIdx = turnStartIdx + 1;
2499
2522
  let t1CollapsesDone = 0;
2500
2523
  // Duplicate policy is scoped to one user query. Only successful, real
@@ -2561,6 +2584,10 @@ export class Engine {
2561
2584
  let primaryModelAtLastBoundary = currentModel;
2562
2585
  let cumulativeInputTokens = 0;
2563
2586
  let cumulativeOutputTokens = 0;
2587
+ let maxContextOccupancyRatio = 0;
2588
+ let peakContextTokens = 0;
2589
+ let peakContextWindow = 0;
2590
+ let postCompactCandidate = null;
2564
2591
  let activeProviderRequest = null;
2565
2592
  // Skill events describe the selection injected into each provider request.
2566
2593
  // The first request must report its initial selection; later loops report
@@ -2592,6 +2619,11 @@ export class Engine {
2592
2619
  let consecutiveRetryableErrors = 0;
2593
2620
  let consecutiveForbiddenErrors = 0;
2594
2621
  let contentPolicyRecoveryAttempts = 0;
2622
+ let contextOverflowRecoveryAttempts = 0;
2623
+ // A provider can know about framing/tokenizer overhead that our monotonic
2624
+ // estimator cannot. Each real overflow retries the same unexecuted request
2625
+ // with a smaller provider-only window; no transcript row is rewritten.
2626
+ let providerContextScale = 1;
2595
2627
 
2596
2628
  while (true) {
2597
2629
  turnNumber++;
@@ -2877,25 +2909,30 @@ export class Engine {
2877
2909
  // query tape remain complete; no summary is generated and no history
2878
2910
  // rows are rewritten. This also protects later tool-loop requests,
2879
2911
  // not just the initial snapshot assembled by the bridge.
2880
- const continuationCost = pendingContinuationForRequest
2881
- ? estimateMessageTokens(pendingContinuationForRequest) : 0;
2882
- const historyBudget = Math.max(1, Math.min(
2883
- requestConfig.messageTokenBudget || 32768,
2884
- Math.floor(currentContextWindow * 0.75) - estimateMessagesTokens(systemPrompt, []),
2885
- ) - continuationCost);
2912
+ // Retry continuation is part of the active turn. Like the opening user
2913
+ // row and tool-loop traffic, it must not consume the 32K history budget.
2914
+ const historyBudget = Math.max(0, requestConfig.messageTokenBudget || 32768);
2886
2915
  const buckets = useMessageHistory ? buildHistoryBuckets(conversationMessages, {
2887
2916
  prompt,
2888
2917
  relatedTurns: relatedHistoryTurns,
2889
- recentTurnCap: Math.max(20, requestConfig.yeaft?.recentTurnsLimit ?? 20),
2918
+ recentTurnCap: Math.max(10, requestConfig.yeaft?.recentTurnsLimit ?? 10),
2890
2919
  relatedTurnCap: Math.min(5, requestConfig.yeaft?.relatedTurnsLimit ?? 5),
2891
2920
  messageTokenBudget: historyBudget,
2892
2921
  currentTurnStartIndex: turnStartIdx,
2893
2922
  language: requestConfig.language,
2894
2923
  }) : null;
2895
- const requestHistory = buckets?.messages || trimSnapshotForBudget(conversationMessages, {
2896
- messageTokenBudget: requestConfig.messageTokenBudget,
2897
- language: requestConfig.language,
2898
- });
2924
+ let historyMessageCount;
2925
+ const requestHistory = buckets?.messages || (() => {
2926
+ const historical = trimSnapshotForBudget(conversationMessages.slice(0, turnStartIdx), {
2927
+ messageTokenBudget: historyBudget,
2928
+ language: requestConfig.language,
2929
+ });
2930
+ historyMessageCount = historical.length;
2931
+ return [...historical, ...conversationMessages.slice(turnStartIdx)];
2932
+ })();
2933
+ if (buckets) {
2934
+ historyMessageCount = requestHistory.length - (buckets.meta?.current?.messageCount || 0);
2935
+ }
2899
2936
  if (buckets) this.#trace.log?.('history_buckets', {
2900
2937
  sessionId: runtimeSessionId, turnId: queryTurnId, ...historyRecallMeta, ...buckets.meta,
2901
2938
  });
@@ -2919,6 +2956,39 @@ export class Engine {
2919
2956
  } catch { /* best-effort */ }
2920
2957
  }
2921
2958
 
2959
+ // Final request boundary: account for every component against the
2960
+ // actual model window. The 32K budget above applies only to historical
2961
+ // rows; current-turn rows are paid here together with system, schemas,
2962
+ // and the model-specific output reserve. This runs on every tool-loop
2963
+ // request and again with tighter headroom after a provider overflow.
2964
+ const requestMaxOutputTokens = Math.max(1, Math.min(
2965
+ requestConfig.maxOutputTokens || resolveMaxOutputTokens(currentModel, requestConfig),
2966
+ resolveMaxOutputTokens(currentModel, requestConfig),
2967
+ ));
2968
+ const toolSchemaTokens = toolDefs.length > 0
2969
+ ? approxTokens(JSON.stringify(toolDefs)) : 0;
2970
+ const fittedRequest = fitProviderRequestToContext(wireMessages, {
2971
+ contextWindow: Math.max(1, Math.floor(currentContextWindow * providerContextScale)),
2972
+ systemTokens: estimateMessagesTokens(systemPrompt, []),
2973
+ toolSchemaTokens,
2974
+ outputReserve: requestMaxOutputTokens,
2975
+ historyMessageCount,
2976
+ historyTokenBudget: historyBudget,
2977
+ language: requestConfig.language,
2978
+ });
2979
+ wireMessages = fittedRequest.messages;
2980
+ if (fittedRequest.meta.droppedHistoryMessages > 0
2981
+ || fittedRequest.meta.droppedCurrentMessages > 0
2982
+ || providerContextScale < 1) {
2983
+ this.#trace.log?.('request_context_trim', {
2984
+ sessionId: runtimeSessionId,
2985
+ turnId: queryTurnId,
2986
+ model: currentModel,
2987
+ recoveryAttempt: contextOverflowRecoveryAttempts,
2988
+ ...fittedRequest.meta,
2989
+ });
2990
+ }
2991
+
2922
2992
  // task-704b: pre-flight total-token guard. Even with the per-tool
2923
2993
  // cap (registry.js: 10% of contextWindow per result), N tool
2924
2994
  // results plus history can still breach the wire limit before we
@@ -3005,7 +3075,7 @@ export class Engine {
3005
3075
  system: systemPrompt,
3006
3076
  messages: wireMessages,
3007
3077
  tools: toolDefs.length > 0 ? toolDefs : undefined,
3008
- maxTokens: requestConfig.maxOutputTokens || 16384,
3078
+ maxTokens: requestMaxOutputTokens,
3009
3079
  effort: resolvedEffort,
3010
3080
  effortConstraint,
3011
3081
  requestIdentity,
@@ -3183,6 +3253,35 @@ export class Engine {
3183
3253
  responseTextBytes: Buffer.byteLength(responseText, 'utf8'),
3184
3254
  },
3185
3255
  });
3256
+ const requestContextOccupancy = (totalUsage.inputTokens || 0)
3257
+ + (totalUsage.cacheInputDeltaTokens || 0)
3258
+ + (totalUsage.outputTokens || 0);
3259
+ const requestContextOccupancyRatio = requestContextOccupancy / currentContextWindow;
3260
+ if (requestContextOccupancyRatio >= maxContextOccupancyRatio) {
3261
+ maxContextOccupancyRatio = requestContextOccupancyRatio;
3262
+ peakContextTokens = requestContextOccupancy;
3263
+ peakContextWindow = currentContextWindow;
3264
+ }
3265
+ // Trigger from the peak request, but summarize the latest provider
3266
+ // state so an earlier, fuller tool loop cannot omit later reflection,
3267
+ // tool results, or the final answer from the derived artifact.
3268
+ postCompactCandidate = {
3269
+ scope: postCompactScope,
3270
+ revision: postCompactState.revision,
3271
+ sessionId: runtimeSessionId,
3272
+ turnId: queryTurnId,
3273
+ model: currentModel,
3274
+ config: requestConfig,
3275
+ adapter: requestAdapter,
3276
+ messages: [
3277
+ ...wireMessages.map(message => ({ ...message })),
3278
+ ...(toolCalls.length === 0 && responseText
3279
+ ? [{ role: 'assistant', content: responseText }]
3280
+ : []),
3281
+ ],
3282
+ contextTokens: peakContextTokens,
3283
+ contextWindow: peakContextWindow,
3284
+ };
3186
3285
  // Stream completed without throwing — reset the retry counter so
3187
3286
  // the next turn starts with a clean budget. In-band adapter errors
3188
3287
  // are converted to throws above so they share the real error path.
@@ -3298,11 +3397,32 @@ export class Engine {
3298
3397
  const earlyIsRateLimit = err instanceof LLMRateLimitError;
3299
3398
  const earlyIsTransient = err instanceof LLMServerError;
3300
3399
  const earlyIsContentPolicy = err instanceof LLMPolicyError;
3400
+ const earlyIsContextOverflow = err instanceof LLMContextError;
3301
3401
  // A completed tool_call has already crossed the streaming boundary to
3302
3402
  // the caller. Replaying that request would publish a duplicate call and
3303
3403
  // leave ambiguous execution ownership, so only pre-tool failures are
3304
3404
  // eligible for transparent retry or model fallback.
3305
3405
  const canReplayProviderRequest = toolCalls.length === 0;
3406
+ if (earlyIsContextOverflow && canReplayProviderRequest
3407
+ && contextOverflowRecoveryAttempts < 3) {
3408
+ contextOverflowRecoveryAttempts += 1;
3409
+ providerContextScale *= 0.75;
3410
+ endAttemptTrace('context_overflow_retry');
3411
+ if (responseText) prepareRetryContinuation();
3412
+ yield {
3413
+ type: 'llm_retry',
3414
+ attempt: contextOverflowRecoveryAttempts,
3415
+ maxRetries: 3,
3416
+ delayMs: 0,
3417
+ reason: 'context_overflow_recovery',
3418
+ recoveryMode: responseText ? 'continue' : 'restart',
3419
+ errorName: err.name,
3420
+ statusCode: err.statusCode ?? null,
3421
+ message: 'Provider rejected the context; retrying with a smaller request copy.',
3422
+ };
3423
+ yield { type: 'turn_end', turnNumber, stopReason: 'llm_retry', threadId };
3424
+ continue;
3425
+ }
3306
3426
  if (earlyIsContentPolicy && canReplayProviderRequest && contentPolicyRecoveryAttempts === 0) {
3307
3427
  contentPolicyRecoveryAttempts = 1;
3308
3428
  endAttemptTrace('llm_retry');
@@ -3896,8 +4016,8 @@ export class Engine {
3896
4016
  // `#applyPendingT2Reflections` carries the result forward.
3897
4017
  //
3898
4018
  // Periodic-T1 fix: gate on `t1CollapsesDone === 0`, NOT
3899
- // `lastT1AtToolCount === 0`. The catch block of T1 bumps
3900
- // `lastT1AtToolCount` after a reflector error to avoid
4019
+ // `lastT1AtLoopCount === 0`. The catch block of T1 advances
4020
+ // `lastT1AtLoopCount` after a reflector error to avoid
3901
4021
  // tight-loop retries — but no collapse happened, so T2 should
3902
4022
  // still be allowed to fall back at end_turn. Fowler-review
3903
4023
  // critical finding.
@@ -4697,29 +4817,28 @@ export class Engine {
4697
4817
  break;
4698
4818
  }
4699
4819
 
4700
- // PR-L: T1 in-turn (synchronous) reflection. Fires once per
4701
- // adapter loop iteration where ≥ TOOL_BATCH_SIZE (30) tool
4702
- // calls have accumulated since the last T1 firing — not just
4703
- // the first batch of the query(). Generates a markdown reflection
4820
+ // PR-L: T1 in-turn (synchronous) reflection. Fires every 30 completed
4821
+ // tool loops. A provider response containing many parallel tool calls is
4822
+ // one loop, not many. Generates a markdown reflection
4704
4823
  // over the assistant+tool arc since the last T1 firing (or
4705
4824
  // since the user prompt for the first batch) and rewrites that
4706
4825
  // range to a SINGLE synthetic user message before the next
4707
4826
  // adapter.stream() runs.
4708
4827
  //
4709
4828
  // Loop semantics:
4710
- // - First batch: arcStartIdx = turnStartIdx + 1, fires when
4711
- // queryToolCount reaches TOOL_BATCH_SIZE.
4712
- // - Each subsequent batch: arcStartIdx is updated to the slot
4829
+ // - First interval: arcStartIdx = turnStartIdx + 1.
4830
+ // - Each subsequent interval: arcStartIdx is updated to the slot
4713
4831
  // right after the just-inserted reflection message; fires
4714
- // again whenever TOOL_BATCH_SIZE more tools have run since
4715
- // lastT1AtToolCount.
4716
- // - The dedup Set key includes `lastT1AtToolCount` so each
4717
- // batch within the same query gets a distinct entry — without
4832
+ // again whenever 30 more tool loops have completed.
4833
+ // - The dedup Set key includes the loop count so each interval
4834
+ // within the same query gets a distinct entry — without
4718
4835
  // this the second batch would be silently skipped.
4719
- const t1BatchDue = queryToolCount - lastT1AtToolCount >= TOOL_BATCH_SIZE;
4720
- if (groupReflectionAllowed && t1BatchDue && !toolBatchBarrier
4836
+ const completedToolLoops = toolLoopTurns + 1;
4837
+ const t1BatchDue = completedToolLoops - lastT1AtLoopCount
4838
+ >= TOOL_LOOP_REFLECTION_INTERVAL;
4839
+ if (t1BatchDue && !toolBatchBarrier
4721
4840
  && !abortedDuringTools && !signal?.aborted) {
4722
- const t1DedupKey = `${queryNumber}:t1:${queryToolCount}`;
4841
+ const t1DedupKey = `${queryNumber}:t1-loop:${completedToolLoops}`;
4723
4842
  if (this.#reflectedTurns.has(t1DedupKey)) {
4724
4843
  // Defensive: should never hit since t1BatchDue gates re-entry
4725
4844
  // and queryNumber namespaces queries. Kept as belt-and-
@@ -4787,10 +4906,10 @@ export class Engine {
4787
4906
  // immediately after it, i.e. at conversationMessages.length
4788
4907
  // (the next assistant message will land here).
4789
4908
  arcStartIdx = conversationMessages.length;
4790
- lastT1AtToolCount = queryToolCount;
4909
+ lastT1AtLoopCount = completedToolLoops;
4791
4910
  // Bump the success counter — used by the T2 schedule check
4792
4911
  // to decide whether T2 still has work to do at end_turn.
4793
- // Distinct from lastT1AtToolCount which the catch block
4912
+ // Distinct from lastT1AtLoopCount which the catch block
4794
4913
  // also bumps (but without rewriting history).
4795
4914
  t1CollapsesDone += 1;
4796
4915
  yield {
@@ -4820,9 +4939,9 @@ export class Engine {
4820
4939
  status: 'error',
4821
4940
  error: err && err.message || String(err),
4822
4941
  };
4823
- // Advance lastT1AtToolCount past this batch so we don't
4942
+ // Advance lastT1AtLoopCount past this interval so we don't
4824
4943
  // tight-loop on a hiccuping reflector. The next attempt is
4825
- // TOOL_BATCH_SIZE tools from now, not immediately. arcStartIdx is
4944
+ // another 30 tool loops from now, not immediately. arcStartIdx is
4826
4945
  // left alone because history wasn't rewritten — the tail still
4827
4946
  // begins where it did. The trade-off: the next batch's
4828
4947
  // reflection will cover the tools that just failed too,
@@ -4831,7 +4950,7 @@ export class Engine {
4831
4950
  // We do NOT bump t1CollapsesDone — see the variable's
4832
4951
  // declaration comment. This keeps the T2 fallback path live
4833
4952
  // when every T1 attempt has errored.
4834
- lastT1AtToolCount = queryToolCount;
4953
+ lastT1AtLoopCount = completedToolLoops;
4835
4954
  }
4836
4955
  }
4837
4956
  }
@@ -4879,6 +4998,76 @@ export class Engine {
4879
4998
  totalTokens: cumulativeInputTokens + cumulativeOutputTokens,
4880
4999
  loopCount: turnNumber,
4881
5000
  };
5001
+
5002
+ // The visible response is complete at the yield above. Only when the
5003
+ // consumer resumes past that boundary do we inspect pressure and launch
5004
+ // best-effort post compact. It never blocks this turn or the next one.
5005
+ if (postCompactCandidate
5006
+ && maxContextOccupancyRatio >= POST_COMPACT_CONTEXT_RATIO) {
5007
+ this.#schedulePostCompact(postCompactCandidate);
5008
+ }
5009
+ }
5010
+
5011
+ #postCompactScope({ sessionId, vpId, threadId }) {
5012
+ if (!this.#yeaftDir || !sessionId) return null;
5013
+ const path = postCompactPath(this.#yeaftDir, { sessionId, vpId, threadId });
5014
+ return { key: path, path };
5015
+ }
5016
+
5017
+ async #beginPostCompactScope(scope) {
5018
+ if (!scope) return { revision: 0, summary: '' };
5019
+ const revision = (this.#postCompactRevisions.get(scope.key) || 0) + 1;
5020
+ this.#postCompactRevisions.set(scope.key, revision);
5021
+ if (!this.#postCompactLoaded.has(scope.key)) {
5022
+ this.#postCompactLoaded.add(scope.key);
5023
+ const artifact = await loadPostCompact(scope.path);
5024
+ if (artifact) this.#postCompactSummaries.set(scope.key, artifact);
5025
+ }
5026
+ return {
5027
+ revision,
5028
+ summary: this.#postCompactSummaries.get(scope.key)?.summary || '',
5029
+ };
5030
+ }
5031
+
5032
+ #schedulePostCompact(candidate) {
5033
+ const { scope, revision } = candidate;
5034
+ if (!scope || this.#postCompactRevisions.get(scope.key) !== revision) return;
5035
+ const compactMaxTokens = Math.max(512, Math.min(4096,
5036
+ resolveMaxOutputTokens(candidate.model, candidate.config)));
5037
+ const task = async () => {
5038
+ try {
5039
+ const summary = await generatePostCompact({
5040
+ adapter: candidate.adapter,
5041
+ model: candidate.model,
5042
+ messages: candidate.messages,
5043
+ maxTokens: compactMaxTokens,
5044
+ });
5045
+ const current = () => this.#postCompactRevisions.get(scope.key) === revision;
5046
+ if (!current()) return;
5047
+ const artifact = {
5048
+ summary,
5049
+ model: candidate.model,
5050
+ sourceTurnId: candidate.turnId,
5051
+ sourceRevision: revision,
5052
+ sourceContextTokens: candidate.contextTokens,
5053
+ contextWindow: candidate.contextWindow,
5054
+ createdAt: new Date().toISOString(),
5055
+ };
5056
+ if (await savePostCompact(scope.path, artifact, current)) {
5057
+ if (current()) this.#postCompactSummaries.set(scope.key, artifact);
5058
+ else await removePostCompactIfSource(scope.path, candidate.turnId);
5059
+ }
5060
+ } catch (error) {
5061
+ // A response already reached the user. Compact failure is diagnostic
5062
+ // only and must not create a late error event.
5063
+ this.#trace.log?.('post_compact_failed', {
5064
+ sessionId: candidate.sessionId,
5065
+ turnId: candidate.turnId,
5066
+ message: String(error?.message || error).slice(0, 200),
5067
+ });
5068
+ }
5069
+ };
5070
+ void task();
4882
5071
  }
4883
5072
 
4884
5073
  /**