mcp-castor 2026.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +487 -0
- package/bin/castor.js +706 -0
- package/index.js +206 -0
- package/package.json +97 -0
- package/skills/canary-test-staging/SKILL.md +24 -0
- package/skills/evo-mutation-rollback/SKILL.md +29 -0
- package/skills/hypothesis-generation/SKILL.md +26 -0
- package/skills/traceback-condensing/SKILL.md +26 -0
- package/src/castor_runner.js +469 -0
- package/src/config.js +1204 -0
- package/src/env.js +10 -0
- package/src/evo_engine.js +214 -0
- package/src/harness/core/events.js +75 -0
- package/src/harness/core/kernel.js +209 -0
- package/src/harness/evo/evaluator.js +156 -0
- package/src/harness/evo/evo_operator.js +550 -0
- package/src/harness/evo/lineage_dag.js +383 -0
- package/src/harness/evo/trace_repair.js +173 -0
- package/src/harness/evo/watchdog.js +72 -0
- package/src/harness/loop_detector.js +135 -0
- package/src/harness/runner.js +1216 -0
- package/src/harness/services/ast_service.js +1813 -0
- package/src/harness/services/event_logger.js +275 -0
- package/src/harness/services/mcp_bridge.js +408 -0
- package/src/harness/services/provider_vllm.js +728 -0
- package/src/harness/services/sandbox_fs.js +1238 -0
- package/src/harness/services/searxng_lifecycle.js +254 -0
- package/src/harness/services/shell_executor.js +264 -0
- package/src/harness/services/shell_validator.js +506 -0
- package/src/harness/services/web_service.js +828 -0
- package/src/platform.js +344 -0
- package/src/repetition_detector.js +139 -0
- package/src/semaphore.js +373 -0
- package/src/server_lifecycle.js +781 -0
- package/src/skills.js +400 -0
- package/src/state_pruner.js +392 -0
- package/src/task_registry.js +1357 -0
- package/src/telemetry.js +638 -0
- package/src/tools.js +997 -0
- package/src/wsl_bridge.js +629 -0
- package/src/wsl_env.js +171 -0
- package/stream_proxy.js +453 -0
|
@@ -0,0 +1,1216 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Castor Runtime Engine (Microkernel Orchestrator)
|
|
3
|
+
*
|
|
4
|
+
* Capabilities:
|
|
5
|
+
* - Pure Node.js runtime (no binary compilation or subprocess shell wrappers)
|
|
6
|
+
* - Castor Context lifecycle with reversible plugin mount/unmount
|
|
7
|
+
* - SSE streaming with real-time token dispatch to Antigravity
|
|
8
|
+
* - Sandboxed ripgrep filesystem & bash toolset
|
|
9
|
+
* - Evo closed-loop evolutionary operators
|
|
10
|
+
* - Append-only JSONL event ledger & session branching
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import path from "node:path";
|
|
14
|
+
import { createRequire } from "node:module";
|
|
15
|
+
import { Context } from "./core/kernel.js";
|
|
16
|
+
import { sandboxFsPlugin } from "./services/sandbox_fs.js";
|
|
17
|
+
import { shellExecutorPlugin } from "./services/shell_executor.js";
|
|
18
|
+
import { eventLoggerPlugin } from "./services/event_logger.js";
|
|
19
|
+
import { vllmProviderPlugin } from "./services/provider_vllm.js";
|
|
20
|
+
import { evoPlugin } from "./evo/evo_operator.js";
|
|
21
|
+
import { astPlugin } from "./services/ast_service.js";
|
|
22
|
+
import { webPlugin } from "./services/web_service.js";
|
|
23
|
+
import { McpBridge } from "./services/mcp_bridge.js";
|
|
24
|
+
import { injectSkills, matchSkills } from "../skills.js";
|
|
25
|
+
import { normalizeWorkspacePath, canonicalizePath } from "../wsl_bridge.js";
|
|
26
|
+
import { LoopDetector } from "./loop_detector.js";
|
|
27
|
+
import {
|
|
28
|
+
MAX_CONTINUATION_TURNS,
|
|
29
|
+
EMPTY_STREAM_RETRIES,
|
|
30
|
+
EMPTY_STREAM_RETRIES_DEEP,
|
|
31
|
+
EMPTY_STREAM_RETRY_DEPTH_CHARS,
|
|
32
|
+
EMPTY_STREAM_RETRY_BACKOFF_BASE_MS,
|
|
33
|
+
EMPTY_STREAM_RETRY_BACKOFF_CAP_MS,
|
|
34
|
+
DEGENERATE_FINAL_SUBSTANTIVE_CHARS,
|
|
35
|
+
DEGENERATE_FINAL_MAX_TURNS,
|
|
36
|
+
getReasoningEffort,
|
|
37
|
+
PROBE_BUDGET,
|
|
38
|
+
SESSION_TURNS_WARN,
|
|
39
|
+
SESSION_TURNS_RECOMMEND,
|
|
40
|
+
CONTEXT_WARN_TOKENS,
|
|
41
|
+
CONTEXT_HIGH_WATERMARK_TOKENS,
|
|
42
|
+
CONTEXT_EMERGENCY_CEILING_TOKENS,
|
|
43
|
+
TOOL_SPILL_BYTES,
|
|
44
|
+
PROMPT_BUDGET_CHARS,
|
|
45
|
+
BASE_TURN_BUDGET,
|
|
46
|
+
MAX_ELASTIC_TURNS,
|
|
47
|
+
KV_CACHE_HEADROOM_CEILING,
|
|
48
|
+
SPEC_ACCEPTANCE_FLOOR,
|
|
49
|
+
LOOP_DETECTION_WINDOW,
|
|
50
|
+
LOOP_DETECTION_REPETITIONS,
|
|
51
|
+
SUPERVISOR_PREVIEW_CHARS,
|
|
52
|
+
SALVAGE_MAX_TOKENS,
|
|
53
|
+
MAX_LEN_HUGE,
|
|
54
|
+
READ_GOVERNOR_MAX_BYTES,
|
|
55
|
+
MODEL,
|
|
56
|
+
MAX_CONTEXT,
|
|
57
|
+
} from "../config.js";
|
|
58
|
+
import { GUARD_MARKER_PREFIX } from "../repetition_detector.js";
|
|
59
|
+
import { recordTurnTelemetry, recordToolExecution, sampleLiveVllmMetrics } from "../telemetry.js";
|
|
60
|
+
|
|
61
|
+
// Single source of truth for the harness version: read from package.json
|
|
62
|
+
// (same createRequire idiom as index.js and mcp_bridge.js).
|
|
63
|
+
const require = createRequire(import.meta.url);
|
|
64
|
+
const PKG_VERSION = require("../../package.json").version;
|
|
65
|
+
|
|
66
|
+
// M4: probe-budget watchdog (issue #11 recs 1+2; F4/F12/F14). On open-ended
|
|
67
|
+
// layout targets the model ran 30+ consecutive inline-python measurement bash
|
|
68
|
+
// calls (~90 min) instead of making the edit. The runner counts CONSECUTIVE
|
|
69
|
+
// non-mutating bash calls (bash/exec_command with no file-mutating tool call in
|
|
70
|
+
// between); when the streak exceeds PROBE_BUDGET it injects an ADVISORY (not an
|
|
71
|
+
// error, not a cancellation) and re-arms the counter. Style reference: the
|
|
72
|
+
// advisory-only EvoWatchdog circuit breaker (src/harness/evo/watchdog.js).
|
|
73
|
+
//
|
|
74
|
+
// MUTATING_TOOLS reset the streak (a file edit means the model is in mutation
|
|
75
|
+
// mode, not probe mode). BASH_TOOLS increment it. Every other tool
|
|
76
|
+
// (read_file, list_dir, search_code, ast_search, evo_evaluate_candidate,
|
|
77
|
+
// evo_status) is neutral — it neither increments nor resets the streak.
|
|
78
|
+
const MUTATING_TOOLS = new Set([
|
|
79
|
+
"write_file",
|
|
80
|
+
"edit_file",
|
|
81
|
+
"apply_patch",
|
|
82
|
+
"ast_replace",
|
|
83
|
+
"ast_replace_batch",
|
|
84
|
+
"evo_propose_candidate",
|
|
85
|
+
"evo_select_candidate",
|
|
86
|
+
"evo_revert_candidate",
|
|
87
|
+
]);
|
|
88
|
+
const BASH_TOOLS = new Set(["bash"]);
|
|
89
|
+
|
|
90
|
+
// Exponential backoff before retry attempt.
|
|
91
|
+
// runner sleeps base * 2^(retryNumber-1) ms, capped at capMs. With the defaults
|
|
92
|
+
// (base 2000ms, cap 30000ms) this is exactly "2^retryNumber seconds capped at
|
|
93
|
+
// 30s": retry 1 waits 2s, retry 2 waits 4s, retry 3 waits 8s, retry 4 waits
|
|
94
|
+
// 16s, retry 5+ waits 30s (capped). The backoff gives a transient empty-stream
|
|
95
|
+
// cluster time to clear before the next (expensive, deep) re-prefill. The
|
|
96
|
+
// computed ms is returned so the caller can record it in the retry event.
|
|
97
|
+
function emptyStreamRetryBackoffMs(retryNumber) {
|
|
98
|
+
const exp = Math.max(0, retryNumber - 1);
|
|
99
|
+
const ms = EMPTY_STREAM_RETRY_BACKOFF_BASE_MS * 2 ** exp;
|
|
100
|
+
return Math.min(ms, EMPTY_STREAM_RETRY_BACKOFF_CAP_MS);
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
export const DEFAULT_SYSTEM_PROMPT = `You are the Autonomous Execution Coworker (${MODEL}) running in the Castor harness.
|
|
104
|
+
You pair with the Lead Architect (Gemini in Antigravity / GLM in Claude Code) as a senior peer engineer. The Lead Architect holds high-level architecture and task decomposition; you hold hands-on execution, empirical testing, and codebase navigation.
|
|
105
|
+
|
|
106
|
+
Operating Principles:
|
|
107
|
+
1. Peer Partnership & Two-Way Discussion:
|
|
108
|
+
- You are an autonomous engineering peer, not a blind batch executor. Discussion and collaborative alignment from both sides is the foundational operating principle.
|
|
109
|
+
- Autonomous Slicing on Clear Tasks: When objectives and acceptance criteria are clearly defined, execute the complete slice (investigate, modify, verify) autonomously across your toolset without micro-confirmations.
|
|
110
|
+
- Collaborative Pause on Ambiguity or Impasse: Never treat a dispatch as "life or death" where you must silently force solutions at all costs. When an empirical test fails an acceptance gate, when requirements are ambiguous, or when multiple paths exist, DO NOT loop in solitary trial-and-error.
|
|
111
|
+
- State your verified findings concisely, present the concrete trade-offs or root causes, and provide your technical recommendation to the Lead Architect in plain text. Concluding your turn with a clear, grounded inquiry or status report IS successful fulfillment of the turn.
|
|
112
|
+
2. Ground Truth in Code & Tests:
|
|
113
|
+
- Ground truth lives exclusively in active source code, test suites, and verifiable build artifacts. Never assume or hallucinate.
|
|
114
|
+
3. Workspace Scratchpads for Audits, Exploration & Empirical Reproduction:
|
|
115
|
+
- You have full, unrestricted write and execution access to '<workspace>/.scratch/' (and repository-local helper scripts) at all times, including during exploration turns.
|
|
116
|
+
- When diagnosing issues, verifying edge cases, or conducting multi-item audits, write minimal reproduction scripts (e.g. '.scratch/repro.py', '.scratch/test_case.js') and dump structured data tables to '.scratch/'.
|
|
117
|
+
- Isolating and verifying a failure empirically with a clean script in '.scratch/' is always preferred over mentally simulating complex logic or running long inline bash one-liners.
|
|
118
|
+
4. Mutation & Tool Discipline:
|
|
119
|
+
- When requirements and reproduction are verified and a dispatch requests a code change, modify production source files directly in ONE targeted pass with native editing tools ('edit_file' / 'apply_patch'). Do NOT run blind measurement probe loops against production files.
|
|
120
|
+
- Reserve 'bash' strictly for compilation, test execution, benchmarks, git operations, package managers, or running project runtimes/binaries.
|
|
121
|
+
- Pure Text-Only Engine: You run in text mode with Universal 245K context. Do NOT call image inspection tools on binary images (.png, .jpg). Multimodal inspection is handled exclusively by the Lead Architect.
|
|
122
|
+
5. Deliverables: Provide concise, direct technical summaries of your actions and findings.`;
|
|
123
|
+
|
|
124
|
+
const EVO_SYSTEM_PROMPT_ADDENDUM = `
|
|
125
|
+
6. When optimizing, refactoring, or evolving procedural skills, use the Evo tools:
|
|
126
|
+
- 'evo_propose_candidate' to snapshot files or skills before modifying.
|
|
127
|
+
- 'evo_evaluate_candidate' to test and compute fitness score (receives compact failure digests on error).
|
|
128
|
+
- 'evo_select_candidate' to accept improvements, or 'evo_revert_candidate' to rollback regressions.`;
|
|
129
|
+
|
|
130
|
+
/**
|
|
131
|
+
* User-role directive injected when the model's output is cut off by the
|
|
132
|
+
* token ceiling (finish_reason: "length"). Instructs the model to resume
|
|
133
|
+
* exactly where it stopped without repeating already-emitted content.
|
|
134
|
+
*/
|
|
135
|
+
export const CONTINUATION_DIRECTIVE =
|
|
136
|
+
"Your previous output was cut off by the token ceiling. " +
|
|
137
|
+
"Resume exactly where you stopped. Do not repeat already-emitted content.";
|
|
138
|
+
|
|
139
|
+
/**
|
|
140
|
+
* Balanced, non-coercive directive injected when the model hits a reasoning ceiling
|
|
141
|
+
* (finish_reason: "length" with reasoningCeilingHit or empty content).
|
|
142
|
+
* Gives permission to conclude or report blockers to the supervisor without hallucinating actions.
|
|
143
|
+
*/
|
|
144
|
+
export const REASONING_CONTINUATION_DIRECTIVE =
|
|
145
|
+
"Your deliberation was paused at the token ceiling. " +
|
|
146
|
+
"If you have reached a resolution, proceed with your tool call or response. " +
|
|
147
|
+
"If you are facing an ambiguous requirement or an impasse, state what you have determined so far and request guidance from the supervisor.";
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Non-coercive salvage directive injected when the reasoning budget is exhausted.
|
|
151
|
+
* Requests that the model persist accumulated findings and incomplete items
|
|
152
|
+
* to a designated scratch file before session termination.
|
|
153
|
+
*/
|
|
154
|
+
export const SALVAGE_DIRECTIVE =
|
|
155
|
+
"Your deliberation has reached the token ceiling and this session is about to conclude. " +
|
|
156
|
+
"Do not reason further. In one short message, record the findings, tables, and conclusions you have " +
|
|
157
|
+
"already accumulated to the file <SALVAGE_PATH>, and briefly state what remains incomplete. " +
|
|
158
|
+
"This is a best-effort salvage of your partial work — if you have nothing concrete to record, simply say so.";
|
|
159
|
+
|
|
160
|
+
/**
|
|
161
|
+
* Advisory message injected when consecutive non-mutating bash executions
|
|
162
|
+
* exceed the configured probe budget threshold.
|
|
163
|
+
*/
|
|
164
|
+
export const PROBE_BUDGET_ADVISORY =
|
|
165
|
+
"[Probe-Budget Advisory] You have run several consecutive shell (bash) calls " +
|
|
166
|
+
"without making progress on your deliverable. Prefer native workspace tools (search_code, read_file, list_dir) over ad-hoc shell inspection. " +
|
|
167
|
+
"Mutation dispatches are single-pass: state a hypothesis, make the edit directly with a native file tool (write_file / edit_file / apply_patch), " +
|
|
168
|
+
"then run the stated verification command ONCE. For read-only or exploration tasks, synthesize your findings and emit your final response now.";
|
|
169
|
+
|
|
170
|
+
/**
|
|
171
|
+
* Advisory message injected when cumulative session turns reach the recommended
|
|
172
|
+
* rollover threshold, advising session consolidation.
|
|
173
|
+
*/
|
|
174
|
+
export const SESSION_ROLLOVER_ADVISORY =
|
|
175
|
+
"[Session-Rollover Advisory] This session has crossed the recommended " +
|
|
176
|
+
"turn-count boundary for a single session. Complete the current task, then " +
|
|
177
|
+
"roll to a FRESH session on the next dispatch — a new session starts with a " +
|
|
178
|
+
"clean, low-cost context instead of re-prefilling this deep one.";
|
|
179
|
+
|
|
180
|
+
/**
|
|
181
|
+
* Spills oversized tool execution results to scratch storage.
|
|
182
|
+
*
|
|
183
|
+
* When a tool output exceeds thresholdBytes, persists the full payload to disk
|
|
184
|
+
* under the workspace scratch directory and replaces the in-band observation
|
|
185
|
+
* with preview metadata and a file pointer.
|
|
186
|
+
*
|
|
187
|
+
* @param {object} params
|
|
188
|
+
* @param {string} params.output Full tool output string.
|
|
189
|
+
* @param {number} params.thresholdBytes Output size threshold before triggering spill.
|
|
190
|
+
* @param {string} params.id Unique tool invocation identifier.
|
|
191
|
+
* @param {string} params.scratchDir Relative scratchpad directory path.
|
|
192
|
+
* @param {object} [params.fsService] Sandboxed filesystem service instance.
|
|
193
|
+
* @returns {Promise<{output: string, spilled: boolean, path?: string, bytes?: number}>}
|
|
194
|
+
*/
|
|
195
|
+
export async function spillToolOutput({
|
|
196
|
+
output,
|
|
197
|
+
thresholdBytes,
|
|
198
|
+
id,
|
|
199
|
+
scratchDir,
|
|
200
|
+
fsService,
|
|
201
|
+
}) {
|
|
202
|
+
const bytes = Buffer.byteLength(output, "utf8");
|
|
203
|
+
if (!fsService || bytes <= thresholdBytes) {
|
|
204
|
+
return { output, spilled: false, bytes };
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
const safeId = String(id).replace(/[^a-zA-Z0-9_-]/g, "_").slice(0, 64) || "out";
|
|
208
|
+
const relPath = `${scratchDir}/tool_out_${safeId}.txt`;
|
|
209
|
+
const head = output.slice(0, 1024);
|
|
210
|
+
const tail = output.slice(-1024);
|
|
211
|
+
|
|
212
|
+
let writtenPath;
|
|
213
|
+
try {
|
|
214
|
+
const res = await fsService.writeFile({ path: relPath, content: output, overwrite: true });
|
|
215
|
+
writtenPath = res && res.path ? res.path : relPath;
|
|
216
|
+
} catch (err) {
|
|
217
|
+
// Fail-fast: a failed spill is surfaced, never silently truncated.
|
|
218
|
+
return {
|
|
219
|
+
output:
|
|
220
|
+
`[ToolOutputSpillError] The full ${bytes}-byte tool output could not be ` +
|
|
221
|
+
`saved to ${relPath} (${err.message}). The payload was NOT truncated in-band; ` +
|
|
222
|
+
`re-run the tool with a narrower query (head/tail/grep, start_line/end_line) ` +
|
|
223
|
+
`to retrieve a smaller result.`,
|
|
224
|
+
spilled: false,
|
|
225
|
+
bytes,
|
|
226
|
+
error: err.message,
|
|
227
|
+
};
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
const pointer =
|
|
231
|
+
`[Tool output spilled to disk: ${bytes} bytes > ${thresholdBytes}-byte threshold. ` +
|
|
232
|
+
`Full payload saved to: ${writtenPath}]\n\n` +
|
|
233
|
+
`--- Head preview (first 1024 bytes) ---\n${head}\n` +
|
|
234
|
+
`... [${bytes - 2048} bytes elided] ...\n` +
|
|
235
|
+
`--- Tail preview (last 1024 bytes) ---\n${tail}\n\n` +
|
|
236
|
+
`Hint: use read_file with start_line/end_line or search_code to inspect ` +
|
|
237
|
+
`specific regions of ${writtenPath}.`;
|
|
238
|
+
|
|
239
|
+
return { output: pointer, spilled: true, path: writtenPath, bytes };
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
export class CastorRunner {
|
|
244
|
+
constructor(options = {}) {
|
|
245
|
+
// Canonicalize working directory through OS symlink/junction layer.
|
|
246
|
+
this.defaultCwd = canonicalizePath(options.cwd ? normalizeWorkspacePath(options.cwd) : process.cwd());
|
|
247
|
+
this.defaultMaxTurns = options.maxTurns || BASE_TURN_BUDGET;
|
|
248
|
+
// Optional injection seams (used by offline tests to substitute a mock
|
|
249
|
+
// LLM / logger without touching the network or the real vLLM provider).
|
|
250
|
+
this._llm = options.llm || null;
|
|
251
|
+
this._logger = options.logger || null;
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
/**
|
|
255
|
+
* Executes a bounded extraction pass to salvage partial deliberation output.
|
|
256
|
+
* Invoked upon reasoning budget exhaustion to record intermediate findings
|
|
257
|
+
* to workspace scratch storage.
|
|
258
|
+
*
|
|
259
|
+
* @param {object} params
|
|
260
|
+
* @param {object} params.llm LLM provider instance.
|
|
261
|
+
* @param {object} params.logger Event logger instance.
|
|
262
|
+
* @param {object} [params.fsService] Sandboxed filesystem service.
|
|
263
|
+
* @param {string} params.sessionId Session identifier.
|
|
264
|
+
* @param {AbortSignal} [params.signal] Cancellation abort signal.
|
|
265
|
+
* @returns {Promise<{ salvaged: boolean, path?: string, error?: string }>}
|
|
266
|
+
*/
|
|
267
|
+
async salvageReasoningBudget({ llm, logger, fsService, sessionId, signal }) {
|
|
268
|
+
const safeId = String(sessionId).replace(/[^a-zA-Z0-9_-]/g, "_").slice(0, 64) || "session";
|
|
269
|
+
const salvagePath = `.scratch/salvage_${safeId}.md`;
|
|
270
|
+
const directive = SALVAGE_DIRECTIVE.replace("<SALVAGE_PATH>", salvagePath);
|
|
271
|
+
|
|
272
|
+
let salvageContent = "";
|
|
273
|
+
try {
|
|
274
|
+
// CRITICAL: use a MINIMAL message set (system prompt + directive only),
|
|
275
|
+
// NOT the full conversation history. The session is about to terminate
|
|
276
|
+
// precisely because the context is deep (near the 245K ceiling); passing
|
|
277
|
+
// the full history would re-trigger ContextExhaustedError. The salvage
|
|
278
|
+
// is a fresh, short extraction turn.
|
|
279
|
+
const salvageMessages = [
|
|
280
|
+
{ role: "system", content: "You are a helpful assistant." },
|
|
281
|
+
{ role: "user", content: directive },
|
|
282
|
+
];
|
|
283
|
+
const salvageResult = await llm.streamChat({
|
|
284
|
+
messages: salvageMessages,
|
|
285
|
+
tools: [],
|
|
286
|
+
reasoningEffort: "low",
|
|
287
|
+
maxTokens: SALVAGE_MAX_TOKENS,
|
|
288
|
+
signal,
|
|
289
|
+
sessionId,
|
|
290
|
+
});
|
|
291
|
+
salvageContent = (salvageResult?.content || "").trim();
|
|
292
|
+
} catch (err) {
|
|
293
|
+
logger.append({
|
|
294
|
+
type: "salvage_failed",
|
|
295
|
+
error: err.message,
|
|
296
|
+
path: salvagePath,
|
|
297
|
+
});
|
|
298
|
+
return { salvaged: false, error: err.message };
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
if (salvageContent === "") {
|
|
302
|
+
logger.append({
|
|
303
|
+
type: "salvage_empty",
|
|
304
|
+
path: salvagePath,
|
|
305
|
+
});
|
|
306
|
+
return { salvaged: false };
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
if (!fsService) {
|
|
310
|
+
logger.append({
|
|
311
|
+
type: "salvage_failed",
|
|
312
|
+
error: "No FS service available",
|
|
313
|
+
path: salvagePath,
|
|
314
|
+
});
|
|
315
|
+
return { salvaged: false, error: "No FS service available" };
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
try {
|
|
319
|
+
const res = await fsService.writeFile({
|
|
320
|
+
path: salvagePath,
|
|
321
|
+
content: salvageContent,
|
|
322
|
+
overwrite: true,
|
|
323
|
+
});
|
|
324
|
+
const writtenPath = res?.path || salvagePath;
|
|
325
|
+
logger.append({
|
|
326
|
+
type: "salvage_extracted",
|
|
327
|
+
path: writtenPath,
|
|
328
|
+
bytes: Buffer.byteLength(salvageContent, "utf8"),
|
|
329
|
+
});
|
|
330
|
+
return { salvaged: true, path: writtenPath };
|
|
331
|
+
} catch (err) {
|
|
332
|
+
logger.append({
|
|
333
|
+
type: "salvage_failed",
|
|
334
|
+
error: err.message,
|
|
335
|
+
path: salvagePath,
|
|
336
|
+
});
|
|
337
|
+
return { salvaged: false, error: err.message };
|
|
338
|
+
}
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
/**
|
|
342
|
+
* Runs an autonomous agent session using the Castor harness.
|
|
343
|
+
*
|
|
344
|
+
* @param {object} params
|
|
345
|
+
* @param {string} params.prompt The user / orchestrator objective
|
|
346
|
+
* @param {string} [params.cwd] Target workspace directory
|
|
347
|
+
* @param {string} [params.sessionId] Unique session ID
|
|
348
|
+
* @param {number} [params.maxTurns] Maximum allowed turns (null = unbounded)
|
|
349
|
+
* @param {string} [params.reasoningEffort] Task-local reasoning-effort tier
|
|
350
|
+
* (one of REASONING_EFFORT_TIERS). Threaded to the provider's streamChat
|
|
351
|
+
* for every turn of this session; when absent the provider falls back to
|
|
352
|
+
* the QWEN_REASONING_EFFORT env default (unchanged behavior).
|
|
353
|
+
* @param {AbortSignal} [params.signal] Cancellation signal
|
|
354
|
+
* @param {(token: string) => void} [params.onToken] Live token streaming callback
|
|
355
|
+
* @param {(metric: object) => void} [params.onMetrics] Performance metric callback
|
|
356
|
+
* @param {(toolCall: object) => void} [params.onToolCall] Live tool call notification
|
|
357
|
+
* @returns {Promise<{
|
|
358
|
+
* finalText: string,
|
|
359
|
+
* turnsTaken: number,
|
|
360
|
+
* status: 'completed' | 'completed_ceiling' | 'aborted' | 'turn_limit_reached'
|
|
361
|
+
* | 'failed' | 'engine_empty_response' | 'reasoning_budget_exhausted'
|
|
362
|
+
* | 'length_limit_reached' | 'degenerate_response_truncated',
|
|
363
|
+
* durationMs: number,
|
|
364
|
+
* totalCompletionTokens: number,
|
|
365
|
+
* sessionId: string,
|
|
366
|
+
* sessionTurns: number,
|
|
367
|
+
* lastPromptTokens: number | null,
|
|
368
|
+
* contextHeadroom: number | null
|
|
369
|
+
* }>}
|
|
370
|
+
*/
|
|
371
|
+
async run({
|
|
372
|
+
prompt,
|
|
373
|
+
cwd,
|
|
374
|
+
sessionId = `evo_${Date.now()}`,
|
|
375
|
+
maxTurns = this.defaultMaxTurns,
|
|
376
|
+
reasoningEffort,
|
|
377
|
+
signal,
|
|
378
|
+
onToken,
|
|
379
|
+
onMetrics,
|
|
380
|
+
onToolCall,
|
|
381
|
+
onActivity,
|
|
382
|
+
getDynamicBudget,
|
|
383
|
+
extensions,
|
|
384
|
+
targetInWsl = false,
|
|
385
|
+
testCommand,
|
|
386
|
+
enableEvo = false,
|
|
387
|
+
}) {
|
|
388
|
+
const t0 = Date.now();
|
|
389
|
+
const loopDetector = new LoopDetector({
|
|
390
|
+
windowSize: LOOP_DETECTION_WINDOW,
|
|
391
|
+
threshold: LOOP_DETECTION_REPETITIONS,
|
|
392
|
+
});
|
|
393
|
+
// Canonicalize working directory through OS symlink/junction layer.
|
|
394
|
+
const effectiveCwd = cwd ? canonicalizePath(normalizeWorkspacePath(cwd)) : this.defaultCwd;
|
|
395
|
+
|
|
396
|
+
// Initialize the Castor microkernel context
|
|
397
|
+
const ctx = new Context(null, `session_${sessionId}`);
|
|
398
|
+
|
|
399
|
+
// Mount core services. The LLM provider and event logger can be injected
|
|
400
|
+
// via the constructor (this._llm / this._logger) for offline testing; when
|
|
401
|
+
// absent we mount the real vLLM provider and on-disk JSONL logger.
|
|
402
|
+
ctx.plugin(sandboxFsPlugin, { root: effectiveCwd });
|
|
403
|
+
ctx.plugin(shellExecutorPlugin, { cwd: effectiveCwd });
|
|
404
|
+
if (this._logger) {
|
|
405
|
+
ctx.provide("logger", this._logger);
|
|
406
|
+
} else {
|
|
407
|
+
ctx.plugin(eventLoggerPlugin, { sessionId });
|
|
408
|
+
}
|
|
409
|
+
if (this._llm) {
|
|
410
|
+
ctx.provide("llm", this._llm);
|
|
411
|
+
} else {
|
|
412
|
+
ctx.plugin(vllmProviderPlugin);
|
|
413
|
+
}
|
|
414
|
+
ctx.plugin(astPlugin, { root: effectiveCwd });
|
|
415
|
+
ctx.plugin(webPlugin);
|
|
416
|
+
|
|
417
|
+
// Conditional Evo mounting: only mount the Evo closed-loop optimization tools
|
|
418
|
+
// when a testCommand / evaluation benchmark or explicit evo flag is active.
|
|
419
|
+
// In standard exploration/editing turns, the toolset remains lean at exactly 8 tools.
|
|
420
|
+
const isEvoActive = Boolean(testCommand || enableEvo);
|
|
421
|
+
if (isEvoActive) {
|
|
422
|
+
ctx.plugin(evoPlugin, { workspaceRoot: effectiveCwd });
|
|
423
|
+
}
|
|
424
|
+
|
|
425
|
+
// Initialize stdio MCP extension bridge before runner execution.
|
|
426
|
+
let mcpBridge = null;
|
|
427
|
+
if (Array.isArray(extensions) && extensions.length > 0) {
|
|
428
|
+
mcpBridge = new McpBridge({
|
|
429
|
+
cwd: effectiveCwd,
|
|
430
|
+
targetInWsl,
|
|
431
|
+
extensions,
|
|
432
|
+
});
|
|
433
|
+
await mcpBridge.start(ctx);
|
|
434
|
+
}
|
|
435
|
+
|
|
436
|
+
const logger = ctx.get("logger");
|
|
437
|
+
const llm = ctx.get("llm");
|
|
438
|
+
|
|
439
|
+
const priorEvents = logger.readAll();
|
|
440
|
+
const hasPriorUserMessages = priorEvents.some((e) => e.type === "user_message");
|
|
441
|
+
const priorSkills = new Set();
|
|
442
|
+
for (const ev of priorEvents) {
|
|
443
|
+
if (ev.type === "skills_injected" && Array.isArray(ev.skills)) {
|
|
444
|
+
for (const s of ev.skills) priorSkills.add(s);
|
|
445
|
+
}
|
|
446
|
+
}
|
|
447
|
+
|
|
448
|
+
// Match and inject applicable workflow skills before entering message list.
|
|
449
|
+
let effectivePrompt = prompt;
|
|
450
|
+
let matchedSkillNames = [];
|
|
451
|
+
if (!hasPriorUserMessages || priorSkills.size === 0) {
|
|
452
|
+
if (!prompt.includes("--- Matching skills (auto-injected from skills/) ---")) {
|
|
453
|
+
matchedSkillNames = matchSkills({ prompt, cwd: effectiveCwd }).map(
|
|
454
|
+
(s) => s.name
|
|
455
|
+
);
|
|
456
|
+
effectivePrompt = injectSkills(prompt, effectiveCwd);
|
|
457
|
+
}
|
|
458
|
+
}
|
|
459
|
+
|
|
460
|
+
logger.append({
|
|
461
|
+
type: "session_start",
|
|
462
|
+
harness: "Castor",
|
|
463
|
+
version: PKG_VERSION,
|
|
464
|
+
cwd: effectiveCwd,
|
|
465
|
+
prompt,
|
|
466
|
+
// Effective reasoning-effort tier for this session (task-local param when
|
|
467
|
+
// provided, else the QWEN_REASONING_EFFORT env default). Surfaced for
|
|
468
|
+
// telemetry; the provider re-resolves the same value per request.
|
|
469
|
+
reasoningEffort: reasoningEffort || getReasoningEffort(),
|
|
470
|
+
});
|
|
471
|
+
|
|
472
|
+
if (matchedSkillNames.length > 0) {
|
|
473
|
+
logger.append({ type: "skills_injected", skills: matchedSkillNames });
|
|
474
|
+
}
|
|
475
|
+
|
|
476
|
+
const systemPrompt = isEvoActive
|
|
477
|
+
? DEFAULT_SYSTEM_PROMPT + EVO_SYSTEM_PROMPT_ADDENDUM
|
|
478
|
+
: DEFAULT_SYSTEM_PROMPT;
|
|
479
|
+
|
|
480
|
+
const messages = [
|
|
481
|
+
{ role: "system", content: systemPrompt },
|
|
482
|
+
...logger.getConversationHistory(),
|
|
483
|
+
];
|
|
484
|
+
|
|
485
|
+
// Add current user prompt (with any auto-injected skills block)
|
|
486
|
+
messages.push({ role: "user", content: effectivePrompt });
|
|
487
|
+
logger.append({ type: "user_message", content: effectivePrompt });
|
|
488
|
+
|
|
489
|
+
let turnsTaken = 0;
|
|
490
|
+
let finalText = "";
|
|
491
|
+
let status = "completed";
|
|
492
|
+
let totalCompletionTokens = 0;
|
|
493
|
+
let continuationsInjected = 0;
|
|
494
|
+
let consecutiveReasoningContinuations = 0;
|
|
495
|
+
let emptyStreamRetries = 0;
|
|
496
|
+
// Track consecutive non-mutating command invocations.
|
|
497
|
+
let probeStreak = 0;
|
|
498
|
+
|
|
499
|
+
// Track session-cumulative turn counts across tasks.
|
|
500
|
+
let sessionTurns = logger
|
|
501
|
+
.readAll()
|
|
502
|
+
.filter((e) => e.type === "assistant_message").length;
|
|
503
|
+
let sessionWarnLatched = false;
|
|
504
|
+
let sessionRecommendLatched = false;
|
|
505
|
+
let contextDepthLatched = false;
|
|
506
|
+
let contextHighWatermarkLatched = false;
|
|
507
|
+
let contextEmergencyCeilingLatched = false;
|
|
508
|
+
let contextEmergencySynthesisEmitted = false;
|
|
509
|
+
let lastPromptTokens = 0;
|
|
510
|
+
|
|
511
|
+
// Emit prompt_over_budget advisory event when prompt exceeds PROMPT_BUDGET_CHARS.
|
|
512
|
+
if (prompt.length > PROMPT_BUDGET_CHARS) {
|
|
513
|
+
logger.append({
|
|
514
|
+
type: "prompt_over_budget",
|
|
515
|
+
promptChars: prompt.length,
|
|
516
|
+
budget: PROMPT_BUDGET_CHARS,
|
|
517
|
+
});
|
|
518
|
+
}
|
|
519
|
+
|
|
520
|
+
try {
|
|
521
|
+
while (true) {
|
|
522
|
+
if (signal?.aborted) {
|
|
523
|
+
status = "aborted";
|
|
524
|
+
break;
|
|
525
|
+
}
|
|
526
|
+
|
|
527
|
+
const currentMaxTurns = typeof getDynamicBudget === "function" ? getDynamicBudget() : (maxTurns || BASE_TURN_BUDGET);
|
|
528
|
+
|
|
529
|
+
if (contextEmergencySynthesisEmitted && continuationsInjected === 0) {
|
|
530
|
+
status = "completed_budget_exhausted";
|
|
531
|
+
break;
|
|
532
|
+
}
|
|
533
|
+
|
|
534
|
+
if (currentMaxTurns && turnsTaken >= currentMaxTurns && continuationsInjected === 0) {
|
|
535
|
+
status = finalText.trim() !== "" ? "completed_budget_exhausted" : "turn_limit_reached";
|
|
536
|
+
break;
|
|
537
|
+
}
|
|
538
|
+
|
|
539
|
+
// Cooperative landing: on the final turn before currentMaxTurns OR when context emergency ceiling is latched,
|
|
540
|
+
// strip tools and mandate synthesis.
|
|
541
|
+
const isCeilingTurn = Boolean(
|
|
542
|
+
contextEmergencyCeilingLatched ||
|
|
543
|
+
(currentMaxTurns && currentMaxTurns > 1 && turnsTaken === currentMaxTurns - 1) ||
|
|
544
|
+
(currentMaxTurns && turnsTaken >= currentMaxTurns && continuationsInjected > 0)
|
|
545
|
+
);
|
|
546
|
+
if (isCeilingTurn && continuationsInjected === 0) {
|
|
547
|
+
if (contextEmergencyCeilingLatched) {
|
|
548
|
+
contextEmergencySynthesisEmitted = true;
|
|
549
|
+
}
|
|
550
|
+
const synthesisPrompt = contextEmergencyCeilingLatched
|
|
551
|
+
? `[Emergency Context Landing (${lastPromptTokens || "215,000+"}/${MAX_CONTEXT.toLocaleString("en-US")} tokens)]: Context space is near capacity. Tools are now disabled to prevent an unhandled engine crash. Synthesize your final deliverable, findings, code changes, and grounded conclusions immediately.`
|
|
552
|
+
: `[Dispatch Budget Notice (${turnsTaken + 1}/${currentMaxTurns})]: You have reached the final turn of your allotted budget for this dispatch. Synthesize your final deliverable, findings, code changes, and grounded conclusions immediately based on the facts gathered so far.`;
|
|
553
|
+
|
|
554
|
+
messages.push({
|
|
555
|
+
role: "user",
|
|
556
|
+
content: synthesisPrompt,
|
|
557
|
+
});
|
|
558
|
+
logger.append({
|
|
559
|
+
type: contextEmergencyCeilingLatched ? "context_emergency_synthesis" : "turn_ceiling_synthesis",
|
|
560
|
+
turnsTaken: turnsTaken + 1,
|
|
561
|
+
maxTurns: currentMaxTurns,
|
|
562
|
+
});
|
|
563
|
+
}
|
|
564
|
+
|
|
565
|
+
turnsTaken++;
|
|
566
|
+
|
|
567
|
+
// Emit session warning telemetry when cumulative turn thresholds are reached.
|
|
568
|
+
sessionTurns = sessionTurns + 1;
|
|
569
|
+
if (!sessionWarnLatched && sessionTurns >= SESSION_TURNS_WARN) {
|
|
570
|
+
sessionWarnLatched = true;
|
|
571
|
+
logger.append({
|
|
572
|
+
type: "session_warning",
|
|
573
|
+
sessionTurns,
|
|
574
|
+
threshold: SESSION_TURNS_WARN,
|
|
575
|
+
});
|
|
576
|
+
}
|
|
577
|
+
if (!sessionRecommendLatched && sessionTurns >= SESSION_TURNS_RECOMMEND) {
|
|
578
|
+
sessionRecommendLatched = true;
|
|
579
|
+
logger.append({
|
|
580
|
+
type: "session_turn_limit_recommended",
|
|
581
|
+
sessionTurns,
|
|
582
|
+
});
|
|
583
|
+
// ONE in-band user-role advisory: complete this task, then roll to a
|
|
584
|
+
// fresh session on the next dispatch.
|
|
585
|
+
messages.push({
|
|
586
|
+
role: "user",
|
|
587
|
+
content: SESSION_ROLLOVER_ADVISORY,
|
|
588
|
+
});
|
|
589
|
+
}
|
|
590
|
+
|
|
591
|
+
const tools = isCeilingTurn ? [] : ctx.listTools();
|
|
592
|
+
|
|
593
|
+
const turnResult = await llm.streamChat({
|
|
594
|
+
messages,
|
|
595
|
+
tools,
|
|
596
|
+
// Task-local reasoning-effort override (per-dispatch). Threaded to
|
|
597
|
+
// every turn of this session; the provider falls back to the
|
|
598
|
+
// QWEN_REASONING_EFFORT env default when it is absent.
|
|
599
|
+
reasoningEffort,
|
|
600
|
+
signal,
|
|
601
|
+
sessionId,
|
|
602
|
+
onToken: (tok) => {
|
|
603
|
+
if (onToken) onToken(tok);
|
|
604
|
+
},
|
|
605
|
+
onMetrics: (m) => {
|
|
606
|
+
totalCompletionTokens += m.completionTokens;
|
|
607
|
+
if (onMetrics) onMetrics(m);
|
|
608
|
+
try {
|
|
609
|
+
recordTurnTelemetry({
|
|
610
|
+
completionTokens: m.completionTokens || 0,
|
|
611
|
+
promptTokens: m.promptTokens || 0,
|
|
612
|
+
reasoningTokens: m.reasoningTokens || 0,
|
|
613
|
+
ttftMs: m.ttftMs,
|
|
614
|
+
prefillMs: m.prefillMs ?? m.ttftMs,
|
|
615
|
+
generationMs: m.generationMs ?? (m.totalMs && m.ttftMs ? Math.max(0, m.totalMs - m.ttftMs) : 0),
|
|
616
|
+
totalMs: m.totalMs,
|
|
617
|
+
prefillTps: m.prefillTps,
|
|
618
|
+
decodeTps: m.decodeTps,
|
|
619
|
+
tpotMs: m.tpotMs,
|
|
620
|
+
effort: reasoningEffort || "medium",
|
|
621
|
+
});
|
|
622
|
+
} catch {}
|
|
623
|
+
},
|
|
624
|
+
});
|
|
625
|
+
|
|
626
|
+
// Hoist re-prefill size and depth-aware empty-stream retry budget.
|
|
627
|
+
const promptChars = JSON.stringify(messages).length;
|
|
628
|
+
const emptyStreamRetryBudget =
|
|
629
|
+
promptChars >= EMPTY_STREAM_RETRY_DEPTH_CHARS
|
|
630
|
+
? EMPTY_STREAM_RETRIES_DEEP
|
|
631
|
+
: EMPTY_STREAM_RETRIES;
|
|
632
|
+
|
|
633
|
+
// Guard against empty generation streams with missing finish reason.
|
|
634
|
+
const isEmptyGeneration =
|
|
635
|
+
(!turnResult.content || turnResult.content.trim() === "") &&
|
|
636
|
+
(!turnResult.toolCalls || turnResult.toolCalls.length === 0) &&
|
|
637
|
+
(turnResult.finishReason === undefined ||
|
|
638
|
+
turnResult.finishReason === null ||
|
|
639
|
+
turnResult.finishReason === "");
|
|
640
|
+
|
|
641
|
+
// Guard against empty stop turns (zero content and zero tool calls with stop finish reason).
|
|
642
|
+
const isEmptyStop =
|
|
643
|
+
turnResult.finishReason === "stop" &&
|
|
644
|
+
(!turnResult.content || turnResult.content.trim() === "") &&
|
|
645
|
+
(!turnResult.toolCalls || turnResult.toolCalls.length === 0);
|
|
646
|
+
|
|
647
|
+
if (isEmptyGeneration || isEmptyStop) {
|
|
648
|
+
// Record task execution coordinates and metrics at point of stream termination.
|
|
649
|
+
const deathContext = {
|
|
650
|
+
turnIndex: turnsTaken,
|
|
651
|
+
promptChars,
|
|
652
|
+
metrics: turnResult.metrics ?? null,
|
|
653
|
+
reasoningTokens: turnResult.reasoningTokens ?? 0,
|
|
654
|
+
};
|
|
655
|
+
if (emptyStreamRetries < emptyStreamRetryBudget) {
|
|
656
|
+
emptyStreamRetries++;
|
|
657
|
+
// Exponential backoff before retry attempt.
|
|
658
|
+
const backoffMs = emptyStreamRetryBackoffMs(emptyStreamRetries);
|
|
659
|
+
logger.append({
|
|
660
|
+
type: "empty_stream_retry",
|
|
661
|
+
retryNumber: emptyStreamRetries,
|
|
662
|
+
maxRetries: emptyStreamRetryBudget,
|
|
663
|
+
// Reason: empty generation stream.
|
|
664
|
+
reason: isEmptyStop ? "empty_stop" : "empty_generation",
|
|
665
|
+
backoffMs,
|
|
666
|
+
...deathContext,
|
|
667
|
+
});
|
|
668
|
+
// If the model completed deliberation inside thinking tags but omitted visible content or tool calls,
|
|
669
|
+
// prompt it directly to emit its conclusion instead of repeating the identical prompt.
|
|
670
|
+
if (isEmptyStop && (turnResult.hadReasoning || (turnResult.reasoning && turnResult.reasoning.trim()))) {
|
|
671
|
+
messages.push({
|
|
672
|
+
role: "user",
|
|
673
|
+
content: "You concluded your internal deliberation without emitting a response or tool call. Please output your conclusion or next action directly now.",
|
|
674
|
+
});
|
|
675
|
+
}
|
|
676
|
+
await new Promise((r) => setTimeout(r, backoffMs));
|
|
677
|
+
continue;
|
|
678
|
+
}
|
|
679
|
+
// Budget exhausted: the engine keeps returning empty generations.
|
|
680
|
+
// Report the honest status instead of a false "completed".
|
|
681
|
+
status = "engine_empty_response";
|
|
682
|
+
logger.append({ type: "engine_empty_response", ...deathContext });
|
|
683
|
+
break;
|
|
684
|
+
}
|
|
685
|
+
|
|
686
|
+
// Degenerate-final guard: catch sentinel-truncated repetition outputs lacking substantive content.
|
|
687
|
+
const guardMarkerIdx =
|
|
688
|
+
typeof turnResult.content === "string"
|
|
689
|
+
? turnResult.content.indexOf(GUARD_MARKER_PREFIX)
|
|
690
|
+
: -1;
|
|
691
|
+
if (guardMarkerIdx !== -1) {
|
|
692
|
+
// Strip the marker (prefix ... closing bracket) and measure the
|
|
693
|
+
// substantive remainder (the real text before/after the marker).
|
|
694
|
+
const closeIdx = turnResult.content.indexOf("]", guardMarkerIdx);
|
|
695
|
+
const substantive =
|
|
696
|
+
closeIdx === -1
|
|
697
|
+
? turnResult.content.slice(0, guardMarkerIdx)
|
|
698
|
+
: turnResult.content.slice(0, guardMarkerIdx) +
|
|
699
|
+
turnResult.content.slice(closeIdx + 1);
|
|
700
|
+
const substantiveLen = substantive.trim().length;
|
|
701
|
+
const noToolCalls =
|
|
702
|
+
!turnResult.toolCalls || turnResult.toolCalls.length === 0;
|
|
703
|
+
const shortSession = turnsTaken <= DEGENERATE_FINAL_MAX_TURNS;
|
|
704
|
+
if (
|
|
705
|
+
substantiveLen < DEGENERATE_FINAL_SUBSTANTIVE_CHARS &&
|
|
706
|
+
noToolCalls &&
|
|
707
|
+
shortSession
|
|
708
|
+
) {
|
|
709
|
+
const deathContext = {
|
|
710
|
+
turnIndex: turnsTaken,
|
|
711
|
+
// Context size hoisted above retry branches.
|
|
712
|
+
promptChars,
|
|
713
|
+
metrics: turnResult.metrics ?? null,
|
|
714
|
+
reasoningTokens: turnResult.reasoningTokens ?? 0,
|
|
715
|
+
substantiveChars: substantiveLen,
|
|
716
|
+
};
|
|
717
|
+
if (emptyStreamRetries < emptyStreamRetryBudget) {
|
|
718
|
+
emptyStreamRetries++;
|
|
719
|
+
// Exponential backoff before retry attempt.
|
|
720
|
+
const backoffMs = emptyStreamRetryBackoffMs(emptyStreamRetries);
|
|
721
|
+
logger.append({
|
|
722
|
+
type: "empty_stream_retry",
|
|
723
|
+
retryNumber: emptyStreamRetries,
|
|
724
|
+
maxRetries: emptyStreamRetryBudget,
|
|
725
|
+
// Reason: degenerate sentinel-truncated final.
|
|
726
|
+
reason: "degenerate_final",
|
|
727
|
+
backoffMs,
|
|
728
|
+
...deathContext,
|
|
729
|
+
});
|
|
730
|
+
await new Promise((r) => setTimeout(r, backoffMs));
|
|
731
|
+
continue;
|
|
732
|
+
}
|
|
733
|
+
// Budget exhausted: the engine keeps returning degenerate
|
|
734
|
+
// guard-truncated finals. Report the honest status instead of a
|
|
735
|
+
// false "completed". Preserve the original partial+marker in
|
|
736
|
+
// finalText for honesty (the client sees exactly what the engine
|
|
737
|
+
// produced, including the guard marker).
|
|
738
|
+
status = "degenerate_response_truncated";
|
|
739
|
+
finalText = turnResult.content;
|
|
740
|
+
logger.append({
|
|
741
|
+
type: "degenerate_response_truncated",
|
|
742
|
+
...deathContext,
|
|
743
|
+
});
|
|
744
|
+
break;
|
|
745
|
+
}
|
|
746
|
+
}
|
|
747
|
+
|
|
748
|
+
// Emit context-depth advisory telemetry when prompt token accumulation reaches warning threshold.
|
|
749
|
+
if (
|
|
750
|
+
!contextDepthLatched &&
|
|
751
|
+
typeof turnResult.metrics?.promptTokens === "number" &&
|
|
752
|
+
turnResult.metrics.promptTokens >= CONTEXT_WARN_TOKENS
|
|
753
|
+
) {
|
|
754
|
+
contextDepthLatched = true;
|
|
755
|
+
logger.append({
|
|
756
|
+
type: "context_depth_warning",
|
|
757
|
+
promptTokens: turnResult.metrics.promptTokens,
|
|
758
|
+
threshold: CONTEXT_WARN_TOKENS,
|
|
759
|
+
...(probeStreak > 0 ? { probeStreakActive: true } : {}),
|
|
760
|
+
});
|
|
761
|
+
}
|
|
762
|
+
|
|
763
|
+
if (typeof turnResult.metrics?.promptTokens === "number") {
|
|
764
|
+
lastPromptTokens = turnResult.metrics.promptTokens;
|
|
765
|
+
}
|
|
766
|
+
|
|
767
|
+
// --- Context High-Watermark Advisory (180,000 tokens) ---------------
|
|
768
|
+
// When promptTokens reaches the 180k high-watermark (~73% of nominal 245K
|
|
769
|
+
// context ceiling), emit a one-shot advisory event and inject an in-band
|
|
770
|
+
// rollover advisory to guide the model to conclude its deliverable rather
|
|
771
|
+
// than crashing with unhandled ContextExhaustedError / 400 Bad Request.
|
|
772
|
+
if (
|
|
773
|
+
!contextHighWatermarkLatched &&
|
|
774
|
+
typeof turnResult.metrics?.promptTokens === "number" &&
|
|
775
|
+
turnResult.metrics.promptTokens >= CONTEXT_HIGH_WATERMARK_TOKENS
|
|
776
|
+
) {
|
|
777
|
+
contextHighWatermarkLatched = true;
|
|
778
|
+
logger.append({
|
|
779
|
+
type: "context_high_watermark",
|
|
780
|
+
promptTokens: turnResult.metrics.promptTokens,
|
|
781
|
+
threshold: CONTEXT_HIGH_WATERMARK_TOKENS,
|
|
782
|
+
});
|
|
783
|
+
// E4: arm the adaptive read-size governor. The context is under
|
|
784
|
+
// pressure (near the 180k high-watermark), so a whole-file read
|
|
785
|
+
// (64KB default) could blow the 245K ceiling (F6.1). Lower the
|
|
786
|
+
// session's read cap to 16KB so subsequent read_file calls are
|
|
787
|
+
// bounded. The governor only LOWERS the cap and is per-session
|
|
788
|
+
// (the SandboxFsService is a fresh per-run instance), so it cannot
|
|
789
|
+
// leak across tasks. Best-effort: a missing FS service is a no-op.
|
|
790
|
+
const fsService = ctx.get("fs");
|
|
791
|
+
if (fsService && typeof fsService.setReadGovernor === "function") {
|
|
792
|
+
fsService.setReadGovernor(READ_GOVERNOR_MAX_BYTES);
|
|
793
|
+
logger.append({
|
|
794
|
+
type: "read_governor_armed",
|
|
795
|
+
maxBytes: READ_GOVERNOR_MAX_BYTES,
|
|
796
|
+
promptTokens: turnResult.metrics.promptTokens,
|
|
797
|
+
});
|
|
798
|
+
}
|
|
799
|
+
messages.push({
|
|
800
|
+
role: "user",
|
|
801
|
+
content:
|
|
802
|
+
`[Context High-Watermark Advisory] Prompt context has reached ${turnResult.metrics.promptTokens} tokens ` +
|
|
803
|
+
`(high-watermark: ${CONTEXT_HIGH_WATERMARK_TOKENS}, max ceiling: ${MAX_CONTEXT.toLocaleString("en-US")}). ` +
|
|
804
|
+
`Wrap up your deliverable and return your final response now. ` +
|
|
805
|
+
`Advise the user/orchestrator to roll into a fresh session_id for subsequent dispatches to prevent context exhaustion.`,
|
|
806
|
+
});
|
|
807
|
+
}
|
|
808
|
+
|
|
809
|
+
// --- Context Emergency Ceiling Latch (215,000 tokens) ----------------
|
|
810
|
+
// When promptTokens approaches the 245K ceiling (~87%), strip tools on the
|
|
811
|
+
// subsequent turn to trigger emergency synthesis and prevent an unhandled
|
|
812
|
+
// context_exhausted crash.
|
|
813
|
+
if (
|
|
814
|
+
!contextEmergencyCeilingLatched &&
|
|
815
|
+
typeof turnResult.metrics?.promptTokens === "number" &&
|
|
816
|
+
turnResult.metrics.promptTokens >= CONTEXT_EMERGENCY_CEILING_TOKENS
|
|
817
|
+
) {
|
|
818
|
+
contextEmergencyCeilingLatched = true;
|
|
819
|
+
logger.append({
|
|
820
|
+
type: "context_emergency_ceiling",
|
|
821
|
+
promptTokens: turnResult.metrics.promptTokens,
|
|
822
|
+
threshold: CONTEXT_EMERGENCY_CEILING_TOKENS,
|
|
823
|
+
});
|
|
824
|
+
}
|
|
825
|
+
|
|
826
|
+
// Record assistant response. reasoningTokens is surfaced as a top-level
|
|
827
|
+
// field (in addition to metrics) so ledgers show thinking volume even
|
|
828
|
+
// when the metrics object is summarized or dropped downstream.
|
|
829
|
+
logger.append({
|
|
830
|
+
type: "assistant_message",
|
|
831
|
+
content: turnResult.content,
|
|
832
|
+
toolCalls: turnResult.toolCalls,
|
|
833
|
+
finishReason: turnResult.finishReason,
|
|
834
|
+
reasoningTokens:
|
|
835
|
+
turnResult.metrics?.reasoningTokens ?? turnResult.reasoningTokens ?? 0,
|
|
836
|
+
metrics: turnResult.metrics,
|
|
837
|
+
});
|
|
838
|
+
|
|
839
|
+
messages.push({
|
|
840
|
+
role: "assistant",
|
|
841
|
+
content: turnResult.content || null,
|
|
842
|
+
tool_calls: turnResult.toolCalls.length > 0 ? turnResult.toolCalls : undefined,
|
|
843
|
+
});
|
|
844
|
+
|
|
845
|
+
if (
|
|
846
|
+
(turnResult.content && turnResult.content.trim() !== "") ||
|
|
847
|
+
(turnResult.toolCalls && turnResult.toolCalls.length > 0)
|
|
848
|
+
) {
|
|
849
|
+
consecutiveReasoningContinuations = 0;
|
|
850
|
+
}
|
|
851
|
+
|
|
852
|
+
if (turnResult.content && turnResult.content.trim() !== "") {
|
|
853
|
+
finalText = turnResult.content;
|
|
854
|
+
if (onActivity) {
|
|
855
|
+
onActivity(turnResult.content.trim().slice(-SUPERVISOR_PREVIEW_CHARS));
|
|
856
|
+
}
|
|
857
|
+
}
|
|
858
|
+
|
|
859
|
+
// If no tool calls, the model concluded its turn - UNLESS the output
|
|
860
|
+
// was cut off by the token ceiling (finish_reason: "length"). In that
|
|
861
|
+
// case the answer is truncated, so we re-prompt the model to resume.
|
|
862
|
+
if (!turnResult.toolCalls || turnResult.toolCalls.length === 0) {
|
|
863
|
+
if (turnResult.finishReason === "length") {
|
|
864
|
+
const hadReasoning =
|
|
865
|
+
turnResult.metrics?.hadReasoning ?? turnResult.hadReasoning ?? false;
|
|
866
|
+
const isReasoningCutoff =
|
|
867
|
+
Boolean(turnResult.metrics?.reasoningCeilingHit) ||
|
|
868
|
+
(hadReasoning && (!turnResult.content || turnResult.content.trim() === ""));
|
|
869
|
+
|
|
870
|
+
if (isReasoningCutoff) {
|
|
871
|
+
consecutiveReasoningContinuations++;
|
|
872
|
+
if (consecutiveReasoningContinuations > 1) {
|
|
873
|
+
status = "reasoning_budget_exhausted";
|
|
874
|
+
finalText = "ReasoningBudgetExhaustedError: The model reached the deliberation ceiling across consecutive continuation turns without taking action or concluding.";
|
|
875
|
+
// E2: bounded salvage extraction pass (F6.3). Best-effort;
|
|
876
|
+
// never alters the honest status above.
|
|
877
|
+
const salvage = await this.salvageReasoningBudget({
|
|
878
|
+
llm,
|
|
879
|
+
logger,
|
|
880
|
+
fsService: ctx.get("fs"),
|
|
881
|
+
sessionId,
|
|
882
|
+
signal,
|
|
883
|
+
});
|
|
884
|
+
if (salvage.salvaged) {
|
|
885
|
+
finalText += `\n\n[Salvage] Partial findings saved to: ${salvage.path}`;
|
|
886
|
+
}
|
|
887
|
+
break;
|
|
888
|
+
}
|
|
889
|
+
continuationsInjected++;
|
|
890
|
+
messages.push({ role: "user", content: REASONING_CONTINUATION_DIRECTIVE });
|
|
891
|
+
logger.append({
|
|
892
|
+
type: "continuation_injected",
|
|
893
|
+
content: REASONING_CONTINUATION_DIRECTIVE,
|
|
894
|
+
continuationNumber: continuationsInjected,
|
|
895
|
+
maxContinuations: MAX_CONTINUATION_TURNS,
|
|
896
|
+
reason: "reasoning_ceiling",
|
|
897
|
+
directive: "balanced_landing",
|
|
898
|
+
hadReasoning: true,
|
|
899
|
+
});
|
|
900
|
+
continue;
|
|
901
|
+
}
|
|
902
|
+
|
|
903
|
+
consecutiveReasoningContinuations = 0;
|
|
904
|
+
if (continuationsInjected < MAX_CONTINUATION_TURNS) {
|
|
905
|
+
continuationsInjected++;
|
|
906
|
+
// Provide clean continuation without artificial stop-thinking directives
|
|
907
|
+
messages.push({ role: "user", content: CONTINUATION_DIRECTIVE });
|
|
908
|
+
logger.append({
|
|
909
|
+
type: "continuation_injected",
|
|
910
|
+
content: CONTINUATION_DIRECTIVE,
|
|
911
|
+
continuationNumber: continuationsInjected,
|
|
912
|
+
maxContinuations: MAX_CONTINUATION_TURNS,
|
|
913
|
+
reason: "length",
|
|
914
|
+
directive: "resume",
|
|
915
|
+
hadReasoning,
|
|
916
|
+
});
|
|
917
|
+
continue;
|
|
918
|
+
}
|
|
919
|
+
// Continuation budget exhausted: report an honest status instead of a
|
|
920
|
+
// false "completed". If the model produced a complete deliverable
|
|
921
|
+
// (non-empty finalText) despite hitting the ceiling, that is
|
|
922
|
+
// "completed_ceiling" — a successful run that merely ran out of room,
|
|
923
|
+
// NOT a failure. Only when nothing usable was produced do we fall
|
|
924
|
+
// through to the honest failure statuses.
|
|
925
|
+
if (isCeilingTurn) {
|
|
926
|
+
status = finalText.trim() !== "" ? "completed_budget_exhausted" : "turn_limit_reached";
|
|
927
|
+
} else if (finalText.trim() !== "") {
|
|
928
|
+
status = "completed_ceiling";
|
|
929
|
+
} else if (hadReasoning) {
|
|
930
|
+
status = "reasoning_budget_exhausted";
|
|
931
|
+
finalText = "ReasoningBudgetExhaustedError: The model exhausted the continuation reasoning budget without emitting visible actions or content.";
|
|
932
|
+
// E2: bounded salvage extraction pass (F6.3). Best-effort;
|
|
933
|
+
// never alters the honest status above.
|
|
934
|
+
const salvage = await this.salvageReasoningBudget({
|
|
935
|
+
llm,
|
|
936
|
+
logger,
|
|
937
|
+
fsService: ctx.get("fs"),
|
|
938
|
+
sessionId,
|
|
939
|
+
signal,
|
|
940
|
+
});
|
|
941
|
+
if (salvage.salvaged) {
|
|
942
|
+
finalText += `\n\n[Salvage] Partial findings saved to: ${salvage.path}`;
|
|
943
|
+
}
|
|
944
|
+
} else {
|
|
945
|
+
status = "length_limit_reached";
|
|
946
|
+
}
|
|
947
|
+
break;
|
|
948
|
+
}
|
|
949
|
+
if (isCeilingTurn) {
|
|
950
|
+
status = "completed_budget_exhausted";
|
|
951
|
+
break;
|
|
952
|
+
}
|
|
953
|
+
// The model concluded its turn with a clean "stop" (or other non-length
|
|
954
|
+
// finish reason). If it had to hit the token ceiling the maximum number
|
|
955
|
+
// of times (continuation budget at the cap) before finally completing,
|
|
956
|
+
// that is "completed_ceiling" — a complete deliverable that only finished
|
|
957
|
+
// after exhausting the continuation budget. A clean "stop" that never
|
|
958
|
+
// exhausted the continuation budget is a plain "completed".
|
|
959
|
+
if (
|
|
960
|
+
continuationsInjected >= MAX_CONTINUATION_TURNS &&
|
|
961
|
+
finalText.trim() !== ""
|
|
962
|
+
) {
|
|
963
|
+
status = "completed_ceiling";
|
|
964
|
+
}
|
|
965
|
+
break;
|
|
966
|
+
}
|
|
967
|
+
|
|
968
|
+
// Execute each requested tool call
|
|
969
|
+
consecutiveReasoningContinuations = 0;
|
|
970
|
+
let droppedTruncatedCalls = 0;
|
|
971
|
+
for (const tc of turnResult.toolCalls) {
|
|
972
|
+
if (signal?.aborted) break;
|
|
973
|
+
|
|
974
|
+
let parsedArgs;
|
|
975
|
+
let argsParseFailed = false;
|
|
976
|
+
try {
|
|
977
|
+
parsedArgs = JSON.parse(tc.function.arguments || "{}");
|
|
978
|
+
} catch {
|
|
979
|
+
// The tool-call arguments were cut off mid-stream (typically by a
|
|
980
|
+
// token-ceiling "length" cutoff). NEVER execute a mangled payload:
|
|
981
|
+
// drop the call, tell the model, and let it re-emit it completely.
|
|
982
|
+
argsParseFailed = true;
|
|
983
|
+
droppedTruncatedCalls++;
|
|
984
|
+
}
|
|
985
|
+
|
|
986
|
+
if (argsParseFailed) {
|
|
987
|
+
// Sanitize the malformed argument string in the assistant message so downstream
|
|
988
|
+
// API parsers (vLLM's qwen3_coder / python json.loads) do not fail with HTTP 400 JSONDecodeError
|
|
989
|
+
tc.function.arguments = "{}";
|
|
990
|
+
|
|
991
|
+
const notice =
|
|
992
|
+
`ToolExecutionError: Tool '${tc.function.name}' (id ${tc.id}) was dropped: its arguments ` +
|
|
993
|
+
`were truncated mid-stream and could not be parsed as JSON (finish_reason: "length"). ` +
|
|
994
|
+
`Please re-emit this tool call with complete, valid JSON arguments.`;
|
|
995
|
+
messages.push({
|
|
996
|
+
role: "tool",
|
|
997
|
+
tool_call_id: tc.id,
|
|
998
|
+
content: notice,
|
|
999
|
+
});
|
|
1000
|
+
logger.append({
|
|
1001
|
+
type: "tool_call_dropped",
|
|
1002
|
+
toolCallId: tc.id,
|
|
1003
|
+
name: tc.function.name,
|
|
1004
|
+
notice,
|
|
1005
|
+
reason: "truncated_arguments",
|
|
1006
|
+
finishReason: turnResult.finishReason,
|
|
1007
|
+
});
|
|
1008
|
+
continue;
|
|
1009
|
+
}
|
|
1010
|
+
|
|
1011
|
+
if (onToolCall) {
|
|
1012
|
+
onToolCall({ id: tc.id, name: tc.function.name, args: parsedArgs });
|
|
1013
|
+
}
|
|
1014
|
+
|
|
1015
|
+
logger.append({
|
|
1016
|
+
type: "tool_call",
|
|
1017
|
+
toolCallId: tc.id,
|
|
1018
|
+
name: tc.function.name,
|
|
1019
|
+
args: parsedArgs,
|
|
1020
|
+
});
|
|
1021
|
+
|
|
1022
|
+
const toolExecution = await ctx.executeTool(tc.function.name, parsedArgs);
|
|
1023
|
+
|
|
1024
|
+
let toolOutputString = toolExecution.isError
|
|
1025
|
+
? `Error: ${toolExecution.error}`
|
|
1026
|
+
: typeof toolExecution.result === "string"
|
|
1027
|
+
? toolExecution.result
|
|
1028
|
+
: JSON.stringify(toolExecution.result ?? "");
|
|
1029
|
+
|
|
1030
|
+
// E1: FS-as-context spillover. Large tool results are written in full
|
|
1031
|
+
// to <workspace>/.scratch/ and the in-band observation is replaced with
|
|
1032
|
+
// a pointer (head + tail preview + re-read hint) instead of being
|
|
1033
|
+
// hard-truncated. Suffix-scoped, so KV prefix stability is preserved.
|
|
1034
|
+
const spill = await spillToolOutput({
|
|
1035
|
+
output: toolOutputString,
|
|
1036
|
+
thresholdBytes: TOOL_SPILL_BYTES,
|
|
1037
|
+
id: tc.id,
|
|
1038
|
+
scratchDir: ".scratch",
|
|
1039
|
+
fsService: ctx.get("fs"),
|
|
1040
|
+
});
|
|
1041
|
+
toolOutputString = spill.output;
|
|
1042
|
+
if (spill.spilled) {
|
|
1043
|
+
logger.append({
|
|
1044
|
+
type: "tool_output_spilled",
|
|
1045
|
+
toolCallId: tc.id,
|
|
1046
|
+
toolName: tc.function.name,
|
|
1047
|
+
bytes: spill.bytes,
|
|
1048
|
+
path: spill.path,
|
|
1049
|
+
});
|
|
1050
|
+
}
|
|
1051
|
+
|
|
1052
|
+
if (onActivity) {
|
|
1053
|
+
const previewSnippet = `[${tc.function.name}] ${toolOutputString.trim().slice(-SUPERVISOR_PREVIEW_CHARS)}`;
|
|
1054
|
+
onActivity(previewSnippet);
|
|
1055
|
+
}
|
|
1056
|
+
|
|
1057
|
+
messages.push({
|
|
1058
|
+
role: "tool",
|
|
1059
|
+
tool_call_id: tc.id,
|
|
1060
|
+
content: toolOutputString,
|
|
1061
|
+
});
|
|
1062
|
+
|
|
1063
|
+
logger.append({
|
|
1064
|
+
type: "tool_result",
|
|
1065
|
+
toolCallId: tc.id,
|
|
1066
|
+
toolName: tc.function.name,
|
|
1067
|
+
result: toolExecution.result,
|
|
1068
|
+
error: toolExecution.error,
|
|
1069
|
+
isError: toolExecution.isError,
|
|
1070
|
+
latencyMs: toolExecution.latencyMs,
|
|
1071
|
+
});
|
|
1072
|
+
|
|
1073
|
+
try {
|
|
1074
|
+
recordToolExecution({
|
|
1075
|
+
toolName: tc.function.name,
|
|
1076
|
+
isError: !!toolExecution.isError,
|
|
1077
|
+
});
|
|
1078
|
+
} catch {}
|
|
1079
|
+
|
|
1080
|
+
// Track consecutive non-mutating command executions and inject advisory when threshold is exceeded.
|
|
1081
|
+
const toolName = tc.function.name;
|
|
1082
|
+
if (MUTATING_TOOLS.has(toolName)) {
|
|
1083
|
+
loopDetector.recordMutation();
|
|
1084
|
+
probeStreak = 0;
|
|
1085
|
+
} else if (BASH_TOOLS.has(toolName)) {
|
|
1086
|
+
probeStreak++;
|
|
1087
|
+
if (probeStreak > PROBE_BUDGET) {
|
|
1088
|
+
messages.push({
|
|
1089
|
+
role: "user",
|
|
1090
|
+
content: PROBE_BUDGET_ADVISORY,
|
|
1091
|
+
});
|
|
1092
|
+
logger.append({
|
|
1093
|
+
type: "probe_budget_warning",
|
|
1094
|
+
consecutiveNonMutatingBash: probeStreak,
|
|
1095
|
+
budget: PROBE_BUDGET,
|
|
1096
|
+
advisory: PROBE_BUDGET_ADVISORY,
|
|
1097
|
+
});
|
|
1098
|
+
// Re-arm: reset for the next run of N consecutive non-mutating
|
|
1099
|
+
// bash calls (the advisory is advisory-only; it does not cancel
|
|
1100
|
+
// or error the session).
|
|
1101
|
+
probeStreak = 0;
|
|
1102
|
+
}
|
|
1103
|
+
}
|
|
1104
|
+
|
|
1105
|
+
// Action-hash loop detection: fingerprint non-mutating repetitions
|
|
1106
|
+
const loopCheck = loopDetector.recordAction(tc.function.name, parsedArgs, toolExecution);
|
|
1107
|
+
if (loopCheck.isLoop) {
|
|
1108
|
+
logger.append({
|
|
1109
|
+
type: "action_loop_detected",
|
|
1110
|
+
fingerprint: loopCheck.fingerprint,
|
|
1111
|
+
repeats: loopCheck.repeats,
|
|
1112
|
+
threshold: loopDetector.threshold,
|
|
1113
|
+
toolName: tc.function.name,
|
|
1114
|
+
});
|
|
1115
|
+
messages.push({
|
|
1116
|
+
role: "user",
|
|
1117
|
+
content: `[Action Loop Detected]: You have executed identical action '${tc.function.name}' ${loopCheck.repeats} times consecutively with no state mutation. Alter your approach, inspect alternative files, or synthesize conclusions.`,
|
|
1118
|
+
});
|
|
1119
|
+
if (loopCheck.repeats >= loopDetector.threshold + 1) {
|
|
1120
|
+
status = "stagnant_action_loop";
|
|
1121
|
+
finalText = `StagnantActionLoopError: Execution terminated after repeated non-mutating action '${tc.function.name}' (${loopCheck.repeats} consecutive calls).`;
|
|
1122
|
+
break;
|
|
1123
|
+
}
|
|
1124
|
+
}
|
|
1125
|
+
}
|
|
1126
|
+
|
|
1127
|
+
if (status === "stagnant_action_loop") {
|
|
1128
|
+
break;
|
|
1129
|
+
}
|
|
1130
|
+
|
|
1131
|
+
// If the turn was cut off by the token ceiling AND it carried tool
|
|
1132
|
+
// calls, the model may have been mid-way through emitting them. Send a
|
|
1133
|
+
// continuation signal so it re-emits any dropped/incomplete calls.
|
|
1134
|
+
if (
|
|
1135
|
+
turnResult.finishReason === "length" &&
|
|
1136
|
+
droppedTruncatedCalls > 0 &&
|
|
1137
|
+
continuationsInjected < MAX_CONTINUATION_TURNS
|
|
1138
|
+
) {
|
|
1139
|
+
continuationsInjected++;
|
|
1140
|
+
const hadReasoning =
|
|
1141
|
+
turnResult.metrics?.hadReasoning ?? turnResult.hadReasoning ?? false;
|
|
1142
|
+
messages.push({ role: "user", content: CONTINUATION_DIRECTIVE });
|
|
1143
|
+
logger.append({
|
|
1144
|
+
type: "continuation_injected",
|
|
1145
|
+
content: CONTINUATION_DIRECTIVE,
|
|
1146
|
+
continuationNumber: continuationsInjected,
|
|
1147
|
+
maxContinuations: MAX_CONTINUATION_TURNS,
|
|
1148
|
+
reason: "length",
|
|
1149
|
+
directive: "resume",
|
|
1150
|
+
hadReasoning,
|
|
1151
|
+
droppedToolCalls: droppedTruncatedCalls,
|
|
1152
|
+
});
|
|
1153
|
+
}
|
|
1154
|
+
}
|
|
1155
|
+
} catch (err) {
|
|
1156
|
+
const isContextExhausted = /maximum context length|context length exceeded|context_exhausted/i.test(err.message || "");
|
|
1157
|
+
if (isContextExhausted) {
|
|
1158
|
+
status = "context_exhausted";
|
|
1159
|
+
finalText =
|
|
1160
|
+
`[Context Exhausted] The session's cumulative context exceeded the model's ${MAX_CONTEXT.toLocaleString("en-US")} token ceiling.\n` +
|
|
1161
|
+
`Prior session events and tool outputs remain intact in the local event ledger.\n` +
|
|
1162
|
+
`Action: Roll into a fresh session_id (e.g. "${sessionId}_stage2") for subsequent dispatches.`;
|
|
1163
|
+
logger.append({
|
|
1164
|
+
type: "session_error",
|
|
1165
|
+
error: "context_exhausted",
|
|
1166
|
+
detail: err.message,
|
|
1167
|
+
});
|
|
1168
|
+
} else {
|
|
1169
|
+
status = "failed";
|
|
1170
|
+
finalText = `Castor execution error: ${err.message}`;
|
|
1171
|
+
logger.append({ type: "session_error", error: err.message, stack: err.stack });
|
|
1172
|
+
}
|
|
1173
|
+
} finally {
|
|
1174
|
+
const durationMs = Date.now() - t0;
|
|
1175
|
+
logger.append({
|
|
1176
|
+
type: "session_end",
|
|
1177
|
+
status,
|
|
1178
|
+
turnsTaken,
|
|
1179
|
+
continuationsInjected,
|
|
1180
|
+
durationMs,
|
|
1181
|
+
totalCompletionTokens,
|
|
1182
|
+
});
|
|
1183
|
+
|
|
1184
|
+
try {
|
|
1185
|
+
sampleLiveVllmMetrics();
|
|
1186
|
+
} catch {}
|
|
1187
|
+
|
|
1188
|
+
// Terminate MCP extension bridge and clean up child processes and registered tools.
|
|
1189
|
+
if (mcpBridge) {
|
|
1190
|
+
try {
|
|
1191
|
+
mcpBridge.dispose();
|
|
1192
|
+
} catch {}
|
|
1193
|
+
}
|
|
1194
|
+
|
|
1195
|
+
// Cleanly dispose microkernel and unmount all plugins
|
|
1196
|
+
ctx.dispose();
|
|
1197
|
+
}
|
|
1198
|
+
|
|
1199
|
+
return {
|
|
1200
|
+
finalText,
|
|
1201
|
+
turnsTaken,
|
|
1202
|
+
// Cumulative turn count across session lifetime.
|
|
1203
|
+
sessionTurns,
|
|
1204
|
+
status,
|
|
1205
|
+
durationMs: Date.now() - t0,
|
|
1206
|
+
totalCompletionTokens,
|
|
1207
|
+
sessionId,
|
|
1208
|
+
// Prompt token count and remaining context headroom under nominal ceiling.
|
|
1209
|
+
lastPromptTokens: lastPromptTokens > 0 ? lastPromptTokens : null,
|
|
1210
|
+
contextHeadroom:
|
|
1211
|
+
lastPromptTokens > 0
|
|
1212
|
+
? Math.max(0, MAX_LEN_HUGE - lastPromptTokens)
|
|
1213
|
+
: null,
|
|
1214
|
+
};
|
|
1215
|
+
}
|
|
1216
|
+
}
|