mcp-castor 2026.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +487 -0
  2. package/bin/castor.js +706 -0
  3. package/index.js +206 -0
  4. package/package.json +97 -0
  5. package/skills/canary-test-staging/SKILL.md +24 -0
  6. package/skills/evo-mutation-rollback/SKILL.md +29 -0
  7. package/skills/hypothesis-generation/SKILL.md +26 -0
  8. package/skills/traceback-condensing/SKILL.md +26 -0
  9. package/src/castor_runner.js +469 -0
  10. package/src/config.js +1204 -0
  11. package/src/env.js +10 -0
  12. package/src/evo_engine.js +214 -0
  13. package/src/harness/core/events.js +75 -0
  14. package/src/harness/core/kernel.js +209 -0
  15. package/src/harness/evo/evaluator.js +156 -0
  16. package/src/harness/evo/evo_operator.js +550 -0
  17. package/src/harness/evo/lineage_dag.js +383 -0
  18. package/src/harness/evo/trace_repair.js +173 -0
  19. package/src/harness/evo/watchdog.js +72 -0
  20. package/src/harness/loop_detector.js +135 -0
  21. package/src/harness/runner.js +1216 -0
  22. package/src/harness/services/ast_service.js +1813 -0
  23. package/src/harness/services/event_logger.js +275 -0
  24. package/src/harness/services/mcp_bridge.js +408 -0
  25. package/src/harness/services/provider_vllm.js +728 -0
  26. package/src/harness/services/sandbox_fs.js +1238 -0
  27. package/src/harness/services/searxng_lifecycle.js +254 -0
  28. package/src/harness/services/shell_executor.js +264 -0
  29. package/src/harness/services/shell_validator.js +506 -0
  30. package/src/harness/services/web_service.js +828 -0
  31. package/src/platform.js +344 -0
  32. package/src/repetition_detector.js +139 -0
  33. package/src/semaphore.js +373 -0
  34. package/src/server_lifecycle.js +781 -0
  35. package/src/skills.js +400 -0
  36. package/src/state_pruner.js +392 -0
  37. package/src/task_registry.js +1357 -0
  38. package/src/telemetry.js +638 -0
  39. package/src/tools.js +997 -0
  40. package/src/wsl_bridge.js +629 -0
  41. package/src/wsl_env.js +171 -0
  42. package/stream_proxy.js +453 -0
@@ -0,0 +1,1216 @@
1
+ /**
2
+ * Castor Runtime Engine (Microkernel Orchestrator)
3
+ *
4
+ * Capabilities:
5
+ * - Pure Node.js runtime (no binary compilation or subprocess shell wrappers)
6
+ * - Castor Context lifecycle with reversible plugin mount/unmount
7
+ * - SSE streaming with real-time token dispatch to Antigravity
8
+ * - Sandboxed ripgrep filesystem & bash toolset
9
+ * - Evo closed-loop evolutionary operators
10
+ * - Append-only JSONL event ledger & session branching
11
+ */
12
+
13
+ import path from "node:path";
14
+ import { createRequire } from "node:module";
15
+ import { Context } from "./core/kernel.js";
16
+ import { sandboxFsPlugin } from "./services/sandbox_fs.js";
17
+ import { shellExecutorPlugin } from "./services/shell_executor.js";
18
+ import { eventLoggerPlugin } from "./services/event_logger.js";
19
+ import { vllmProviderPlugin } from "./services/provider_vllm.js";
20
+ import { evoPlugin } from "./evo/evo_operator.js";
21
+ import { astPlugin } from "./services/ast_service.js";
22
+ import { webPlugin } from "./services/web_service.js";
23
+ import { McpBridge } from "./services/mcp_bridge.js";
24
+ import { injectSkills, matchSkills } from "../skills.js";
25
+ import { normalizeWorkspacePath, canonicalizePath } from "../wsl_bridge.js";
26
+ import { LoopDetector } from "./loop_detector.js";
27
+ import {
28
+ MAX_CONTINUATION_TURNS,
29
+ EMPTY_STREAM_RETRIES,
30
+ EMPTY_STREAM_RETRIES_DEEP,
31
+ EMPTY_STREAM_RETRY_DEPTH_CHARS,
32
+ EMPTY_STREAM_RETRY_BACKOFF_BASE_MS,
33
+ EMPTY_STREAM_RETRY_BACKOFF_CAP_MS,
34
+ DEGENERATE_FINAL_SUBSTANTIVE_CHARS,
35
+ DEGENERATE_FINAL_MAX_TURNS,
36
+ getReasoningEffort,
37
+ PROBE_BUDGET,
38
+ SESSION_TURNS_WARN,
39
+ SESSION_TURNS_RECOMMEND,
40
+ CONTEXT_WARN_TOKENS,
41
+ CONTEXT_HIGH_WATERMARK_TOKENS,
42
+ CONTEXT_EMERGENCY_CEILING_TOKENS,
43
+ TOOL_SPILL_BYTES,
44
+ PROMPT_BUDGET_CHARS,
45
+ BASE_TURN_BUDGET,
46
+ MAX_ELASTIC_TURNS,
47
+ KV_CACHE_HEADROOM_CEILING,
48
+ SPEC_ACCEPTANCE_FLOOR,
49
+ LOOP_DETECTION_WINDOW,
50
+ LOOP_DETECTION_REPETITIONS,
51
+ SUPERVISOR_PREVIEW_CHARS,
52
+ SALVAGE_MAX_TOKENS,
53
+ MAX_LEN_HUGE,
54
+ READ_GOVERNOR_MAX_BYTES,
55
+ MODEL,
56
+ MAX_CONTEXT,
57
+ } from "../config.js";
58
+ import { GUARD_MARKER_PREFIX } from "../repetition_detector.js";
59
+ import { recordTurnTelemetry, recordToolExecution, sampleLiveVllmMetrics } from "../telemetry.js";
60
+
61
+ // Single source of truth for the harness version: read from package.json
62
+ // (same createRequire idiom as index.js and mcp_bridge.js).
63
+ const require = createRequire(import.meta.url);
64
+ const PKG_VERSION = require("../../package.json").version;
65
+
66
+ // M4: probe-budget watchdog (issue #11 recs 1+2; F4/F12/F14). On open-ended
67
+ // layout targets the model ran 30+ consecutive inline-python measurement bash
68
+ // calls (~90 min) instead of making the edit. The runner counts CONSECUTIVE
69
+ // non-mutating bash calls (bash/exec_command with no file-mutating tool call in
70
+ // between); when the streak exceeds PROBE_BUDGET it injects an ADVISORY (not an
71
+ // error, not a cancellation) and re-arms the counter. Style reference: the
72
+ // advisory-only EvoWatchdog circuit breaker (src/harness/evo/watchdog.js).
73
+ //
74
+ // MUTATING_TOOLS reset the streak (a file edit means the model is in mutation
75
+ // mode, not probe mode). BASH_TOOLS increment it. Every other tool
76
+ // (read_file, list_dir, search_code, ast_search, evo_evaluate_candidate,
77
+ // evo_status) is neutral — it neither increments nor resets the streak.
78
+ const MUTATING_TOOLS = new Set([
79
+ "write_file",
80
+ "edit_file",
81
+ "apply_patch",
82
+ "ast_replace",
83
+ "ast_replace_batch",
84
+ "evo_propose_candidate",
85
+ "evo_select_candidate",
86
+ "evo_revert_candidate",
87
+ ]);
88
+ const BASH_TOOLS = new Set(["bash"]);
89
+
90
+ // Exponential backoff before retry attempt.
91
+ // runner sleeps base * 2^(retryNumber-1) ms, capped at capMs. With the defaults
92
+ // (base 2000ms, cap 30000ms) this is exactly "2^retryNumber seconds capped at
93
+ // 30s": retry 1 waits 2s, retry 2 waits 4s, retry 3 waits 8s, retry 4 waits
94
+ // 16s, retry 5+ waits 30s (capped). The backoff gives a transient empty-stream
95
+ // cluster time to clear before the next (expensive, deep) re-prefill. The
96
+ // computed ms is returned so the caller can record it in the retry event.
97
+ function emptyStreamRetryBackoffMs(retryNumber) {
98
+ const exp = Math.max(0, retryNumber - 1);
99
+ const ms = EMPTY_STREAM_RETRY_BACKOFF_BASE_MS * 2 ** exp;
100
+ return Math.min(ms, EMPTY_STREAM_RETRY_BACKOFF_CAP_MS);
101
+ }
102
+
103
+ export const DEFAULT_SYSTEM_PROMPT = `You are the Autonomous Execution Coworker (${MODEL}) running in the Castor harness.
104
+ You pair with the Lead Architect (Gemini in Antigravity / GLM in Claude Code) as a senior peer engineer. The Lead Architect holds high-level architecture and task decomposition; you hold hands-on execution, empirical testing, and codebase navigation.
105
+
106
+ Operating Principles:
107
+ 1. Peer Partnership & Two-Way Discussion:
108
+ - You are an autonomous engineering peer, not a blind batch executor. Discussion and collaborative alignment from both sides is the foundational operating principle.
109
+ - Autonomous Slicing on Clear Tasks: When objectives and acceptance criteria are clearly defined, execute the complete slice (investigate, modify, verify) autonomously across your toolset without micro-confirmations.
110
+ - Collaborative Pause on Ambiguity or Impasse: Never treat a dispatch as "life or death" where you must silently force solutions at all costs. When an empirical test fails an acceptance gate, when requirements are ambiguous, or when multiple paths exist, DO NOT loop in solitary trial-and-error.
111
+ - State your verified findings concisely, present the concrete trade-offs or root causes, and provide your technical recommendation to the Lead Architect in plain text. Concluding your turn with a clear, grounded inquiry or status report IS successful fulfillment of the turn.
112
+ 2. Ground Truth in Code & Tests:
113
+ - Ground truth lives exclusively in active source code, test suites, and verifiable build artifacts. Never assume or hallucinate.
114
+ 3. Workspace Scratchpads for Audits, Exploration & Empirical Reproduction:
115
+ - You have full, unrestricted write and execution access to '<workspace>/.scratch/' (and repository-local helper scripts) at all times, including during exploration turns.
116
+ - When diagnosing issues, verifying edge cases, or conducting multi-item audits, write minimal reproduction scripts (e.g. '.scratch/repro.py', '.scratch/test_case.js') and dump structured data tables to '.scratch/'.
117
+ - Isolating and verifying a failure empirically with a clean script in '.scratch/' is always preferred over mentally simulating complex logic or running long inline bash one-liners.
118
+ 4. Mutation & Tool Discipline:
119
+ - When requirements and reproduction are verified and a dispatch requests a code change, modify production source files directly in ONE targeted pass with native editing tools ('edit_file' / 'apply_patch'). Do NOT run blind measurement probe loops against production files.
120
+ - Reserve 'bash' strictly for compilation, test execution, benchmarks, git operations, package managers, or running project runtimes/binaries.
121
+ - Pure Text-Only Engine: You run in text mode with Universal 245K context. Do NOT call image inspection tools on binary images (.png, .jpg). Multimodal inspection is handled exclusively by the Lead Architect.
122
+ 5. Deliverables: Provide concise, direct technical summaries of your actions and findings.`;
123
+
124
+ const EVO_SYSTEM_PROMPT_ADDENDUM = `
125
+ 6. When optimizing, refactoring, or evolving procedural skills, use the Evo tools:
126
+ - 'evo_propose_candidate' to snapshot files or skills before modifying.
127
+ - 'evo_evaluate_candidate' to test and compute fitness score (receives compact failure digests on error).
128
+ - 'evo_select_candidate' to accept improvements, or 'evo_revert_candidate' to rollback regressions.`;
129
+
130
+ /**
131
+ * User-role directive injected when the model's output is cut off by the
132
+ * token ceiling (finish_reason: "length"). Instructs the model to resume
133
+ * exactly where it stopped without repeating already-emitted content.
134
+ */
135
+ export const CONTINUATION_DIRECTIVE =
136
+ "Your previous output was cut off by the token ceiling. " +
137
+ "Resume exactly where you stopped. Do not repeat already-emitted content.";
138
+
139
+ /**
140
+ * Balanced, non-coercive directive injected when the model hits a reasoning ceiling
141
+ * (finish_reason: "length" with reasoningCeilingHit or empty content).
142
+ * Gives permission to conclude or report blockers to the supervisor without hallucinating actions.
143
+ */
144
+ export const REASONING_CONTINUATION_DIRECTIVE =
145
+ "Your deliberation was paused at the token ceiling. " +
146
+ "If you have reached a resolution, proceed with your tool call or response. " +
147
+ "If you are facing an ambiguous requirement or an impasse, state what you have determined so far and request guidance from the supervisor.";
148
+
149
+ /**
150
+ * Non-coercive salvage directive injected when the reasoning budget is exhausted.
151
+ * Requests that the model persist accumulated findings and incomplete items
152
+ * to a designated scratch file before session termination.
153
+ */
154
+ export const SALVAGE_DIRECTIVE =
155
+ "Your deliberation has reached the token ceiling and this session is about to conclude. " +
156
+ "Do not reason further. In one short message, record the findings, tables, and conclusions you have " +
157
+ "already accumulated to the file <SALVAGE_PATH>, and briefly state what remains incomplete. " +
158
+ "This is a best-effort salvage of your partial work — if you have nothing concrete to record, simply say so.";
159
+
160
+ /**
161
+ * Advisory message injected when consecutive non-mutating bash executions
162
+ * exceed the configured probe budget threshold.
163
+ */
164
+ export const PROBE_BUDGET_ADVISORY =
165
+ "[Probe-Budget Advisory] You have run several consecutive shell (bash) calls " +
166
+ "without making progress on your deliverable. Prefer native workspace tools (search_code, read_file, list_dir) over ad-hoc shell inspection. " +
167
+ "Mutation dispatches are single-pass: state a hypothesis, make the edit directly with a native file tool (write_file / edit_file / apply_patch), " +
168
+ "then run the stated verification command ONCE. For read-only or exploration tasks, synthesize your findings and emit your final response now.";
169
+
170
+ /**
171
+ * Advisory message injected when cumulative session turns reach the recommended
172
+ * rollover threshold, advising session consolidation.
173
+ */
174
+ export const SESSION_ROLLOVER_ADVISORY =
175
+ "[Session-Rollover Advisory] This session has crossed the recommended " +
176
+ "turn-count boundary for a single session. Complete the current task, then " +
177
+ "roll to a FRESH session on the next dispatch — a new session starts with a " +
178
+ "clean, low-cost context instead of re-prefilling this deep one.";
179
+
180
+ /**
181
+ * Spills oversized tool execution results to scratch storage.
182
+ *
183
+ * When a tool output exceeds thresholdBytes, persists the full payload to disk
184
+ * under the workspace scratch directory and replaces the in-band observation
185
+ * with preview metadata and a file pointer.
186
+ *
187
+ * @param {object} params
188
+ * @param {string} params.output Full tool output string.
189
+ * @param {number} params.thresholdBytes Output size threshold before triggering spill.
190
+ * @param {string} params.id Unique tool invocation identifier.
191
+ * @param {string} params.scratchDir Relative scratchpad directory path.
192
+ * @param {object} [params.fsService] Sandboxed filesystem service instance.
193
+ * @returns {Promise<{output: string, spilled: boolean, path?: string, bytes?: number}>}
194
+ */
195
+ export async function spillToolOutput({
196
+ output,
197
+ thresholdBytes,
198
+ id,
199
+ scratchDir,
200
+ fsService,
201
+ }) {
202
+ const bytes = Buffer.byteLength(output, "utf8");
203
+ if (!fsService || bytes <= thresholdBytes) {
204
+ return { output, spilled: false, bytes };
205
+ }
206
+
207
+ const safeId = String(id).replace(/[^a-zA-Z0-9_-]/g, "_").slice(0, 64) || "out";
208
+ const relPath = `${scratchDir}/tool_out_${safeId}.txt`;
209
+ const head = output.slice(0, 1024);
210
+ const tail = output.slice(-1024);
211
+
212
+ let writtenPath;
213
+ try {
214
+ const res = await fsService.writeFile({ path: relPath, content: output, overwrite: true });
215
+ writtenPath = res && res.path ? res.path : relPath;
216
+ } catch (err) {
217
+ // Fail-fast: a failed spill is surfaced, never silently truncated.
218
+ return {
219
+ output:
220
+ `[ToolOutputSpillError] The full ${bytes}-byte tool output could not be ` +
221
+ `saved to ${relPath} (${err.message}). The payload was NOT truncated in-band; ` +
222
+ `re-run the tool with a narrower query (head/tail/grep, start_line/end_line) ` +
223
+ `to retrieve a smaller result.`,
224
+ spilled: false,
225
+ bytes,
226
+ error: err.message,
227
+ };
228
+ }
229
+
230
+ const pointer =
231
+ `[Tool output spilled to disk: ${bytes} bytes > ${thresholdBytes}-byte threshold. ` +
232
+ `Full payload saved to: ${writtenPath}]\n\n` +
233
+ `--- Head preview (first 1024 bytes) ---\n${head}\n` +
234
+ `... [${bytes - 2048} bytes elided] ...\n` +
235
+ `--- Tail preview (last 1024 bytes) ---\n${tail}\n\n` +
236
+ `Hint: use read_file with start_line/end_line or search_code to inspect ` +
237
+ `specific regions of ${writtenPath}.`;
238
+
239
+ return { output: pointer, spilled: true, path: writtenPath, bytes };
240
+ }
241
+
242
+
243
+ export class CastorRunner {
244
+ constructor(options = {}) {
245
+ // Canonicalize working directory through OS symlink/junction layer.
246
+ this.defaultCwd = canonicalizePath(options.cwd ? normalizeWorkspacePath(options.cwd) : process.cwd());
247
+ this.defaultMaxTurns = options.maxTurns || BASE_TURN_BUDGET;
248
+ // Optional injection seams (used by offline tests to substitute a mock
249
+ // LLM / logger without touching the network or the real vLLM provider).
250
+ this._llm = options.llm || null;
251
+ this._logger = options.logger || null;
252
+ }
253
+
254
+ /**
255
+ * Executes a bounded extraction pass to salvage partial deliberation output.
256
+ * Invoked upon reasoning budget exhaustion to record intermediate findings
257
+ * to workspace scratch storage.
258
+ *
259
+ * @param {object} params
260
+ * @param {object} params.llm LLM provider instance.
261
+ * @param {object} params.logger Event logger instance.
262
+ * @param {object} [params.fsService] Sandboxed filesystem service.
263
+ * @param {string} params.sessionId Session identifier.
264
+ * @param {AbortSignal} [params.signal] Cancellation abort signal.
265
+ * @returns {Promise<{ salvaged: boolean, path?: string, error?: string }>}
266
+ */
267
+ async salvageReasoningBudget({ llm, logger, fsService, sessionId, signal }) {
268
+ const safeId = String(sessionId).replace(/[^a-zA-Z0-9_-]/g, "_").slice(0, 64) || "session";
269
+ const salvagePath = `.scratch/salvage_${safeId}.md`;
270
+ const directive = SALVAGE_DIRECTIVE.replace("<SALVAGE_PATH>", salvagePath);
271
+
272
+ let salvageContent = "";
273
+ try {
274
+ // CRITICAL: use a MINIMAL message set (system prompt + directive only),
275
+ // NOT the full conversation history. The session is about to terminate
276
+ // precisely because the context is deep (near the 245K ceiling); passing
277
+ // the full history would re-trigger ContextExhaustedError. The salvage
278
+ // is a fresh, short extraction turn.
279
+ const salvageMessages = [
280
+ { role: "system", content: "You are a helpful assistant." },
281
+ { role: "user", content: directive },
282
+ ];
283
+ const salvageResult = await llm.streamChat({
284
+ messages: salvageMessages,
285
+ tools: [],
286
+ reasoningEffort: "low",
287
+ maxTokens: SALVAGE_MAX_TOKENS,
288
+ signal,
289
+ sessionId,
290
+ });
291
+ salvageContent = (salvageResult?.content || "").trim();
292
+ } catch (err) {
293
+ logger.append({
294
+ type: "salvage_failed",
295
+ error: err.message,
296
+ path: salvagePath,
297
+ });
298
+ return { salvaged: false, error: err.message };
299
+ }
300
+
301
+ if (salvageContent === "") {
302
+ logger.append({
303
+ type: "salvage_empty",
304
+ path: salvagePath,
305
+ });
306
+ return { salvaged: false };
307
+ }
308
+
309
+ if (!fsService) {
310
+ logger.append({
311
+ type: "salvage_failed",
312
+ error: "No FS service available",
313
+ path: salvagePath,
314
+ });
315
+ return { salvaged: false, error: "No FS service available" };
316
+ }
317
+
318
+ try {
319
+ const res = await fsService.writeFile({
320
+ path: salvagePath,
321
+ content: salvageContent,
322
+ overwrite: true,
323
+ });
324
+ const writtenPath = res?.path || salvagePath;
325
+ logger.append({
326
+ type: "salvage_extracted",
327
+ path: writtenPath,
328
+ bytes: Buffer.byteLength(salvageContent, "utf8"),
329
+ });
330
+ return { salvaged: true, path: writtenPath };
331
+ } catch (err) {
332
+ logger.append({
333
+ type: "salvage_failed",
334
+ error: err.message,
335
+ path: salvagePath,
336
+ });
337
+ return { salvaged: false, error: err.message };
338
+ }
339
+ }
340
+
341
+ /**
342
+ * Runs an autonomous agent session using the Castor harness.
343
+ *
344
+ * @param {object} params
345
+ * @param {string} params.prompt The user / orchestrator objective
346
+ * @param {string} [params.cwd] Target workspace directory
347
+ * @param {string} [params.sessionId] Unique session ID
348
+ * @param {number} [params.maxTurns] Maximum allowed turns (null = unbounded)
349
+ * @param {string} [params.reasoningEffort] Task-local reasoning-effort tier
350
+ * (one of REASONING_EFFORT_TIERS). Threaded to the provider's streamChat
351
+ * for every turn of this session; when absent the provider falls back to
352
+ * the QWEN_REASONING_EFFORT env default (unchanged behavior).
353
+ * @param {AbortSignal} [params.signal] Cancellation signal
354
+ * @param {(token: string) => void} [params.onToken] Live token streaming callback
355
+ * @param {(metric: object) => void} [params.onMetrics] Performance metric callback
356
+ * @param {(toolCall: object) => void} [params.onToolCall] Live tool call notification
357
+ * @returns {Promise<{
358
+ * finalText: string,
359
+ * turnsTaken: number,
360
+ * status: 'completed' | 'completed_ceiling' | 'aborted' | 'turn_limit_reached'
361
+ * | 'failed' | 'engine_empty_response' | 'reasoning_budget_exhausted'
362
+ * | 'length_limit_reached' | 'degenerate_response_truncated',
363
+ * durationMs: number,
364
+ * totalCompletionTokens: number,
365
+ * sessionId: string,
366
+ * sessionTurns: number,
367
+ * lastPromptTokens: number | null,
368
+ * contextHeadroom: number | null
369
+ * }>}
370
+ */
371
+ async run({
372
+ prompt,
373
+ cwd,
374
+ sessionId = `evo_${Date.now()}`,
375
+ maxTurns = this.defaultMaxTurns,
376
+ reasoningEffort,
377
+ signal,
378
+ onToken,
379
+ onMetrics,
380
+ onToolCall,
381
+ onActivity,
382
+ getDynamicBudget,
383
+ extensions,
384
+ targetInWsl = false,
385
+ testCommand,
386
+ enableEvo = false,
387
+ }) {
388
+ const t0 = Date.now();
389
+ const loopDetector = new LoopDetector({
390
+ windowSize: LOOP_DETECTION_WINDOW,
391
+ threshold: LOOP_DETECTION_REPETITIONS,
392
+ });
393
+ // Canonicalize working directory through OS symlink/junction layer.
394
+ const effectiveCwd = cwd ? canonicalizePath(normalizeWorkspacePath(cwd)) : this.defaultCwd;
395
+
396
+ // Initialize the Castor microkernel context
397
+ const ctx = new Context(null, `session_${sessionId}`);
398
+
399
+ // Mount core services. The LLM provider and event logger can be injected
400
+ // via the constructor (this._llm / this._logger) for offline testing; when
401
+ // absent we mount the real vLLM provider and on-disk JSONL logger.
402
+ ctx.plugin(sandboxFsPlugin, { root: effectiveCwd });
403
+ ctx.plugin(shellExecutorPlugin, { cwd: effectiveCwd });
404
+ if (this._logger) {
405
+ ctx.provide("logger", this._logger);
406
+ } else {
407
+ ctx.plugin(eventLoggerPlugin, { sessionId });
408
+ }
409
+ if (this._llm) {
410
+ ctx.provide("llm", this._llm);
411
+ } else {
412
+ ctx.plugin(vllmProviderPlugin);
413
+ }
414
+ ctx.plugin(astPlugin, { root: effectiveCwd });
415
+ ctx.plugin(webPlugin);
416
+
417
+ // Conditional Evo mounting: only mount the Evo closed-loop optimization tools
418
+ // when a testCommand / evaluation benchmark or explicit evo flag is active.
419
+ // In standard exploration/editing turns, the toolset remains lean at exactly 8 tools.
420
+ const isEvoActive = Boolean(testCommand || enableEvo);
421
+ if (isEvoActive) {
422
+ ctx.plugin(evoPlugin, { workspaceRoot: effectiveCwd });
423
+ }
424
+
425
+ // Initialize stdio MCP extension bridge before runner execution.
426
+ let mcpBridge = null;
427
+ if (Array.isArray(extensions) && extensions.length > 0) {
428
+ mcpBridge = new McpBridge({
429
+ cwd: effectiveCwd,
430
+ targetInWsl,
431
+ extensions,
432
+ });
433
+ await mcpBridge.start(ctx);
434
+ }
435
+
436
+ const logger = ctx.get("logger");
437
+ const llm = ctx.get("llm");
438
+
439
+ const priorEvents = logger.readAll();
440
+ const hasPriorUserMessages = priorEvents.some((e) => e.type === "user_message");
441
+ const priorSkills = new Set();
442
+ for (const ev of priorEvents) {
443
+ if (ev.type === "skills_injected" && Array.isArray(ev.skills)) {
444
+ for (const s of ev.skills) priorSkills.add(s);
445
+ }
446
+ }
447
+
448
+ // Match and inject applicable workflow skills before entering message list.
449
+ let effectivePrompt = prompt;
450
+ let matchedSkillNames = [];
451
+ if (!hasPriorUserMessages || priorSkills.size === 0) {
452
+ if (!prompt.includes("--- Matching skills (auto-injected from skills/) ---")) {
453
+ matchedSkillNames = matchSkills({ prompt, cwd: effectiveCwd }).map(
454
+ (s) => s.name
455
+ );
456
+ effectivePrompt = injectSkills(prompt, effectiveCwd);
457
+ }
458
+ }
459
+
460
+ logger.append({
461
+ type: "session_start",
462
+ harness: "Castor",
463
+ version: PKG_VERSION,
464
+ cwd: effectiveCwd,
465
+ prompt,
466
+ // Effective reasoning-effort tier for this session (task-local param when
467
+ // provided, else the QWEN_REASONING_EFFORT env default). Surfaced for
468
+ // telemetry; the provider re-resolves the same value per request.
469
+ reasoningEffort: reasoningEffort || getReasoningEffort(),
470
+ });
471
+
472
+ if (matchedSkillNames.length > 0) {
473
+ logger.append({ type: "skills_injected", skills: matchedSkillNames });
474
+ }
475
+
476
+ const systemPrompt = isEvoActive
477
+ ? DEFAULT_SYSTEM_PROMPT + EVO_SYSTEM_PROMPT_ADDENDUM
478
+ : DEFAULT_SYSTEM_PROMPT;
479
+
480
+ const messages = [
481
+ { role: "system", content: systemPrompt },
482
+ ...logger.getConversationHistory(),
483
+ ];
484
+
485
+ // Add current user prompt (with any auto-injected skills block)
486
+ messages.push({ role: "user", content: effectivePrompt });
487
+ logger.append({ type: "user_message", content: effectivePrompt });
488
+
489
+ let turnsTaken = 0;
490
+ let finalText = "";
491
+ let status = "completed";
492
+ let totalCompletionTokens = 0;
493
+ let continuationsInjected = 0;
494
+ let consecutiveReasoningContinuations = 0;
495
+ let emptyStreamRetries = 0;
496
+ // Track consecutive non-mutating command invocations.
497
+ let probeStreak = 0;
498
+
499
+ // Track session-cumulative turn counts across tasks.
500
+ let sessionTurns = logger
501
+ .readAll()
502
+ .filter((e) => e.type === "assistant_message").length;
503
+ let sessionWarnLatched = false;
504
+ let sessionRecommendLatched = false;
505
+ let contextDepthLatched = false;
506
+ let contextHighWatermarkLatched = false;
507
+ let contextEmergencyCeilingLatched = false;
508
+ let contextEmergencySynthesisEmitted = false;
509
+ let lastPromptTokens = 0;
510
+
511
+ // Emit prompt_over_budget advisory event when prompt exceeds PROMPT_BUDGET_CHARS.
512
+ if (prompt.length > PROMPT_BUDGET_CHARS) {
513
+ logger.append({
514
+ type: "prompt_over_budget",
515
+ promptChars: prompt.length,
516
+ budget: PROMPT_BUDGET_CHARS,
517
+ });
518
+ }
519
+
520
+ try {
521
+ while (true) {
522
+ if (signal?.aborted) {
523
+ status = "aborted";
524
+ break;
525
+ }
526
+
527
+ const currentMaxTurns = typeof getDynamicBudget === "function" ? getDynamicBudget() : (maxTurns || BASE_TURN_BUDGET);
528
+
529
+ if (contextEmergencySynthesisEmitted && continuationsInjected === 0) {
530
+ status = "completed_budget_exhausted";
531
+ break;
532
+ }
533
+
534
+ if (currentMaxTurns && turnsTaken >= currentMaxTurns && continuationsInjected === 0) {
535
+ status = finalText.trim() !== "" ? "completed_budget_exhausted" : "turn_limit_reached";
536
+ break;
537
+ }
538
+
539
+ // Cooperative landing: on the final turn before currentMaxTurns OR when context emergency ceiling is latched,
540
+ // strip tools and mandate synthesis.
541
+ const isCeilingTurn = Boolean(
542
+ contextEmergencyCeilingLatched ||
543
+ (currentMaxTurns && currentMaxTurns > 1 && turnsTaken === currentMaxTurns - 1) ||
544
+ (currentMaxTurns && turnsTaken >= currentMaxTurns && continuationsInjected > 0)
545
+ );
546
+ if (isCeilingTurn && continuationsInjected === 0) {
547
+ if (contextEmergencyCeilingLatched) {
548
+ contextEmergencySynthesisEmitted = true;
549
+ }
550
+ const synthesisPrompt = contextEmergencyCeilingLatched
551
+ ? `[Emergency Context Landing (${lastPromptTokens || "215,000+"}/${MAX_CONTEXT.toLocaleString("en-US")} tokens)]: Context space is near capacity. Tools are now disabled to prevent an unhandled engine crash. Synthesize your final deliverable, findings, code changes, and grounded conclusions immediately.`
552
+ : `[Dispatch Budget Notice (${turnsTaken + 1}/${currentMaxTurns})]: You have reached the final turn of your allotted budget for this dispatch. Synthesize your final deliverable, findings, code changes, and grounded conclusions immediately based on the facts gathered so far.`;
553
+
554
+ messages.push({
555
+ role: "user",
556
+ content: synthesisPrompt,
557
+ });
558
+ logger.append({
559
+ type: contextEmergencyCeilingLatched ? "context_emergency_synthesis" : "turn_ceiling_synthesis",
560
+ turnsTaken: turnsTaken + 1,
561
+ maxTurns: currentMaxTurns,
562
+ });
563
+ }
564
+
565
+ turnsTaken++;
566
+
567
+ // Emit session warning telemetry when cumulative turn thresholds are reached.
568
+ sessionTurns = sessionTurns + 1;
569
+ if (!sessionWarnLatched && sessionTurns >= SESSION_TURNS_WARN) {
570
+ sessionWarnLatched = true;
571
+ logger.append({
572
+ type: "session_warning",
573
+ sessionTurns,
574
+ threshold: SESSION_TURNS_WARN,
575
+ });
576
+ }
577
+ if (!sessionRecommendLatched && sessionTurns >= SESSION_TURNS_RECOMMEND) {
578
+ sessionRecommendLatched = true;
579
+ logger.append({
580
+ type: "session_turn_limit_recommended",
581
+ sessionTurns,
582
+ });
583
+ // ONE in-band user-role advisory: complete this task, then roll to a
584
+ // fresh session on the next dispatch.
585
+ messages.push({
586
+ role: "user",
587
+ content: SESSION_ROLLOVER_ADVISORY,
588
+ });
589
+ }
590
+
591
+ const tools = isCeilingTurn ? [] : ctx.listTools();
592
+
593
+ const turnResult = await llm.streamChat({
594
+ messages,
595
+ tools,
596
+ // Task-local reasoning-effort override (per-dispatch). Threaded to
597
+ // every turn of this session; the provider falls back to the
598
+ // QWEN_REASONING_EFFORT env default when it is absent.
599
+ reasoningEffort,
600
+ signal,
601
+ sessionId,
602
+ onToken: (tok) => {
603
+ if (onToken) onToken(tok);
604
+ },
605
+ onMetrics: (m) => {
606
+ totalCompletionTokens += m.completionTokens;
607
+ if (onMetrics) onMetrics(m);
608
+ try {
609
+ recordTurnTelemetry({
610
+ completionTokens: m.completionTokens || 0,
611
+ promptTokens: m.promptTokens || 0,
612
+ reasoningTokens: m.reasoningTokens || 0,
613
+ ttftMs: m.ttftMs,
614
+ prefillMs: m.prefillMs ?? m.ttftMs,
615
+ generationMs: m.generationMs ?? (m.totalMs && m.ttftMs ? Math.max(0, m.totalMs - m.ttftMs) : 0),
616
+ totalMs: m.totalMs,
617
+ prefillTps: m.prefillTps,
618
+ decodeTps: m.decodeTps,
619
+ tpotMs: m.tpotMs,
620
+ effort: reasoningEffort || "medium",
621
+ });
622
+ } catch {}
623
+ },
624
+ });
625
+
626
+ // Hoist re-prefill size and depth-aware empty-stream retry budget.
627
+ const promptChars = JSON.stringify(messages).length;
628
+ const emptyStreamRetryBudget =
629
+ promptChars >= EMPTY_STREAM_RETRY_DEPTH_CHARS
630
+ ? EMPTY_STREAM_RETRIES_DEEP
631
+ : EMPTY_STREAM_RETRIES;
632
+
633
+ // Guard against empty generation streams with missing finish reason.
634
+ const isEmptyGeneration =
635
+ (!turnResult.content || turnResult.content.trim() === "") &&
636
+ (!turnResult.toolCalls || turnResult.toolCalls.length === 0) &&
637
+ (turnResult.finishReason === undefined ||
638
+ turnResult.finishReason === null ||
639
+ turnResult.finishReason === "");
640
+
641
+ // Guard against empty stop turns (zero content and zero tool calls with stop finish reason).
642
+ const isEmptyStop =
643
+ turnResult.finishReason === "stop" &&
644
+ (!turnResult.content || turnResult.content.trim() === "") &&
645
+ (!turnResult.toolCalls || turnResult.toolCalls.length === 0);
646
+
647
+ if (isEmptyGeneration || isEmptyStop) {
648
+ // Record task execution coordinates and metrics at point of stream termination.
649
+ const deathContext = {
650
+ turnIndex: turnsTaken,
651
+ promptChars,
652
+ metrics: turnResult.metrics ?? null,
653
+ reasoningTokens: turnResult.reasoningTokens ?? 0,
654
+ };
655
+ if (emptyStreamRetries < emptyStreamRetryBudget) {
656
+ emptyStreamRetries++;
657
+ // Exponential backoff before retry attempt.
658
+ const backoffMs = emptyStreamRetryBackoffMs(emptyStreamRetries);
659
+ logger.append({
660
+ type: "empty_stream_retry",
661
+ retryNumber: emptyStreamRetries,
662
+ maxRetries: emptyStreamRetryBudget,
663
+ // Reason: empty generation stream.
664
+ reason: isEmptyStop ? "empty_stop" : "empty_generation",
665
+ backoffMs,
666
+ ...deathContext,
667
+ });
668
+ // If the model completed deliberation inside thinking tags but omitted visible content or tool calls,
669
+ // prompt it directly to emit its conclusion instead of repeating the identical prompt.
670
+ if (isEmptyStop && (turnResult.hadReasoning || (turnResult.reasoning && turnResult.reasoning.trim()))) {
671
+ messages.push({
672
+ role: "user",
673
+ content: "You concluded your internal deliberation without emitting a response or tool call. Please output your conclusion or next action directly now.",
674
+ });
675
+ }
676
+ await new Promise((r) => setTimeout(r, backoffMs));
677
+ continue;
678
+ }
679
+ // Budget exhausted: the engine keeps returning empty generations.
680
+ // Report the honest status instead of a false "completed".
681
+ status = "engine_empty_response";
682
+ logger.append({ type: "engine_empty_response", ...deathContext });
683
+ break;
684
+ }
685
+
686
+ // Degenerate-final guard: catch sentinel-truncated repetition outputs lacking substantive content.
687
+ const guardMarkerIdx =
688
+ typeof turnResult.content === "string"
689
+ ? turnResult.content.indexOf(GUARD_MARKER_PREFIX)
690
+ : -1;
691
+ if (guardMarkerIdx !== -1) {
692
+ // Strip the marker (prefix ... closing bracket) and measure the
693
+ // substantive remainder (the real text before/after the marker).
694
+ const closeIdx = turnResult.content.indexOf("]", guardMarkerIdx);
695
+ const substantive =
696
+ closeIdx === -1
697
+ ? turnResult.content.slice(0, guardMarkerIdx)
698
+ : turnResult.content.slice(0, guardMarkerIdx) +
699
+ turnResult.content.slice(closeIdx + 1);
700
+ const substantiveLen = substantive.trim().length;
701
+ const noToolCalls =
702
+ !turnResult.toolCalls || turnResult.toolCalls.length === 0;
703
+ const shortSession = turnsTaken <= DEGENERATE_FINAL_MAX_TURNS;
704
+ if (
705
+ substantiveLen < DEGENERATE_FINAL_SUBSTANTIVE_CHARS &&
706
+ noToolCalls &&
707
+ shortSession
708
+ ) {
709
+ const deathContext = {
710
+ turnIndex: turnsTaken,
711
+ // Context size hoisted above retry branches.
712
+ promptChars,
713
+ metrics: turnResult.metrics ?? null,
714
+ reasoningTokens: turnResult.reasoningTokens ?? 0,
715
+ substantiveChars: substantiveLen,
716
+ };
717
+ if (emptyStreamRetries < emptyStreamRetryBudget) {
718
+ emptyStreamRetries++;
719
+ // Exponential backoff before retry attempt.
720
+ const backoffMs = emptyStreamRetryBackoffMs(emptyStreamRetries);
721
+ logger.append({
722
+ type: "empty_stream_retry",
723
+ retryNumber: emptyStreamRetries,
724
+ maxRetries: emptyStreamRetryBudget,
725
+ // Reason: degenerate sentinel-truncated final.
726
+ reason: "degenerate_final",
727
+ backoffMs,
728
+ ...deathContext,
729
+ });
730
+ await new Promise((r) => setTimeout(r, backoffMs));
731
+ continue;
732
+ }
733
+ // Budget exhausted: the engine keeps returning degenerate
734
+ // guard-truncated finals. Report the honest status instead of a
735
+ // false "completed". Preserve the original partial+marker in
736
+ // finalText for honesty (the client sees exactly what the engine
737
+ // produced, including the guard marker).
738
+ status = "degenerate_response_truncated";
739
+ finalText = turnResult.content;
740
+ logger.append({
741
+ type: "degenerate_response_truncated",
742
+ ...deathContext,
743
+ });
744
+ break;
745
+ }
746
+ }
747
+
748
+ // Emit context-depth advisory telemetry when prompt token accumulation reaches warning threshold.
749
+ if (
750
+ !contextDepthLatched &&
751
+ typeof turnResult.metrics?.promptTokens === "number" &&
752
+ turnResult.metrics.promptTokens >= CONTEXT_WARN_TOKENS
753
+ ) {
754
+ contextDepthLatched = true;
755
+ logger.append({
756
+ type: "context_depth_warning",
757
+ promptTokens: turnResult.metrics.promptTokens,
758
+ threshold: CONTEXT_WARN_TOKENS,
759
+ ...(probeStreak > 0 ? { probeStreakActive: true } : {}),
760
+ });
761
+ }
762
+
763
+ if (typeof turnResult.metrics?.promptTokens === "number") {
764
+ lastPromptTokens = turnResult.metrics.promptTokens;
765
+ }
766
+
767
+ // --- Context High-Watermark Advisory (180,000 tokens) ---------------
768
+ // When promptTokens reaches the 180k high-watermark (~73% of nominal 245K
769
+ // context ceiling), emit a one-shot advisory event and inject an in-band
770
+ // rollover advisory to guide the model to conclude its deliverable rather
771
+ // than crashing with unhandled ContextExhaustedError / 400 Bad Request.
772
+ if (
773
+ !contextHighWatermarkLatched &&
774
+ typeof turnResult.metrics?.promptTokens === "number" &&
775
+ turnResult.metrics.promptTokens >= CONTEXT_HIGH_WATERMARK_TOKENS
776
+ ) {
777
+ contextHighWatermarkLatched = true;
778
+ logger.append({
779
+ type: "context_high_watermark",
780
+ promptTokens: turnResult.metrics.promptTokens,
781
+ threshold: CONTEXT_HIGH_WATERMARK_TOKENS,
782
+ });
783
+ // E4: arm the adaptive read-size governor. The context is under
784
+ // pressure (near the 180k high-watermark), so a whole-file read
785
+ // (64KB default) could blow the 245K ceiling (F6.1). Lower the
786
+ // session's read cap to 16KB so subsequent read_file calls are
787
+ // bounded. The governor only LOWERS the cap and is per-session
788
+ // (the SandboxFsService is a fresh per-run instance), so it cannot
789
+ // leak across tasks. Best-effort: a missing FS service is a no-op.
790
+ const fsService = ctx.get("fs");
791
+ if (fsService && typeof fsService.setReadGovernor === "function") {
792
+ fsService.setReadGovernor(READ_GOVERNOR_MAX_BYTES);
793
+ logger.append({
794
+ type: "read_governor_armed",
795
+ maxBytes: READ_GOVERNOR_MAX_BYTES,
796
+ promptTokens: turnResult.metrics.promptTokens,
797
+ });
798
+ }
799
+ messages.push({
800
+ role: "user",
801
+ content:
802
+ `[Context High-Watermark Advisory] Prompt context has reached ${turnResult.metrics.promptTokens} tokens ` +
803
+ `(high-watermark: ${CONTEXT_HIGH_WATERMARK_TOKENS}, max ceiling: ${MAX_CONTEXT.toLocaleString("en-US")}). ` +
804
+ `Wrap up your deliverable and return your final response now. ` +
805
+ `Advise the user/orchestrator to roll into a fresh session_id for subsequent dispatches to prevent context exhaustion.`,
806
+ });
807
+ }
808
+
809
+ // --- Context Emergency Ceiling Latch (215,000 tokens) ----------------
810
+ // When promptTokens approaches the 245K ceiling (~87%), strip tools on the
811
+ // subsequent turn to trigger emergency synthesis and prevent an unhandled
812
+ // context_exhausted crash.
813
+ if (
814
+ !contextEmergencyCeilingLatched &&
815
+ typeof turnResult.metrics?.promptTokens === "number" &&
816
+ turnResult.metrics.promptTokens >= CONTEXT_EMERGENCY_CEILING_TOKENS
817
+ ) {
818
+ contextEmergencyCeilingLatched = true;
819
+ logger.append({
820
+ type: "context_emergency_ceiling",
821
+ promptTokens: turnResult.metrics.promptTokens,
822
+ threshold: CONTEXT_EMERGENCY_CEILING_TOKENS,
823
+ });
824
+ }
825
+
826
+ // Record assistant response. reasoningTokens is surfaced as a top-level
827
+ // field (in addition to metrics) so ledgers show thinking volume even
828
+ // when the metrics object is summarized or dropped downstream.
829
+ logger.append({
830
+ type: "assistant_message",
831
+ content: turnResult.content,
832
+ toolCalls: turnResult.toolCalls,
833
+ finishReason: turnResult.finishReason,
834
+ reasoningTokens:
835
+ turnResult.metrics?.reasoningTokens ?? turnResult.reasoningTokens ?? 0,
836
+ metrics: turnResult.metrics,
837
+ });
838
+
839
+ messages.push({
840
+ role: "assistant",
841
+ content: turnResult.content || null,
842
+ tool_calls: turnResult.toolCalls.length > 0 ? turnResult.toolCalls : undefined,
843
+ });
844
+
845
+ if (
846
+ (turnResult.content && turnResult.content.trim() !== "") ||
847
+ (turnResult.toolCalls && turnResult.toolCalls.length > 0)
848
+ ) {
849
+ consecutiveReasoningContinuations = 0;
850
+ }
851
+
852
+ if (turnResult.content && turnResult.content.trim() !== "") {
853
+ finalText = turnResult.content;
854
+ if (onActivity) {
855
+ onActivity(turnResult.content.trim().slice(-SUPERVISOR_PREVIEW_CHARS));
856
+ }
857
+ }
858
+
859
+ // If no tool calls, the model concluded its turn - UNLESS the output
860
+ // was cut off by the token ceiling (finish_reason: "length"). In that
861
+ // case the answer is truncated, so we re-prompt the model to resume.
862
+ if (!turnResult.toolCalls || turnResult.toolCalls.length === 0) {
863
+ if (turnResult.finishReason === "length") {
864
+ const hadReasoning =
865
+ turnResult.metrics?.hadReasoning ?? turnResult.hadReasoning ?? false;
866
+ const isReasoningCutoff =
867
+ Boolean(turnResult.metrics?.reasoningCeilingHit) ||
868
+ (hadReasoning && (!turnResult.content || turnResult.content.trim() === ""));
869
+
870
+ if (isReasoningCutoff) {
871
+ consecutiveReasoningContinuations++;
872
+ if (consecutiveReasoningContinuations > 1) {
873
+ status = "reasoning_budget_exhausted";
874
+ finalText = "ReasoningBudgetExhaustedError: The model reached the deliberation ceiling across consecutive continuation turns without taking action or concluding.";
875
+ // E2: bounded salvage extraction pass (F6.3). Best-effort;
876
+ // never alters the honest status above.
877
+ const salvage = await this.salvageReasoningBudget({
878
+ llm,
879
+ logger,
880
+ fsService: ctx.get("fs"),
881
+ sessionId,
882
+ signal,
883
+ });
884
+ if (salvage.salvaged) {
885
+ finalText += `\n\n[Salvage] Partial findings saved to: ${salvage.path}`;
886
+ }
887
+ break;
888
+ }
889
+ continuationsInjected++;
890
+ messages.push({ role: "user", content: REASONING_CONTINUATION_DIRECTIVE });
891
+ logger.append({
892
+ type: "continuation_injected",
893
+ content: REASONING_CONTINUATION_DIRECTIVE,
894
+ continuationNumber: continuationsInjected,
895
+ maxContinuations: MAX_CONTINUATION_TURNS,
896
+ reason: "reasoning_ceiling",
897
+ directive: "balanced_landing",
898
+ hadReasoning: true,
899
+ });
900
+ continue;
901
+ }
902
+
903
+ consecutiveReasoningContinuations = 0;
904
+ if (continuationsInjected < MAX_CONTINUATION_TURNS) {
905
+ continuationsInjected++;
906
+ // Provide clean continuation without artificial stop-thinking directives
907
+ messages.push({ role: "user", content: CONTINUATION_DIRECTIVE });
908
+ logger.append({
909
+ type: "continuation_injected",
910
+ content: CONTINUATION_DIRECTIVE,
911
+ continuationNumber: continuationsInjected,
912
+ maxContinuations: MAX_CONTINUATION_TURNS,
913
+ reason: "length",
914
+ directive: "resume",
915
+ hadReasoning,
916
+ });
917
+ continue;
918
+ }
919
+ // Continuation budget exhausted: report an honest status instead of a
920
+ // false "completed". If the model produced a complete deliverable
921
+ // (non-empty finalText) despite hitting the ceiling, that is
922
+ // "completed_ceiling" — a successful run that merely ran out of room,
923
+ // NOT a failure. Only when nothing usable was produced do we fall
924
+ // through to the honest failure statuses.
925
+ if (isCeilingTurn) {
926
+ status = finalText.trim() !== "" ? "completed_budget_exhausted" : "turn_limit_reached";
927
+ } else if (finalText.trim() !== "") {
928
+ status = "completed_ceiling";
929
+ } else if (hadReasoning) {
930
+ status = "reasoning_budget_exhausted";
931
+ finalText = "ReasoningBudgetExhaustedError: The model exhausted the continuation reasoning budget without emitting visible actions or content.";
932
+ // E2: bounded salvage extraction pass (F6.3). Best-effort;
933
+ // never alters the honest status above.
934
+ const salvage = await this.salvageReasoningBudget({
935
+ llm,
936
+ logger,
937
+ fsService: ctx.get("fs"),
938
+ sessionId,
939
+ signal,
940
+ });
941
+ if (salvage.salvaged) {
942
+ finalText += `\n\n[Salvage] Partial findings saved to: ${salvage.path}`;
943
+ }
944
+ } else {
945
+ status = "length_limit_reached";
946
+ }
947
+ break;
948
+ }
949
+ if (isCeilingTurn) {
950
+ status = "completed_budget_exhausted";
951
+ break;
952
+ }
953
+ // The model concluded its turn with a clean "stop" (or other non-length
954
+ // finish reason). If it had to hit the token ceiling the maximum number
955
+ // of times (continuation budget at the cap) before finally completing,
956
+ // that is "completed_ceiling" — a complete deliverable that only finished
957
+ // after exhausting the continuation budget. A clean "stop" that never
958
+ // exhausted the continuation budget is a plain "completed".
959
+ if (
960
+ continuationsInjected >= MAX_CONTINUATION_TURNS &&
961
+ finalText.trim() !== ""
962
+ ) {
963
+ status = "completed_ceiling";
964
+ }
965
+ break;
966
+ }
967
+
968
+ // Execute each requested tool call
969
+ consecutiveReasoningContinuations = 0;
970
+ let droppedTruncatedCalls = 0;
971
+ for (const tc of turnResult.toolCalls) {
972
+ if (signal?.aborted) break;
973
+
974
+ let parsedArgs;
975
+ let argsParseFailed = false;
976
+ try {
977
+ parsedArgs = JSON.parse(tc.function.arguments || "{}");
978
+ } catch {
979
+ // The tool-call arguments were cut off mid-stream (typically by a
980
+ // token-ceiling "length" cutoff). NEVER execute a mangled payload:
981
+ // drop the call, tell the model, and let it re-emit it completely.
982
+ argsParseFailed = true;
983
+ droppedTruncatedCalls++;
984
+ }
985
+
986
+ if (argsParseFailed) {
987
+ // Sanitize the malformed argument string in the assistant message so downstream
988
+ // API parsers (vLLM's qwen3_coder / python json.loads) do not fail with HTTP 400 JSONDecodeError
989
+ tc.function.arguments = "{}";
990
+
991
+ const notice =
992
+ `ToolExecutionError: Tool '${tc.function.name}' (id ${tc.id}) was dropped: its arguments ` +
993
+ `were truncated mid-stream and could not be parsed as JSON (finish_reason: "length"). ` +
994
+ `Please re-emit this tool call with complete, valid JSON arguments.`;
995
+ messages.push({
996
+ role: "tool",
997
+ tool_call_id: tc.id,
998
+ content: notice,
999
+ });
1000
+ logger.append({
1001
+ type: "tool_call_dropped",
1002
+ toolCallId: tc.id,
1003
+ name: tc.function.name,
1004
+ notice,
1005
+ reason: "truncated_arguments",
1006
+ finishReason: turnResult.finishReason,
1007
+ });
1008
+ continue;
1009
+ }
1010
+
1011
+ if (onToolCall) {
1012
+ onToolCall({ id: tc.id, name: tc.function.name, args: parsedArgs });
1013
+ }
1014
+
1015
+ logger.append({
1016
+ type: "tool_call",
1017
+ toolCallId: tc.id,
1018
+ name: tc.function.name,
1019
+ args: parsedArgs,
1020
+ });
1021
+
1022
+ const toolExecution = await ctx.executeTool(tc.function.name, parsedArgs);
1023
+
1024
+ let toolOutputString = toolExecution.isError
1025
+ ? `Error: ${toolExecution.error}`
1026
+ : typeof toolExecution.result === "string"
1027
+ ? toolExecution.result
1028
+ : JSON.stringify(toolExecution.result ?? "");
1029
+
1030
+ // E1: FS-as-context spillover. Large tool results are written in full
1031
+ // to <workspace>/.scratch/ and the in-band observation is replaced with
1032
+ // a pointer (head + tail preview + re-read hint) instead of being
1033
+ // hard-truncated. Suffix-scoped, so KV prefix stability is preserved.
1034
+ const spill = await spillToolOutput({
1035
+ output: toolOutputString,
1036
+ thresholdBytes: TOOL_SPILL_BYTES,
1037
+ id: tc.id,
1038
+ scratchDir: ".scratch",
1039
+ fsService: ctx.get("fs"),
1040
+ });
1041
+ toolOutputString = spill.output;
1042
+ if (spill.spilled) {
1043
+ logger.append({
1044
+ type: "tool_output_spilled",
1045
+ toolCallId: tc.id,
1046
+ toolName: tc.function.name,
1047
+ bytes: spill.bytes,
1048
+ path: spill.path,
1049
+ });
1050
+ }
1051
+
1052
+ if (onActivity) {
1053
+ const previewSnippet = `[${tc.function.name}] ${toolOutputString.trim().slice(-SUPERVISOR_PREVIEW_CHARS)}`;
1054
+ onActivity(previewSnippet);
1055
+ }
1056
+
1057
+ messages.push({
1058
+ role: "tool",
1059
+ tool_call_id: tc.id,
1060
+ content: toolOutputString,
1061
+ });
1062
+
1063
+ logger.append({
1064
+ type: "tool_result",
1065
+ toolCallId: tc.id,
1066
+ toolName: tc.function.name,
1067
+ result: toolExecution.result,
1068
+ error: toolExecution.error,
1069
+ isError: toolExecution.isError,
1070
+ latencyMs: toolExecution.latencyMs,
1071
+ });
1072
+
1073
+ try {
1074
+ recordToolExecution({
1075
+ toolName: tc.function.name,
1076
+ isError: !!toolExecution.isError,
1077
+ });
1078
+ } catch {}
1079
+
1080
+ // Track consecutive non-mutating command executions and inject advisory when threshold is exceeded.
1081
+ const toolName = tc.function.name;
1082
+ if (MUTATING_TOOLS.has(toolName)) {
1083
+ loopDetector.recordMutation();
1084
+ probeStreak = 0;
1085
+ } else if (BASH_TOOLS.has(toolName)) {
1086
+ probeStreak++;
1087
+ if (probeStreak > PROBE_BUDGET) {
1088
+ messages.push({
1089
+ role: "user",
1090
+ content: PROBE_BUDGET_ADVISORY,
1091
+ });
1092
+ logger.append({
1093
+ type: "probe_budget_warning",
1094
+ consecutiveNonMutatingBash: probeStreak,
1095
+ budget: PROBE_BUDGET,
1096
+ advisory: PROBE_BUDGET_ADVISORY,
1097
+ });
1098
+ // Re-arm: reset for the next run of N consecutive non-mutating
1099
+ // bash calls (the advisory is advisory-only; it does not cancel
1100
+ // or error the session).
1101
+ probeStreak = 0;
1102
+ }
1103
+ }
1104
+
1105
+ // Action-hash loop detection: fingerprint non-mutating repetitions
1106
+ const loopCheck = loopDetector.recordAction(tc.function.name, parsedArgs, toolExecution);
1107
+ if (loopCheck.isLoop) {
1108
+ logger.append({
1109
+ type: "action_loop_detected",
1110
+ fingerprint: loopCheck.fingerprint,
1111
+ repeats: loopCheck.repeats,
1112
+ threshold: loopDetector.threshold,
1113
+ toolName: tc.function.name,
1114
+ });
1115
+ messages.push({
1116
+ role: "user",
1117
+ content: `[Action Loop Detected]: You have executed identical action '${tc.function.name}' ${loopCheck.repeats} times consecutively with no state mutation. Alter your approach, inspect alternative files, or synthesize conclusions.`,
1118
+ });
1119
+ if (loopCheck.repeats >= loopDetector.threshold + 1) {
1120
+ status = "stagnant_action_loop";
1121
+ finalText = `StagnantActionLoopError: Execution terminated after repeated non-mutating action '${tc.function.name}' (${loopCheck.repeats} consecutive calls).`;
1122
+ break;
1123
+ }
1124
+ }
1125
+ }
1126
+
1127
+ if (status === "stagnant_action_loop") {
1128
+ break;
1129
+ }
1130
+
1131
+ // If the turn was cut off by the token ceiling AND it carried tool
1132
+ // calls, the model may have been mid-way through emitting them. Send a
1133
+ // continuation signal so it re-emits any dropped/incomplete calls.
1134
+ if (
1135
+ turnResult.finishReason === "length" &&
1136
+ droppedTruncatedCalls > 0 &&
1137
+ continuationsInjected < MAX_CONTINUATION_TURNS
1138
+ ) {
1139
+ continuationsInjected++;
1140
+ const hadReasoning =
1141
+ turnResult.metrics?.hadReasoning ?? turnResult.hadReasoning ?? false;
1142
+ messages.push({ role: "user", content: CONTINUATION_DIRECTIVE });
1143
+ logger.append({
1144
+ type: "continuation_injected",
1145
+ content: CONTINUATION_DIRECTIVE,
1146
+ continuationNumber: continuationsInjected,
1147
+ maxContinuations: MAX_CONTINUATION_TURNS,
1148
+ reason: "length",
1149
+ directive: "resume",
1150
+ hadReasoning,
1151
+ droppedToolCalls: droppedTruncatedCalls,
1152
+ });
1153
+ }
1154
+ }
1155
+ } catch (err) {
1156
+ const isContextExhausted = /maximum context length|context length exceeded|context_exhausted/i.test(err.message || "");
1157
+ if (isContextExhausted) {
1158
+ status = "context_exhausted";
1159
+ finalText =
1160
+ `[Context Exhausted] The session's cumulative context exceeded the model's ${MAX_CONTEXT.toLocaleString("en-US")} token ceiling.\n` +
1161
+ `Prior session events and tool outputs remain intact in the local event ledger.\n` +
1162
+ `Action: Roll into a fresh session_id (e.g. "${sessionId}_stage2") for subsequent dispatches.`;
1163
+ logger.append({
1164
+ type: "session_error",
1165
+ error: "context_exhausted",
1166
+ detail: err.message,
1167
+ });
1168
+ } else {
1169
+ status = "failed";
1170
+ finalText = `Castor execution error: ${err.message}`;
1171
+ logger.append({ type: "session_error", error: err.message, stack: err.stack });
1172
+ }
1173
+ } finally {
1174
+ const durationMs = Date.now() - t0;
1175
+ logger.append({
1176
+ type: "session_end",
1177
+ status,
1178
+ turnsTaken,
1179
+ continuationsInjected,
1180
+ durationMs,
1181
+ totalCompletionTokens,
1182
+ });
1183
+
1184
+ try {
1185
+ sampleLiveVllmMetrics();
1186
+ } catch {}
1187
+
1188
+ // Terminate MCP extension bridge and clean up child processes and registered tools.
1189
+ if (mcpBridge) {
1190
+ try {
1191
+ mcpBridge.dispose();
1192
+ } catch {}
1193
+ }
1194
+
1195
+ // Cleanly dispose microkernel and unmount all plugins
1196
+ ctx.dispose();
1197
+ }
1198
+
1199
+ return {
1200
+ finalText,
1201
+ turnsTaken,
1202
+ // Cumulative turn count across session lifetime.
1203
+ sessionTurns,
1204
+ status,
1205
+ durationMs: Date.now() - t0,
1206
+ totalCompletionTokens,
1207
+ sessionId,
1208
+ // Prompt token count and remaining context headroom under nominal ceiling.
1209
+ lastPromptTokens: lastPromptTokens > 0 ? lastPromptTokens : null,
1210
+ contextHeadroom:
1211
+ lastPromptTokens > 0
1212
+ ? Math.max(0, MAX_LEN_HUGE - lastPromptTokens)
1213
+ : null,
1214
+ };
1215
+ }
1216
+ }