headlesscode 1.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/ATTRIBUTION.md +53 -0
- package/CODE_OF_CONDUCT.md +130 -0
- package/CONTRIBUTING.md +107 -0
- package/LICENSE +202 -0
- package/README.md +486 -0
- package/SECURITY.md +211 -0
- package/bin/headlesscode.mjs +83 -0
- package/package.json +63 -0
- package/shared/prompts/review-mode-prompt-short.md +93 -0
- package/shared/prompts/review-mode-prompt.md +281 -0
- package/shared/rules-code/rules.md +22 -0
- package/shared/stacks/cpp/rules.md +30 -0
- package/shared/stacks/fastapi/rules.md +30 -0
- package/shared/stacks/javascript/rules.md +37 -0
- package/shared/stacks/postgresql/rules.md +31 -0
- package/shared/stacks/python/rules.md +35 -0
- package/shared/stacks/react/rules.md +11 -0
- package/shared/stacks/typescript/rules.md +10 -0
- package/src/budget/budget.ts +221 -0
- package/src/budget/concurrency.ts +126 -0
- package/src/budget/cost.ts +309 -0
- package/src/budget/index.ts +8 -0
- package/src/checkpoints/cli.ts +256 -0
- package/src/checkpoints/service.ts +227 -0
- package/src/cli.ts +1535 -0
- package/src/cloud/docker-provider.ts +334 -0
- package/src/cloud/provider.ts +300 -0
- package/src/codeintel/call-graph.ts +78 -0
- package/src/codeintel/find-references.ts +123 -0
- package/src/codeintel/go-to-definition.ts +193 -0
- package/src/codeintel/handlers.ts +190 -0
- package/src/codeintel/import-graph.ts +173 -0
- package/src/codeintel/outline.ts +180 -0
- package/src/codeintel/position.ts +77 -0
- package/src/codeintel/program.ts +350 -0
- package/src/codeintel/rename-symbol.ts +213 -0
- package/src/codeintel/tools.ts +280 -0
- package/src/codemap/build.ts +135 -0
- package/src/codemap/cli.ts +190 -0
- package/src/codemap/extract.ts +339 -0
- package/src/codemap/files.ts +236 -0
- package/src/codemap/fingerprint.ts +65 -0
- package/src/codemap/flows.ts +62 -0
- package/src/codemap/html.ts +451 -0
- package/src/codemap/lock.ts +80 -0
- package/src/codemap/types.ts +101 -0
- package/src/codesearch/airunner-embedder.ts +185 -0
- package/src/codesearch/chunk.ts +339 -0
- package/src/codesearch/cli.ts +223 -0
- package/src/codesearch/embedder.ts +332 -0
- package/src/codesearch/files.ts +280 -0
- package/src/codesearch/index.ts +469 -0
- package/src/codesearch/ollama-embedder.ts +205 -0
- package/src/codesearch/search.ts +141 -0
- package/src/codesearch/types.ts +100 -0
- package/src/config/mode-models.ts +218 -0
- package/src/dashboard/aggregate.ts +364 -0
- package/src/dashboard/chat-thread.ts +141 -0
- package/src/dashboard/checkpoints.ts +124 -0
- package/src/dashboard/cli.ts +193 -0
- package/src/dashboard/codemap.ts +44 -0
- package/src/dashboard/files.ts +121 -0
- package/src/dashboard/page.ts +2803 -0
- package/src/dashboard/self-improvement-metrics.ts +282 -0
- package/src/dashboard/server.ts +1103 -0
- package/src/dashboard/session-launch.ts +310 -0
- package/src/dashboard/timeline.ts +273 -0
- package/src/dashboard/tool-exec.ts +107 -0
- package/src/dashboard/trend-cli.ts +141 -0
- package/src/dashboard/trend.ts +413 -0
- package/src/decision-proxy/cli.ts +261 -0
- package/src/decision-proxy/proxy.ts +569 -0
- package/src/deploy/gate-cli.ts +147 -0
- package/src/deploy/gate.ts +254 -0
- package/src/engine/condense.ts +512 -0
- package/src/engine/events.ts +428 -0
- package/src/engine/handoff.ts +71 -0
- package/src/engine/lazy-tools.ts +160 -0
- package/src/engine/local-explore.ts +653 -0
- package/src/engine/logger.ts +96 -0
- package/src/engine/loop.ts +5517 -0
- package/src/engine/parser.ts +347 -0
- package/src/engine/prompt.ts +860 -0
- package/src/engine/reports.ts +47 -0
- package/src/engine/stacks.ts +448 -0
- package/src/engine/types.ts +291 -0
- package/src/engine/usage.ts +186 -0
- package/src/github/app-auth.ts +161 -0
- package/src/github/cli.ts +448 -0
- package/src/github/installations.ts +133 -0
- package/src/github/pr.ts +321 -0
- package/src/github/provision.ts +118 -0
- package/src/github/push.ts +122 -0
- package/src/index-util.ts +50 -0
- package/src/index.ts +81 -0
- package/src/init/cli.ts +248 -0
- package/src/init/gitignore.ts +74 -0
- package/src/llm/ollama.ts +308 -0
- package/src/llm/openrouter.ts +868 -0
- package/src/llm/preflight.ts +367 -0
- package/src/llm/transcript-capture.ts +84 -0
- package/src/memory/embed.ts +110 -0
- package/src/memory/index.ts +22 -0
- package/src/memory/local.ts +259 -0
- package/src/memory/summarizer.ts +283 -0
- package/src/memory/types.ts +153 -0
- package/src/memory/uwuchat.ts +157 -0
- package/src/migrate/cli.ts +115 -0
- package/src/orchestrator/analyze-cli.ts +104 -0
- package/src/orchestrator/auto-split.ts +206 -0
- package/src/orchestrator/cleanup.ts +1003 -0
- package/src/orchestrator/cli.ts +3571 -0
- package/src/orchestrator/cost-estimate.ts +564 -0
- package/src/orchestrator/cost-history-cli.ts +242 -0
- package/src/orchestrator/cost-history.ts +397 -0
- package/src/orchestrator/git-sync.ts +250 -0
- package/src/orchestrator/index.ts +153 -0
- package/src/orchestrator/log-analysis.ts +0 -0
- package/src/orchestrator/merge-check.ts +108 -0
- package/src/orchestrator/pipeline.ts +411 -0
- package/src/orchestrator/resume.ts +1940 -0
- package/src/orchestrator/reviewer.ts +503 -0
- package/src/orchestrator/split.ts +296 -0
- package/src/orchestrator/state.ts +542 -0
- package/src/orchestrator/status.ts +697 -0
- package/src/orchestrator/verification-gate.ts +134 -0
- package/src/orchestrator/watch.ts +898 -0
- package/src/permissions/commands.ts +1083 -0
- package/src/permissions/config.ts +241 -0
- package/src/permissions/index.ts +12 -0
- package/src/permissions/protected-files.ts +96 -0
- package/src/permissions/store-protection.ts +272 -0
- package/src/project-store.ts +648 -0
- package/src/projects/cli.ts +382 -0
- package/src/qa/qa.ts +487 -0
- package/src/tools/browser/handler.ts +346 -0
- package/src/tools/browser/service.ts +406 -0
- package/src/tools/browser/smoke.ts +78 -0
- package/src/tools/browser/tool.ts +99 -0
- package/src/tools/executor.ts +2575 -0
- package/src/tools/language-detect.ts +183 -0
- package/src/tools/output-summarizer.ts +369 -0
- package/src/tools/run-tests.ts +302 -0
- package/src/tools/set-indentation-tool.ts +49 -0
- package/src/tools/test-selection.ts +160 -0
- package/src/vendor/tests/smoke.ts +103 -0
- package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
- package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
- package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
- package/src/vendor/zoo-code/shim/os-name.ts +18 -0
- package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
- package/src/vendor/zoo-code/shim/vscode.ts +76 -0
- package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
- package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
- package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
- package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
- package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
- package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
- package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
- package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
- package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
- package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
- package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
- package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
- package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
- package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
- package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
- package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
- package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
- package/src/vendor/zoo-code/src/shared/language.ts +43 -0
- package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
- package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
- package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
- package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
- package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
- package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
- package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
- package/src/vendor/zoo-code/src/utils/object.ts +18 -0
- package/src/vendor/zoo-code/src/utils/path.ts +94 -0
- package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
- package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
- package/src/vendor/zoo-code/types/global-settings.ts +19 -0
- package/src/vendor/zoo-code/types/index.ts +22 -0
- package/src/vendor/zoo-code/types/message.ts +375 -0
- package/src/vendor/zoo-code/types/mode.ts +241 -0
- package/src/vendor/zoo-code/types/todo.ts +19 -0
- package/src/vendor/zoo-code/types/tool-params.ts +116 -0
- package/src/vendor/zoo-code/types/tool.ts +67 -0
- package/src/vendor/zoo-code/types/vscode.ts +84 -0
- package/src/vision/describe.ts +242 -0
- package/src/vision/tool.ts +91 -0
- package/src/watcher/cli.ts +369 -0
- package/src/watcher/github.ts +304 -0
- package/src/watcher/index.ts +59 -0
- package/src/watcher/state.ts +254 -0
- package/src/watcher/watch.ts +562 -0
- package/tsconfig.json +18 -0
|
@@ -0,0 +1,653 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Opt-in LOCAL EXPLORATION PHASE (default OFF) — plans/local-explore-phase-experiment.md.
|
|
3
|
+
*
|
|
4
|
+
* A bounded, strictly read-only local-model pass that runs BEFORE the cloud
|
|
5
|
+
* model's first turn: a local Ollama model (default qwen3.5:9b) gets a
|
|
6
|
+
* chance to explore the repository with a deliberately narrowed tool set
|
|
7
|
+
* (read_file + list_files + codebase_search — no execute_command, no write
|
|
8
|
+
* tools). codebase_search was originally excluded because the two-phase
|
|
9
|
+
* design avoids co-resident VRAM: the exploration model AND a local
|
|
10
|
+
* embedding model can't share limited VRAM. But the actually-deployed
|
|
11
|
+
* embedder is cloud-side (OpenRouter, qwen/qwen3-embedding-4b — see
|
|
12
|
+
* src/codesearch/cli.ts), NOT local Ollama, so codebase_search touches no
|
|
13
|
+
* local VRAM at all and is safe in the local phase (2026-08-04). When the
|
|
14
|
+
* phase concludes, its transcript is folded into the cloud session's initial
|
|
15
|
+
* context as a clearly-labeled synthetic message and the cloud model takes
|
|
16
|
+
* over completely.
|
|
17
|
+
*
|
|
18
|
+
* The phase is bounded by BOTH an iteration cap and a context-token budget
|
|
19
|
+
* (the measured safe solo-model VRAM ceiling at the configured num_ctx — see
|
|
20
|
+
* the plans doc's live-measured table). Token estimation reuses
|
|
21
|
+
* `estimateMessageChars` from condense.ts (4 chars/token — conservative, so
|
|
22
|
+
* we stop BEFORE the real ceiling, not after).
|
|
23
|
+
*
|
|
24
|
+
* FAIL-OPEN CONTRACT: `runLocalExplorePhase` NEVER throws. Any local-model
|
|
25
|
+
* error (Ollama unreachable, malformed response, model not pulled, ...)
|
|
26
|
+
* returns a result with `terminatedBy: "error"` and a null handoff — the
|
|
27
|
+
* caller then proceeds exactly as today (cloud-only). A broken experimental
|
|
28
|
+
* feature must never be able to break a real session. This mirrors the
|
|
29
|
+
* non-fatal-degradation idiom used for memory/checkpoints in loop.ts.
|
|
30
|
+
*/
|
|
31
|
+
|
|
32
|
+
import { createLocalExploreExecutor } from "../tools/executor.js"
|
|
33
|
+
import { DEFAULT_OLLAMA_URL, OLLAMA_URL_ENV } from "../tools/output-summarizer.js"
|
|
34
|
+
import { getNativeTools } from "../vendor/zoo-code/src/core/prompts/tools/native-tools/index.js"
|
|
35
|
+
import { estimateMessageChars } from "./condense.js"
|
|
36
|
+
import { Logger } from "./logger.js"
|
|
37
|
+
import { parseToolCall } from "./parser.js"
|
|
38
|
+
import type { ChatMessage, ChatTool, ChatToolCall, ToolResult } from "./types.js"
|
|
39
|
+
|
|
40
|
+
// ─── Configuration (env-var resolvable, like output-summarizer.ts) ─────────
|
|
41
|
+
|
|
42
|
+
/** Gate env var — HEADLESSCODE_LOCAL_EXPLORE=1 enables the phase (also --local-explore). */
|
|
43
|
+
export const LOCAL_EXPLORE_ENV = "HEADLESSCODE_LOCAL_EXPLORE"
|
|
44
|
+
/** Local model used for the exploration phase. */
|
|
45
|
+
export const LOCAL_EXPLORE_MODEL_ENV = "HEADLESSCODE_LOCAL_EXPLORE_MODEL"
|
|
46
|
+
export const DEFAULT_LOCAL_EXPLORE_MODEL = "qwen3.5:9b"
|
|
47
|
+
/**
|
|
48
|
+
* Iteration cap (default 15). Justification: each iteration is one local
|
|
49
|
+
* model call (a few seconds warm) + one or more read tool executions. 15
|
|
50
|
+
* calls is comfortably enough for a real exploration arc — list the tree,
|
|
51
|
+
* read the entry point, follow references — while guaranteeing the phase
|
|
52
|
+
* always terminates in well under a minute of warm inference even if the
|
|
53
|
+
* model never calls attempt_completion. The context budget below is the
|
|
54
|
+
* hard safety net that actually protects VRAM; this cap bounds TIME.
|
|
55
|
+
*/
|
|
56
|
+
export const LOCAL_EXPLORE_MAX_ITERATIONS_ENV = "HEADLESSCODE_LOCAL_EXPLORE_MAX_ITERATIONS"
|
|
57
|
+
export const DEFAULT_LOCAL_EXPLORE_MAX_ITERATIONS = 15
|
|
58
|
+
/**
|
|
59
|
+
* Context-token budget (default 131,072 = the measured safe solo-model
|
|
60
|
+
* ceiling). Live-measured on the project owner's RTX 5080 (16GB), Ollama
|
|
61
|
+
* 0.24.0 with OLLAMA_FLASH_ATTENTION=1 + OLLAMA_KV_CACHE_TYPE=q8_0:
|
|
62
|
+
* num_ctx 131,072 → ~12GB VRAM for qwen3.5:9b alone, leaving ~4GB headroom
|
|
63
|
+
* on the card. 262,144 (nominal max) is 17GB → CPU spillover, not usable.
|
|
64
|
+
* This is a HARD constraint (the phase stops before a request would exceed
|
|
65
|
+
* it), not a soft default. Verified live 2026-08-02; re-verify with `ollama
|
|
66
|
+
* ps` if driver/Ollama versions shift.
|
|
67
|
+
*/
|
|
68
|
+
export const LOCAL_EXPLORE_CONTEXT_TOKENS_ENV = "HEADLESSCODE_LOCAL_EXPLORE_CONTEXT_TOKENS"
|
|
69
|
+
export const DEFAULT_LOCAL_EXPLORE_CONTEXT_TOKENS = 131_072
|
|
70
|
+
/** Per-call HTTP timeout, ms. */
|
|
71
|
+
export const LOCAL_EXPLORE_TIMEOUT_MS_ENV = "HEADLESSCODE_LOCAL_EXPLORE_TIMEOUT_MS"
|
|
72
|
+
export const DEFAULT_LOCAL_EXPLORE_TIMEOUT_MS = 120_000
|
|
73
|
+
/** Max output tokens (num_predict) per local call — tool decisions are short. */
|
|
74
|
+
export const DEFAULT_LOCAL_EXPLORE_MAX_TOKENS = 2048
|
|
75
|
+
/** Consecutive mistakes (empty replies / failed tool calls) before the phase gives up. */
|
|
76
|
+
export const LOCAL_EXPLORE_MISTAKE_LIMIT = 3
|
|
77
|
+
/**
|
|
78
|
+
* Cap on the rendered handoff transcript fed to the cloud model (chars).
|
|
79
|
+
* The transcript is a DIGEST of the exploration record — each tool result is
|
|
80
|
+
* truncated to LOCAL_EXPLORE_TOOL_RESULT_TRANSCRIPT_CHARS (below), so a
|
|
81
|
+
* session's handoff is a few KB, not a verbatim file dump. This is
|
|
82
|
+
* deliberate: the cloud model can re-read any file itself (read_file /
|
|
83
|
+
* codebase_search) at lower effective cost than paying ~4 chars/token for
|
|
84
|
+
* raw source it mostly does not need, and the cloud-side input-token bill is
|
|
85
|
+
* exactly what the "free local exploration" cost story must not blow up.
|
|
86
|
+
* Verbatim tool output still lives in the phase's OWN context — the phase's
|
|
87
|
+
* loop uses full results, only the rendered handoff is bounded. Kept as a
|
|
88
|
+
* digest rather than an LLM summarization pass: summarization risks
|
|
89
|
+
* needle-loss (see plans/local-output-summarization.md) and costs latency.
|
|
90
|
+
*/
|
|
91
|
+
export const MAX_HANDOFF_CHARS = 12_000
|
|
92
|
+
/**
|
|
93
|
+
* Per-tool-result cap in the rendered handoff transcript (chars). Long
|
|
94
|
+
* enough to keep the read_file header ("File: X, Showing lines A-B of Y")
|
|
95
|
+
* plus a hint of the body — the verification anchor the cloud model needs —
|
|
96
|
+
* without the bulk.
|
|
97
|
+
*/
|
|
98
|
+
export const LOCAL_EXPLORE_TOOL_RESULT_TRANSCRIPT_CHARS = 400
|
|
99
|
+
/** Chars per estimated token — conservative (lower than the ~8 English rule of thumb) so we stop early, not late. */
|
|
100
|
+
export const CHARS_PER_TOKEN = 4
|
|
101
|
+
|
|
102
|
+
// ─── Types ──────────────────────────────────────────────────────────────────
|
|
103
|
+
|
|
104
|
+
export type LocalExploreTermination =
|
|
105
|
+
| "attempt_completion"
|
|
106
|
+
| "text-only-reply"
|
|
107
|
+
| "iteration-cap"
|
|
108
|
+
| "context-budget"
|
|
109
|
+
| "error"
|
|
110
|
+
|
|
111
|
+
export interface LocalExploreOptions {
|
|
112
|
+
workspaceRoot: string
|
|
113
|
+
/** The session task text — also the local phase's mission statement. */
|
|
114
|
+
taskText?: string
|
|
115
|
+
/** Local model id (default env HEADLESSCODE_LOCAL_EXPLORE_MODEL or qwen3.5:9b). */
|
|
116
|
+
model?: string
|
|
117
|
+
/** Ollama base URL (default env HEADLESSCODE_OLLAMA_URL or http://localhost:11434). */
|
|
118
|
+
baseUrl?: string
|
|
119
|
+
/** Iteration cap (default HEADLESSCODE_LOCAL_EXPLORE_MAX_ITERATIONS or 15). */
|
|
120
|
+
maxIterations?: number
|
|
121
|
+
/** Context-token budget (default HEADLESSCODE_LOCAL_EXPLORE_CONTEXT_TOKENS or 131072). */
|
|
122
|
+
contextTokens?: number
|
|
123
|
+
/** Per-call timeout ms (default HEADLESSCODE_LOCAL_EXPLORE_TIMEOUT_MS or 120s). */
|
|
124
|
+
timeoutMs?: number
|
|
125
|
+
/** Max output tokens per local call (default 2048). */
|
|
126
|
+
maxTokens?: number
|
|
127
|
+
/** Handoff transcript cap in chars (default MAX_HANDOFF_CHARS). */
|
|
128
|
+
maxHandoffChars?: number
|
|
129
|
+
logger?: Logger
|
|
130
|
+
/** Injectable chat client (tests use a fake; default: real Ollama client). */
|
|
131
|
+
client?: LocalChatClient
|
|
132
|
+
/** Injectable executor (tests; default: createLocalExploreExecutor(root)). */
|
|
133
|
+
executor?: ReturnType<typeof createLocalExploreExecutor>
|
|
134
|
+
/** Injectable fetch (tests; default: global fetch). */
|
|
135
|
+
fetchImpl?: typeof fetch
|
|
136
|
+
/** System prompt override (tests; default EXPLORE_SYSTEM_PROMPT). */
|
|
137
|
+
systemPrompt?: string
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
export interface LocalExploreResult {
|
|
141
|
+
/** Why the phase ended. "error" = fail open, no handoff (cloud proceeds as today). */
|
|
142
|
+
terminatedBy: LocalExploreTermination
|
|
143
|
+
/** Number of model calls actually made. */
|
|
144
|
+
iterations: number
|
|
145
|
+
/** The full local transcript: [system, user(task), assistant, tool, ...]. */
|
|
146
|
+
messages: ChatMessage[]
|
|
147
|
+
/** Estimated prompt tokens of the final transcript (chars/4). */
|
|
148
|
+
estimatedPromptTokens: number
|
|
149
|
+
/** Human-readable termination detail (e.g. the error message). */
|
|
150
|
+
detail?: string
|
|
151
|
+
/**
|
|
152
|
+
* The synthetic cloud-context message to fold into the cloud session, or
|
|
153
|
+
* null when the phase produced nothing usable (fail open → cloud-only).
|
|
154
|
+
*/
|
|
155
|
+
handoffMessage: ChatMessage | null
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
/** Request shape for a local chat call. */
|
|
159
|
+
export interface LocalExploreRequest {
|
|
160
|
+
model: string
|
|
161
|
+
messages: ChatMessage[]
|
|
162
|
+
tools: ChatTool[]
|
|
163
|
+
numCtx: number
|
|
164
|
+
maxTokens: number
|
|
165
|
+
signal?: AbortSignal
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
export interface LocalChatResponse {
|
|
169
|
+
/** Assistant message in engine format (tool_call arguments as JSON strings). */
|
|
170
|
+
message: ChatMessage
|
|
171
|
+
usage: { promptTokens: number; completionTokens: number }
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
export interface LocalChatClient {
|
|
175
|
+
chat(request: LocalExploreRequest): Promise<LocalChatResponse>
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/** Any local-chat failure — thrown by the client, caught inside runLocalExplorePhase. */
|
|
179
|
+
export class LocalExploreError extends Error {}
|
|
180
|
+
|
|
181
|
+
// ─── The local exploration loop ─────────────────────────────────────────────
|
|
182
|
+
|
|
183
|
+
/**
|
|
184
|
+
* Run the bounded local exploration phase. NEVER throws: every failure mode
|
|
185
|
+
* (Ollama down, malformed response, timeout, model not pulled) returns a
|
|
186
|
+
* result with `terminatedBy: "error"` and a null handoff so the caller fails
|
|
187
|
+
* open to today's cloud-only behavior.
|
|
188
|
+
*/
|
|
189
|
+
export async function runLocalExplorePhase(options: LocalExploreOptions): Promise<LocalExploreResult> {
|
|
190
|
+
const logger = options.logger ?? new Logger()
|
|
191
|
+
const model = options.model ?? resolveLocalExploreModel(process.env)
|
|
192
|
+
const baseUrl = options.baseUrl ?? resolveOllamaUrl(process.env)
|
|
193
|
+
const maxIterations = options.maxIterations ?? resolveLocalExploreMaxIterations(process.env)
|
|
194
|
+
const contextTokens = options.contextTokens ?? resolveLocalExploreContextTokens(process.env)
|
|
195
|
+
const timeoutMs = options.timeoutMs ?? resolveLocalExploreTimeoutMs(process.env)
|
|
196
|
+
const maxTokens = options.maxTokens ?? DEFAULT_LOCAL_EXPLORE_MAX_TOKENS
|
|
197
|
+
const maxHandoffChars = options.maxHandoffChars ?? MAX_HANDOFF_CHARS
|
|
198
|
+
|
|
199
|
+
const client = options.client ?? new OllamaLocalChatClient({ baseUrl, timeoutMs, fetchImpl: options.fetchImpl })
|
|
200
|
+
const executor = options.executor ?? createLocalExploreExecutor(options.workspaceRoot)
|
|
201
|
+
const tools = buildLocalExploreTools()
|
|
202
|
+
const systemPrompt = options.systemPrompt ?? EXPLORE_SYSTEM_PROMPT
|
|
203
|
+
|
|
204
|
+
const messages: ChatMessage[] = [
|
|
205
|
+
{ role: "system", content: systemPrompt },
|
|
206
|
+
{ role: "user", content: options.taskText ?? "Explore this repository and gather the context needed to complete the task." },
|
|
207
|
+
]
|
|
208
|
+
|
|
209
|
+
let mistakes = 0
|
|
210
|
+
let iterations = 0
|
|
211
|
+
let terminatedBy: LocalExploreTermination = "iteration-cap"
|
|
212
|
+
let detail: string | undefined
|
|
213
|
+
|
|
214
|
+
logger.info("[local-explore] phase start", { model, baseUrl, maxIterations, contextTokens })
|
|
215
|
+
|
|
216
|
+
for (; iterations < maxIterations; iterations++) {
|
|
217
|
+
// Context-budget gate: stop BEFORE a request would exceed the ceiling.
|
|
218
|
+
const estimated = estimatePromptTokens(messages)
|
|
219
|
+
if (estimated > contextTokens) {
|
|
220
|
+
terminatedBy = "context-budget"
|
|
221
|
+
detail = `estimated ${estimated} prompt tokens exceeds the ${contextTokens}-token budget`
|
|
222
|
+
logger.warn("[local-explore] context budget reached — stopping", { estimated, contextTokens })
|
|
223
|
+
break
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
let response: LocalChatResponse
|
|
227
|
+
try {
|
|
228
|
+
response = await client.chat({ model, messages, tools, numCtx: contextTokens, maxTokens })
|
|
229
|
+
} catch (err) {
|
|
230
|
+
logger.warn(`[local-explore] local model call failed — failing open to cloud-only: ${err instanceof Error ? err.message : String(err)}`)
|
|
231
|
+
return {
|
|
232
|
+
terminatedBy: "error",
|
|
233
|
+
iterations,
|
|
234
|
+
messages,
|
|
235
|
+
estimatedPromptTokens: estimatePromptTokens(messages),
|
|
236
|
+
detail: err instanceof Error ? err.message : String(err),
|
|
237
|
+
handoffMessage: null,
|
|
238
|
+
}
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
const assistant = response.message
|
|
242
|
+
messages.push(assistant)
|
|
243
|
+
|
|
244
|
+
// The explicit "I'm done exploring, hand off" signal. Distinct from
|
|
245
|
+
// actually completing the task: the local phase is read-only, and the
|
|
246
|
+
// cloud model still does the real work.
|
|
247
|
+
const completionCall = (assistant.tool_calls ?? []).find((c) => c.function.name === "attempt_completion")
|
|
248
|
+
if (completionCall) {
|
|
249
|
+
terminatedBy = "attempt_completion"
|
|
250
|
+
detail = extractCompletionResult(completionCall)
|
|
251
|
+
logger.info("[local-explore] local model signaled completion", { iterations: iterations + 1 })
|
|
252
|
+
break
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
const calls = (assistant.tool_calls ?? []).filter((c) => c.function.name !== "attempt_completion")
|
|
256
|
+
|
|
257
|
+
if (calls.length === 0) {
|
|
258
|
+
// Text-only reply (non-empty): the model handed its findings back
|
|
259
|
+
// directly — treat it as the exploration result.
|
|
260
|
+
if (assistant.content && assistant.content.trim().length > 0) {
|
|
261
|
+
terminatedBy = "text-only-reply"
|
|
262
|
+
logger.info("[local-explore] local model replied with text (no tool call)", { iterations: iterations + 1 })
|
|
263
|
+
break
|
|
264
|
+
}
|
|
265
|
+
// Empty reply: nudge, like the main loop does. Repeated empties are
|
|
266
|
+
// a local-model failure — abandon the phase (fail open).
|
|
267
|
+
mistakes += 1
|
|
268
|
+
if (mistakes >= LOCAL_EXPLORE_MISTAKE_LIMIT) {
|
|
269
|
+
terminatedBy = "error"
|
|
270
|
+
detail = `local model produced ${mistakes} consecutive empty replies`
|
|
271
|
+
logger.warn(`[local-explore] ${detail}`)
|
|
272
|
+
break
|
|
273
|
+
}
|
|
274
|
+
messages.push({
|
|
275
|
+
role: "user",
|
|
276
|
+
content: "[System: your last response contained no tool calls and no text. Continue exploring: call read_file or list_files, or call attempt_completion when you have enough context.]",
|
|
277
|
+
})
|
|
278
|
+
continue
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
// Execute every tool call and feed the results back.
|
|
282
|
+
for (const call of calls) {
|
|
283
|
+
let result: ToolResult
|
|
284
|
+
try {
|
|
285
|
+
result = await executor.execute(call.function.name, parseToolCall(call).args)
|
|
286
|
+
} catch (err) {
|
|
287
|
+
result = { content: `[Error] ${err instanceof Error ? err.message : String(err)}`, isError: true }
|
|
288
|
+
}
|
|
289
|
+
if (result.isError) {
|
|
290
|
+
mistakes += 1
|
|
291
|
+
}
|
|
292
|
+
messages.push({
|
|
293
|
+
role: "tool",
|
|
294
|
+
content: result.content,
|
|
295
|
+
tool_call_id: call.id,
|
|
296
|
+
name: call.function.name,
|
|
297
|
+
})
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
// A single huge tool result (e.g. a 30K-char read) can jump the budget
|
|
301
|
+
// even though the request that triggered it was under it — stop before
|
|
302
|
+
// any NEXT request would exceed the ceiling.
|
|
303
|
+
const afterTools = estimatePromptTokens(messages)
|
|
304
|
+
if (afterTools > contextTokens) {
|
|
305
|
+
terminatedBy = "context-budget"
|
|
306
|
+
detail = `estimated ${afterTools} prompt tokens exceeds the ${contextTokens}-token budget after tool results`
|
|
307
|
+
logger.warn("[local-explore] context budget reached after tool results — stopping", { estimated: afterTools, contextTokens })
|
|
308
|
+
break
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
if (terminatedBy === "iteration-cap" && iterations >= maxIterations) {
|
|
313
|
+
detail = `iteration cap (${maxIterations}) reached`
|
|
314
|
+
logger.info(`[local-explore] ${detail}`)
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
const result: LocalExploreResult = {
|
|
318
|
+
terminatedBy,
|
|
319
|
+
iterations,
|
|
320
|
+
messages,
|
|
321
|
+
estimatedPromptTokens: estimatePromptTokens(messages),
|
|
322
|
+
detail,
|
|
323
|
+
handoffMessage: null,
|
|
324
|
+
}
|
|
325
|
+
result.handoffMessage = buildLocalExploreHandoffMessage(result, maxHandoffChars)
|
|
326
|
+
logger.info("[local-explore] phase end", { terminatedBy, iterations: result.iterations, estimatedPromptTokens: result.estimatedPromptTokens })
|
|
327
|
+
return result
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
// ─── Handoff ────────────────────────────────────────────────────────────────
|
|
331
|
+
|
|
332
|
+
/** Extract the completion result text from an attempt_completion call. */
|
|
333
|
+
function extractCompletionResult(call: ChatToolCall): string {
|
|
334
|
+
try {
|
|
335
|
+
const args = JSON.parse(call.function.arguments) as { result?: unknown }
|
|
336
|
+
if (typeof args.result === "string" && args.result.trim()) {
|
|
337
|
+
return args.result
|
|
338
|
+
}
|
|
339
|
+
return JSON.stringify(args)
|
|
340
|
+
} catch {
|
|
341
|
+
return call.function.arguments
|
|
342
|
+
}
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
/**
|
|
346
|
+
* Build the clearly-labeled synthetic message folded into the cloud session's
|
|
347
|
+
* initial context. Returns null when there is nothing usable to hand off —
|
|
348
|
+
* i.e. the phase failed (fail open → cloud-only) or gathered no activity
|
|
349
|
+
* beyond the seed messages.
|
|
350
|
+
*/
|
|
351
|
+
export function buildLocalExploreHandoffMessage(
|
|
352
|
+
result: Pick<LocalExploreResult, "terminatedBy" | "iterations" | "messages" | "detail">,
|
|
353
|
+
maxHandoffChars = MAX_HANDOFF_CHARS,
|
|
354
|
+
): ChatMessage | null {
|
|
355
|
+
if (result.terminatedBy === "error") {
|
|
356
|
+
return null
|
|
357
|
+
}
|
|
358
|
+
const transcript = renderTranscript(result.messages, maxHandoffChars)
|
|
359
|
+
if (!transcript) {
|
|
360
|
+
return null
|
|
361
|
+
}
|
|
362
|
+
const reason = terminationReason(result)
|
|
363
|
+
const header = `[Local exploration phase — a separate LOCAL model (read-only, could call read_file, list_files and codebase_search: no commands, no writes) ran a pre-pass on this repository BEFORE your first turn. This is NOT your own prior work. Treat its file claims as a starting point to verify, not ground truth. The phase stopped because: ${reason}.]`
|
|
364
|
+
const footer = "=== END LOCAL EXPLORATION TRANSCRIPT ==="
|
|
365
|
+
return {
|
|
366
|
+
role: "user",
|
|
367
|
+
content: `${header}\n\n=== LOCAL EXPLORATION TRANSCRIPT ===\n${transcript}\n${footer}`,
|
|
368
|
+
}
|
|
369
|
+
}
|
|
370
|
+
|
|
371
|
+
function terminationReason(result: Pick<LocalExploreResult, "terminatedBy" | "iterations" | "detail">): string {
|
|
372
|
+
switch (result.terminatedBy) {
|
|
373
|
+
case "attempt_completion":
|
|
374
|
+
return "the local model decided it had enough context"
|
|
375
|
+
case "text-only-reply":
|
|
376
|
+
return "the local model replied directly with its findings"
|
|
377
|
+
case "iteration-cap":
|
|
378
|
+
return `iteration cap (${result.iterations}) reached`
|
|
379
|
+
case "context-budget":
|
|
380
|
+
return result.detail ?? "its context budget was reached"
|
|
381
|
+
default:
|
|
382
|
+
return result.detail ?? result.terminatedBy
|
|
383
|
+
}
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
/**
|
|
387
|
+
* Render the exploration activity into a compact transcript. Skips the seed
|
|
388
|
+
* system + user task messages (the cloud history already carries the task) —
|
|
389
|
+
* only the model's calls, tool results and final reply are included. Capped
|
|
390
|
+
* at maxHandoffChars with a truncation marker.
|
|
391
|
+
*/
|
|
392
|
+
function renderTranscript(messages: ChatMessage[], maxHandoffChars: number): string {
|
|
393
|
+
const lines: string[] = []
|
|
394
|
+
for (let i = 2; i < messages.length; i++) {
|
|
395
|
+
lines.push(renderMessage(messages[i]))
|
|
396
|
+
}
|
|
397
|
+
if (lines.length === 0) {
|
|
398
|
+
return ""
|
|
399
|
+
}
|
|
400
|
+
let out = lines.join("\n")
|
|
401
|
+
if (out.length > maxHandoffChars) {
|
|
402
|
+
out = `${out.slice(0, maxHandoffChars)}\n...[transcript truncated at ${maxHandoffChars} chars]`
|
|
403
|
+
}
|
|
404
|
+
return out
|
|
405
|
+
}
|
|
406
|
+
|
|
407
|
+
function renderMessage(m: ChatMessage): string {
|
|
408
|
+
switch (m.role) {
|
|
409
|
+
case "assistant":
|
|
410
|
+
if (m.tool_calls && m.tool_calls.length > 0) {
|
|
411
|
+
return m.tool_calls
|
|
412
|
+
.map((c) => {
|
|
413
|
+
if (c.function.name === "attempt_completion") {
|
|
414
|
+
return `[local model] attempt_completion: ${extractCompletionResult(c)}`
|
|
415
|
+
}
|
|
416
|
+
return `[local model] called ${c.function.name}(${c.function.arguments})`
|
|
417
|
+
})
|
|
418
|
+
.join("\n")
|
|
419
|
+
}
|
|
420
|
+
return `[local model] ${m.content ?? ""}`
|
|
421
|
+
case "tool": {
|
|
422
|
+
const body = m.content ?? ""
|
|
423
|
+
const shown = body.length > LOCAL_EXPLORE_TOOL_RESULT_TRANSCRIPT_CHARS
|
|
424
|
+
? `${body.slice(0, LOCAL_EXPLORE_TOOL_RESULT_TRANSCRIPT_CHARS)}\n…[tool result truncated in handoff at ${LOCAL_EXPLORE_TOOL_RESULT_TRANSCRIPT_CHARS} chars — re-read the file yourself to verify]`
|
|
425
|
+
: body
|
|
426
|
+
return `[tool result for ${m.name ?? m.tool_call_id ?? "?"}] ${shown}`
|
|
427
|
+
}
|
|
428
|
+
default:
|
|
429
|
+
return `[${m.role}] ${m.content ?? ""}`
|
|
430
|
+
}
|
|
431
|
+
}
|
|
432
|
+
|
|
433
|
+
// ─── Ollama client ──────────────────────────────────────────────────────────
|
|
434
|
+
|
|
435
|
+
/**
|
|
436
|
+
* Ollama /api/chat client. WIRE-FORMAT NOTE (verified live 2026-08-02 against
|
|
437
|
+
* Ollama 0.24.0 + qwen3.5:9b): Ollama wants `function.arguments` as a parsed
|
|
438
|
+
* OBJECT in BOTH outgoing history and the response — the JSON-string form that
|
|
439
|
+
* OpenAI/OpenRouter use makes Ollama reject the request with "Value looks like
|
|
440
|
+
* object, but can't find closing '}' symbol". So:
|
|
441
|
+
* - outgoing: engine ChatToolCall.arguments (JSON string) → parsed object;
|
|
442
|
+
* - incoming: Ollama's object arguments → engine ChatToolCall.arguments
|
|
443
|
+
* (JSON string), so the rest of the harness (parseToolCall etc.) works
|
|
444
|
+
* unchanged.
|
|
445
|
+
* qwen3-class models also reason by default and leave message.content empty —
|
|
446
|
+
* we send `think: false` (same verified finding as output-summarizer.ts).
|
|
447
|
+
*/
|
|
448
|
+
export class OllamaLocalChatClient implements LocalChatClient {
|
|
449
|
+
private readonly baseUrl: string
|
|
450
|
+
private readonly timeoutMs: number
|
|
451
|
+
private readonly fetchImpl: typeof fetch
|
|
452
|
+
|
|
453
|
+
constructor(options: { baseUrl: string; timeoutMs?: number; fetchImpl?: typeof fetch }) {
|
|
454
|
+
this.baseUrl = options.baseUrl.replace(/\/+$/, "")
|
|
455
|
+
this.timeoutMs = options.timeoutMs ?? DEFAULT_LOCAL_EXPLORE_TIMEOUT_MS
|
|
456
|
+
this.fetchImpl = options.fetchImpl ?? ((...args) => fetch(...args))
|
|
457
|
+
}
|
|
458
|
+
|
|
459
|
+
async chat(request: LocalExploreRequest): Promise<LocalChatResponse> {
|
|
460
|
+
const controller = new AbortController()
|
|
461
|
+
const timer = setTimeout(() => controller.abort(), this.timeoutMs)
|
|
462
|
+
try {
|
|
463
|
+
const res = await this.fetchImpl(`${this.baseUrl}/api/chat`, {
|
|
464
|
+
method: "POST",
|
|
465
|
+
headers: { "content-type": "application/json" },
|
|
466
|
+
signal: controller.signal,
|
|
467
|
+
body: JSON.stringify({
|
|
468
|
+
model: request.model,
|
|
469
|
+
messages: toOllamaMessages(request.messages),
|
|
470
|
+
stream: false,
|
|
471
|
+
think: false,
|
|
472
|
+
tools: request.tools,
|
|
473
|
+
options: {
|
|
474
|
+
num_ctx: request.numCtx,
|
|
475
|
+
num_predict: request.maxTokens,
|
|
476
|
+
temperature: 0,
|
|
477
|
+
},
|
|
478
|
+
}),
|
|
479
|
+
})
|
|
480
|
+
if (!res.ok) {
|
|
481
|
+
const body = await res.text().catch(() => "")
|
|
482
|
+
throw new LocalExploreError(`Ollama /api/chat returned HTTP ${res.status}: ${body.slice(0, 500)}`)
|
|
483
|
+
}
|
|
484
|
+
const data = (await res.json()) as {
|
|
485
|
+
message?: {
|
|
486
|
+
role?: string
|
|
487
|
+
content?: string | null
|
|
488
|
+
tool_calls?: Array<{
|
|
489
|
+
id?: string
|
|
490
|
+
function?: { name?: string; arguments?: unknown }
|
|
491
|
+
}>
|
|
492
|
+
}
|
|
493
|
+
prompt_eval_count?: number
|
|
494
|
+
eval_count?: number
|
|
495
|
+
}
|
|
496
|
+
if (!data.message) {
|
|
497
|
+
throw new LocalExploreError("Ollama /api/chat response had no message field")
|
|
498
|
+
}
|
|
499
|
+
return {
|
|
500
|
+
message: {
|
|
501
|
+
role: "assistant",
|
|
502
|
+
content: data.message.content ?? null,
|
|
503
|
+
tool_calls: (data.message.tool_calls ?? []).map((c, i) => ({
|
|
504
|
+
id: c.id ?? `call_${i}`,
|
|
505
|
+
type: "function" as const,
|
|
506
|
+
function: {
|
|
507
|
+
name: c.function?.name ?? "",
|
|
508
|
+
arguments: normalizeToolCallArguments(c.function?.arguments),
|
|
509
|
+
},
|
|
510
|
+
})),
|
|
511
|
+
},
|
|
512
|
+
usage: {
|
|
513
|
+
promptTokens: data.prompt_eval_count ?? 0,
|
|
514
|
+
completionTokens: data.eval_count ?? 0,
|
|
515
|
+
},
|
|
516
|
+
}
|
|
517
|
+
} catch (err) {
|
|
518
|
+
if (err instanceof LocalExploreError) {
|
|
519
|
+
throw err
|
|
520
|
+
}
|
|
521
|
+
if (err instanceof Error && err.name === "AbortError") {
|
|
522
|
+
throw new LocalExploreError(`Ollama /api/chat timed out after ${this.timeoutMs}ms`)
|
|
523
|
+
}
|
|
524
|
+
throw new LocalExploreError(
|
|
525
|
+
`Ollama /api/chat request failed: ${err instanceof Error ? err.message : String(err)} (is Ollama running at ${this.baseUrl}? is ${request.model} pulled?)`,
|
|
526
|
+
)
|
|
527
|
+
} finally {
|
|
528
|
+
clearTimeout(timer)
|
|
529
|
+
}
|
|
530
|
+
}
|
|
531
|
+
}
|
|
532
|
+
|
|
533
|
+
/** Engine ChatMessage[] → Ollama wire format (tool_call arguments as OBJECTS). */
|
|
534
|
+
function toOllamaMessages(messages: ChatMessage[]): unknown[] {
|
|
535
|
+
return messages.map((m) => {
|
|
536
|
+
if (m.role === "assistant" && m.tool_calls && m.tool_calls.length > 0) {
|
|
537
|
+
return {
|
|
538
|
+
role: "assistant",
|
|
539
|
+
content: m.content ?? "",
|
|
540
|
+
tool_calls: m.tool_calls.map((c) => ({
|
|
541
|
+
id: c.id,
|
|
542
|
+
type: "function",
|
|
543
|
+
function: { name: c.function.name, arguments: toOllamaArgs(c.function.arguments) },
|
|
544
|
+
})),
|
|
545
|
+
}
|
|
546
|
+
}
|
|
547
|
+
if (m.role === "tool") {
|
|
548
|
+
return { role: "tool", content: m.content ?? "", tool_call_id: m.tool_call_id, name: m.name }
|
|
549
|
+
}
|
|
550
|
+
return { role: m.role, content: m.content ?? "" }
|
|
551
|
+
})
|
|
552
|
+
}
|
|
553
|
+
|
|
554
|
+
/** Engine arguments JSON string → Ollama object. Parse failure → {} (Ollama rejects strings here). */
|
|
555
|
+
function toOllamaArgs(args: string): unknown {
|
|
556
|
+
try {
|
|
557
|
+
return JSON.parse(args)
|
|
558
|
+
} catch {
|
|
559
|
+
return {}
|
|
560
|
+
}
|
|
561
|
+
}
|
|
562
|
+
|
|
563
|
+
/** Ollama arguments (object, or occasionally string) → engine JSON-string form. */
|
|
564
|
+
function normalizeToolCallArguments(args: unknown): string {
|
|
565
|
+
if (typeof args === "string") {
|
|
566
|
+
// Already a JSON string (some models return it stringified) — keep it
|
|
567
|
+
// parseable by the harness's parseToolCall.
|
|
568
|
+
try {
|
|
569
|
+
JSON.parse(args)
|
|
570
|
+
return args
|
|
571
|
+
} catch {
|
|
572
|
+
return JSON.stringify({ raw: args })
|
|
573
|
+
}
|
|
574
|
+
}
|
|
575
|
+
return JSON.stringify(args ?? {})
|
|
576
|
+
}
|
|
577
|
+
|
|
578
|
+
// ─── Tools + prompt ─────────────────────────────────────────────────────────
|
|
579
|
+
|
|
580
|
+
/**
|
|
581
|
+
* The local phase's tool schemas: read_file, list_files, codebase_search,
|
|
582
|
+
* attempt_completion — taken from the vendored native-tool definitions so the
|
|
583
|
+
* schema matches what the cloud model sees. codebase_search is included even
|
|
584
|
+
* though the phase is local because its embedding call is cloud-side
|
|
585
|
+
* (OpenRouter), not local Ollama — it consumes no local VRAM (see the
|
|
586
|
+
* file-top comment). Everything else is deliberately absent (the executor
|
|
587
|
+
* stubs it if called anyway).
|
|
588
|
+
*/
|
|
589
|
+
export function buildLocalExploreTools(): ChatTool[] {
|
|
590
|
+
const names = new Set(["read_file", "list_files", "codebase_search", "attempt_completion"])
|
|
591
|
+
return getNativeTools()
|
|
592
|
+
.filter((t) => t.type === "function" && names.has(t.function.name))
|
|
593
|
+
.map((t) => t as unknown as ChatTool)
|
|
594
|
+
}
|
|
595
|
+
|
|
596
|
+
export const EXPLORE_SYSTEM_PROMPT = `You are a LOCAL, read-only repository exploration agent. You run BEFORE the main coding model's turn: your ONLY job is to gather context that will help the main model complete the task. You never implement anything.
|
|
597
|
+
|
|
598
|
+
Available tools:
|
|
599
|
+
- read_file — read a file (path is relative to the workspace root).
|
|
600
|
+
- list_files — list files/directories (path is relative to the workspace root; recursive: true for a full tree).
|
|
601
|
+
- codebase_search — semantic search over the indexed codebase (a query describing what you need; returns matching file excerpts). Use it for targeted search instead of blind tree-walking.
|
|
602
|
+
- attempt_completion — call this when you have gathered ENOUGH context. Its result argument is your exploration report: what you found, which files are relevant and why. This does NOT complete the task — it hands off to the main model.
|
|
603
|
+
|
|
604
|
+
Rules:
|
|
605
|
+
- Read-only: you cannot write files or run commands. You may use read_file, list_files and codebase_search only.
|
|
606
|
+
- Prefer codebase_search for targeted lookups ("where is X?" / "how does Y work?") over recursively listing and reading whole trees.
|
|
607
|
+
- Do not fabricate file contents or paths. Report exactly what the tools return. If a read fails, say so.
|
|
608
|
+
- Explore purposefully: list the top level first, then drill into the directories and files that matter for the task. Prefer reading the actual entry points (package.json, src/index.ts, README) before guessing.
|
|
609
|
+
- When you have enough context to hand the main model a focused map of the relevant code, call attempt_completion with your findings. If you cannot make progress, call attempt_completion anyway with what you have — never loop forever.`
|
|
610
|
+
|
|
611
|
+
// ─── Env resolution (mirrors output-summarizer.ts's pattern) ────────────────
|
|
612
|
+
|
|
613
|
+
export function isLocalExploreEnabled(env: NodeJS.ProcessEnv = process.env): boolean {
|
|
614
|
+
const v = env[LOCAL_EXPLORE_ENV]
|
|
615
|
+
return v !== undefined && v !== "" && v !== "0" && v.toLowerCase() !== "false"
|
|
616
|
+
}
|
|
617
|
+
|
|
618
|
+
export function resolveLocalExploreModel(env: NodeJS.ProcessEnv = process.env): string {
|
|
619
|
+
return env[LOCAL_EXPLORE_MODEL_ENV]?.trim() || DEFAULT_LOCAL_EXPLORE_MODEL
|
|
620
|
+
}
|
|
621
|
+
|
|
622
|
+
export function resolveOllamaUrl(env: NodeJS.ProcessEnv = process.env): string {
|
|
623
|
+
return env[OLLAMA_URL_ENV]?.trim() || DEFAULT_OLLAMA_URL
|
|
624
|
+
}
|
|
625
|
+
|
|
626
|
+
export function resolveLocalExploreMaxIterations(env: NodeJS.ProcessEnv = process.env): number {
|
|
627
|
+
return parsePositiveInt(env[LOCAL_EXPLORE_MAX_ITERATIONS_ENV], DEFAULT_LOCAL_EXPLORE_MAX_ITERATIONS, LOCAL_EXPLORE_MAX_ITERATIONS_ENV)
|
|
628
|
+
}
|
|
629
|
+
|
|
630
|
+
export function resolveLocalExploreContextTokens(env: NodeJS.ProcessEnv = process.env): number {
|
|
631
|
+
return parsePositiveInt(env[LOCAL_EXPLORE_CONTEXT_TOKENS_ENV], DEFAULT_LOCAL_EXPLORE_CONTEXT_TOKENS, LOCAL_EXPLORE_CONTEXT_TOKENS_ENV)
|
|
632
|
+
}
|
|
633
|
+
|
|
634
|
+
export function resolveLocalExploreTimeoutMs(env: NodeJS.ProcessEnv = process.env): number {
|
|
635
|
+
return parsePositiveInt(env[LOCAL_EXPLORE_TIMEOUT_MS_ENV], DEFAULT_LOCAL_EXPLORE_TIMEOUT_MS, LOCAL_EXPLORE_TIMEOUT_MS_ENV)
|
|
636
|
+
}
|
|
637
|
+
|
|
638
|
+
function parsePositiveInt(raw: string | undefined, fallback: number, envName: string): number {
|
|
639
|
+
if (raw === undefined || raw.trim() === "") {
|
|
640
|
+
return fallback
|
|
641
|
+
}
|
|
642
|
+
const n = Number(raw)
|
|
643
|
+
if (!Number.isInteger(n) || n <= 0) {
|
|
644
|
+
throw new LocalExploreError(`${envName} must be a positive integer, got "${raw}"`)
|
|
645
|
+
}
|
|
646
|
+
return n
|
|
647
|
+
}
|
|
648
|
+
|
|
649
|
+
/** Estimated prompt tokens for a message list (reuses condense.ts's estimator). */
|
|
650
|
+
export function estimatePromptTokens(messages: ChatMessage[]): number {
|
|
651
|
+
const chars = messages.reduce((sum, m) => sum + estimateMessageChars(m), 0)
|
|
652
|
+
return Math.ceil(chars / CHARS_PER_TOKEN)
|
|
653
|
+
}
|