headlesscode 1.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (232) hide show
  1. package/ATTRIBUTION.md +53 -0
  2. package/CODE_OF_CONDUCT.md +130 -0
  3. package/CONTRIBUTING.md +107 -0
  4. package/LICENSE +202 -0
  5. package/README.md +486 -0
  6. package/SECURITY.md +211 -0
  7. package/bin/headlesscode.mjs +83 -0
  8. package/package.json +63 -0
  9. package/shared/prompts/review-mode-prompt-short.md +93 -0
  10. package/shared/prompts/review-mode-prompt.md +281 -0
  11. package/shared/rules-code/rules.md +22 -0
  12. package/shared/stacks/cpp/rules.md +30 -0
  13. package/shared/stacks/fastapi/rules.md +30 -0
  14. package/shared/stacks/javascript/rules.md +37 -0
  15. package/shared/stacks/postgresql/rules.md +31 -0
  16. package/shared/stacks/python/rules.md +35 -0
  17. package/shared/stacks/react/rules.md +11 -0
  18. package/shared/stacks/typescript/rules.md +10 -0
  19. package/src/budget/budget.ts +221 -0
  20. package/src/budget/concurrency.ts +126 -0
  21. package/src/budget/cost.ts +309 -0
  22. package/src/budget/index.ts +8 -0
  23. package/src/checkpoints/cli.ts +256 -0
  24. package/src/checkpoints/service.ts +227 -0
  25. package/src/cli.ts +1535 -0
  26. package/src/cloud/docker-provider.ts +334 -0
  27. package/src/cloud/provider.ts +300 -0
  28. package/src/codeintel/call-graph.ts +78 -0
  29. package/src/codeintel/find-references.ts +123 -0
  30. package/src/codeintel/go-to-definition.ts +193 -0
  31. package/src/codeintel/handlers.ts +190 -0
  32. package/src/codeintel/import-graph.ts +173 -0
  33. package/src/codeintel/outline.ts +180 -0
  34. package/src/codeintel/position.ts +77 -0
  35. package/src/codeintel/program.ts +350 -0
  36. package/src/codeintel/rename-symbol.ts +213 -0
  37. package/src/codeintel/tools.ts +280 -0
  38. package/src/codemap/build.ts +135 -0
  39. package/src/codemap/cli.ts +190 -0
  40. package/src/codemap/extract.ts +339 -0
  41. package/src/codemap/files.ts +236 -0
  42. package/src/codemap/fingerprint.ts +65 -0
  43. package/src/codemap/flows.ts +62 -0
  44. package/src/codemap/html.ts +451 -0
  45. package/src/codemap/lock.ts +80 -0
  46. package/src/codemap/types.ts +101 -0
  47. package/src/codesearch/airunner-embedder.ts +185 -0
  48. package/src/codesearch/chunk.ts +339 -0
  49. package/src/codesearch/cli.ts +223 -0
  50. package/src/codesearch/embedder.ts +332 -0
  51. package/src/codesearch/files.ts +280 -0
  52. package/src/codesearch/index.ts +469 -0
  53. package/src/codesearch/ollama-embedder.ts +205 -0
  54. package/src/codesearch/search.ts +141 -0
  55. package/src/codesearch/types.ts +100 -0
  56. package/src/config/mode-models.ts +218 -0
  57. package/src/dashboard/aggregate.ts +364 -0
  58. package/src/dashboard/chat-thread.ts +141 -0
  59. package/src/dashboard/checkpoints.ts +124 -0
  60. package/src/dashboard/cli.ts +193 -0
  61. package/src/dashboard/codemap.ts +44 -0
  62. package/src/dashboard/files.ts +121 -0
  63. package/src/dashboard/page.ts +2803 -0
  64. package/src/dashboard/self-improvement-metrics.ts +282 -0
  65. package/src/dashboard/server.ts +1103 -0
  66. package/src/dashboard/session-launch.ts +310 -0
  67. package/src/dashboard/timeline.ts +273 -0
  68. package/src/dashboard/tool-exec.ts +107 -0
  69. package/src/dashboard/trend-cli.ts +141 -0
  70. package/src/dashboard/trend.ts +413 -0
  71. package/src/decision-proxy/cli.ts +261 -0
  72. package/src/decision-proxy/proxy.ts +569 -0
  73. package/src/deploy/gate-cli.ts +147 -0
  74. package/src/deploy/gate.ts +254 -0
  75. package/src/engine/condense.ts +512 -0
  76. package/src/engine/events.ts +428 -0
  77. package/src/engine/handoff.ts +71 -0
  78. package/src/engine/lazy-tools.ts +160 -0
  79. package/src/engine/local-explore.ts +653 -0
  80. package/src/engine/logger.ts +96 -0
  81. package/src/engine/loop.ts +5517 -0
  82. package/src/engine/parser.ts +347 -0
  83. package/src/engine/prompt.ts +860 -0
  84. package/src/engine/reports.ts +47 -0
  85. package/src/engine/stacks.ts +448 -0
  86. package/src/engine/types.ts +291 -0
  87. package/src/engine/usage.ts +186 -0
  88. package/src/github/app-auth.ts +161 -0
  89. package/src/github/cli.ts +448 -0
  90. package/src/github/installations.ts +133 -0
  91. package/src/github/pr.ts +321 -0
  92. package/src/github/provision.ts +118 -0
  93. package/src/github/push.ts +122 -0
  94. package/src/index-util.ts +50 -0
  95. package/src/index.ts +81 -0
  96. package/src/init/cli.ts +248 -0
  97. package/src/init/gitignore.ts +74 -0
  98. package/src/llm/ollama.ts +308 -0
  99. package/src/llm/openrouter.ts +868 -0
  100. package/src/llm/preflight.ts +367 -0
  101. package/src/llm/transcript-capture.ts +84 -0
  102. package/src/memory/embed.ts +110 -0
  103. package/src/memory/index.ts +22 -0
  104. package/src/memory/local.ts +259 -0
  105. package/src/memory/summarizer.ts +283 -0
  106. package/src/memory/types.ts +153 -0
  107. package/src/memory/uwuchat.ts +157 -0
  108. package/src/migrate/cli.ts +115 -0
  109. package/src/orchestrator/analyze-cli.ts +104 -0
  110. package/src/orchestrator/auto-split.ts +206 -0
  111. package/src/orchestrator/cleanup.ts +1003 -0
  112. package/src/orchestrator/cli.ts +3571 -0
  113. package/src/orchestrator/cost-estimate.ts +564 -0
  114. package/src/orchestrator/cost-history-cli.ts +242 -0
  115. package/src/orchestrator/cost-history.ts +397 -0
  116. package/src/orchestrator/git-sync.ts +250 -0
  117. package/src/orchestrator/index.ts +153 -0
  118. package/src/orchestrator/log-analysis.ts +0 -0
  119. package/src/orchestrator/merge-check.ts +108 -0
  120. package/src/orchestrator/pipeline.ts +411 -0
  121. package/src/orchestrator/resume.ts +1940 -0
  122. package/src/orchestrator/reviewer.ts +503 -0
  123. package/src/orchestrator/split.ts +296 -0
  124. package/src/orchestrator/state.ts +542 -0
  125. package/src/orchestrator/status.ts +697 -0
  126. package/src/orchestrator/verification-gate.ts +134 -0
  127. package/src/orchestrator/watch.ts +898 -0
  128. package/src/permissions/commands.ts +1083 -0
  129. package/src/permissions/config.ts +241 -0
  130. package/src/permissions/index.ts +12 -0
  131. package/src/permissions/protected-files.ts +96 -0
  132. package/src/permissions/store-protection.ts +272 -0
  133. package/src/project-store.ts +648 -0
  134. package/src/projects/cli.ts +382 -0
  135. package/src/qa/qa.ts +487 -0
  136. package/src/tools/browser/handler.ts +346 -0
  137. package/src/tools/browser/service.ts +406 -0
  138. package/src/tools/browser/smoke.ts +78 -0
  139. package/src/tools/browser/tool.ts +99 -0
  140. package/src/tools/executor.ts +2575 -0
  141. package/src/tools/language-detect.ts +183 -0
  142. package/src/tools/output-summarizer.ts +369 -0
  143. package/src/tools/run-tests.ts +302 -0
  144. package/src/tools/set-indentation-tool.ts +49 -0
  145. package/src/tools/test-selection.ts +160 -0
  146. package/src/vendor/tests/smoke.ts +103 -0
  147. package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
  148. package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
  149. package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
  150. package/src/vendor/zoo-code/shim/os-name.ts +18 -0
  151. package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
  152. package/src/vendor/zoo-code/shim/vscode.ts +76 -0
  153. package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
  154. package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
  155. package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
  156. package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
  157. package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
  158. package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
  159. package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
  160. package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
  161. package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
  162. package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
  163. package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
  164. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
  165. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
  166. package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
  167. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
  168. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
  169. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
  170. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
  171. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
  172. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
  173. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
  174. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
  175. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
  176. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
  177. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
  178. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
  179. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
  180. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
  181. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
  182. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
  183. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
  184. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
  185. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
  186. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
  187. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
  188. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
  189. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
  190. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
  191. package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
  192. package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
  193. package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
  194. package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
  195. package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
  196. package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
  197. package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
  198. package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
  199. package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
  200. package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
  201. package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
  202. package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
  203. package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
  204. package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
  205. package/src/vendor/zoo-code/src/shared/language.ts +43 -0
  206. package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
  207. package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
  208. package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
  209. package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
  210. package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
  211. package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
  212. package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
  213. package/src/vendor/zoo-code/src/utils/object.ts +18 -0
  214. package/src/vendor/zoo-code/src/utils/path.ts +94 -0
  215. package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
  216. package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
  217. package/src/vendor/zoo-code/types/global-settings.ts +19 -0
  218. package/src/vendor/zoo-code/types/index.ts +22 -0
  219. package/src/vendor/zoo-code/types/message.ts +375 -0
  220. package/src/vendor/zoo-code/types/mode.ts +241 -0
  221. package/src/vendor/zoo-code/types/todo.ts +19 -0
  222. package/src/vendor/zoo-code/types/tool-params.ts +116 -0
  223. package/src/vendor/zoo-code/types/tool.ts +67 -0
  224. package/src/vendor/zoo-code/types/vscode.ts +84 -0
  225. package/src/vision/describe.ts +242 -0
  226. package/src/vision/tool.ts +91 -0
  227. package/src/watcher/cli.ts +369 -0
  228. package/src/watcher/github.ts +304 -0
  229. package/src/watcher/index.ts +59 -0
  230. package/src/watcher/state.ts +254 -0
  231. package/src/watcher/watch.ts +562 -0
  232. package/tsconfig.json +18 -0
@@ -0,0 +1,183 @@
1
+ /**
2
+ * Cheap workspace-language detection used to CONDITIONALLY register tools that
3
+ * only work for specific languages (Part D of the central-store round):
4
+ *
5
+ * - The `ts.Program`-based code-intelligence tools (outline, go_to_definition,
6
+ * find_references, import_graph, rename_symbol — src/codeintel/) only scan
7
+ * TS/JS files, and
8
+ * - `run_tests`'s direct-match discovery + `tsx` invocation are TS-specific.
9
+ *
10
+ * On a real target project like a C++ or Python repo these tools are dead
11
+ * weight — advertised on every request, costing prompt tokens, silently
12
+ * useless if tried. `codebase_search` already handles many languages (its
13
+ * chunker has a broad extension list), so the bar is: gate the TS-only tools
14
+ * to TS/JS workspaces the same way `codebase_search` is language-generic.
15
+ *
16
+ * The detector is deliberately crude — marker files at the root plus a
17
+ * BOUNDED extension-frequency walk (see MAX_SCAN_FILES) — accurate enough to
18
+ * gate on, and cheap enough to run synchronously at every executor creation.
19
+ * It is NOT a language taxonomy: it only needs to answer "is this workspace
20
+ * TS/JS (so the TS-only tools make sense)?" and, secondarily, what other
21
+ * languages are present for future generic-tool gating.
22
+ */
23
+
24
+ import * as fs from "node:fs"
25
+ import * as path from "node:path"
26
+
27
+ /** Workspace languages the detector can recognize. */
28
+ export type WorkspaceLanguage = "typescript" | "python" | "cpp" | "rust" | "go" | "java"
29
+
30
+ /**
31
+ * Directory names never scanned (same spirit as src/codesearch/files.ts's
32
+ * FALLBACK_EXCLUDES — generated artifacts and VCS dirs). Anything not listed
33
+ * is fair game: the detector is extension-frequency-based and cheap.
34
+ */
35
+ const SKIP_DIR_NAMES = new Set([
36
+ ".git",
37
+ ".headlesscode",
38
+ ".worktrees",
39
+ ".roo",
40
+ ".idea",
41
+ ".vscode",
42
+ ".next",
43
+ ".nuxt",
44
+ ".terraform",
45
+ ".terragrunt-cache",
46
+ ".pytest_cache",
47
+ "node_modules",
48
+ "dist",
49
+ "build",
50
+ "out",
51
+ "coverage",
52
+ "target",
53
+ "__pycache__",
54
+ "venv",
55
+ ".venv",
56
+ "env",
57
+ "vendor",
58
+ "Pods",
59
+ "bin",
60
+ "obj",
61
+ ])
62
+
63
+ /** Root marker files that imply a language even with zero source files yet. */
64
+ const MARKER_FILES: ReadonlyArray<{ name: string; language: WorkspaceLanguage }> = [
65
+ { name: "package.json", language: "typescript" },
66
+ { name: "tsconfig.json", language: "typescript" },
67
+ { name: "jsconfig.json", language: "typescript" },
68
+ { name: "pyproject.toml", language: "python" },
69
+ { name: "setup.py", language: "python" },
70
+ { name: "setup.cfg", language: "python" },
71
+ { name: "requirements.txt", language: "python" },
72
+ { name: "Pipfile", language: "python" },
73
+ { name: "CMakeLists.txt", language: "cpp" },
74
+ { name: "Cargo.toml", language: "rust" },
75
+ { name: "go.mod", language: "go" },
76
+ { name: "pom.xml", language: "java" },
77
+ { name: "build.gradle", language: "java" },
78
+ { name: "build.gradle.kts", language: "java" },
79
+ { name: "settings.gradle", language: "java" },
80
+ ]
81
+
82
+ /** Extension → language. `.h`/`.c` alone do NOT imply C++ (plain C is common). */
83
+ const EXTENSION_LANGUAGES: ReadonlyArray<{ exts: readonly string[]; language: WorkspaceLanguage }> = [
84
+ { exts: [".ts", ".tsx", ".mts", ".cts", ".js", ".jsx", ".mjs", ".cjs"], language: "typescript" },
85
+ { exts: [".py"], language: "python" },
86
+ { exts: [".cpp", ".cc", ".cxx", ".c++", ".hpp", ".hh", ".hxx"], language: "cpp" },
87
+ { exts: [".rs"], language: "rust" },
88
+ { exts: [".go"], language: "go" },
89
+ { exts: [".java"], language: "java" },
90
+ ]
91
+
92
+ /** Cap on files examined during the extension walk (cheap + deterministic). */
93
+ export const MAX_SCAN_FILES = 4_000
94
+
95
+ /** Cap on directories descended during the walk (prevents pathological trees). */
96
+ export const MAX_SCAN_DIRS = 2_000
97
+
98
+ /** Number of same-extension hits required to count a language (1 is enough to gate on). */
99
+ const EXTENSION_COUNT_THRESHOLD = 1
100
+
101
+ /**
102
+ * Detect the languages present in a workspace. Root marker files are checked
103
+ * first (cheap + authoritative for tooling-heavy repos); then a bounded
104
+ * recursive walk counts source extensions. Never throws — a missing/unreadable
105
+ * workspace returns the empty set.
106
+ */
107
+ export function detectWorkspaceLanguages(workspaceRoot: string): Set<WorkspaceLanguage> {
108
+ const root = path.resolve(workspaceRoot)
109
+ const detected = new Set<WorkspaceLanguage>()
110
+
111
+ // 1. Root marker files.
112
+ let rootEntries: fs.Dirent[]
113
+ try {
114
+ rootEntries = fs.readdirSync(root, { withFileTypes: true })
115
+ } catch {
116
+ return detected
117
+ }
118
+ for (const ent of rootEntries) {
119
+ if (!ent.isFile() && !ent.isSymbolicLink()) {
120
+ continue
121
+ }
122
+ for (const { name, language } of MARKER_FILES) {
123
+ if (ent.name === name) {
124
+ detected.add(language)
125
+ }
126
+ }
127
+ }
128
+
129
+ // 2. Bounded extension-frequency walk. Once every language we care about is
130
+ // found, stop early.
131
+ const wanted = new Set<WorkspaceLanguage>(["typescript", "python", "cpp", "rust", "go", "java"])
132
+ const counts = new Map<WorkspaceLanguage, number>()
133
+ let scanned = 0
134
+ let dirs = 0
135
+
136
+ const walk = (dir: string): void => {
137
+ if (dirs >= MAX_SCAN_DIRS || scanned >= MAX_SCAN_FILES) {
138
+ return
139
+ }
140
+ dirs++
141
+ let entries: fs.Dirent[]
142
+ try {
143
+ entries = fs.readdirSync(dir, { withFileTypes: true })
144
+ } catch {
145
+ return
146
+ }
147
+ for (const ent of entries) {
148
+ if (scanned >= MAX_SCAN_FILES) {
149
+ return
150
+ }
151
+ if (ent.isDirectory()) {
152
+ if (!SKIP_DIR_NAMES.has(ent.name) && !ent.name.startsWith(".")) {
153
+ walk(path.join(dir, ent.name))
154
+ }
155
+ continue
156
+ }
157
+ if (!ent.isFile()) {
158
+ continue
159
+ }
160
+ scanned++
161
+ const ext = path.extname(ent.name).toLowerCase()
162
+ for (const { exts, language } of EXTENSION_LANGUAGES) {
163
+ if (exts.includes(ext)) {
164
+ counts.set(language, (counts.get(language) ?? 0) + 1)
165
+ break
166
+ }
167
+ }
168
+ }
169
+ }
170
+ walk(root)
171
+
172
+ for (const lang of wanted) {
173
+ if ((counts.get(lang) ?? 0) >= EXTENSION_COUNT_THRESHOLD) {
174
+ detected.add(lang)
175
+ }
176
+ }
177
+ return detected
178
+ }
179
+
180
+ /** Convenience: is this workspace TS/JS (the codeintel/run_tests gate)? */
181
+ export function isTypeScriptWorkspace(workspaceRoot: string): boolean {
182
+ return detectWorkspaceLanguages(workspaceRoot).has("typescript")
183
+ }
@@ -0,0 +1,369 @@
1
+ /**
2
+ * Local-model output summarization for oversized tool results
3
+ * (opt-in — see the `HEADLESSCODE_LOCAL_SUMMARIZATION` env var).
4
+ *
5
+ * ─── Why this exists ─────────────────────────────────────────────────────────
6
+ *
7
+ * `src/tools/executor.ts`'s MAX_RESULT_CHARS (30,000) hard-truncates any tool
8
+ * result before it reaches the model. For large, mostly-noisy command output
9
+ * (a verbose test run, a big `npm install` log, a large `grep -r`) that blunt
10
+ * cut discards whatever sat past the cutoff even when it contained the one
11
+ * relevant line. This module gives an opted-in session a way to have a small
12
+ * LOCAL model (via Ollama) compress that output instead — fast, zero marginal
13
+ * $ cost, and deliberately NOT involved in any coding decision: it only
14
+ * rewrites text the cloud model will read.
15
+ *
16
+ * ─── Ollama chat API — verified shape (2026-08-01, real local calls) ────────
17
+ *
18
+ * Verified live against `http://localhost:11434` (Ollama 0.24.0) — this is
19
+ * the CHAT endpoint, distinct from the `/api/embed` embeddings endpoint that
20
+ * `src/codesearch/` uses (see src/codesearch/ollama-embedder.ts if present):
21
+ *
22
+ * POST /api/chat
23
+ * body: {
24
+ * model: "qwen3:8b",
25
+ * messages: [{ role: "system", content: ... }, { role: "user", content: ... }],
26
+ * stream: false,
27
+ * think: false, // qwen3-class models only (see below)
28
+ * options: { num_predict, temperature }
29
+ * }
30
+ * 200 response: {
31
+ * model, created_at,
32
+ * message: { role: "assistant", content: "...", thinking?: "..." },
33
+ * done: true, done_reason: "stop" | "length",
34
+ * total_duration, load_duration, prompt_eval_count, eval_count, ...
35
+ * }
36
+ *
37
+ * VERIFIED FINDING 1 — qwen3-class models reason by default: with a plain
38
+ * request (no `think` field), `qwen3:8b` fills `message.thinking` and leaves
39
+ * `message.content` EMPTY until the token budget is exhausted. The request
40
+ * MUST send `think: false` for those models (llama3.1 ignores the field).
41
+ *
42
+ * VERIFIED FINDING 2 — a cold model takes ~3-6s to load into VRAM before the
43
+ * first token; a warm call is ~1-2s for a ~7KB input. Ollama holds the model
44
+ * resident after the first call, so consecutive oversized outputs in one
45
+ * session are cheap.
46
+ *
47
+ * VERIFIED FINDING 3 — a summarization prompt that only says "keep errors
48
+ * verbatim" makes the model DISCARD listing content (grep output) entirely.
49
+ * The prompt must tell it what the output IS (error log vs listing) and that
50
+ * it must never invent labels/content (llama3.1:8b fabricated "Error:",
51
+ * "Exit code: 1" and merged code lines when not explicitly forbidden).
52
+ *
53
+ * ─── Failure contract (non-fatal, matches loop.ts's idiom) ──────────────────
54
+ *
55
+ * EVERY failure path in here throws; the executor catches and falls back to
56
+ * its existing blunt truncation. A summarizer must NEVER turn a tool result
57
+ * into an error or block the session. Timeout, unreachable Ollama, HTTP
58
+ * error, non-JSON body, missing `message.content`, empty content — all
59
+ * throw, all fall back.
60
+ */
61
+
62
+ import type { Logger } from "../engine/logger.js"
63
+
64
+ /** Env var that gates the whole feature (default OFF — see executor.ts). */
65
+ export const LOCAL_SUMMARIZATION_ENV = "HEADLESSCODE_LOCAL_SUMMARIZATION"
66
+
67
+ /** Ollama base URL override. */
68
+ export const OLLAMA_URL_ENV = "HEADLESSCODE_OLLAMA_URL"
69
+
70
+ /** Default Ollama base URL (local default install). */
71
+ export const DEFAULT_OLLAMA_URL = "http://localhost:11434"
72
+
73
+ /** Default local chat model used for summarization. */
74
+ export const DEFAULT_SUMMARIZATION_MODEL = "qwen3:8b"
75
+
76
+ /** Model override env var. */
77
+ export const SUMMARIZATION_MODEL_ENV = "HEADLESSCODE_SUMMARIZATION_MODEL"
78
+
79
+ /** Request timeout: a slow/broken local model must never block the loop. */
80
+ export const DEFAULT_SUMMARIZATION_TIMEOUT_MS = 15_000
81
+
82
+ /**
83
+ * Hard cap on the raw content we are willing to SEND to the local model. A
84
+ * pathological multi-MB output should not be uploaded to the local server at
85
+ * full size; beyond this the model sees only the first `MAX_RESULT_CHARS`
86
+ * window (matching the blunt truncation) and summarizes that window. Keeps a
87
+ * 500KB command log from becoming a 500KB local request.
88
+ */
89
+ export const MAX_SUMMARIZER_INPUT_CHARS = 60_000
90
+
91
+ /** Blunt-truncation cap used by the summarizer's own fallback (matches executor MAX_RESULT_CHARS). */
92
+ export const MAX_RESULT_CHARS_FOR_FALLBACK = 30_000
93
+
94
+ /** Soft cap on summary length the model is asked to stay under. */
95
+ export const SUMMARIZATION_TARGET_CHARS = 400
96
+
97
+ /** Hard safety cap on what the summarizer is ALLOWED to return. */
98
+ export const MAX_SUMMARY_CHARS = 8_000
99
+
100
+ /**
101
+ * Whether local output summarization is enabled when
102
+ * HEADLESSCODE_LOCAL_SUMMARIZATION is unset. Measured decision (2026-08-15,
103
+ * r3-summarize round): oversized execute_command results that would engage
104
+ * the summarizer occur in ~1.8% of real exec results (~0.5/session), saving
105
+ * ~4k input tokens/session (mostly provider-cache-covered) at +2.5-3.4s
106
+ * latency per result — not material enough to impose on every deployment,
107
+ * and summary quality is model-dependent. Flipping to true is a one-line
108
+ * default change; the env var then acts as the opt-out ("0"/"false").
109
+ */
110
+ export const LOCAL_SUMMARIZATION_DEFAULT_ENABLED = false
111
+
112
+ /** Env-var gate: is local summarization enabled for this process? */
113
+ export function isLocalSummarizationEnabled(env: NodeJS.ProcessEnv = process.env): boolean {
114
+ if (env[LOCAL_SUMMARIZATION_ENV] !== undefined) {
115
+ const v = env[LOCAL_SUMMARIZATION_ENV].toLowerCase()
116
+ return v === "1" || v === "true"
117
+ }
118
+ return LOCAL_SUMMARIZATION_DEFAULT_ENABLED
119
+ }
120
+
121
+ /** Resolve the Ollama base URL: env → default. */
122
+ export function resolveOllamaUrl(env: NodeJS.ProcessEnv = process.env): string {
123
+ return (env[OLLAMA_URL_ENV]?.trim() || DEFAULT_OLLAMA_URL).replace(/\/+$/, "")
124
+ }
125
+
126
+ /** Resolve the summarization model: env → default. */
127
+ export function resolveSummarizationModel(env: NodeJS.ProcessEnv = process.env): string {
128
+ return env[SUMMARIZATION_MODEL_ENV]?.trim() || DEFAULT_SUMMARIZATION_MODEL
129
+ }
130
+
131
+ /** Result of a successful local summarization. */
132
+ export interface SummarizeResult {
133
+ /** The compressed output (never exceeds MAX_SUMMARY_CHARS). */
134
+ summary: string
135
+ /** Original length in characters (for the transparency header). */
136
+ originalChars: number
137
+ /** True when the response was cut off by the model's token budget. */
138
+ truncated: boolean
139
+ /** Wall-clock time the local call took, ms. */
140
+ elapsedMs: number
141
+ }
142
+
143
+ /** Options for OllamaOutputSummarizer. */
144
+ export interface OutputSummarizerOptions {
145
+ baseUrl?: string
146
+ model?: string
147
+ timeoutMs?: number
148
+ /**
149
+ * Injectable fetch for tests. Must accept a RequestInfo/URL + init and
150
+ * return a Response-like object (the real `fetch` signature).
151
+ */
152
+ fetchImpl?: typeof fetch
153
+ }
154
+
155
+ /**
156
+ * Summarize large tool output via a local Ollama chat model.
157
+ *
158
+ * System prompt: extractive, anti-hallucination, output-type-aware — the
159
+ * evaluation found this exact combination is what makes a small local model
160
+ * keep the needle instead of inventing one or discarding the whole listing.
161
+ */
162
+ export class OllamaOutputSummarizer {
163
+ /**
164
+ * Optional explicit base URL/model (used by tests / non-env callers).
165
+ * When absent, resolved from process.env on EACH summarize() call, so a
166
+ * process whose env changes (tests) always hits the right endpoint.
167
+ */
168
+ private readonly baseUrlOverride: string | undefined
169
+ private readonly modelOverride: string | undefined
170
+ /** Model id used for summarization (surfaced in the transparency header). */
171
+ readonly model: string
172
+ private readonly timeoutMs: number
173
+ private readonly fetchImpl: typeof fetch
174
+
175
+ constructor(options: OutputSummarizerOptions = {}) {
176
+ this.baseUrlOverride = options.baseUrl
177
+ this.modelOverride = options.model
178
+ // When no explicit model was given, resolve now so `readonly model` is
179
+ // stable for the header even if env changes later; per-call resolution
180
+ // below only affects the URL when no override is present.
181
+ this.model = options.model ?? resolveSummarizationModel()
182
+ this.timeoutMs = options.timeoutMs ?? DEFAULT_SUMMARIZATION_TIMEOUT_MS
183
+ this.fetchImpl = options.fetchImpl ?? fetch
184
+ }
185
+
186
+ /** Base URL used for the next request: explicit override, else env-per-call. */
187
+ private currentBaseUrl(): string {
188
+ return this.baseUrlOverride ?? resolveOllamaUrl()
189
+ }
190
+
191
+ /**
192
+ * Compress `rawOutput`. Throws on any failure — the caller (the executor)
193
+ * catches and falls back to blunt truncation. Never resolves with an empty
194
+ * or oversized summary.
195
+ */
196
+ async summarize(rawOutput: string): Promise<SummarizeResult> {
197
+ const started = Date.now()
198
+ const controller = new AbortController()
199
+ const timer = setTimeout(() => controller.abort(), this.timeoutMs)
200
+ const baseUrl = this.currentBaseUrl()
201
+
202
+ let response: Response
203
+ try {
204
+ response = await this.fetchImpl(`${baseUrl}/api/chat`, {
205
+ method: "POST",
206
+ headers: { "Content-Type": "application/json" },
207
+ body: JSON.stringify({
208
+ model: this.model,
209
+ messages: [
210
+ { role: "system", content: SUMMARIZATION_SYSTEM_PROMPT },
211
+ { role: "user", content: buildSummarizeUserPrompt(rawOutput) },
212
+ ],
213
+ stream: false,
214
+ // qwen3-class models reason by default and leave content
215
+ // empty; think:false forces a direct answer (verified live,
216
+ // see header). Harmless for models that ignore the field.
217
+ think: false,
218
+ options: { num_predict: 700, temperature: 0 },
219
+ }),
220
+ signal: controller.signal,
221
+ })
222
+ } catch (error) {
223
+ throw new SummarizerError(
224
+ error instanceof Error && error.name === "AbortError"
225
+ ? `local summarization timed out after ${this.timeoutMs}ms (${this.model} @ ${baseUrl})`
226
+ : `local summarization request failed (Ollama unreachable?): ${
227
+ error instanceof Error ? error.message : String(error)
228
+ }`,
229
+ )
230
+ } finally {
231
+ clearTimeout(timer)
232
+ }
233
+
234
+ if (!response.ok) {
235
+ const body = await response.text().catch(() => "")
236
+ throw new SummarizerError(
237
+ `local summarization returned HTTP ${response.status} from ${baseUrl} (model ${this.model}): ${excerpt(body)}`,
238
+ )
239
+ }
240
+
241
+ const rawBody = await response.text()
242
+ let data:
243
+ | {
244
+ message?: { content?: string; thinking?: string }
245
+ done_reason?: string
246
+ }
247
+ | undefined
248
+ try {
249
+ data = JSON.parse(rawBody) as { message?: { content?: string }; done_reason?: string }
250
+ } catch {
251
+ throw new SummarizerError(
252
+ `local summarization returned a non-JSON body from ${baseUrl}: ${excerpt(rawBody) || "(empty)"}`,
253
+ )
254
+ }
255
+
256
+ // A qwen3-class model that ignored `think: false` (or an old Ollama that
257
+ // doesn't support the field) leaves content empty — treat as a failure
258
+ // rather than sending the model an empty summary.
259
+ const content = data?.message?.content
260
+ if (typeof content !== "string" || content.trim() === "") {
261
+ throw new SummarizerError(
262
+ `local summarization returned an empty message.content (model ${this.model} may have put everything in 'thinking'; ${
263
+ this.model.startsWith("qwen3") ? "is think:false supported by this Ollama version?" : ""
264
+ }). Raw body: ${excerpt(rawBody)}`,
265
+ )
266
+ }
267
+
268
+ // Hard safety cap: a runaway model response must never blow the context
269
+ // budget this feature exists to protect. If it exceeds the cap we fall
270
+ // back to blunt truncation rather than serving a summary bigger than
271
+ // the original truncation.
272
+ if (content.length > MAX_SUMMARY_CHARS) {
273
+ throw new SummarizerError(
274
+ `local summarization produced ${content.length} chars (cap ${MAX_SUMMARY_CHARS}) — falling back to blunt truncation`,
275
+ )
276
+ }
277
+
278
+ return {
279
+ summary: content.trim(),
280
+ originalChars: rawOutput.length,
281
+ truncated: data.done_reason === "length",
282
+ elapsedMs: Date.now() - started,
283
+ }
284
+ }
285
+ }
286
+
287
+ /** Typed error for every summarizer failure (caught by the executor). */
288
+ export class SummarizerError extends Error {
289
+ constructor(message: string) {
290
+ super(message)
291
+ this.name = "SummarizerError"
292
+ }
293
+ }
294
+
295
+ /**
296
+ * System prompt — the anti-hallucination + output-type rules are load-bearing
297
+ * (see header VERIFIED FINDING 3). Extractive, not generative.
298
+ */
299
+ export const SUMMARIZATION_SYSTEM_PROMPT =
300
+ "Your job: compress LARGE command output for a software engineering agent. This is a TOOL RESULT " +
301
+ "(stdout/stderr of a command the agent ran). Compress it while preserving the information a coding " +
302
+ "agent needs. HARD RULES:\n" +
303
+ "1. NEVER invent or add content. Do NOT add 'Error:', 'Warning:', 'Exit code:' labels, do NOT claim " +
304
+ "something is an error unless the output literally contains that error, do NOT reorder or synthesize. " +
305
+ "You may only quote, condense, or omit.\n" +
306
+ "2. Keep failures, errors, warnings and their exact messages VERBATIM — this is the single most " +
307
+ "important thing.\n" +
308
+ "3. If the output is a listing (grep matches, file lists, test names), keep the listed entries with " +
309
+ "their locations — the entries ARE the content.\n" +
310
+ "4. If there is nothing important, say so in one short line — do not fabricate.\n" +
311
+ "5. No meta-commentary, no preamble like 'Here is', no advice. Output the compressed content only."
312
+
313
+ /** Build the user prompt for a given raw output. */
314
+ export function buildSummarizeUserPrompt(rawOutput: string): string {
315
+ return (
316
+ `The agent ran a command and got this tool output (${rawOutput.length} chars). ` +
317
+ `Compress it to roughly ${SUMMARIZATION_TARGET_CHARS} chars or fewer, keeping anything important verbatim:\n\n` +
318
+ `=== OUTPUT BEGIN ===\n${rawOutput}\n=== OUTPUT END ===`
319
+ )
320
+ }
321
+
322
+ /** Truncate a raw body to a bounded excerpt for error messages. */
323
+ function excerpt(body: string): string {
324
+ return body.length > 500 ? `${body.slice(0, 500)}…` : body
325
+ }
326
+
327
+ /**
328
+ * Wrapper used by the executor's result path: run the summarizer, log the
329
+ * outcome, and fall back to blunt truncation on ANY failure. Mirrors the
330
+ * checkpoint/memory "non-fatal" idiom — the local model must never be able to
331
+ * turn a tool result into an error or block the session.
332
+ */
333
+ export async function summarizeToolResult(
334
+ content: string,
335
+ summarizer: OllamaOutputSummarizer,
336
+ logger: Pick<Logger, "debug" | "warn">,
337
+ ): Promise<string> {
338
+ try {
339
+ const result = await summarizer.summarize(content)
340
+ const header =
341
+ `[Output summarized by local model (${summarizer.model}) — original was ${result.originalChars} chars; ` +
342
+ `summary is ${result.summary.length} chars${result.truncated ? "; model hit its output budget, summary may be incomplete" : ""}]`
343
+ logger.debug(`[local-summ] summarized ${result.originalChars} chars -> ${result.summary.length} chars`, {
344
+ model: summarizer.model,
345
+ elapsedMs: result.elapsedMs,
346
+ truncated: result.truncated,
347
+ })
348
+ return `${header}\n${result.summary}`
349
+ } catch (error) {
350
+ // Non-fatal: fall back to today's blunt truncation, never an error.
351
+ logger.warn(
352
+ `[local-summ] summarization failed (non-fatal; falling back to blunt truncation): ${
353
+ error instanceof Error ? error.message : String(error)
354
+ }`,
355
+ )
356
+ return truncateFallback(content)
357
+ }
358
+ }
359
+
360
+ /** Today's exact blunt-truncation behavior (also exported for tests). */
361
+ export function truncateFallback(content: string, maxChars = MAX_RESULT_CHARS_FOR_FALLBACK): string {
362
+ if (content.length <= maxChars) {
363
+ return content
364
+ }
365
+ return (
366
+ content.slice(0, maxChars) +
367
+ `\n…[output truncated at ${maxChars} chars to keep context bounded]`
368
+ )
369
+ }