headlesscode 1.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/ATTRIBUTION.md +53 -0
- package/CODE_OF_CONDUCT.md +130 -0
- package/CONTRIBUTING.md +107 -0
- package/LICENSE +202 -0
- package/README.md +486 -0
- package/SECURITY.md +211 -0
- package/bin/headlesscode.mjs +83 -0
- package/package.json +63 -0
- package/shared/prompts/review-mode-prompt-short.md +93 -0
- package/shared/prompts/review-mode-prompt.md +281 -0
- package/shared/rules-code/rules.md +22 -0
- package/shared/stacks/cpp/rules.md +30 -0
- package/shared/stacks/fastapi/rules.md +30 -0
- package/shared/stacks/javascript/rules.md +37 -0
- package/shared/stacks/postgresql/rules.md +31 -0
- package/shared/stacks/python/rules.md +35 -0
- package/shared/stacks/react/rules.md +11 -0
- package/shared/stacks/typescript/rules.md +10 -0
- package/src/budget/budget.ts +221 -0
- package/src/budget/concurrency.ts +126 -0
- package/src/budget/cost.ts +309 -0
- package/src/budget/index.ts +8 -0
- package/src/checkpoints/cli.ts +256 -0
- package/src/checkpoints/service.ts +227 -0
- package/src/cli.ts +1535 -0
- package/src/cloud/docker-provider.ts +334 -0
- package/src/cloud/provider.ts +300 -0
- package/src/codeintel/call-graph.ts +78 -0
- package/src/codeintel/find-references.ts +123 -0
- package/src/codeintel/go-to-definition.ts +193 -0
- package/src/codeintel/handlers.ts +190 -0
- package/src/codeintel/import-graph.ts +173 -0
- package/src/codeintel/outline.ts +180 -0
- package/src/codeintel/position.ts +77 -0
- package/src/codeintel/program.ts +350 -0
- package/src/codeintel/rename-symbol.ts +213 -0
- package/src/codeintel/tools.ts +280 -0
- package/src/codemap/build.ts +135 -0
- package/src/codemap/cli.ts +190 -0
- package/src/codemap/extract.ts +339 -0
- package/src/codemap/files.ts +236 -0
- package/src/codemap/fingerprint.ts +65 -0
- package/src/codemap/flows.ts +62 -0
- package/src/codemap/html.ts +451 -0
- package/src/codemap/lock.ts +80 -0
- package/src/codemap/types.ts +101 -0
- package/src/codesearch/airunner-embedder.ts +185 -0
- package/src/codesearch/chunk.ts +339 -0
- package/src/codesearch/cli.ts +223 -0
- package/src/codesearch/embedder.ts +332 -0
- package/src/codesearch/files.ts +280 -0
- package/src/codesearch/index.ts +469 -0
- package/src/codesearch/ollama-embedder.ts +205 -0
- package/src/codesearch/search.ts +141 -0
- package/src/codesearch/types.ts +100 -0
- package/src/config/mode-models.ts +218 -0
- package/src/dashboard/aggregate.ts +364 -0
- package/src/dashboard/chat-thread.ts +141 -0
- package/src/dashboard/checkpoints.ts +124 -0
- package/src/dashboard/cli.ts +193 -0
- package/src/dashboard/codemap.ts +44 -0
- package/src/dashboard/files.ts +121 -0
- package/src/dashboard/page.ts +2803 -0
- package/src/dashboard/self-improvement-metrics.ts +282 -0
- package/src/dashboard/server.ts +1103 -0
- package/src/dashboard/session-launch.ts +310 -0
- package/src/dashboard/timeline.ts +273 -0
- package/src/dashboard/tool-exec.ts +107 -0
- package/src/dashboard/trend-cli.ts +141 -0
- package/src/dashboard/trend.ts +413 -0
- package/src/decision-proxy/cli.ts +261 -0
- package/src/decision-proxy/proxy.ts +569 -0
- package/src/deploy/gate-cli.ts +147 -0
- package/src/deploy/gate.ts +254 -0
- package/src/engine/condense.ts +512 -0
- package/src/engine/events.ts +428 -0
- package/src/engine/handoff.ts +71 -0
- package/src/engine/lazy-tools.ts +160 -0
- package/src/engine/local-explore.ts +653 -0
- package/src/engine/logger.ts +96 -0
- package/src/engine/loop.ts +5517 -0
- package/src/engine/parser.ts +347 -0
- package/src/engine/prompt.ts +860 -0
- package/src/engine/reports.ts +47 -0
- package/src/engine/stacks.ts +448 -0
- package/src/engine/types.ts +291 -0
- package/src/engine/usage.ts +186 -0
- package/src/github/app-auth.ts +161 -0
- package/src/github/cli.ts +448 -0
- package/src/github/installations.ts +133 -0
- package/src/github/pr.ts +321 -0
- package/src/github/provision.ts +118 -0
- package/src/github/push.ts +122 -0
- package/src/index-util.ts +50 -0
- package/src/index.ts +81 -0
- package/src/init/cli.ts +248 -0
- package/src/init/gitignore.ts +74 -0
- package/src/llm/ollama.ts +308 -0
- package/src/llm/openrouter.ts +868 -0
- package/src/llm/preflight.ts +367 -0
- package/src/llm/transcript-capture.ts +84 -0
- package/src/memory/embed.ts +110 -0
- package/src/memory/index.ts +22 -0
- package/src/memory/local.ts +259 -0
- package/src/memory/summarizer.ts +283 -0
- package/src/memory/types.ts +153 -0
- package/src/memory/uwuchat.ts +157 -0
- package/src/migrate/cli.ts +115 -0
- package/src/orchestrator/analyze-cli.ts +104 -0
- package/src/orchestrator/auto-split.ts +206 -0
- package/src/orchestrator/cleanup.ts +1003 -0
- package/src/orchestrator/cli.ts +3571 -0
- package/src/orchestrator/cost-estimate.ts +564 -0
- package/src/orchestrator/cost-history-cli.ts +242 -0
- package/src/orchestrator/cost-history.ts +397 -0
- package/src/orchestrator/git-sync.ts +250 -0
- package/src/orchestrator/index.ts +153 -0
- package/src/orchestrator/log-analysis.ts +0 -0
- package/src/orchestrator/merge-check.ts +108 -0
- package/src/orchestrator/pipeline.ts +411 -0
- package/src/orchestrator/resume.ts +1940 -0
- package/src/orchestrator/reviewer.ts +503 -0
- package/src/orchestrator/split.ts +296 -0
- package/src/orchestrator/state.ts +542 -0
- package/src/orchestrator/status.ts +697 -0
- package/src/orchestrator/verification-gate.ts +134 -0
- package/src/orchestrator/watch.ts +898 -0
- package/src/permissions/commands.ts +1083 -0
- package/src/permissions/config.ts +241 -0
- package/src/permissions/index.ts +12 -0
- package/src/permissions/protected-files.ts +96 -0
- package/src/permissions/store-protection.ts +272 -0
- package/src/project-store.ts +648 -0
- package/src/projects/cli.ts +382 -0
- package/src/qa/qa.ts +487 -0
- package/src/tools/browser/handler.ts +346 -0
- package/src/tools/browser/service.ts +406 -0
- package/src/tools/browser/smoke.ts +78 -0
- package/src/tools/browser/tool.ts +99 -0
- package/src/tools/executor.ts +2575 -0
- package/src/tools/language-detect.ts +183 -0
- package/src/tools/output-summarizer.ts +369 -0
- package/src/tools/run-tests.ts +302 -0
- package/src/tools/set-indentation-tool.ts +49 -0
- package/src/tools/test-selection.ts +160 -0
- package/src/vendor/tests/smoke.ts +103 -0
- package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
- package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
- package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
- package/src/vendor/zoo-code/shim/os-name.ts +18 -0
- package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
- package/src/vendor/zoo-code/shim/vscode.ts +76 -0
- package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
- package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
- package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
- package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
- package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
- package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
- package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
- package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
- package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
- package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
- package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
- package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
- package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
- package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
- package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
- package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
- package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
- package/src/vendor/zoo-code/src/shared/language.ts +43 -0
- package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
- package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
- package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
- package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
- package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
- package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
- package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
- package/src/vendor/zoo-code/src/utils/object.ts +18 -0
- package/src/vendor/zoo-code/src/utils/path.ts +94 -0
- package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
- package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
- package/src/vendor/zoo-code/types/global-settings.ts +19 -0
- package/src/vendor/zoo-code/types/index.ts +22 -0
- package/src/vendor/zoo-code/types/message.ts +375 -0
- package/src/vendor/zoo-code/types/mode.ts +241 -0
- package/src/vendor/zoo-code/types/todo.ts +19 -0
- package/src/vendor/zoo-code/types/tool-params.ts +116 -0
- package/src/vendor/zoo-code/types/tool.ts +67 -0
- package/src/vendor/zoo-code/types/vscode.ts +84 -0
- package/src/vision/describe.ts +242 -0
- package/src/vision/tool.ts +91 -0
- package/src/watcher/cli.ts +369 -0
- package/src/watcher/github.ts +304 -0
- package/src/watcher/index.ts +59 -0
- package/src/watcher/state.ts +254 -0
- package/src/watcher/watch.ts +562 -0
- package/tsconfig.json +18 -0
|
@@ -0,0 +1,469 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Persisted codebase-search index: storage + incremental build.
|
|
3
|
+
*
|
|
4
|
+
* Storage: the CENTRAL per-project data store's
|
|
5
|
+
* `codesearch/index.jsonl` (see src/project-store.ts) — one JSON line per
|
|
6
|
+
* chunk — { file, startLine, endLine, embedding, hash } (see
|
|
7
|
+
* src/codesearch/types.ts). Plain JSONL, no vector DB: brute-force cosine
|
|
8
|
+
* over every stored vector at query time is the deliberate scope for this
|
|
9
|
+
* project's scale. SCALING LIMIT (documented, accepted): cosine search is
|
|
10
|
+
* O(N) over all chunks in the index, so on very large repos (hundreds of
|
|
11
|
+
* thousands of chunks) queries get slow. A real vector DB (pgvector/Qdrant)
|
|
12
|
+
* or an HNSW index is the known future path if that ever becomes a problem
|
|
13
|
+
* — it is deliberately NOT solved here.
|
|
14
|
+
*
|
|
15
|
+
* The JSONL itself is read and written STREAMING (bounded blocks / batches),
|
|
16
|
+
* never as one monolithic string: a 2560-dim embedding is ~21KB serialized,
|
|
17
|
+
* so a repo like airunner (37k chunks) produces a ~780MB index that far
|
|
18
|
+
* exceeds V8's ~1GB max string length if materialized whole (see the
|
|
19
|
+
* `Invalid string length` crash this guards against). Stored embedding
|
|
20
|
+
* floats are rounded to 6 decimal places (float32-grade precision — cosine
|
|
21
|
+
* similarity is unaffected, and it cuts each vector's serialized size by
|
|
22
|
+
* ~60%).
|
|
23
|
+
*
|
|
24
|
+
* Build semantics: the content hash lets a re-index skip unchanged chunks
|
|
25
|
+
* instead of re-embedding the whole repo every time (the cost-control
|
|
26
|
+
* mechanism). A chunk is re-embedded only when its content changed, is new,
|
|
27
|
+
* or was previously indexed with a DIFFERENT backend or model (a switch
|
|
28
|
+
* would otherwise leave a mixed-dimension index that compares apples to
|
|
29
|
+
* oranges — local `qwen3-embedding:8b` = 4096 dims vs cloud
|
|
30
|
+
* `qwen/qwen3-embedding-4b` = 2560 dims). Chunks whose file was deleted are
|
|
31
|
+
* dropped from the new index.
|
|
32
|
+
*/
|
|
33
|
+
|
|
34
|
+
import * as crypto from "node:crypto"
|
|
35
|
+
import * as fs from "node:fs"
|
|
36
|
+
import * as fsp from "node:fs/promises"
|
|
37
|
+
import * as path from "node:path"
|
|
38
|
+
|
|
39
|
+
import { chunkFile, type Chunk } from "./chunk.js"
|
|
40
|
+
import {
|
|
41
|
+
DEFAULT_EMBEDDING_BACKEND,
|
|
42
|
+
EMBEDDING_BACKENDS,
|
|
43
|
+
METADATA_BACKEND_KEY,
|
|
44
|
+
METADATA_MODEL_KEY,
|
|
45
|
+
type Embedder,
|
|
46
|
+
type EmbeddingBackend,
|
|
47
|
+
} from "./embedder.js"
|
|
48
|
+
import { walkSourceFiles } from "./files.js"
|
|
49
|
+
import { resolveProjectDataDir } from "../project-store.js"
|
|
50
|
+
import { INDEX_RELATIVE_PATH, type IndexEntry } from "./types.js"
|
|
51
|
+
|
|
52
|
+
export interface IndexBuildResult {
|
|
53
|
+
/** Total files scanned (source files found by the walker). */
|
|
54
|
+
filesScanned: number
|
|
55
|
+
/** Chunks embedded in this build. */
|
|
56
|
+
chunksEmbedded: number
|
|
57
|
+
/** Chunks skipped because their content hash was unchanged (and model same). */
|
|
58
|
+
chunksSkipped: number
|
|
59
|
+
/** Chunks dropped because their file disappeared. */
|
|
60
|
+
chunksRemoved: number
|
|
61
|
+
/** Total chunks in the written index. */
|
|
62
|
+
totalChunks: number
|
|
63
|
+
/** Real prompt tokens consumed by embedding (for cost reporting). */
|
|
64
|
+
promptTokens: number
|
|
65
|
+
/** Model used for embedding. */
|
|
66
|
+
model: string
|
|
67
|
+
/** Backend used for embedding (openrouter | ollama). */
|
|
68
|
+
backend: EmbeddingBackend
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/** Absolute path of the index file for a workspace (central project store). */
|
|
72
|
+
export function indexFilePath(workspaceRoot: string): string {
|
|
73
|
+
return path.join(resolveProjectDataDir(workspaceRoot), INDEX_RELATIVE_PATH)
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/** Legacy workspace-relative index path (pre-central-store), for grace reads. */
|
|
77
|
+
export function legacyIndexFilePath(workspaceRoot: string): string {
|
|
78
|
+
return path.join(workspaceRoot, ".headlesscode", INDEX_RELATIVE_PATH)
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Size of each read block when loading the index. A bounded block keeps peak
|
|
83
|
+
* memory flat regardless of index size; reading the whole file as one string
|
|
84
|
+
* is impossible above V8's ~1GB string cap anyway.
|
|
85
|
+
*/
|
|
86
|
+
const LOAD_BLOCK_SIZE = 64 * 1024 * 1024
|
|
87
|
+
|
|
88
|
+
/** Parse one JSONL line into an IndexEntry; undefined for blank/malformed lines. */
|
|
89
|
+
function parseIndexLine(line: string): IndexEntry | undefined {
|
|
90
|
+
const trimmed = line.trim()
|
|
91
|
+
if (trimmed === "") {
|
|
92
|
+
return undefined
|
|
93
|
+
}
|
|
94
|
+
try {
|
|
95
|
+
const parsed = JSON.parse(trimmed) as Partial<IndexEntry>
|
|
96
|
+
if (
|
|
97
|
+
typeof parsed.file === "string" &&
|
|
98
|
+
typeof parsed.startLine === "number" &&
|
|
99
|
+
typeof parsed.endLine === "number" &&
|
|
100
|
+
Array.isArray(parsed.embedding) &&
|
|
101
|
+
parsed.embedding.every((v) => typeof v === "number") &&
|
|
102
|
+
typeof parsed.hash === "string"
|
|
103
|
+
) {
|
|
104
|
+
// Entries written before the Ollama/AIRunner backends existed
|
|
105
|
+
// carry no `backend` field — they were necessarily built with the
|
|
106
|
+
// only backend that existed then (openrouter). Any other backend
|
|
107
|
+
// value passes through as-is (validated against EMBEDDING_BACKENDS).
|
|
108
|
+
const backend =
|
|
109
|
+
typeof parsed.backend === "string" && (EMBEDDING_BACKENDS as readonly string[]).includes(parsed.backend)
|
|
110
|
+
? (parsed.backend as EmbeddingBackend)
|
|
111
|
+
: "openrouter"
|
|
112
|
+
return { ...parsed, backend } as IndexEntry
|
|
113
|
+
}
|
|
114
|
+
// Malformed lines are skipped (loose validation, matching the
|
|
115
|
+
// project's read-side idiom — see src/memory/local.ts's readJsonl).
|
|
116
|
+
} catch {
|
|
117
|
+
// skip unparseable line
|
|
118
|
+
}
|
|
119
|
+
return undefined
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/**
|
|
123
|
+
* Read a JSONL file's entries in bounded blocks ([] when missing/unreadable).
|
|
124
|
+
*
|
|
125
|
+
* `blockSize` is a test seam — production uses LOAD_BLOCK_SIZE. A line is
|
|
126
|
+
* never split across blocks (partial content is carried in a pending buffer),
|
|
127
|
+
* so lines arbitrarily larger than the block size still load correctly. One
|
|
128
|
+
* accepted imperfection: a multi-byte UTF-8 char straddling a block boundary
|
|
129
|
+
* decodes as a replacement char and the line is dropped by the loose
|
|
130
|
+
* validation above — irrelevant in practice because index lines are almost
|
|
131
|
+
* entirely ASCII (only file paths are strings).
|
|
132
|
+
*/
|
|
133
|
+
export function readJsonlEntries(file: string, blockSize: number = LOAD_BLOCK_SIZE): IndexEntry[] {
|
|
134
|
+
let fd: number
|
|
135
|
+
try {
|
|
136
|
+
fd = fs.openSync(file, "r")
|
|
137
|
+
} catch {
|
|
138
|
+
return []
|
|
139
|
+
}
|
|
140
|
+
const entries: IndexEntry[] = []
|
|
141
|
+
const buffer = Buffer.alloc(blockSize)
|
|
142
|
+
let pending = ""
|
|
143
|
+
let offset = 0
|
|
144
|
+
try {
|
|
145
|
+
for (;;) {
|
|
146
|
+
const bytesRead = fs.readSync(fd, buffer, 0, blockSize, offset)
|
|
147
|
+
if (bytesRead === 0) {
|
|
148
|
+
break
|
|
149
|
+
}
|
|
150
|
+
offset += bytesRead
|
|
151
|
+
pending += buffer.subarray(0, bytesRead).toString("utf-8")
|
|
152
|
+
let nl = pending.indexOf("\n")
|
|
153
|
+
while (nl !== -1) {
|
|
154
|
+
const entry = parseIndexLine(pending.slice(0, nl))
|
|
155
|
+
if (entry) {
|
|
156
|
+
entries.push(entry)
|
|
157
|
+
}
|
|
158
|
+
pending = pending.slice(nl + 1)
|
|
159
|
+
nl = pending.indexOf("\n")
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
// Final line when the file doesn't end in a newline.
|
|
163
|
+
if (pending.trim() !== "") {
|
|
164
|
+
const entry = parseIndexLine(pending)
|
|
165
|
+
if (entry) {
|
|
166
|
+
entries.push(entry)
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
} finally {
|
|
170
|
+
fs.closeSync(fd)
|
|
171
|
+
}
|
|
172
|
+
return entries
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
/** Load all entries from the index file ([] when missing/unreadable). */
|
|
176
|
+
export function loadIndex(workspaceRoot: string): IndexEntry[] {
|
|
177
|
+
const central = indexFilePath(workspaceRoot)
|
|
178
|
+
if (!fs.existsSync(central)) {
|
|
179
|
+
// Pre-migration grace: a legacy workspace-relative index is still
|
|
180
|
+
// READABLE (though not writable) until the real-machine migration
|
|
181
|
+
// moves it into the central store.
|
|
182
|
+
const legacy = legacyIndexFilePath(workspaceRoot)
|
|
183
|
+
if (fs.existsSync(legacy)) {
|
|
184
|
+
return readJsonlEntries(legacy)
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
return readJsonlEntries(central)
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
/** Absolute path of the index METADATA file (sibling of index.jsonl). */
|
|
191
|
+
export function indexMetadataFilePath(workspaceRoot: string): string {
|
|
192
|
+
return indexFilePath(workspaceRoot) + ".meta.json"
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
/** Legacy workspace-relative index metadata path (pre-central-store grace). */
|
|
196
|
+
export function legacyIndexMetadataFilePath(workspaceRoot: string): string {
|
|
197
|
+
return legacyIndexFilePath(workspaceRoot) + ".meta.json"
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
/** The backend + model an on-disk index was built with. */
|
|
201
|
+
export interface IndexMetadata {
|
|
202
|
+
/** Embedding backend that built the index. */
|
|
203
|
+
backend: EmbeddingBackend
|
|
204
|
+
/** Embedding model that built the index. */
|
|
205
|
+
model: string
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
/** Load the metadata sibling file; undefined when missing/unreadable/malformed. */
|
|
209
|
+
export function loadIndexMetadata(workspaceRoot: string): IndexMetadata | undefined {
|
|
210
|
+
let raw: string
|
|
211
|
+
try {
|
|
212
|
+
raw = fs.readFileSync(indexMetadataFilePath(workspaceRoot), "utf-8")
|
|
213
|
+
} catch {
|
|
214
|
+
// Pre-migration grace: read the legacy sibling when the central one is
|
|
215
|
+
// absent (see loadIndex).
|
|
216
|
+
try {
|
|
217
|
+
raw = fs.readFileSync(legacyIndexMetadataFilePath(workspaceRoot), "utf-8")
|
|
218
|
+
} catch {
|
|
219
|
+
return undefined
|
|
220
|
+
}
|
|
221
|
+
}
|
|
222
|
+
try {
|
|
223
|
+
const parsed = JSON.parse(raw) as Record<string, unknown>
|
|
224
|
+
// Generic 3-way mapping driven off EMBEDDING_BACKENDS, NOT a hardcoded
|
|
225
|
+
// binary (ollama ?: openrouter) ternary — an index built with the
|
|
226
|
+
// airunner backend would otherwise be silently misread as openrouter.
|
|
227
|
+
const rawBackend = parsed[METADATA_BACKEND_KEY]
|
|
228
|
+
const backend =
|
|
229
|
+
typeof rawBackend === "string" && (EMBEDDING_BACKENDS as readonly string[]).includes(rawBackend)
|
|
230
|
+
? (rawBackend as EmbeddingBackend)
|
|
231
|
+
: "openrouter"
|
|
232
|
+
const model = typeof parsed[METADATA_MODEL_KEY] === "string" ? parsed[METADATA_MODEL_KEY] : undefined
|
|
233
|
+
if (model === undefined) {
|
|
234
|
+
return undefined
|
|
235
|
+
}
|
|
236
|
+
return { backend, model }
|
|
237
|
+
} catch {
|
|
238
|
+
return undefined
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
/** Validate a new batch of embeddings before it enters the index. */
|
|
243
|
+
function checkEmbeddingDimensions(batchEmbeddings: number[][], expectedDim: number | undefined, batchStart: number): void {
|
|
244
|
+
const first = batchEmbeddings[0]
|
|
245
|
+
if (!first) {
|
|
246
|
+
return
|
|
247
|
+
}
|
|
248
|
+
const dim = first.length
|
|
249
|
+
if (dim === 0) {
|
|
250
|
+
throw new Error(
|
|
251
|
+
`codebase index: embedder returned a zero-dimensional embedding for chunk ${batchStart + 1} — the model produced no usable vector`,
|
|
252
|
+
)
|
|
253
|
+
}
|
|
254
|
+
if (expectedDim !== undefined && dim !== expectedDim) {
|
|
255
|
+
throw new Error(
|
|
256
|
+
`codebase index: embedder returned a ${dim}-dim vector where a ${expectedDim}-dim vector was expected (chunk ${batchStart + 1}) — ` +
|
|
257
|
+
`the embedding model changed dimension mid-build?`,
|
|
258
|
+
)
|
|
259
|
+
}
|
|
260
|
+
for (let j = 1; j < batchEmbeddings.length; j++) {
|
|
261
|
+
if (batchEmbeddings[j].length !== dim) {
|
|
262
|
+
throw new Error(
|
|
263
|
+
`codebase index: embedder returned vectors of mixed dimensions (${dim} and ${batchEmbeddings[j].length}) in one batch — refusing to write a corrupt index`,
|
|
264
|
+
)
|
|
265
|
+
}
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
/** sha256 hex of a chunk's text (the dedupe key). */
|
|
270
|
+
export function chunkHash(content: string): string {
|
|
271
|
+
return crypto.createHash("sha256").update(content, "utf-8").digest("hex")
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
/** Group index entries by (file, startLine) for lookup during incremental build. */
|
|
275
|
+
function indexLookup(entries: IndexEntry[]): Map<string, IndexEntry> {
|
|
276
|
+
const map = new Map<string, IndexEntry>()
|
|
277
|
+
for (const entry of entries) {
|
|
278
|
+
map.set(`${entry.file}:${entry.startLine}`, entry)
|
|
279
|
+
}
|
|
280
|
+
return map
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
/**
|
|
284
|
+
* Significant decimal places stored per embedding float. Full double
|
|
285
|
+
* precision is wasted here: embeddings are compared by cosine similarity,
|
|
286
|
+
* which tolerates float32-grade precision (~7 sig figs) with no measurable
|
|
287
|
+
* quality change, and each stored float costs ~1 char per significant digit
|
|
288
|
+
* in the JSONL. Rounding to 6 places shrinks a 2560-dim vector's serialized
|
|
289
|
+
* form by ~60% (53KB → 21KB), which is the difference between a ~780MB and
|
|
290
|
+
* a ~1.9GB index on a repo the size of airunner.
|
|
291
|
+
*/
|
|
292
|
+
export const EMBEDDING_FLOAT_DECIMALS = 6
|
|
293
|
+
|
|
294
|
+
/** Round an entry's embedding for serialization (does not mutate the entry). */
|
|
295
|
+
function serializeEntry(entry: IndexEntry): IndexEntry {
|
|
296
|
+
if (entry.embedding.every((v) => Math.abs(v * 10 ** EMBEDDING_FLOAT_DECIMALS - Math.round(v * 10 ** EMBEDDING_FLOAT_DECIMALS)) < 1e-9)) {
|
|
297
|
+
// Already rounded to the storage precision (reused unchanged entries
|
|
298
|
+
// from an earlier build) — skip the copy.
|
|
299
|
+
return entry
|
|
300
|
+
}
|
|
301
|
+
const scale = 10 ** EMBEDDING_FLOAT_DECIMALS
|
|
302
|
+
const embedding = entry.embedding.map((v) => Math.round(v * scale) / scale)
|
|
303
|
+
return { ...entry, embedding }
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
/** How many lines to buffer before flushing to the index file's write stream. */
|
|
307
|
+
const WRITE_BATCH_SIZE = 512
|
|
308
|
+
|
|
309
|
+
/**
|
|
310
|
+
* Write index entries to `file` as JSONL, streaming in bounded batches.
|
|
311
|
+
*
|
|
312
|
+
* Building one giant `entries.map(JSON.stringify).join("\n")` string crashes
|
|
313
|
+
* with `Invalid string length` once the serialized index exceeds V8's ~1GB
|
|
314
|
+
* string cap (reached at ~37k chunks × 2560-dim embeddings — see the header
|
|
315
|
+
* comment). Each flushed batch is at most ~11MB (512 × ~21KB), so any repo
|
|
316
|
+
* size works.
|
|
317
|
+
*/
|
|
318
|
+
export async function writeIndexEntries(file: string, entries: IndexEntry[]): Promise<void> {
|
|
319
|
+
const stream = fs.createWriteStream(file, { encoding: "utf-8" })
|
|
320
|
+
const writeError = new Promise<never>((_, reject) => {
|
|
321
|
+
stream.once("error", reject)
|
|
322
|
+
})
|
|
323
|
+
try {
|
|
324
|
+
for (let i = 0; i < entries.length; i += WRITE_BATCH_SIZE) {
|
|
325
|
+
const batch = entries.slice(i, i + WRITE_BATCH_SIZE)
|
|
326
|
+
const lines = batch.map((e) => JSON.stringify(serializeEntry(e))).join("\n") + "\n"
|
|
327
|
+
if (!stream.write(lines)) {
|
|
328
|
+
// Backpressure: wait for the stream to drain before writing more.
|
|
329
|
+
await Promise.race([
|
|
330
|
+
new Promise<void>((resolve) => stream.once("drain", resolve)),
|
|
331
|
+
writeError,
|
|
332
|
+
])
|
|
333
|
+
}
|
|
334
|
+
}
|
|
335
|
+
} finally {
|
|
336
|
+
await new Promise<void>((resolve) => stream.end(resolve))
|
|
337
|
+
// Surface a write error (e.g. ENOSPC) instead of silently truncating.
|
|
338
|
+
await Promise.race([
|
|
339
|
+
new Promise<void>((resolve) => stream.once("close", resolve)),
|
|
340
|
+
writeError,
|
|
341
|
+
])
|
|
342
|
+
}
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
/**
|
|
346
|
+
* Build (or incrementally refresh) the codebase index for a workspace.
|
|
347
|
+
*
|
|
348
|
+
* @param embedder embedder to use for changed/new chunks
|
|
349
|
+
* @param workspaceRoot workspace root
|
|
350
|
+
* @param onProgress optional progress callback (files scanned so far)
|
|
351
|
+
*/
|
|
352
|
+
export async function buildIndex(
|
|
353
|
+
embedder: Embedder,
|
|
354
|
+
workspaceRoot: string,
|
|
355
|
+
onProgress?: (scanned: number, total: number) => void,
|
|
356
|
+
backend: EmbeddingBackend = DEFAULT_EMBEDDING_BACKEND,
|
|
357
|
+
onEmbedProgress?: (embedded: number, total: number) => void,
|
|
358
|
+
): Promise<IndexBuildResult> {
|
|
359
|
+
const root = path.resolve(workspaceRoot)
|
|
360
|
+
const files = await walkSourceFiles(root)
|
|
361
|
+
|
|
362
|
+
const oldEntries = loadIndex(root)
|
|
363
|
+
const oldByKey = indexLookup(oldEntries)
|
|
364
|
+
|
|
365
|
+
const newEntries: IndexEntry[] = []
|
|
366
|
+
const chunksToEmbed: Array<{ chunk: Chunk; hash: string }> = []
|
|
367
|
+
let chunksSkipped = 0
|
|
368
|
+
let chunksRemoved = 0
|
|
369
|
+
// Dimension of the vectors this build produces — checked once per batch
|
|
370
|
+
// so a mid-build dimension change can never silently write a corrupt
|
|
371
|
+
// (mixed-dimension) index.
|
|
372
|
+
let embeddingDim: number | undefined
|
|
373
|
+
|
|
374
|
+
// Track which old (file:startLine) keys survive so we can drop entries for
|
|
375
|
+
// deleted/rewritten files.
|
|
376
|
+
const survivingKeys = new Set<string>()
|
|
377
|
+
|
|
378
|
+
for (let fi = 0; fi < files.length; fi++) {
|
|
379
|
+
const file = files[fi]
|
|
380
|
+
onProgress?.(fi + 1, files.length)
|
|
381
|
+
const chunks = chunkFile(file.abs, file.rel)
|
|
382
|
+
for (const chunk of chunks) {
|
|
383
|
+
if (chunk.content.trim() === "") {
|
|
384
|
+
// Whitespace-only chunks produce no useful vector and are
|
|
385
|
+
// rejected outright by embedding providers (live crash
|
|
386
|
+
// 2026-08-16: OpenRouter HTTP 400 "too_small" for an empty
|
|
387
|
+
// string at input[90]). Skip entirely — and deliberately do
|
|
388
|
+
// NOT add the key to survivingKeys, so a stale entry for it
|
|
389
|
+
// (from an older build that didn't filter) is dropped on write.
|
|
390
|
+
continue
|
|
391
|
+
}
|
|
392
|
+
const hash = chunkHash(chunk.content)
|
|
393
|
+
const key = `${chunk.file}:${chunk.startLine}`
|
|
394
|
+
survivingKeys.add(key)
|
|
395
|
+
const old = oldByKey.get(key)
|
|
396
|
+
if (old && old.hash === hash && old.embedding.length > 0 && (old.backend ?? "openrouter") === backend) {
|
|
397
|
+
// Unchanged chunk, same content hash AND same backend → keep the
|
|
398
|
+
// stored vector (backends produce different-dimension vectors, so
|
|
399
|
+
// a stored vector from the other backend is never reusable).
|
|
400
|
+
newEntries.push(old)
|
|
401
|
+
chunksSkipped++
|
|
402
|
+
} else {
|
|
403
|
+
chunksToEmbed.push({ chunk, hash })
|
|
404
|
+
}
|
|
405
|
+
}
|
|
406
|
+
}
|
|
407
|
+
|
|
408
|
+
// Drop entries whose file:startLine no longer exists (deleted/rewritten).
|
|
409
|
+
for (const entry of oldEntries) {
|
|
410
|
+
if (!survivingKeys.has(`${entry.file}:${entry.startLine}`)) {
|
|
411
|
+
chunksRemoved++
|
|
412
|
+
}
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
// Embed all changed/new chunks in batches.
|
|
416
|
+
let promptTokens = 0
|
|
417
|
+
for (let i = 0; i < chunksToEmbed.length; ) {
|
|
418
|
+
const batch = chunksToEmbed.slice(i, i + 128)
|
|
419
|
+
const result = await embedder.embedBatch(batch.map((b) => b.chunk.content))
|
|
420
|
+
promptTokens += result.promptTokens
|
|
421
|
+
if (result.embeddings.length !== batch.length) {
|
|
422
|
+
throw new Error(
|
|
423
|
+
`codebase index: embedder returned ${result.embeddings.length} embeddings for ${batch.length} chunks`,
|
|
424
|
+
)
|
|
425
|
+
}
|
|
426
|
+
checkEmbeddingDimensions(result.embeddings, embeddingDim, i)
|
|
427
|
+
if (embeddingDim === undefined) {
|
|
428
|
+
embeddingDim = result.embeddings[0]?.length
|
|
429
|
+
}
|
|
430
|
+
for (let j = 0; j < batch.length; j++) {
|
|
431
|
+
newEntries.push({
|
|
432
|
+
file: batch[j].chunk.file,
|
|
433
|
+
startLine: batch[j].chunk.startLine,
|
|
434
|
+
endLine: batch[j].chunk.endLine,
|
|
435
|
+
embedding: result.embeddings[j],
|
|
436
|
+
hash: batch[j].hash,
|
|
437
|
+
backend,
|
|
438
|
+
})
|
|
439
|
+
}
|
|
440
|
+
i += batch.length
|
|
441
|
+
onEmbedProgress?.(Math.min(i, chunksToEmbed.length), chunksToEmbed.length)
|
|
442
|
+
}
|
|
443
|
+
|
|
444
|
+
// Deterministic order for the on-disk file (stable across rebuilds).
|
|
445
|
+
newEntries.sort((a, b) => (a.file === b.file ? a.startLine - b.startLine : a.file.localeCompare(b.file)))
|
|
446
|
+
|
|
447
|
+
const file = indexFilePath(root)
|
|
448
|
+
await fsp.mkdir(path.dirname(file), { recursive: true })
|
|
449
|
+
await writeIndexEntries(file, newEntries)
|
|
450
|
+
// Sibling metadata records which backend/model built this index, so a
|
|
451
|
+
// query can refuse a backend mismatch with a clear error instead of
|
|
452
|
+
// comparing different-dimension vectors (see loadIndexMetadata).
|
|
453
|
+
await fsp.writeFile(
|
|
454
|
+
indexMetadataFilePath(root),
|
|
455
|
+
JSON.stringify({ [METADATA_BACKEND_KEY]: backend, [METADATA_MODEL_KEY]: embedder.model }, null, 2) + "\n",
|
|
456
|
+
"utf-8",
|
|
457
|
+
)
|
|
458
|
+
|
|
459
|
+
return {
|
|
460
|
+
filesScanned: files.length,
|
|
461
|
+
chunksEmbedded: chunksToEmbed.length,
|
|
462
|
+
chunksSkipped,
|
|
463
|
+
chunksRemoved,
|
|
464
|
+
totalChunks: newEntries.length,
|
|
465
|
+
promptTokens,
|
|
466
|
+
model: embedder.model,
|
|
467
|
+
backend,
|
|
468
|
+
}
|
|
469
|
+
}
|
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Local Ollama embedder for the codebase index.
|
|
3
|
+
*
|
|
4
|
+
* An alternative backend to the OpenRouter embedder (src/codesearch/embedder.ts)
|
|
5
|
+
* for environments where Ollama runs locally with an embedding model pulled
|
|
6
|
+
* (the project owner's machine: `qwen3-embedding:8b` on an RTX 5080). Opt-in
|
|
7
|
+
* via `HEADLESSCODE_EMBEDDING_BACKEND=ollama` or `--embedding-backend ollama`
|
|
8
|
+
* on `headlesscode index`; the default stays OpenRouter, because hosted/headless
|
|
9
|
+
* workers don't have a local GPU.
|
|
10
|
+
*
|
|
11
|
+
* ─── Ollama embeddings — verified findings (2026-08-01, real local calls) ──
|
|
12
|
+
*
|
|
13
|
+
* 1. ENDPOINT + BATCHING: POST {OLLAMA_URL}/api/embed accepts an ARRAY of
|
|
14
|
+
* inputs in one call — `{"model": "...", "input": ["a", "b", ...]}` returns
|
|
15
|
+
* `embeddings: [vector, vector, ...]` in input order, one HTTP call (verified
|
|
16
|
+
* live: 2 inputs → 2 vectors in one response). The older
|
|
17
|
+
* POST /api/embeddings endpoint takes ONE `prompt` per call and returns
|
|
18
|
+
* `{embedding: [...]}` (also verified live). We use the batch `/api/embed`
|
|
19
|
+
* endpoint — same one-request-per-batch property as OpenRouter's
|
|
20
|
+
* /api/v1/embeddings.
|
|
21
|
+
* 2. RESPONSE SHAPE: `{ model, embeddings: number[][], total_duration,
|
|
22
|
+
* load_duration, prompt_eval_count, ... }`. The embeddings are plain JSON
|
|
23
|
+
* number arrays; `prompt_eval_count` is the total prompt tokens across all
|
|
24
|
+
* inputs (verified: a 2-input call reported 6 = 2 + 4). There is NO
|
|
25
|
+
* `usage` object and no cost — the call is free and fully local, so we do
|
|
26
|
+
* NOT feed the budget/cost path (src/budget/cost.ts) for this backend.
|
|
27
|
+
* 3. DIMENSION: `qwen3-embedding:8b` returns 4096-dim vectors (measured live),
|
|
28
|
+
* vs 2560 for `qwen/qwen3-embedding-4b` (OpenRouter) — NOT interchangeable.
|
|
29
|
+
* An index built with one backend is unusable for search with the other,
|
|
30
|
+
* so the index build records `{backend, model}` metadata and the
|
|
31
|
+
* `codebase_search` tool refuses a backend mismatch with a clear error
|
|
32
|
+
* (see src/codesearch/index.ts + the executor handler).
|
|
33
|
+
* 4. COLD vs WARM LATENCY: a cold call (model freshly loaded into VRAM)
|
|
34
|
+
* measures ~48ms load + ~40ms eval for one short input (~130ms wall);
|
|
35
|
+
* warm calls drop the load duration to ~0 and run at roughly
|
|
36
|
+
* ~14ms/input in a batch of 16 and ~13ms/input in a batch of 64
|
|
37
|
+
* (verified live). The 8B model stays loaded after the first call
|
|
38
|
+
* (default keep_alive), so a large first-time index build can batch
|
|
39
|
+
* aggressively (the same 128-chunk batch size as the cloud path) —
|
|
40
|
+
* the per-batch cost is wall-clock only, and there is no spend to
|
|
41
|
+
* control.
|
|
42
|
+
*
|
|
43
|
+
* The embedder interface (Embedder) is shared with the OpenRouter path; only
|
|
44
|
+
* the HTTP client differs. No batching/hashing logic is duplicated here — the
|
|
45
|
+
* caller (src/codesearch/index.ts buildIndex) slices batches exactly as it
|
|
46
|
+
* does for the cloud path.
|
|
47
|
+
*/
|
|
48
|
+
|
|
49
|
+
import type { EmbedResult } from "./embedder.js"
|
|
50
|
+
|
|
51
|
+
/** Default Ollama server base URL (Ollama's standard local port). */
|
|
52
|
+
export const DEFAULT_OLLAMA_URL = "http://localhost:11434"
|
|
53
|
+
|
|
54
|
+
/** Default local embedding model (the one the project owner has pulled). */
|
|
55
|
+
export const DEFAULT_OLLAMA_EMBEDDING_MODEL = "qwen3-embedding:8b"
|
|
56
|
+
|
|
57
|
+
/** Env var that overrides the Ollama server URL. */
|
|
58
|
+
export const OLLAMA_URL_ENV = "HEADLESSCODE_OLLAMA_URL"
|
|
59
|
+
|
|
60
|
+
/** Env var that overrides the local embedding model name. */
|
|
61
|
+
export const OLLAMA_EMBEDDING_MODEL_ENV = "HEADLESSCODE_OLLAMA_EMBEDDING_MODEL"
|
|
62
|
+
|
|
63
|
+
/** Resolve the Ollama URL: env override → default. */
|
|
64
|
+
export function resolveOllamaUrl(env: NodeJS.ProcessEnv = process.env): string {
|
|
65
|
+
return env[OLLAMA_URL_ENV]?.trim() || DEFAULT_OLLAMA_URL
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/** Resolve the local embedding model: env override → default. */
|
|
69
|
+
export function resolveOllamaEmbeddingModel(env: NodeJS.ProcessEnv = process.env): string {
|
|
70
|
+
return env[OLLAMA_EMBEDDING_MODEL_ENV]?.trim() || DEFAULT_OLLAMA_EMBEDDING_MODEL
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/** Typed error for Ollama failures, carrying a pre-built actionable message. */
|
|
74
|
+
export class OllamaEmbedError extends Error {
|
|
75
|
+
/** HTTP status when Ollama answered (undefined for network-level failures). */
|
|
76
|
+
readonly status?: number
|
|
77
|
+
/** The raw error body Ollama returned, when one exists. */
|
|
78
|
+
readonly body?: string
|
|
79
|
+
|
|
80
|
+
constructor(message: string, status?: number, body?: string) {
|
|
81
|
+
super(message)
|
|
82
|
+
this.name = "OllamaEmbedError"
|
|
83
|
+
this.status = status
|
|
84
|
+
this.body = body
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/** Distinguish Ollama's "model not found" error from everything else. */
|
|
89
|
+
const MODEL_NOT_FOUND_RE = /not found|pull/i
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* Ollama-backed embedder. One HTTP call per batch (POST /api/embed), matching
|
|
93
|
+
* the OpenRouter path's batching. Failures are wrapped in actionable errors —
|
|
94
|
+
* never a silent fallback to the cloud backend (that would surprise the user
|
|
95
|
+
* with unexpected spend).
|
|
96
|
+
*/
|
|
97
|
+
export class OllamaEmbedder {
|
|
98
|
+
readonly backend = "ollama" as const
|
|
99
|
+
readonly model: string
|
|
100
|
+
private readonly baseUrl: string
|
|
101
|
+
|
|
102
|
+
constructor(model: string = resolveOllamaEmbeddingModel(), baseUrl: string = resolveOllamaUrl()) {
|
|
103
|
+
this.model = model
|
|
104
|
+
this.baseUrl = baseUrl.replace(/\/+$/, "")
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
async embedBatch(texts: string[]): Promise<EmbedResult> {
|
|
108
|
+
if (texts.length === 0) {
|
|
109
|
+
return { embeddings: [], model: this.model, promptTokens: 0, totalTokens: 0 }
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
// Same guard as OpenRouterEmbedder: empty inputs are rejected upstream
|
|
113
|
+
// and produce no useful vector anyway.
|
|
114
|
+
const emptyIndex = texts.findIndex((t) => t.trim() === "")
|
|
115
|
+
if (emptyIndex !== -1) {
|
|
116
|
+
throw new OllamaEmbedError(`cannot embed empty/whitespace-only input at index ${emptyIndex} — providers reject it`)
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
const url = `${this.baseUrl}/api/embed`
|
|
120
|
+
let response: Response
|
|
121
|
+
try {
|
|
122
|
+
response = await fetch(url, {
|
|
123
|
+
method: "POST",
|
|
124
|
+
headers: { "Content-Type": "application/json" },
|
|
125
|
+
body: JSON.stringify({ model: this.model, input: texts }),
|
|
126
|
+
})
|
|
127
|
+
} catch (err) {
|
|
128
|
+
// Network-level failure (connection refused, DNS, ...) — Ollama is
|
|
129
|
+
// either not running or unreachable. Actionable, not generic.
|
|
130
|
+
throw new OllamaEmbedError(
|
|
131
|
+
`Ollama backend selected but ${this.baseUrl} is unreachable — is \`ollama serve\` running? ` +
|
|
132
|
+
`(or set HEADLESSCODE_OLLAMA_URL if Ollama listens elsewhere): ` +
|
|
133
|
+
`${err instanceof Error ? err.message : String(err)}`,
|
|
134
|
+
)
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
const rawBody = await response.text().catch(() => "")
|
|
138
|
+
if (!response.ok) {
|
|
139
|
+
// Ollama answers HTTP 400/404 with `{"error": "model \"...\" not found,
|
|
140
|
+
// try pulling it first"}` for a missing model — surface that
|
|
141
|
+
// specifically, since it's the second most common misconfiguration.
|
|
142
|
+
if (MODEL_NOT_FOUND_RE.test(rawBody)) {
|
|
143
|
+
throw new OllamaEmbedError(
|
|
144
|
+
`Ollama backend selected but model "${this.model}" is not pulled — run ` +
|
|
145
|
+
`\`ollama pull ${this.model}\``,
|
|
146
|
+
response.status,
|
|
147
|
+
excerpt(rawBody),
|
|
148
|
+
)
|
|
149
|
+
}
|
|
150
|
+
throw new OllamaEmbedError(
|
|
151
|
+
`Ollama embeddings returned HTTP ${response.status}: ${excerpt(rawBody || "(empty body)")}`,
|
|
152
|
+
response.status,
|
|
153
|
+
excerpt(rawBody),
|
|
154
|
+
)
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
let data:
|
|
158
|
+
| {
|
|
159
|
+
embeddings?: unknown
|
|
160
|
+
error?: { message?: string }
|
|
161
|
+
}
|
|
162
|
+
| undefined
|
|
163
|
+
try {
|
|
164
|
+
data = JSON.parse(rawBody)
|
|
165
|
+
} catch {
|
|
166
|
+
throw new OllamaEmbedError(
|
|
167
|
+
`Ollama embeddings returned HTTP 200 with a non-JSON/unparseable body: ${excerpt(rawBody || "(empty)")}`,
|
|
168
|
+
)
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
if (!Array.isArray(data?.embeddings) || data.embeddings.length !== texts.length) {
|
|
172
|
+
const n = Array.isArray(data?.embeddings) ? data.embeddings.length : 0
|
|
173
|
+
throw new OllamaEmbedError(
|
|
174
|
+
`Ollama embeddings response contained ${n} embeddings for ${texts.length} inputs. ` +
|
|
175
|
+
`Raw body: ${excerpt(rawBody || "(empty)")}`,
|
|
176
|
+
)
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
// Validate every entry is a numeric array — garbage here would silently
|
|
180
|
+
// poison the index (dimension is whatever the model returns, so the
|
|
181
|
+
// per-vector length must at least be self-consistent).
|
|
182
|
+
const embeddings: number[][] = []
|
|
183
|
+
let promptTokens = 0
|
|
184
|
+
for (const item of data.embeddings) {
|
|
185
|
+
if (!Array.isArray(item) || item.some((v) => typeof v !== "number")) {
|
|
186
|
+
throw new OllamaEmbedError(
|
|
187
|
+
`Ollama embeddings response contained a non-numeric embedding vector. Raw body: ${excerpt(rawBody)}`,
|
|
188
|
+
)
|
|
189
|
+
}
|
|
190
|
+
embeddings.push(item as number[])
|
|
191
|
+
promptTokens += (item as number[]).length
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
// No usage/cost to report for a fully local call — promptTokens is
|
|
195
|
+
// approximate (vector length, not true token count) and only used for
|
|
196
|
+
// display; totalTokens mirrors it so accounting call sites never see
|
|
197
|
+
// NaN/undefined.
|
|
198
|
+
return { embeddings, model: this.model, promptTokens, totalTokens: promptTokens }
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
/** Truncate a raw body to a bounded excerpt for error messages. */
|
|
203
|
+
function excerpt(body: string): string {
|
|
204
|
+
return body.length > 500 ? `${body.slice(0, 500)}…` : body
|
|
205
|
+
}
|