algo-cli-runtime 0.14.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. algo_cli/__init__.py +3 -0
  2. algo_cli/__main__.py +7 -0
  3. algo_cli/_internal/__init__.py +12 -0
  4. algo_cli/_internal/policy_chain.py +259 -0
  5. algo_cli/action_registry.py +1047 -0
  6. algo_cli/agent_blocks.py +550 -0
  7. algo_cli/agent_pipeline.py +1457 -0
  8. algo_cli/agent_threads.py +308 -0
  9. algo_cli/animations.py +316 -0
  10. algo_cli/cache_admission.py +209 -0
  11. algo_cli/capability_mask.py +66 -0
  12. algo_cli/chat_protocol.py +116 -0
  13. algo_cli/chatgpt_auth.py +510 -0
  14. algo_cli/chatgpt_client.py +657 -0
  15. algo_cli/code_rag.py +479 -0
  16. algo_cli/config.py +651 -0
  17. algo_cli/context_budget.py +679 -0
  18. algo_cli/credential_helpers.py +315 -0
  19. algo_cli/deliberation.py +29 -0
  20. algo_cli/display.py +1470 -0
  21. algo_cli/evals/__init__.py +21 -0
  22. algo_cli/evals/algorithm_effectiveness.py +560 -0
  23. algo_cli/evals/competitive_harness_rating.py +702 -0
  24. algo_cli/evals/cot_quality.py +220 -0
  25. algo_cli/evals/harness_retrieval_benchmark.py +401 -0
  26. algo_cli/evals/performance_regression.py +136 -0
  27. algo_cli/evals/scorecard_grading.py +308 -0
  28. algo_cli/evals/session_distribution.py +84 -0
  29. algo_cli/execution_guardrails.py +806 -0
  30. algo_cli/extensions_manifest.py +84 -0
  31. algo_cli/git_evidence.py +227 -0
  32. algo_cli/google_workspace.py +407 -0
  33. algo_cli/google_workspace_auth.py +523 -0
  34. algo_cli/harness.py +2587 -0
  35. algo_cli/identity.py +557 -0
  36. algo_cli/index_compute_lab.py +228 -0
  37. algo_cli/inference_harness.py +70 -0
  38. algo_cli/intelligence/__init__.py +1103 -0
  39. algo_cli/intelligence/acrobat_config.py +307 -0
  40. algo_cli/intelligence/acrobat_manifests.py +338 -0
  41. algo_cli/intelligence/acrobat_models.py +195 -0
  42. algo_cli/intelligence/acrobat_pipeline.py +295 -0
  43. algo_cli/intelligence/acrobat_runtime.py +302 -0
  44. algo_cli/intelligence/acrobat_security.py +261 -0
  45. algo_cli/intelligence/acrobat_workflows.py +226 -0
  46. algo_cli/intelligence/actionability.py +165 -0
  47. algo_cli/intelligence/adversarial_audit.py +136 -0
  48. algo_cli/intelligence/agent_arena.py +92 -0
  49. algo_cli/intelligence/agent_benchmark.py +236 -0
  50. algo_cli/intelligence/agent_runtime.py +171 -0
  51. algo_cli/intelligence/agents_as_tools.py +70 -0
  52. algo_cli/intelligence/artifact_binding.py +80 -0
  53. algo_cli/intelligence/autonomous_engineer.py +1976 -0
  54. algo_cli/intelligence/backpressure.py +99 -0
  55. algo_cli/intelligence/bloom_filter.py +186 -0
  56. algo_cli/intelligence/bonferroni.py +66 -0
  57. algo_cli/intelligence/boundary_compaction.py +98 -0
  58. algo_cli/intelligence/catalog_verifier.py +172 -0
  59. algo_cli/intelligence/cavecrew.py +118 -0
  60. algo_cli/intelligence/changelog.py +176 -0
  61. algo_cli/intelligence/checkpoint_resume.py +92 -0
  62. algo_cli/intelligence/circuit_breaker.py +88 -0
  63. algo_cli/intelligence/clarification_gate.py +101 -0
  64. algo_cli/intelligence/code_graph.py +180 -0
  65. algo_cli/intelligence/coderank.py +97 -0
  66. algo_cli/intelligence/consistent_hash.py +150 -0
  67. algo_cli/intelligence/consortium_synthesis.py +139 -0
  68. algo_cli/intelligence/construction/__init__.py +241 -0
  69. algo_cli/intelligence/construction/common.py +273 -0
  70. algo_cli/intelligence/construction/documents.py +496 -0
  71. algo_cli/intelligence/construction/labor_units.py +1395 -0
  72. algo_cli/intelligence/construction/payments.py +470 -0
  73. algo_cli/intelligence/construction/risk.py +784 -0
  74. algo_cli/intelligence/content_extractor.py +132 -0
  75. algo_cli/intelligence/context_adaptive.py +102 -0
  76. algo_cli/intelligence/context_ops.py +95 -0
  77. algo_cli/intelligence/count_min.py +145 -0
  78. algo_cli/intelligence/cow_state.py +103 -0
  79. algo_cli/intelligence/critic_loop.py +119 -0
  80. algo_cli/intelligence/cross_source.py +113 -0
  81. algo_cli/intelligence/daemon_mode.py +99 -0
  82. algo_cli/intelligence/dag_orchestration.py +151 -0
  83. algo_cli/intelligence/deep_research.py +155 -0
  84. algo_cli/intelligence/degenerate_detector.py +78 -0
  85. algo_cli/intelligence/delta_report.py +92 -0
  86. algo_cli/intelligence/discovery_event_log.py +92 -0
  87. algo_cli/intelligence/document_ingest.py +298 -0
  88. algo_cli/intelligence/dual_layer_validate.py +151 -0
  89. algo_cli/intelligence/echo_fidelity.py +73 -0
  90. algo_cli/intelligence/ema_tuning.py +104 -0
  91. algo_cli/intelligence/event_log.py +92 -0
  92. algo_cli/intelligence/evidence_graph.py +114 -0
  93. algo_cli/intelligence/extension_host.py +162 -0
  94. algo_cli/intelligence/extension_manifest.py +115 -0
  95. algo_cli/intelligence/falsification_suite.py +178 -0
  96. algo_cli/intelligence/finance/__init__.py +169 -0
  97. algo_cli/intelligence/finance/anomalies.py +135 -0
  98. algo_cli/intelligence/finance/ap_ar.py +351 -0
  99. algo_cli/intelligence/finance/cash.py +162 -0
  100. algo_cli/intelligence/finance/close.py +332 -0
  101. algo_cli/intelligence/finance/common.py +244 -0
  102. algo_cli/intelligence/finance/construction.py +135 -0
  103. algo_cli/intelligence/finance/controls.py +172 -0
  104. algo_cli/intelligence/finance/evidence.py +119 -0
  105. algo_cli/intelligence/finance/exceptions.py +157 -0
  106. algo_cli/intelligence/finance/reconciliations.py +254 -0
  107. algo_cli/intelligence/finance/revenue.py +109 -0
  108. algo_cli/intelligence/finance/tax.py +74 -0
  109. algo_cli/intelligence/finance/workpapers.py +111 -0
  110. algo_cli/intelligence/finding_record.py +120 -0
  111. algo_cli/intelligence/flow_dag.py +267 -0
  112. algo_cli/intelligence/gatherer.py +223 -0
  113. algo_cli/intelligence/golden_master.py +98 -0
  114. algo_cli/intelligence/graph_rag.py +195 -0
  115. algo_cli/intelligence/group_chat.py +143 -0
  116. algo_cli/intelligence/hash_dedup.py +145 -0
  117. algo_cli/intelligence/hyperloglog.py +128 -0
  118. algo_cli/intelligence/incremental_index.py +316 -0
  119. algo_cli/intelligence/index_store.py +16 -0
  120. algo_cli/intelligence/iteration_plan.py +133 -0
  121. algo_cli/intelligence/kernel_plugins.py +167 -0
  122. algo_cli/intelligence/lesson_catalog.py +135 -0
  123. algo_cli/intelligence/llm_fallback.py +169 -0
  124. algo_cli/intelligence/log2_histogram.py +267 -0
  125. algo_cli/intelligence/lsp_integration.py +147 -0
  126. algo_cli/intelligence/memory_evolution.py +117 -0
  127. algo_cli/intelligence/minhash_lsh.py +182 -0
  128. algo_cli/intelligence/multi_model_score.py +174 -0
  129. algo_cli/intelligence/multi_tier_grade.py +211 -0
  130. algo_cli/intelligence/negative_controls.py +113 -0
  131. algo_cli/intelligence/numeric_clamp.py +63 -0
  132. algo_cli/intelligence/occ_editor.py +66 -0
  133. algo_cli/intelligence/output_normalize.py +112 -0
  134. algo_cli/intelligence/parallel_delegation.py +98 -0
  135. algo_cli/intelligence/parallel_fanout.py +104 -0
  136. algo_cli/intelligence/permission_modes.py +105 -0
  137. algo_cli/intelligence/pre_push_gate.py +68 -0
  138. algo_cli/intelligence/prefetch.py +171 -0
  139. algo_cli/intelligence/process_framework.py +217 -0
  140. algo_cli/intelligence/project_graph.py +387 -0
  141. algo_cli/intelligence/query_expansion.py +146 -0
  142. algo_cli/intelligence/ralph_loop.py +117 -0
  143. algo_cli/intelligence/rate_limiter.py +153 -0
  144. algo_cli/intelligence/refactor_transaction.py +94 -0
  145. algo_cli/intelligence/research_workspace.py +108 -0
  146. algo_cli/intelligence/retraction_ledger.py +72 -0
  147. algo_cli/intelligence/saga_pattern.py +88 -0
  148. algo_cli/intelligence/session_fork.py +100 -0
  149. algo_cli/intelligence/shadow_editor.py +67 -0
  150. algo_cli/intelligence/shell_session.py +213 -0
  151. algo_cli/intelligence/source_registry.py +143 -0
  152. algo_cli/intelligence/spawn_scales.py +99 -0
  153. algo_cli/intelligence/stat_stability.py +104 -0
  154. algo_cli/intelligence/structural_validator.py +148 -0
  155. algo_cli/intelligence/subagent_spawner.py +111 -0
  156. algo_cli/intelligence/symmetric_verify.py +70 -0
  157. algo_cli/intelligence/task_classifier.py +129 -0
  158. algo_cli/intelligence/team_execution.py +122 -0
  159. algo_cli/intelligence/tiered_access.py +121 -0
  160. algo_cli/intelligence/utility_registry.py +159 -0
  161. algo_cli/intuition_engine.py +560 -0
  162. algo_cli/intuition_injector.py +82 -0
  163. algo_cli/kernels/__init__.py +5 -0
  164. algo_cli/kernels/manifest.py +763 -0
  165. algo_cli/main.py +3903 -0
  166. algo_cli/memory_candidates.py +541 -0
  167. algo_cli/memory_echo_veil.py +394 -0
  168. algo_cli/memory_runtime.py +112 -0
  169. algo_cli/model_info.py +548 -0
  170. algo_cli/model_profile.py +160 -0
  171. algo_cli/model_routing.py +74 -0
  172. algo_cli/oneshot.py +331 -0
  173. algo_cli/perf_telemetry.py +389 -0
  174. algo_cli/plugins.py +245 -0
  175. algo_cli/private_event_store.py +654 -0
  176. algo_cli/quantization/__init__.py +24 -0
  177. algo_cli/quantization/lloyd_max.py +98 -0
  178. algo_cli/quantization/turbo_quant.py +308 -0
  179. algo_cli/reasoning/__init__.py +46 -0
  180. algo_cli/reasoning/combinatorial.py +356 -0
  181. algo_cli/reasoning/graph_of_thought.py +297 -0
  182. algo_cli/reasoning/mcts.py +220 -0
  183. algo_cli/reasoning/neuro_symbolic.py +250 -0
  184. algo_cli/reasoning/react.py +246 -0
  185. algo_cli/reasoning/reflexion.py +225 -0
  186. algo_cli/reasoning/tree_of_thought.py +241 -0
  187. algo_cli/reasoning_bridge.py +150 -0
  188. algo_cli/reconciliation.py +284 -0
  189. algo_cli/reflex.py +385 -0
  190. algo_cli/resources/docs/ALGO.md +13958 -0
  191. algo_cli/resources/docs/algo-cli-algorithm-evidence-contract.md +60 -0
  192. algo_cli/resources/docs/algo-cli-execution-verification-contract.md +59 -0
  193. algo_cli/resources/docs/algo-cli-memory-lifecycle-contract.md +72 -0
  194. algo_cli/resources/docs/harness-extension-cleanup-recommendation.md +41 -0
  195. algo_cli/resources/docs/index-compute-lab-integration.md +32 -0
  196. algo_cli/resources/docs/inference-harness-loop-blueprint-2026-06.md +55 -0
  197. algo_cli/resources/docs/main-split-map.md +35 -0
  198. algo_cli/resources/docs/privacy-and-context.md +48 -0
  199. algo_cli/resources/docs/reflex-loop-v0.2.md +354 -0
  200. algo_cli/resources/skills/README.md +26 -0
  201. algo_cli/resources/skills/algo-cli.md +59 -0
  202. algo_cli/resources/skills/edit-file-precision.md +49 -0
  203. algo_cli/resources/skills/harness-search-first.md +47 -0
  204. algo_cli/resources/skills/memory-recall-ritual.md +51 -0
  205. algo_cli/resources/skills/qol-algorithms.md +224 -0
  206. algo_cli/resources/skills/smart-error-recovery.md +56 -0
  207. algo_cli/resources/skills/tool-selection-cheatsheet.md +65 -0
  208. algo_cli/retrieval_algorithms.py +127 -0
  209. algo_cli/runtime_qos.py +236 -0
  210. algo_cli/runtime_services.py +320 -0
  211. algo_cli/session_commands.py +95 -0
  212. algo_cli/session_mode.py +113 -0
  213. algo_cli/skills.py +430 -0
  214. algo_cli/slash_dispatch.py +1265 -0
  215. algo_cli/small_context.py +206 -0
  216. algo_cli/spawn_budget.py +89 -0
  217. algo_cli/task_ledger.py +84 -0
  218. algo_cli/task_router.py +197 -0
  219. algo_cli/tool_context.py +94 -0
  220. algo_cli/tool_contract.py +99 -0
  221. algo_cli/tool_policy.py +357 -0
  222. algo_cli/tool_runtime.py +647 -0
  223. algo_cli/tools.py +3056 -0
  224. algo_cli/url_scheme.py +174 -0
  225. algo_cli/verify.py +154 -0
  226. algo_cli/version_manifest.py +178 -0
  227. algo_cli/vision_screenshot_verify.py +76 -0
  228. algo_cli/workspace_resolver.py +68 -0
  229. algo_cli/x_account.py +209 -0
  230. algo_cli/xai_auth.py +374 -0
  231. algo_cli/xai_client.py +600 -0
  232. algo_cli_runtime-0.14.0.dist-info/METADATA +369 -0
  233. algo_cli_runtime-0.14.0.dist-info/RECORD +237 -0
  234. algo_cli_runtime-0.14.0.dist-info/WHEEL +4 -0
  235. algo_cli_runtime-0.14.0.dist-info/entry_points.txt +3 -0
  236. algo_cli_runtime-0.14.0.dist-info/licenses/LICENSE +21 -0
  237. ollama_cli/__init__.py +67 -0
algo_cli/code_rag.py ADDED
@@ -0,0 +1,479 @@
1
+ """Working-directory code retrieval (RAG over cfg.cwd source files).
2
+
3
+ Harness RAG covers skills/wiki/memory but never the project the user is
4
+ actually working in. A small local model's biggest weakness is not knowing the
5
+ codebase; this module gives it line-anchored code chunks relevant to the turn.
6
+
7
+ Design (deliberately close to harness.py, which is battle-tested):
8
+ - Per-cwd JSON index at CONFIG_DIR/code_index/<digest>.json.
9
+ - Incremental: chunks are keyed by (relative_path, start_line); a file whose
10
+ size+mtime are unchanged reuses its chunks and embeddings. Changed files
11
+ reuse embeddings for content-identical chunks by stable content hash.
12
+ - Embedding is capped per turn (like harness EMBED_PER_TURN_CAP) so the first
13
+ few turns in a new project don't stall on a full-project embed.
14
+ - Retrieval is cosine top-k, numpy fast-path with a scalar fallback.
15
+
16
+ Best-effort throughout: any failure returns empty and the turn proceeds
17
+ without code context.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import hashlib
23
+ import json
24
+ import os
25
+ import re
26
+ import time
27
+ from pathlib import Path
28
+ from typing import Any, Callable
29
+
30
+ try:
31
+ import numpy as _np
32
+ _NUMPY = True
33
+ except ImportError:
34
+ _np = None # type: ignore[assignment]
35
+ _NUMPY = False
36
+
37
+ from .config import CONFIG_DIR, _atomic_write_text
38
+ from .retrieval_algorithms import stable_top_k
39
+
40
+ EmbedFn = Callable[[list[str]], list[list[float]]]
41
+
42
+ CODE_INDEX_DIR = CONFIG_DIR / "code_index"
43
+ CHUNK_LINES = 60
44
+ CHUNK_OVERLAP = 10
45
+ MAX_FILES = 600
46
+ MAX_FILE_BYTES = 400_000
47
+ MAX_CHUNKS = 4000
48
+ EMBED_PER_TURN_CAP = 64
49
+ SNIPPET_CHARS = 600
50
+
51
+ CODE_EXTENSIONS: frozenset[str] = frozenset({
52
+ ".py", ".pyi", ".js", ".jsx", ".ts", ".tsx", ".go", ".rs", ".java", ".kt",
53
+ ".c", ".h", ".cpp", ".hpp", ".cc", ".cs", ".rb", ".php", ".swift", ".scala",
54
+ ".sh", ".ps1", ".sql", ".lua", ".r", ".jl", ".ml", ".ex", ".exs",
55
+ ".toml", ".cfg", ".ini", ".yaml", ".yml", ".json", ".md",
56
+ })
57
+ SKIP_DIRS: frozenset[str] = frozenset({
58
+ ".git", "node_modules", ".venv", "venv", "env", "__pycache__", "dist",
59
+ "build", "target", ".next", ".mypy_cache", ".pytest_cache", ".ruff_cache",
60
+ "site-packages", ".tox", ".idea", ".vscode", "coverage", ".cache",
61
+ })
62
+
63
+ # Same policy as harness.SECRET_RE: never index files whose names suggest
64
+ # credentials. Their contents would otherwise be embedded, persisted under
65
+ # ~/.algo_cli/code_index/, and injected into prompts (off-machine in cloud mode).
66
+ SECRET_RE = re.compile(
67
+ r"(?:^|[/\\._-])"
68
+ r"(?:secrets?|tokens?|credentials?|auth(?:orization)?|passwords?|passwd|"
69
+ r"api[_-]?keys?|access[_-]?tokens?|private[_-]?keys?|\.env)"
70
+ r"(?:[/\\._-]|s?$|s?[/\\._-])",
71
+ re.IGNORECASE,
72
+ )
73
+
74
+ # Some local embedders only use a short prefix of each input, so the text we
75
+ # embed must front-load the salient content.
76
+ EMBED_TEXT_CHARS = 280
77
+ _SYMBOL_LINE_RE = re.compile(
78
+ r"^\s*(?:def |class |function |func |fn |pub fn |impl |interface |type \w+ |const |export )",
79
+ )
80
+
81
+ # Per-process index cache + rescan throttle: without these, every turn
82
+ # re-parses a multi-MB JSON index and re-walks up to MAX_FILES files.
83
+ _INDEX_MEM: dict[str, dict[str, Any]] = {}
84
+ _LAST_SCAN: dict[str, float] = {}
85
+ SCAN_TTL_SECONDS = 15.0
86
+
87
+
88
+ def _index_path_for(cwd: str) -> Path:
89
+ digest = hashlib.sha1(str(Path(cwd).resolve()).lower().encode("utf-8")).hexdigest()[:16]
90
+ return CODE_INDEX_DIR / f"{digest}.json"
91
+
92
+
93
+ def _iter_source_files(root: Path) -> list[Path]:
94
+ resolved_root = root.resolve()
95
+ found: list[Path] = []
96
+ for current, dirs, files in os.walk(root):
97
+ dirs[:] = [
98
+ d for d in dirs
99
+ if d not in SKIP_DIRS and not d.startswith(".") and not SECRET_RE.search(d)
100
+ ]
101
+ for name in files:
102
+ if Path(name).suffix.lower() not in CODE_EXTENSIONS:
103
+ continue
104
+ path = Path(current) / name
105
+ try:
106
+ rel = path.relative_to(root).as_posix()
107
+ except ValueError:
108
+ rel = name
109
+ if SECRET_RE.search(rel):
110
+ continue
111
+ try:
112
+ resolved = path.resolve(strict=True)
113
+ resolved_rel = resolved.relative_to(resolved_root).as_posix()
114
+ except (OSError, ValueError):
115
+ # Broken links, permission failures, and links escaping cwd are
116
+ # skipped. In-root file symlinks are allowed after this check.
117
+ continue
118
+ if SECRET_RE.search(resolved_rel):
119
+ continue
120
+ try:
121
+ if path.stat().st_size > MAX_FILE_BYTES:
122
+ continue
123
+ except OSError:
124
+ continue
125
+ found.append(path)
126
+ if len(found) >= MAX_FILES:
127
+ return found
128
+ return found
129
+
130
+
131
+ def _chunk_file(path: Path, root: Path) -> list[dict[str, Any]]:
132
+ try:
133
+ text = path.read_text(encoding="utf-8", errors="replace")
134
+ except OSError:
135
+ return []
136
+ lines = text.splitlines()
137
+ if not lines:
138
+ return []
139
+ try:
140
+ rel = path.relative_to(root).as_posix()
141
+ except ValueError:
142
+ rel = path.name
143
+ chunks: list[dict[str, Any]] = []
144
+ step = max(1, CHUNK_LINES - CHUNK_OVERLAP)
145
+ for start in range(0, len(lines), step):
146
+ block = lines[start:start + CHUNK_LINES]
147
+ body = "\n".join(block).strip()
148
+ if not body:
149
+ continue
150
+ chunk_text = f"{rel}:{start + 1}\n{body}"
151
+ chunks.append({
152
+ "relative_path": rel,
153
+ "start_line": start + 1,
154
+ "end_line": min(len(lines), start + CHUNK_LINES),
155
+ "text": chunk_text,
156
+ "content_hash": _chunk_content_hash(chunk_text),
157
+ })
158
+ if start + CHUNK_LINES >= len(lines):
159
+ break
160
+ return chunks
161
+
162
+
163
+ def _chunk_body(text: str) -> str:
164
+ """Exclude the mutable line-location header from semantic chunk identity."""
165
+ _header, separator, body = str(text or "").partition("\n")
166
+ return body if separator else str(text or "")
167
+
168
+
169
+ def _chunk_content_hash(text: str) -> str:
170
+ return hashlib.sha256(_chunk_body(text).encode("utf-8", errors="replace")).hexdigest()
171
+
172
+
173
+ def _reuse_content_embeddings(
174
+ fresh: list[dict[str, Any]],
175
+ previous: list[dict[str, Any]],
176
+ ) -> int:
177
+ """Copy embeddings onto content-identical chunks after line/mtime changes."""
178
+ reusable: dict[str, list[dict[str, Any]]] = {}
179
+ for chunk in previous:
180
+ if not chunk.get("embedding") or not chunk.get("embedding_model"):
181
+ continue
182
+ content_hash = str(chunk.get("content_hash") or _chunk_content_hash(str(chunk.get("text") or "")))
183
+ reusable.setdefault(content_hash, []).append(chunk)
184
+ reused = 0
185
+ for chunk in fresh:
186
+ content_hash = str(chunk.get("content_hash") or _chunk_content_hash(str(chunk.get("text") or "")))
187
+ chunk["content_hash"] = content_hash
188
+ matches = reusable.get(content_hash) or []
189
+ match_index = next(
190
+ (
191
+ index
192
+ for index, candidate in enumerate(matches)
193
+ if _chunk_body(str(candidate.get("text") or "")) == _chunk_body(str(chunk.get("text") or ""))
194
+ ),
195
+ None,
196
+ )
197
+ if match_index is None:
198
+ continue
199
+ prior = matches.pop(match_index)
200
+ chunk["embedding"] = prior["embedding"]
201
+ chunk["embedding_model"] = prior["embedding_model"]
202
+ reused += 1
203
+ return reused
204
+
205
+
206
+ def _load_index(cwd: str) -> dict[str, Any]:
207
+ path = _index_path_for(cwd)
208
+ if not path.exists():
209
+ return {"cwd": str(Path(cwd).resolve()), "files": {}, "chunks": []}
210
+ try:
211
+ data = json.loads(path.read_text(encoding="utf-8"))
212
+ if isinstance(data, dict) and isinstance(data.get("chunks"), list):
213
+ return data
214
+ except (OSError, json.JSONDecodeError):
215
+ pass
216
+ return {"cwd": str(Path(cwd).resolve()), "files": {}, "chunks": []}
217
+
218
+
219
+ def _save_index(cwd: str, index: dict[str, Any]) -> None:
220
+ CODE_INDEX_DIR.mkdir(parents=True, exist_ok=True)
221
+ _atomic_write_text(_index_path_for(cwd), json.dumps(index, separators=(",", ":")))
222
+ _INDEX_MEM[str(Path(cwd).resolve())] = index
223
+
224
+
225
+ def embed_text_for(chunk: dict[str, Any]) -> str:
226
+ """Salient embed text for a chunk, front-loaded for short-input embedders.
227
+
228
+ Priority: location header, then symbol-definition lines (def/class/fn/...),
229
+ then leading body lines — packed into EMBED_TEXT_CHARS.
230
+ """
231
+ text = str(chunk.get("text", ""))
232
+ lines = text.splitlines()
233
+ if not lines:
234
+ return text[:EMBED_TEXT_CHARS]
235
+ header = lines[0] # "rel:start" location line
236
+ body = lines[1:]
237
+ symbols = [ln.strip() for ln in body if _SYMBOL_LINE_RE.match(ln)]
238
+ leading = [ln.strip() for ln in body if ln.strip() and ln.strip() not in symbols]
239
+ out: list[str] = [header]
240
+ budget = EMBED_TEXT_CHARS - len(header) - 1
241
+ for line in symbols + leading:
242
+ if budget - (len(line) + 1) < 0:
243
+ break
244
+ out.append(line)
245
+ budget -= len(line) + 1
246
+ return "\n".join(out)
247
+
248
+
249
+ def invalidate_cache(cwd: str | None = None) -> None:
250
+ """Drop the in-memory index cache (all cwds when None). For tests/reload."""
251
+ if cwd is None:
252
+ _INDEX_MEM.clear()
253
+ _LAST_SCAN.clear()
254
+ return
255
+ key = str(Path(cwd).resolve())
256
+ _INDEX_MEM.pop(key, None)
257
+ _LAST_SCAN.pop(key, None)
258
+
259
+
260
+ def persisted_index_count() -> int:
261
+ """Return the number of persisted code-index files without creating state."""
262
+
263
+ if CODE_INDEX_DIR.is_symlink() or CODE_INDEX_DIR.is_file():
264
+ return 1
265
+ try:
266
+ return sum(1 for path in CODE_INDEX_DIR.iterdir() if path.is_file() or path.is_symlink())
267
+ except OSError:
268
+ return 0
269
+
270
+
271
+ def purge_persisted_indexes() -> int:
272
+ """Delete every persisted code-index file and clear process-local caches."""
273
+
274
+ invalidate_cache()
275
+ # Never follow a user-created directory symlink while deleting generated
276
+ # state. Remove only the link (or an unexpected file at the index path).
277
+ if CODE_INDEX_DIR.is_symlink() or CODE_INDEX_DIR.is_file():
278
+ CODE_INDEX_DIR.unlink()
279
+ return 1
280
+ try:
281
+ paths = tuple(CODE_INDEX_DIR.iterdir())
282
+ except FileNotFoundError:
283
+ return 0
284
+ removed = 0
285
+ for path in paths:
286
+ if not (path.is_file() or path.is_symlink()):
287
+ continue
288
+ path.unlink()
289
+ removed += 1
290
+ try:
291
+ CODE_INDEX_DIR.rmdir()
292
+ except OSError:
293
+ pass
294
+ return removed
295
+
296
+
297
+ def build_or_update_index(cwd: str, *, force: bool = False) -> dict[str, Any]:
298
+ """Rescan cwd, reusing chunks for unchanged files (size+mtime). No embedding.
299
+
300
+ Rescans are throttled to SCAN_TTL_SECONDS per cwd; within the window the
301
+ in-memory index is returned as-is (a fresh edit shows up on the next scan).
302
+ """
303
+ root = Path(cwd).resolve()
304
+ key = str(root)
305
+ now = time.monotonic()
306
+ if not force and key in _INDEX_MEM and (now - _LAST_SCAN.get(key, 0.0)) < SCAN_TTL_SECONDS:
307
+ return _INDEX_MEM[key]
308
+ _LAST_SCAN[key] = now
309
+ index = _INDEX_MEM.get(key) or _load_index(cwd)
310
+ old_files: dict[str, Any] = index.get("files", {}) if index.get("cwd") == str(root) else {}
311
+ old_chunks_by_file: dict[str, list[dict[str, Any]]] = {}
312
+ for chunk in index.get("chunks", []) if index.get("cwd") == str(root) else []:
313
+ old_chunks_by_file.setdefault(chunk.get("relative_path", ""), []).append(chunk)
314
+
315
+ new_files: dict[str, Any] = {}
316
+ new_chunks: list[dict[str, Any]] = []
317
+ reused_files = 0
318
+ reused_chunk_embeddings = 0
319
+ rebuilt_chunks = 0
320
+ for path in _iter_source_files(root):
321
+ try:
322
+ st = path.stat()
323
+ except OSError:
324
+ continue
325
+ rel = path.relative_to(root).as_posix() if root in path.parents or path.parent == root else path.name
326
+ sig = {"size": int(st.st_size), "mtime_ns": int(st.st_mtime_ns)}
327
+ prior = old_files.get(rel)
328
+ if prior and prior.get("size") == sig["size"] and prior.get("mtime_ns") == sig["mtime_ns"] and rel in old_chunks_by_file:
329
+ reused = old_chunks_by_file[rel]
330
+ new_chunks.extend(reused)
331
+ new_files[rel] = sig
332
+ reused_files += 1
333
+ else:
334
+ fresh = _chunk_file(path, root)
335
+ reused_chunk_embeddings += _reuse_content_embeddings(
336
+ fresh,
337
+ old_chunks_by_file.get(rel, []),
338
+ )
339
+ rebuilt_chunks += len(fresh)
340
+ new_chunks.extend(fresh)
341
+ new_files[rel] = sig
342
+ if len(new_chunks) >= MAX_CHUNKS:
343
+ break
344
+
345
+ index = {
346
+ "cwd": str(root),
347
+ "files": new_files,
348
+ "chunks": new_chunks,
349
+ "refresh_stats": {
350
+ "reused_files": reused_files,
351
+ "content_reused_embeddings": reused_chunk_embeddings,
352
+ "rebuilt_chunks": rebuilt_chunks,
353
+ },
354
+ }
355
+ _save_index(cwd, index) # also refreshes _INDEX_MEM
356
+ return index
357
+
358
+
359
+ def ensure_embeddings(cwd: str, embed_fn: EmbedFn, model: str, *, cap: int = EMBED_PER_TURN_CAP) -> dict[str, Any]:
360
+ """Embed up to `cap` chunks missing an embedding for `model`. Returns the index."""
361
+ index = build_or_update_index(cwd)
362
+ chunks = index.get("chunks", [])
363
+ pending = [c for c in chunks if not c.get("embedding") or c.get("embedding_model") != model]
364
+ if not pending:
365
+ return index
366
+ batch = pending[:cap]
367
+ try:
368
+ vectors = embed_fn([embed_text_for(c) for c in batch])
369
+ except Exception:
370
+ return index
371
+ if len(vectors) != len(batch):
372
+ return index
373
+ for chunk, vec in zip(batch, vectors):
374
+ chunk["embedding"] = vec
375
+ chunk["embedding_model"] = model
376
+ _save_index(cwd, index)
377
+ return index
378
+
379
+
380
+ def _cosine(a: list[float], b: list[float]) -> float:
381
+ if not a or not b or len(a) != len(b):
382
+ return 0.0
383
+ dot = na = nb = 0.0
384
+ for x, y in zip(a, b):
385
+ dot += x * y
386
+ na += x * x
387
+ nb += y * y
388
+ if na == 0.0 or nb == 0.0:
389
+ return 0.0
390
+ return dot / ((na ** 0.5) * (nb ** 0.5))
391
+
392
+
393
+ def retrieve(cwd: str, query: str, embed_fn: EmbedFn, model: str, *, k: int = 4) -> list[dict[str, Any]]:
394
+ query = (query or "").strip()
395
+ if not query:
396
+ return []
397
+ index = ensure_embeddings(cwd, embed_fn, model)
398
+ candidates = [c for c in index.get("chunks", []) if c.get("embedding") and c.get("embedding_model") == model]
399
+ if not candidates:
400
+ return []
401
+ try:
402
+ qvecs = embed_fn([query])
403
+ except Exception:
404
+ return []
405
+ if not qvecs:
406
+ return []
407
+ qvec = qvecs[0]
408
+ if _NUMPY:
409
+ # Normalize both sides so this path computes true cosine and agrees
410
+ # with the scalar fallback even for non-unit embedders.
411
+ mat = _np.array([c["embedding"] for c in candidates], dtype=_np.float32)
412
+ norms = _np.linalg.norm(mat, axis=1)
413
+ norms[norms == 0.0] = 1.0
414
+ mat = mat / norms[:, None]
415
+ qv = _np.array(qvec, dtype=_np.float32)
416
+ qnorm = float(_np.linalg.norm(qv))
417
+ if qnorm > 0.0:
418
+ qv = qv / qnorm
419
+ sims = (mat @ qv).tolist()
420
+ scored = [(float(s), candidates[i]) for i, s in enumerate(sims) if s > 0.0]
421
+ else:
422
+ scored = [(_cosine(qvec, c["embedding"]), c) for c in candidates]
423
+ scored = [(s, c) for s, c in scored if s > 0.0]
424
+ scored = stable_top_k(scored, k, score=lambda pair: pair[0])
425
+ out: list[dict[str, Any]] = []
426
+ for sim, chunk in scored:
427
+ out.append({
428
+ "relative_path": chunk.get("relative_path", ""),
429
+ "start_line": chunk.get("start_line", 1),
430
+ "end_line": chunk.get("end_line", 1),
431
+ "text": chunk.get("text", ""),
432
+ "score": round(float(sim), 4),
433
+ })
434
+ return out
435
+
436
+
437
+ def format_code_context(results: list[dict[str, Any]]) -> str:
438
+ if not results:
439
+ return ""
440
+ lines = [
441
+ "Relevant code from the working directory (read_file the path for full context):",
442
+ "",
443
+ ]
444
+ for r in results:
445
+ body = r.get("text", "")
446
+ if len(body) > SNIPPET_CHARS:
447
+ body = body[:SNIPPET_CHARS].rstrip() + "\n…"
448
+ loc = f"{r.get('relative_path', '?')}:{r.get('start_line', 1)}-{r.get('end_line', 1)}"
449
+ lines.append(f"### {loc}")
450
+ lines.append("```")
451
+ lines.append(body)
452
+ lines.append("```")
453
+ lines.append("")
454
+ return "\n".join(lines).rstrip()
455
+
456
+
457
+ def looks_like_code_project(cwd: str) -> bool:
458
+ """Cheap gate: does cwd contain enough source to be worth indexing?"""
459
+ root = Path(cwd)
460
+ if not root.is_dir():
461
+ return False
462
+ markers = (
463
+ "pyproject.toml", "setup.py", "package.json", "Cargo.toml", "go.mod",
464
+ "pom.xml", "build.gradle", ".git", "requirements.txt", "tsconfig.json",
465
+ )
466
+ try:
467
+ for marker in markers:
468
+ if (root / marker).exists():
469
+ return True
470
+ # Otherwise require at least a few source files at the top two levels.
471
+ count = 0
472
+ for path in root.iterdir():
473
+ if path.is_file() and path.suffix.lower() in CODE_EXTENSIONS:
474
+ count += 1
475
+ if count >= 3:
476
+ return True
477
+ except OSError:
478
+ return False
479
+ return False