algo-cli-runtime 0.14.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- algo_cli/__init__.py +3 -0
- algo_cli/__main__.py +7 -0
- algo_cli/_internal/__init__.py +12 -0
- algo_cli/_internal/policy_chain.py +259 -0
- algo_cli/action_registry.py +1047 -0
- algo_cli/agent_blocks.py +550 -0
- algo_cli/agent_pipeline.py +1457 -0
- algo_cli/agent_threads.py +308 -0
- algo_cli/animations.py +316 -0
- algo_cli/cache_admission.py +209 -0
- algo_cli/capability_mask.py +66 -0
- algo_cli/chat_protocol.py +116 -0
- algo_cli/chatgpt_auth.py +510 -0
- algo_cli/chatgpt_client.py +657 -0
- algo_cli/code_rag.py +479 -0
- algo_cli/config.py +651 -0
- algo_cli/context_budget.py +679 -0
- algo_cli/credential_helpers.py +315 -0
- algo_cli/deliberation.py +29 -0
- algo_cli/display.py +1470 -0
- algo_cli/evals/__init__.py +21 -0
- algo_cli/evals/algorithm_effectiveness.py +560 -0
- algo_cli/evals/competitive_harness_rating.py +702 -0
- algo_cli/evals/cot_quality.py +220 -0
- algo_cli/evals/harness_retrieval_benchmark.py +401 -0
- algo_cli/evals/performance_regression.py +136 -0
- algo_cli/evals/scorecard_grading.py +308 -0
- algo_cli/evals/session_distribution.py +84 -0
- algo_cli/execution_guardrails.py +806 -0
- algo_cli/extensions_manifest.py +84 -0
- algo_cli/git_evidence.py +227 -0
- algo_cli/google_workspace.py +407 -0
- algo_cli/google_workspace_auth.py +523 -0
- algo_cli/harness.py +2587 -0
- algo_cli/identity.py +557 -0
- algo_cli/index_compute_lab.py +228 -0
- algo_cli/inference_harness.py +70 -0
- algo_cli/intelligence/__init__.py +1103 -0
- algo_cli/intelligence/acrobat_config.py +307 -0
- algo_cli/intelligence/acrobat_manifests.py +338 -0
- algo_cli/intelligence/acrobat_models.py +195 -0
- algo_cli/intelligence/acrobat_pipeline.py +295 -0
- algo_cli/intelligence/acrobat_runtime.py +302 -0
- algo_cli/intelligence/acrobat_security.py +261 -0
- algo_cli/intelligence/acrobat_workflows.py +226 -0
- algo_cli/intelligence/actionability.py +165 -0
- algo_cli/intelligence/adversarial_audit.py +136 -0
- algo_cli/intelligence/agent_arena.py +92 -0
- algo_cli/intelligence/agent_benchmark.py +236 -0
- algo_cli/intelligence/agent_runtime.py +171 -0
- algo_cli/intelligence/agents_as_tools.py +70 -0
- algo_cli/intelligence/artifact_binding.py +80 -0
- algo_cli/intelligence/autonomous_engineer.py +1976 -0
- algo_cli/intelligence/backpressure.py +99 -0
- algo_cli/intelligence/bloom_filter.py +186 -0
- algo_cli/intelligence/bonferroni.py +66 -0
- algo_cli/intelligence/boundary_compaction.py +98 -0
- algo_cli/intelligence/catalog_verifier.py +172 -0
- algo_cli/intelligence/cavecrew.py +118 -0
- algo_cli/intelligence/changelog.py +176 -0
- algo_cli/intelligence/checkpoint_resume.py +92 -0
- algo_cli/intelligence/circuit_breaker.py +88 -0
- algo_cli/intelligence/clarification_gate.py +101 -0
- algo_cli/intelligence/code_graph.py +180 -0
- algo_cli/intelligence/coderank.py +97 -0
- algo_cli/intelligence/consistent_hash.py +150 -0
- algo_cli/intelligence/consortium_synthesis.py +139 -0
- algo_cli/intelligence/construction/__init__.py +241 -0
- algo_cli/intelligence/construction/common.py +273 -0
- algo_cli/intelligence/construction/documents.py +496 -0
- algo_cli/intelligence/construction/labor_units.py +1395 -0
- algo_cli/intelligence/construction/payments.py +470 -0
- algo_cli/intelligence/construction/risk.py +784 -0
- algo_cli/intelligence/content_extractor.py +132 -0
- algo_cli/intelligence/context_adaptive.py +102 -0
- algo_cli/intelligence/context_ops.py +95 -0
- algo_cli/intelligence/count_min.py +145 -0
- algo_cli/intelligence/cow_state.py +103 -0
- algo_cli/intelligence/critic_loop.py +119 -0
- algo_cli/intelligence/cross_source.py +113 -0
- algo_cli/intelligence/daemon_mode.py +99 -0
- algo_cli/intelligence/dag_orchestration.py +151 -0
- algo_cli/intelligence/deep_research.py +155 -0
- algo_cli/intelligence/degenerate_detector.py +78 -0
- algo_cli/intelligence/delta_report.py +92 -0
- algo_cli/intelligence/discovery_event_log.py +92 -0
- algo_cli/intelligence/document_ingest.py +298 -0
- algo_cli/intelligence/dual_layer_validate.py +151 -0
- algo_cli/intelligence/echo_fidelity.py +73 -0
- algo_cli/intelligence/ema_tuning.py +104 -0
- algo_cli/intelligence/event_log.py +92 -0
- algo_cli/intelligence/evidence_graph.py +114 -0
- algo_cli/intelligence/extension_host.py +162 -0
- algo_cli/intelligence/extension_manifest.py +115 -0
- algo_cli/intelligence/falsification_suite.py +178 -0
- algo_cli/intelligence/finance/__init__.py +169 -0
- algo_cli/intelligence/finance/anomalies.py +135 -0
- algo_cli/intelligence/finance/ap_ar.py +351 -0
- algo_cli/intelligence/finance/cash.py +162 -0
- algo_cli/intelligence/finance/close.py +332 -0
- algo_cli/intelligence/finance/common.py +244 -0
- algo_cli/intelligence/finance/construction.py +135 -0
- algo_cli/intelligence/finance/controls.py +172 -0
- algo_cli/intelligence/finance/evidence.py +119 -0
- algo_cli/intelligence/finance/exceptions.py +157 -0
- algo_cli/intelligence/finance/reconciliations.py +254 -0
- algo_cli/intelligence/finance/revenue.py +109 -0
- algo_cli/intelligence/finance/tax.py +74 -0
- algo_cli/intelligence/finance/workpapers.py +111 -0
- algo_cli/intelligence/finding_record.py +120 -0
- algo_cli/intelligence/flow_dag.py +267 -0
- algo_cli/intelligence/gatherer.py +223 -0
- algo_cli/intelligence/golden_master.py +98 -0
- algo_cli/intelligence/graph_rag.py +195 -0
- algo_cli/intelligence/group_chat.py +143 -0
- algo_cli/intelligence/hash_dedup.py +145 -0
- algo_cli/intelligence/hyperloglog.py +128 -0
- algo_cli/intelligence/incremental_index.py +316 -0
- algo_cli/intelligence/index_store.py +16 -0
- algo_cli/intelligence/iteration_plan.py +133 -0
- algo_cli/intelligence/kernel_plugins.py +167 -0
- algo_cli/intelligence/lesson_catalog.py +135 -0
- algo_cli/intelligence/llm_fallback.py +169 -0
- algo_cli/intelligence/log2_histogram.py +267 -0
- algo_cli/intelligence/lsp_integration.py +147 -0
- algo_cli/intelligence/memory_evolution.py +117 -0
- algo_cli/intelligence/minhash_lsh.py +182 -0
- algo_cli/intelligence/multi_model_score.py +174 -0
- algo_cli/intelligence/multi_tier_grade.py +211 -0
- algo_cli/intelligence/negative_controls.py +113 -0
- algo_cli/intelligence/numeric_clamp.py +63 -0
- algo_cli/intelligence/occ_editor.py +66 -0
- algo_cli/intelligence/output_normalize.py +112 -0
- algo_cli/intelligence/parallel_delegation.py +98 -0
- algo_cli/intelligence/parallel_fanout.py +104 -0
- algo_cli/intelligence/permission_modes.py +105 -0
- algo_cli/intelligence/pre_push_gate.py +68 -0
- algo_cli/intelligence/prefetch.py +171 -0
- algo_cli/intelligence/process_framework.py +217 -0
- algo_cli/intelligence/project_graph.py +387 -0
- algo_cli/intelligence/query_expansion.py +146 -0
- algo_cli/intelligence/ralph_loop.py +117 -0
- algo_cli/intelligence/rate_limiter.py +153 -0
- algo_cli/intelligence/refactor_transaction.py +94 -0
- algo_cli/intelligence/research_workspace.py +108 -0
- algo_cli/intelligence/retraction_ledger.py +72 -0
- algo_cli/intelligence/saga_pattern.py +88 -0
- algo_cli/intelligence/session_fork.py +100 -0
- algo_cli/intelligence/shadow_editor.py +67 -0
- algo_cli/intelligence/shell_session.py +213 -0
- algo_cli/intelligence/source_registry.py +143 -0
- algo_cli/intelligence/spawn_scales.py +99 -0
- algo_cli/intelligence/stat_stability.py +104 -0
- algo_cli/intelligence/structural_validator.py +148 -0
- algo_cli/intelligence/subagent_spawner.py +111 -0
- algo_cli/intelligence/symmetric_verify.py +70 -0
- algo_cli/intelligence/task_classifier.py +129 -0
- algo_cli/intelligence/team_execution.py +122 -0
- algo_cli/intelligence/tiered_access.py +121 -0
- algo_cli/intelligence/utility_registry.py +159 -0
- algo_cli/intuition_engine.py +560 -0
- algo_cli/intuition_injector.py +82 -0
- algo_cli/kernels/__init__.py +5 -0
- algo_cli/kernels/manifest.py +763 -0
- algo_cli/main.py +3903 -0
- algo_cli/memory_candidates.py +541 -0
- algo_cli/memory_echo_veil.py +394 -0
- algo_cli/memory_runtime.py +112 -0
- algo_cli/model_info.py +548 -0
- algo_cli/model_profile.py +160 -0
- algo_cli/model_routing.py +74 -0
- algo_cli/oneshot.py +331 -0
- algo_cli/perf_telemetry.py +389 -0
- algo_cli/plugins.py +245 -0
- algo_cli/private_event_store.py +654 -0
- algo_cli/quantization/__init__.py +24 -0
- algo_cli/quantization/lloyd_max.py +98 -0
- algo_cli/quantization/turbo_quant.py +308 -0
- algo_cli/reasoning/__init__.py +46 -0
- algo_cli/reasoning/combinatorial.py +356 -0
- algo_cli/reasoning/graph_of_thought.py +297 -0
- algo_cli/reasoning/mcts.py +220 -0
- algo_cli/reasoning/neuro_symbolic.py +250 -0
- algo_cli/reasoning/react.py +246 -0
- algo_cli/reasoning/reflexion.py +225 -0
- algo_cli/reasoning/tree_of_thought.py +241 -0
- algo_cli/reasoning_bridge.py +150 -0
- algo_cli/reconciliation.py +284 -0
- algo_cli/reflex.py +385 -0
- algo_cli/resources/docs/ALGO.md +13958 -0
- algo_cli/resources/docs/algo-cli-algorithm-evidence-contract.md +60 -0
- algo_cli/resources/docs/algo-cli-execution-verification-contract.md +59 -0
- algo_cli/resources/docs/algo-cli-memory-lifecycle-contract.md +72 -0
- algo_cli/resources/docs/harness-extension-cleanup-recommendation.md +41 -0
- algo_cli/resources/docs/index-compute-lab-integration.md +32 -0
- algo_cli/resources/docs/inference-harness-loop-blueprint-2026-06.md +55 -0
- algo_cli/resources/docs/main-split-map.md +35 -0
- algo_cli/resources/docs/privacy-and-context.md +48 -0
- algo_cli/resources/docs/reflex-loop-v0.2.md +354 -0
- algo_cli/resources/skills/README.md +26 -0
- algo_cli/resources/skills/algo-cli.md +59 -0
- algo_cli/resources/skills/edit-file-precision.md +49 -0
- algo_cli/resources/skills/harness-search-first.md +47 -0
- algo_cli/resources/skills/memory-recall-ritual.md +51 -0
- algo_cli/resources/skills/qol-algorithms.md +224 -0
- algo_cli/resources/skills/smart-error-recovery.md +56 -0
- algo_cli/resources/skills/tool-selection-cheatsheet.md +65 -0
- algo_cli/retrieval_algorithms.py +127 -0
- algo_cli/runtime_qos.py +236 -0
- algo_cli/runtime_services.py +320 -0
- algo_cli/session_commands.py +95 -0
- algo_cli/session_mode.py +113 -0
- algo_cli/skills.py +430 -0
- algo_cli/slash_dispatch.py +1265 -0
- algo_cli/small_context.py +206 -0
- algo_cli/spawn_budget.py +89 -0
- algo_cli/task_ledger.py +84 -0
- algo_cli/task_router.py +197 -0
- algo_cli/tool_context.py +94 -0
- algo_cli/tool_contract.py +99 -0
- algo_cli/tool_policy.py +357 -0
- algo_cli/tool_runtime.py +647 -0
- algo_cli/tools.py +3056 -0
- algo_cli/url_scheme.py +174 -0
- algo_cli/verify.py +154 -0
- algo_cli/version_manifest.py +178 -0
- algo_cli/vision_screenshot_verify.py +76 -0
- algo_cli/workspace_resolver.py +68 -0
- algo_cli/x_account.py +209 -0
- algo_cli/xai_auth.py +374 -0
- algo_cli/xai_client.py +600 -0
- algo_cli_runtime-0.14.0.dist-info/METADATA +369 -0
- algo_cli_runtime-0.14.0.dist-info/RECORD +237 -0
- algo_cli_runtime-0.14.0.dist-info/WHEEL +4 -0
- algo_cli_runtime-0.14.0.dist-info/entry_points.txt +3 -0
- algo_cli_runtime-0.14.0.dist-info/licenses/LICENSE +21 -0
- ollama_cli/__init__.py +67 -0
algo_cli/code_rag.py
ADDED
|
@@ -0,0 +1,479 @@
|
|
|
1
|
+
"""Working-directory code retrieval (RAG over cfg.cwd source files).
|
|
2
|
+
|
|
3
|
+
Harness RAG covers skills/wiki/memory but never the project the user is
|
|
4
|
+
actually working in. A small local model's biggest weakness is not knowing the
|
|
5
|
+
codebase; this module gives it line-anchored code chunks relevant to the turn.
|
|
6
|
+
|
|
7
|
+
Design (deliberately close to harness.py, which is battle-tested):
|
|
8
|
+
- Per-cwd JSON index at CONFIG_DIR/code_index/<digest>.json.
|
|
9
|
+
- Incremental: chunks are keyed by (relative_path, start_line); a file whose
|
|
10
|
+
size+mtime are unchanged reuses its chunks and embeddings. Changed files
|
|
11
|
+
reuse embeddings for content-identical chunks by stable content hash.
|
|
12
|
+
- Embedding is capped per turn (like harness EMBED_PER_TURN_CAP) so the first
|
|
13
|
+
few turns in a new project don't stall on a full-project embed.
|
|
14
|
+
- Retrieval is cosine top-k, numpy fast-path with a scalar fallback.
|
|
15
|
+
|
|
16
|
+
Best-effort throughout: any failure returns empty and the turn proceeds
|
|
17
|
+
without code context.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import hashlib
|
|
23
|
+
import json
|
|
24
|
+
import os
|
|
25
|
+
import re
|
|
26
|
+
import time
|
|
27
|
+
from pathlib import Path
|
|
28
|
+
from typing import Any, Callable
|
|
29
|
+
|
|
30
|
+
try:
|
|
31
|
+
import numpy as _np
|
|
32
|
+
_NUMPY = True
|
|
33
|
+
except ImportError:
|
|
34
|
+
_np = None # type: ignore[assignment]
|
|
35
|
+
_NUMPY = False
|
|
36
|
+
|
|
37
|
+
from .config import CONFIG_DIR, _atomic_write_text
|
|
38
|
+
from .retrieval_algorithms import stable_top_k
|
|
39
|
+
|
|
40
|
+
EmbedFn = Callable[[list[str]], list[list[float]]]
|
|
41
|
+
|
|
42
|
+
CODE_INDEX_DIR = CONFIG_DIR / "code_index"
|
|
43
|
+
CHUNK_LINES = 60
|
|
44
|
+
CHUNK_OVERLAP = 10
|
|
45
|
+
MAX_FILES = 600
|
|
46
|
+
MAX_FILE_BYTES = 400_000
|
|
47
|
+
MAX_CHUNKS = 4000
|
|
48
|
+
EMBED_PER_TURN_CAP = 64
|
|
49
|
+
SNIPPET_CHARS = 600
|
|
50
|
+
|
|
51
|
+
CODE_EXTENSIONS: frozenset[str] = frozenset({
|
|
52
|
+
".py", ".pyi", ".js", ".jsx", ".ts", ".tsx", ".go", ".rs", ".java", ".kt",
|
|
53
|
+
".c", ".h", ".cpp", ".hpp", ".cc", ".cs", ".rb", ".php", ".swift", ".scala",
|
|
54
|
+
".sh", ".ps1", ".sql", ".lua", ".r", ".jl", ".ml", ".ex", ".exs",
|
|
55
|
+
".toml", ".cfg", ".ini", ".yaml", ".yml", ".json", ".md",
|
|
56
|
+
})
|
|
57
|
+
SKIP_DIRS: frozenset[str] = frozenset({
|
|
58
|
+
".git", "node_modules", ".venv", "venv", "env", "__pycache__", "dist",
|
|
59
|
+
"build", "target", ".next", ".mypy_cache", ".pytest_cache", ".ruff_cache",
|
|
60
|
+
"site-packages", ".tox", ".idea", ".vscode", "coverage", ".cache",
|
|
61
|
+
})
|
|
62
|
+
|
|
63
|
+
# Same policy as harness.SECRET_RE: never index files whose names suggest
|
|
64
|
+
# credentials. Their contents would otherwise be embedded, persisted under
|
|
65
|
+
# ~/.algo_cli/code_index/, and injected into prompts (off-machine in cloud mode).
|
|
66
|
+
SECRET_RE = re.compile(
|
|
67
|
+
r"(?:^|[/\\._-])"
|
|
68
|
+
r"(?:secrets?|tokens?|credentials?|auth(?:orization)?|passwords?|passwd|"
|
|
69
|
+
r"api[_-]?keys?|access[_-]?tokens?|private[_-]?keys?|\.env)"
|
|
70
|
+
r"(?:[/\\._-]|s?$|s?[/\\._-])",
|
|
71
|
+
re.IGNORECASE,
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
# Some local embedders only use a short prefix of each input, so the text we
|
|
75
|
+
# embed must front-load the salient content.
|
|
76
|
+
EMBED_TEXT_CHARS = 280
|
|
77
|
+
_SYMBOL_LINE_RE = re.compile(
|
|
78
|
+
r"^\s*(?:def |class |function |func |fn |pub fn |impl |interface |type \w+ |const |export )",
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
# Per-process index cache + rescan throttle: without these, every turn
|
|
82
|
+
# re-parses a multi-MB JSON index and re-walks up to MAX_FILES files.
|
|
83
|
+
_INDEX_MEM: dict[str, dict[str, Any]] = {}
|
|
84
|
+
_LAST_SCAN: dict[str, float] = {}
|
|
85
|
+
SCAN_TTL_SECONDS = 15.0
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _index_path_for(cwd: str) -> Path:
|
|
89
|
+
digest = hashlib.sha1(str(Path(cwd).resolve()).lower().encode("utf-8")).hexdigest()[:16]
|
|
90
|
+
return CODE_INDEX_DIR / f"{digest}.json"
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _iter_source_files(root: Path) -> list[Path]:
|
|
94
|
+
resolved_root = root.resolve()
|
|
95
|
+
found: list[Path] = []
|
|
96
|
+
for current, dirs, files in os.walk(root):
|
|
97
|
+
dirs[:] = [
|
|
98
|
+
d for d in dirs
|
|
99
|
+
if d not in SKIP_DIRS and not d.startswith(".") and not SECRET_RE.search(d)
|
|
100
|
+
]
|
|
101
|
+
for name in files:
|
|
102
|
+
if Path(name).suffix.lower() not in CODE_EXTENSIONS:
|
|
103
|
+
continue
|
|
104
|
+
path = Path(current) / name
|
|
105
|
+
try:
|
|
106
|
+
rel = path.relative_to(root).as_posix()
|
|
107
|
+
except ValueError:
|
|
108
|
+
rel = name
|
|
109
|
+
if SECRET_RE.search(rel):
|
|
110
|
+
continue
|
|
111
|
+
try:
|
|
112
|
+
resolved = path.resolve(strict=True)
|
|
113
|
+
resolved_rel = resolved.relative_to(resolved_root).as_posix()
|
|
114
|
+
except (OSError, ValueError):
|
|
115
|
+
# Broken links, permission failures, and links escaping cwd are
|
|
116
|
+
# skipped. In-root file symlinks are allowed after this check.
|
|
117
|
+
continue
|
|
118
|
+
if SECRET_RE.search(resolved_rel):
|
|
119
|
+
continue
|
|
120
|
+
try:
|
|
121
|
+
if path.stat().st_size > MAX_FILE_BYTES:
|
|
122
|
+
continue
|
|
123
|
+
except OSError:
|
|
124
|
+
continue
|
|
125
|
+
found.append(path)
|
|
126
|
+
if len(found) >= MAX_FILES:
|
|
127
|
+
return found
|
|
128
|
+
return found
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _chunk_file(path: Path, root: Path) -> list[dict[str, Any]]:
|
|
132
|
+
try:
|
|
133
|
+
text = path.read_text(encoding="utf-8", errors="replace")
|
|
134
|
+
except OSError:
|
|
135
|
+
return []
|
|
136
|
+
lines = text.splitlines()
|
|
137
|
+
if not lines:
|
|
138
|
+
return []
|
|
139
|
+
try:
|
|
140
|
+
rel = path.relative_to(root).as_posix()
|
|
141
|
+
except ValueError:
|
|
142
|
+
rel = path.name
|
|
143
|
+
chunks: list[dict[str, Any]] = []
|
|
144
|
+
step = max(1, CHUNK_LINES - CHUNK_OVERLAP)
|
|
145
|
+
for start in range(0, len(lines), step):
|
|
146
|
+
block = lines[start:start + CHUNK_LINES]
|
|
147
|
+
body = "\n".join(block).strip()
|
|
148
|
+
if not body:
|
|
149
|
+
continue
|
|
150
|
+
chunk_text = f"{rel}:{start + 1}\n{body}"
|
|
151
|
+
chunks.append({
|
|
152
|
+
"relative_path": rel,
|
|
153
|
+
"start_line": start + 1,
|
|
154
|
+
"end_line": min(len(lines), start + CHUNK_LINES),
|
|
155
|
+
"text": chunk_text,
|
|
156
|
+
"content_hash": _chunk_content_hash(chunk_text),
|
|
157
|
+
})
|
|
158
|
+
if start + CHUNK_LINES >= len(lines):
|
|
159
|
+
break
|
|
160
|
+
return chunks
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _chunk_body(text: str) -> str:
|
|
164
|
+
"""Exclude the mutable line-location header from semantic chunk identity."""
|
|
165
|
+
_header, separator, body = str(text or "").partition("\n")
|
|
166
|
+
return body if separator else str(text or "")
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _chunk_content_hash(text: str) -> str:
|
|
170
|
+
return hashlib.sha256(_chunk_body(text).encode("utf-8", errors="replace")).hexdigest()
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _reuse_content_embeddings(
|
|
174
|
+
fresh: list[dict[str, Any]],
|
|
175
|
+
previous: list[dict[str, Any]],
|
|
176
|
+
) -> int:
|
|
177
|
+
"""Copy embeddings onto content-identical chunks after line/mtime changes."""
|
|
178
|
+
reusable: dict[str, list[dict[str, Any]]] = {}
|
|
179
|
+
for chunk in previous:
|
|
180
|
+
if not chunk.get("embedding") or not chunk.get("embedding_model"):
|
|
181
|
+
continue
|
|
182
|
+
content_hash = str(chunk.get("content_hash") or _chunk_content_hash(str(chunk.get("text") or "")))
|
|
183
|
+
reusable.setdefault(content_hash, []).append(chunk)
|
|
184
|
+
reused = 0
|
|
185
|
+
for chunk in fresh:
|
|
186
|
+
content_hash = str(chunk.get("content_hash") or _chunk_content_hash(str(chunk.get("text") or "")))
|
|
187
|
+
chunk["content_hash"] = content_hash
|
|
188
|
+
matches = reusable.get(content_hash) or []
|
|
189
|
+
match_index = next(
|
|
190
|
+
(
|
|
191
|
+
index
|
|
192
|
+
for index, candidate in enumerate(matches)
|
|
193
|
+
if _chunk_body(str(candidate.get("text") or "")) == _chunk_body(str(chunk.get("text") or ""))
|
|
194
|
+
),
|
|
195
|
+
None,
|
|
196
|
+
)
|
|
197
|
+
if match_index is None:
|
|
198
|
+
continue
|
|
199
|
+
prior = matches.pop(match_index)
|
|
200
|
+
chunk["embedding"] = prior["embedding"]
|
|
201
|
+
chunk["embedding_model"] = prior["embedding_model"]
|
|
202
|
+
reused += 1
|
|
203
|
+
return reused
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _load_index(cwd: str) -> dict[str, Any]:
|
|
207
|
+
path = _index_path_for(cwd)
|
|
208
|
+
if not path.exists():
|
|
209
|
+
return {"cwd": str(Path(cwd).resolve()), "files": {}, "chunks": []}
|
|
210
|
+
try:
|
|
211
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
212
|
+
if isinstance(data, dict) and isinstance(data.get("chunks"), list):
|
|
213
|
+
return data
|
|
214
|
+
except (OSError, json.JSONDecodeError):
|
|
215
|
+
pass
|
|
216
|
+
return {"cwd": str(Path(cwd).resolve()), "files": {}, "chunks": []}
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def _save_index(cwd: str, index: dict[str, Any]) -> None:
|
|
220
|
+
CODE_INDEX_DIR.mkdir(parents=True, exist_ok=True)
|
|
221
|
+
_atomic_write_text(_index_path_for(cwd), json.dumps(index, separators=(",", ":")))
|
|
222
|
+
_INDEX_MEM[str(Path(cwd).resolve())] = index
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def embed_text_for(chunk: dict[str, Any]) -> str:
|
|
226
|
+
"""Salient embed text for a chunk, front-loaded for short-input embedders.
|
|
227
|
+
|
|
228
|
+
Priority: location header, then symbol-definition lines (def/class/fn/...),
|
|
229
|
+
then leading body lines — packed into EMBED_TEXT_CHARS.
|
|
230
|
+
"""
|
|
231
|
+
text = str(chunk.get("text", ""))
|
|
232
|
+
lines = text.splitlines()
|
|
233
|
+
if not lines:
|
|
234
|
+
return text[:EMBED_TEXT_CHARS]
|
|
235
|
+
header = lines[0] # "rel:start" location line
|
|
236
|
+
body = lines[1:]
|
|
237
|
+
symbols = [ln.strip() for ln in body if _SYMBOL_LINE_RE.match(ln)]
|
|
238
|
+
leading = [ln.strip() for ln in body if ln.strip() and ln.strip() not in symbols]
|
|
239
|
+
out: list[str] = [header]
|
|
240
|
+
budget = EMBED_TEXT_CHARS - len(header) - 1
|
|
241
|
+
for line in symbols + leading:
|
|
242
|
+
if budget - (len(line) + 1) < 0:
|
|
243
|
+
break
|
|
244
|
+
out.append(line)
|
|
245
|
+
budget -= len(line) + 1
|
|
246
|
+
return "\n".join(out)
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def invalidate_cache(cwd: str | None = None) -> None:
|
|
250
|
+
"""Drop the in-memory index cache (all cwds when None). For tests/reload."""
|
|
251
|
+
if cwd is None:
|
|
252
|
+
_INDEX_MEM.clear()
|
|
253
|
+
_LAST_SCAN.clear()
|
|
254
|
+
return
|
|
255
|
+
key = str(Path(cwd).resolve())
|
|
256
|
+
_INDEX_MEM.pop(key, None)
|
|
257
|
+
_LAST_SCAN.pop(key, None)
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def persisted_index_count() -> int:
|
|
261
|
+
"""Return the number of persisted code-index files without creating state."""
|
|
262
|
+
|
|
263
|
+
if CODE_INDEX_DIR.is_symlink() or CODE_INDEX_DIR.is_file():
|
|
264
|
+
return 1
|
|
265
|
+
try:
|
|
266
|
+
return sum(1 for path in CODE_INDEX_DIR.iterdir() if path.is_file() or path.is_symlink())
|
|
267
|
+
except OSError:
|
|
268
|
+
return 0
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def purge_persisted_indexes() -> int:
|
|
272
|
+
"""Delete every persisted code-index file and clear process-local caches."""
|
|
273
|
+
|
|
274
|
+
invalidate_cache()
|
|
275
|
+
# Never follow a user-created directory symlink while deleting generated
|
|
276
|
+
# state. Remove only the link (or an unexpected file at the index path).
|
|
277
|
+
if CODE_INDEX_DIR.is_symlink() or CODE_INDEX_DIR.is_file():
|
|
278
|
+
CODE_INDEX_DIR.unlink()
|
|
279
|
+
return 1
|
|
280
|
+
try:
|
|
281
|
+
paths = tuple(CODE_INDEX_DIR.iterdir())
|
|
282
|
+
except FileNotFoundError:
|
|
283
|
+
return 0
|
|
284
|
+
removed = 0
|
|
285
|
+
for path in paths:
|
|
286
|
+
if not (path.is_file() or path.is_symlink()):
|
|
287
|
+
continue
|
|
288
|
+
path.unlink()
|
|
289
|
+
removed += 1
|
|
290
|
+
try:
|
|
291
|
+
CODE_INDEX_DIR.rmdir()
|
|
292
|
+
except OSError:
|
|
293
|
+
pass
|
|
294
|
+
return removed
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def build_or_update_index(cwd: str, *, force: bool = False) -> dict[str, Any]:
|
|
298
|
+
"""Rescan cwd, reusing chunks for unchanged files (size+mtime). No embedding.
|
|
299
|
+
|
|
300
|
+
Rescans are throttled to SCAN_TTL_SECONDS per cwd; within the window the
|
|
301
|
+
in-memory index is returned as-is (a fresh edit shows up on the next scan).
|
|
302
|
+
"""
|
|
303
|
+
root = Path(cwd).resolve()
|
|
304
|
+
key = str(root)
|
|
305
|
+
now = time.monotonic()
|
|
306
|
+
if not force and key in _INDEX_MEM and (now - _LAST_SCAN.get(key, 0.0)) < SCAN_TTL_SECONDS:
|
|
307
|
+
return _INDEX_MEM[key]
|
|
308
|
+
_LAST_SCAN[key] = now
|
|
309
|
+
index = _INDEX_MEM.get(key) or _load_index(cwd)
|
|
310
|
+
old_files: dict[str, Any] = index.get("files", {}) if index.get("cwd") == str(root) else {}
|
|
311
|
+
old_chunks_by_file: dict[str, list[dict[str, Any]]] = {}
|
|
312
|
+
for chunk in index.get("chunks", []) if index.get("cwd") == str(root) else []:
|
|
313
|
+
old_chunks_by_file.setdefault(chunk.get("relative_path", ""), []).append(chunk)
|
|
314
|
+
|
|
315
|
+
new_files: dict[str, Any] = {}
|
|
316
|
+
new_chunks: list[dict[str, Any]] = []
|
|
317
|
+
reused_files = 0
|
|
318
|
+
reused_chunk_embeddings = 0
|
|
319
|
+
rebuilt_chunks = 0
|
|
320
|
+
for path in _iter_source_files(root):
|
|
321
|
+
try:
|
|
322
|
+
st = path.stat()
|
|
323
|
+
except OSError:
|
|
324
|
+
continue
|
|
325
|
+
rel = path.relative_to(root).as_posix() if root in path.parents or path.parent == root else path.name
|
|
326
|
+
sig = {"size": int(st.st_size), "mtime_ns": int(st.st_mtime_ns)}
|
|
327
|
+
prior = old_files.get(rel)
|
|
328
|
+
if prior and prior.get("size") == sig["size"] and prior.get("mtime_ns") == sig["mtime_ns"] and rel in old_chunks_by_file:
|
|
329
|
+
reused = old_chunks_by_file[rel]
|
|
330
|
+
new_chunks.extend(reused)
|
|
331
|
+
new_files[rel] = sig
|
|
332
|
+
reused_files += 1
|
|
333
|
+
else:
|
|
334
|
+
fresh = _chunk_file(path, root)
|
|
335
|
+
reused_chunk_embeddings += _reuse_content_embeddings(
|
|
336
|
+
fresh,
|
|
337
|
+
old_chunks_by_file.get(rel, []),
|
|
338
|
+
)
|
|
339
|
+
rebuilt_chunks += len(fresh)
|
|
340
|
+
new_chunks.extend(fresh)
|
|
341
|
+
new_files[rel] = sig
|
|
342
|
+
if len(new_chunks) >= MAX_CHUNKS:
|
|
343
|
+
break
|
|
344
|
+
|
|
345
|
+
index = {
|
|
346
|
+
"cwd": str(root),
|
|
347
|
+
"files": new_files,
|
|
348
|
+
"chunks": new_chunks,
|
|
349
|
+
"refresh_stats": {
|
|
350
|
+
"reused_files": reused_files,
|
|
351
|
+
"content_reused_embeddings": reused_chunk_embeddings,
|
|
352
|
+
"rebuilt_chunks": rebuilt_chunks,
|
|
353
|
+
},
|
|
354
|
+
}
|
|
355
|
+
_save_index(cwd, index) # also refreshes _INDEX_MEM
|
|
356
|
+
return index
|
|
357
|
+
|
|
358
|
+
|
|
359
|
+
def ensure_embeddings(cwd: str, embed_fn: EmbedFn, model: str, *, cap: int = EMBED_PER_TURN_CAP) -> dict[str, Any]:
|
|
360
|
+
"""Embed up to `cap` chunks missing an embedding for `model`. Returns the index."""
|
|
361
|
+
index = build_or_update_index(cwd)
|
|
362
|
+
chunks = index.get("chunks", [])
|
|
363
|
+
pending = [c for c in chunks if not c.get("embedding") or c.get("embedding_model") != model]
|
|
364
|
+
if not pending:
|
|
365
|
+
return index
|
|
366
|
+
batch = pending[:cap]
|
|
367
|
+
try:
|
|
368
|
+
vectors = embed_fn([embed_text_for(c) for c in batch])
|
|
369
|
+
except Exception:
|
|
370
|
+
return index
|
|
371
|
+
if len(vectors) != len(batch):
|
|
372
|
+
return index
|
|
373
|
+
for chunk, vec in zip(batch, vectors):
|
|
374
|
+
chunk["embedding"] = vec
|
|
375
|
+
chunk["embedding_model"] = model
|
|
376
|
+
_save_index(cwd, index)
|
|
377
|
+
return index
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
def _cosine(a: list[float], b: list[float]) -> float:
|
|
381
|
+
if not a or not b or len(a) != len(b):
|
|
382
|
+
return 0.0
|
|
383
|
+
dot = na = nb = 0.0
|
|
384
|
+
for x, y in zip(a, b):
|
|
385
|
+
dot += x * y
|
|
386
|
+
na += x * x
|
|
387
|
+
nb += y * y
|
|
388
|
+
if na == 0.0 or nb == 0.0:
|
|
389
|
+
return 0.0
|
|
390
|
+
return dot / ((na ** 0.5) * (nb ** 0.5))
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
def retrieve(cwd: str, query: str, embed_fn: EmbedFn, model: str, *, k: int = 4) -> list[dict[str, Any]]:
|
|
394
|
+
query = (query or "").strip()
|
|
395
|
+
if not query:
|
|
396
|
+
return []
|
|
397
|
+
index = ensure_embeddings(cwd, embed_fn, model)
|
|
398
|
+
candidates = [c for c in index.get("chunks", []) if c.get("embedding") and c.get("embedding_model") == model]
|
|
399
|
+
if not candidates:
|
|
400
|
+
return []
|
|
401
|
+
try:
|
|
402
|
+
qvecs = embed_fn([query])
|
|
403
|
+
except Exception:
|
|
404
|
+
return []
|
|
405
|
+
if not qvecs:
|
|
406
|
+
return []
|
|
407
|
+
qvec = qvecs[0]
|
|
408
|
+
if _NUMPY:
|
|
409
|
+
# Normalize both sides so this path computes true cosine and agrees
|
|
410
|
+
# with the scalar fallback even for non-unit embedders.
|
|
411
|
+
mat = _np.array([c["embedding"] for c in candidates], dtype=_np.float32)
|
|
412
|
+
norms = _np.linalg.norm(mat, axis=1)
|
|
413
|
+
norms[norms == 0.0] = 1.0
|
|
414
|
+
mat = mat / norms[:, None]
|
|
415
|
+
qv = _np.array(qvec, dtype=_np.float32)
|
|
416
|
+
qnorm = float(_np.linalg.norm(qv))
|
|
417
|
+
if qnorm > 0.0:
|
|
418
|
+
qv = qv / qnorm
|
|
419
|
+
sims = (mat @ qv).tolist()
|
|
420
|
+
scored = [(float(s), candidates[i]) for i, s in enumerate(sims) if s > 0.0]
|
|
421
|
+
else:
|
|
422
|
+
scored = [(_cosine(qvec, c["embedding"]), c) for c in candidates]
|
|
423
|
+
scored = [(s, c) for s, c in scored if s > 0.0]
|
|
424
|
+
scored = stable_top_k(scored, k, score=lambda pair: pair[0])
|
|
425
|
+
out: list[dict[str, Any]] = []
|
|
426
|
+
for sim, chunk in scored:
|
|
427
|
+
out.append({
|
|
428
|
+
"relative_path": chunk.get("relative_path", ""),
|
|
429
|
+
"start_line": chunk.get("start_line", 1),
|
|
430
|
+
"end_line": chunk.get("end_line", 1),
|
|
431
|
+
"text": chunk.get("text", ""),
|
|
432
|
+
"score": round(float(sim), 4),
|
|
433
|
+
})
|
|
434
|
+
return out
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
def format_code_context(results: list[dict[str, Any]]) -> str:
|
|
438
|
+
if not results:
|
|
439
|
+
return ""
|
|
440
|
+
lines = [
|
|
441
|
+
"Relevant code from the working directory (read_file the path for full context):",
|
|
442
|
+
"",
|
|
443
|
+
]
|
|
444
|
+
for r in results:
|
|
445
|
+
body = r.get("text", "")
|
|
446
|
+
if len(body) > SNIPPET_CHARS:
|
|
447
|
+
body = body[:SNIPPET_CHARS].rstrip() + "\n…"
|
|
448
|
+
loc = f"{r.get('relative_path', '?')}:{r.get('start_line', 1)}-{r.get('end_line', 1)}"
|
|
449
|
+
lines.append(f"### {loc}")
|
|
450
|
+
lines.append("```")
|
|
451
|
+
lines.append(body)
|
|
452
|
+
lines.append("```")
|
|
453
|
+
lines.append("")
|
|
454
|
+
return "\n".join(lines).rstrip()
|
|
455
|
+
|
|
456
|
+
|
|
457
|
+
def looks_like_code_project(cwd: str) -> bool:
|
|
458
|
+
"""Cheap gate: does cwd contain enough source to be worth indexing?"""
|
|
459
|
+
root = Path(cwd)
|
|
460
|
+
if not root.is_dir():
|
|
461
|
+
return False
|
|
462
|
+
markers = (
|
|
463
|
+
"pyproject.toml", "setup.py", "package.json", "Cargo.toml", "go.mod",
|
|
464
|
+
"pom.xml", "build.gradle", ".git", "requirements.txt", "tsconfig.json",
|
|
465
|
+
)
|
|
466
|
+
try:
|
|
467
|
+
for marker in markers:
|
|
468
|
+
if (root / marker).exists():
|
|
469
|
+
return True
|
|
470
|
+
# Otherwise require at least a few source files at the top two levels.
|
|
471
|
+
count = 0
|
|
472
|
+
for path in root.iterdir():
|
|
473
|
+
if path.is_file() and path.suffix.lower() in CODE_EXTENSIONS:
|
|
474
|
+
count += 1
|
|
475
|
+
if count >= 3:
|
|
476
|
+
return True
|
|
477
|
+
except OSError:
|
|
478
|
+
return False
|
|
479
|
+
return False
|