algo-cli-runtime 0.14.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. algo_cli/__init__.py +3 -0
  2. algo_cli/__main__.py +7 -0
  3. algo_cli/_internal/__init__.py +12 -0
  4. algo_cli/_internal/policy_chain.py +259 -0
  5. algo_cli/action_registry.py +1047 -0
  6. algo_cli/agent_blocks.py +550 -0
  7. algo_cli/agent_pipeline.py +1457 -0
  8. algo_cli/agent_threads.py +308 -0
  9. algo_cli/animations.py +316 -0
  10. algo_cli/cache_admission.py +209 -0
  11. algo_cli/capability_mask.py +66 -0
  12. algo_cli/chat_protocol.py +116 -0
  13. algo_cli/chatgpt_auth.py +510 -0
  14. algo_cli/chatgpt_client.py +657 -0
  15. algo_cli/code_rag.py +479 -0
  16. algo_cli/config.py +651 -0
  17. algo_cli/context_budget.py +679 -0
  18. algo_cli/credential_helpers.py +315 -0
  19. algo_cli/deliberation.py +29 -0
  20. algo_cli/display.py +1470 -0
  21. algo_cli/evals/__init__.py +21 -0
  22. algo_cli/evals/algorithm_effectiveness.py +560 -0
  23. algo_cli/evals/competitive_harness_rating.py +702 -0
  24. algo_cli/evals/cot_quality.py +220 -0
  25. algo_cli/evals/harness_retrieval_benchmark.py +401 -0
  26. algo_cli/evals/performance_regression.py +136 -0
  27. algo_cli/evals/scorecard_grading.py +308 -0
  28. algo_cli/evals/session_distribution.py +84 -0
  29. algo_cli/execution_guardrails.py +806 -0
  30. algo_cli/extensions_manifest.py +84 -0
  31. algo_cli/git_evidence.py +227 -0
  32. algo_cli/google_workspace.py +407 -0
  33. algo_cli/google_workspace_auth.py +523 -0
  34. algo_cli/harness.py +2587 -0
  35. algo_cli/identity.py +557 -0
  36. algo_cli/index_compute_lab.py +228 -0
  37. algo_cli/inference_harness.py +70 -0
  38. algo_cli/intelligence/__init__.py +1103 -0
  39. algo_cli/intelligence/acrobat_config.py +307 -0
  40. algo_cli/intelligence/acrobat_manifests.py +338 -0
  41. algo_cli/intelligence/acrobat_models.py +195 -0
  42. algo_cli/intelligence/acrobat_pipeline.py +295 -0
  43. algo_cli/intelligence/acrobat_runtime.py +302 -0
  44. algo_cli/intelligence/acrobat_security.py +261 -0
  45. algo_cli/intelligence/acrobat_workflows.py +226 -0
  46. algo_cli/intelligence/actionability.py +165 -0
  47. algo_cli/intelligence/adversarial_audit.py +136 -0
  48. algo_cli/intelligence/agent_arena.py +92 -0
  49. algo_cli/intelligence/agent_benchmark.py +236 -0
  50. algo_cli/intelligence/agent_runtime.py +171 -0
  51. algo_cli/intelligence/agents_as_tools.py +70 -0
  52. algo_cli/intelligence/artifact_binding.py +80 -0
  53. algo_cli/intelligence/autonomous_engineer.py +1976 -0
  54. algo_cli/intelligence/backpressure.py +99 -0
  55. algo_cli/intelligence/bloom_filter.py +186 -0
  56. algo_cli/intelligence/bonferroni.py +66 -0
  57. algo_cli/intelligence/boundary_compaction.py +98 -0
  58. algo_cli/intelligence/catalog_verifier.py +172 -0
  59. algo_cli/intelligence/cavecrew.py +118 -0
  60. algo_cli/intelligence/changelog.py +176 -0
  61. algo_cli/intelligence/checkpoint_resume.py +92 -0
  62. algo_cli/intelligence/circuit_breaker.py +88 -0
  63. algo_cli/intelligence/clarification_gate.py +101 -0
  64. algo_cli/intelligence/code_graph.py +180 -0
  65. algo_cli/intelligence/coderank.py +97 -0
  66. algo_cli/intelligence/consistent_hash.py +150 -0
  67. algo_cli/intelligence/consortium_synthesis.py +139 -0
  68. algo_cli/intelligence/construction/__init__.py +241 -0
  69. algo_cli/intelligence/construction/common.py +273 -0
  70. algo_cli/intelligence/construction/documents.py +496 -0
  71. algo_cli/intelligence/construction/labor_units.py +1395 -0
  72. algo_cli/intelligence/construction/payments.py +470 -0
  73. algo_cli/intelligence/construction/risk.py +784 -0
  74. algo_cli/intelligence/content_extractor.py +132 -0
  75. algo_cli/intelligence/context_adaptive.py +102 -0
  76. algo_cli/intelligence/context_ops.py +95 -0
  77. algo_cli/intelligence/count_min.py +145 -0
  78. algo_cli/intelligence/cow_state.py +103 -0
  79. algo_cli/intelligence/critic_loop.py +119 -0
  80. algo_cli/intelligence/cross_source.py +113 -0
  81. algo_cli/intelligence/daemon_mode.py +99 -0
  82. algo_cli/intelligence/dag_orchestration.py +151 -0
  83. algo_cli/intelligence/deep_research.py +155 -0
  84. algo_cli/intelligence/degenerate_detector.py +78 -0
  85. algo_cli/intelligence/delta_report.py +92 -0
  86. algo_cli/intelligence/discovery_event_log.py +92 -0
  87. algo_cli/intelligence/document_ingest.py +298 -0
  88. algo_cli/intelligence/dual_layer_validate.py +151 -0
  89. algo_cli/intelligence/echo_fidelity.py +73 -0
  90. algo_cli/intelligence/ema_tuning.py +104 -0
  91. algo_cli/intelligence/event_log.py +92 -0
  92. algo_cli/intelligence/evidence_graph.py +114 -0
  93. algo_cli/intelligence/extension_host.py +162 -0
  94. algo_cli/intelligence/extension_manifest.py +115 -0
  95. algo_cli/intelligence/falsification_suite.py +178 -0
  96. algo_cli/intelligence/finance/__init__.py +169 -0
  97. algo_cli/intelligence/finance/anomalies.py +135 -0
  98. algo_cli/intelligence/finance/ap_ar.py +351 -0
  99. algo_cli/intelligence/finance/cash.py +162 -0
  100. algo_cli/intelligence/finance/close.py +332 -0
  101. algo_cli/intelligence/finance/common.py +244 -0
  102. algo_cli/intelligence/finance/construction.py +135 -0
  103. algo_cli/intelligence/finance/controls.py +172 -0
  104. algo_cli/intelligence/finance/evidence.py +119 -0
  105. algo_cli/intelligence/finance/exceptions.py +157 -0
  106. algo_cli/intelligence/finance/reconciliations.py +254 -0
  107. algo_cli/intelligence/finance/revenue.py +109 -0
  108. algo_cli/intelligence/finance/tax.py +74 -0
  109. algo_cli/intelligence/finance/workpapers.py +111 -0
  110. algo_cli/intelligence/finding_record.py +120 -0
  111. algo_cli/intelligence/flow_dag.py +267 -0
  112. algo_cli/intelligence/gatherer.py +223 -0
  113. algo_cli/intelligence/golden_master.py +98 -0
  114. algo_cli/intelligence/graph_rag.py +195 -0
  115. algo_cli/intelligence/group_chat.py +143 -0
  116. algo_cli/intelligence/hash_dedup.py +145 -0
  117. algo_cli/intelligence/hyperloglog.py +128 -0
  118. algo_cli/intelligence/incremental_index.py +316 -0
  119. algo_cli/intelligence/index_store.py +16 -0
  120. algo_cli/intelligence/iteration_plan.py +133 -0
  121. algo_cli/intelligence/kernel_plugins.py +167 -0
  122. algo_cli/intelligence/lesson_catalog.py +135 -0
  123. algo_cli/intelligence/llm_fallback.py +169 -0
  124. algo_cli/intelligence/log2_histogram.py +267 -0
  125. algo_cli/intelligence/lsp_integration.py +147 -0
  126. algo_cli/intelligence/memory_evolution.py +117 -0
  127. algo_cli/intelligence/minhash_lsh.py +182 -0
  128. algo_cli/intelligence/multi_model_score.py +174 -0
  129. algo_cli/intelligence/multi_tier_grade.py +211 -0
  130. algo_cli/intelligence/negative_controls.py +113 -0
  131. algo_cli/intelligence/numeric_clamp.py +63 -0
  132. algo_cli/intelligence/occ_editor.py +66 -0
  133. algo_cli/intelligence/output_normalize.py +112 -0
  134. algo_cli/intelligence/parallel_delegation.py +98 -0
  135. algo_cli/intelligence/parallel_fanout.py +104 -0
  136. algo_cli/intelligence/permission_modes.py +105 -0
  137. algo_cli/intelligence/pre_push_gate.py +68 -0
  138. algo_cli/intelligence/prefetch.py +171 -0
  139. algo_cli/intelligence/process_framework.py +217 -0
  140. algo_cli/intelligence/project_graph.py +387 -0
  141. algo_cli/intelligence/query_expansion.py +146 -0
  142. algo_cli/intelligence/ralph_loop.py +117 -0
  143. algo_cli/intelligence/rate_limiter.py +153 -0
  144. algo_cli/intelligence/refactor_transaction.py +94 -0
  145. algo_cli/intelligence/research_workspace.py +108 -0
  146. algo_cli/intelligence/retraction_ledger.py +72 -0
  147. algo_cli/intelligence/saga_pattern.py +88 -0
  148. algo_cli/intelligence/session_fork.py +100 -0
  149. algo_cli/intelligence/shadow_editor.py +67 -0
  150. algo_cli/intelligence/shell_session.py +213 -0
  151. algo_cli/intelligence/source_registry.py +143 -0
  152. algo_cli/intelligence/spawn_scales.py +99 -0
  153. algo_cli/intelligence/stat_stability.py +104 -0
  154. algo_cli/intelligence/structural_validator.py +148 -0
  155. algo_cli/intelligence/subagent_spawner.py +111 -0
  156. algo_cli/intelligence/symmetric_verify.py +70 -0
  157. algo_cli/intelligence/task_classifier.py +129 -0
  158. algo_cli/intelligence/team_execution.py +122 -0
  159. algo_cli/intelligence/tiered_access.py +121 -0
  160. algo_cli/intelligence/utility_registry.py +159 -0
  161. algo_cli/intuition_engine.py +560 -0
  162. algo_cli/intuition_injector.py +82 -0
  163. algo_cli/kernels/__init__.py +5 -0
  164. algo_cli/kernels/manifest.py +763 -0
  165. algo_cli/main.py +3903 -0
  166. algo_cli/memory_candidates.py +541 -0
  167. algo_cli/memory_echo_veil.py +394 -0
  168. algo_cli/memory_runtime.py +112 -0
  169. algo_cli/model_info.py +548 -0
  170. algo_cli/model_profile.py +160 -0
  171. algo_cli/model_routing.py +74 -0
  172. algo_cli/oneshot.py +331 -0
  173. algo_cli/perf_telemetry.py +389 -0
  174. algo_cli/plugins.py +245 -0
  175. algo_cli/private_event_store.py +654 -0
  176. algo_cli/quantization/__init__.py +24 -0
  177. algo_cli/quantization/lloyd_max.py +98 -0
  178. algo_cli/quantization/turbo_quant.py +308 -0
  179. algo_cli/reasoning/__init__.py +46 -0
  180. algo_cli/reasoning/combinatorial.py +356 -0
  181. algo_cli/reasoning/graph_of_thought.py +297 -0
  182. algo_cli/reasoning/mcts.py +220 -0
  183. algo_cli/reasoning/neuro_symbolic.py +250 -0
  184. algo_cli/reasoning/react.py +246 -0
  185. algo_cli/reasoning/reflexion.py +225 -0
  186. algo_cli/reasoning/tree_of_thought.py +241 -0
  187. algo_cli/reasoning_bridge.py +150 -0
  188. algo_cli/reconciliation.py +284 -0
  189. algo_cli/reflex.py +385 -0
  190. algo_cli/resources/docs/ALGO.md +13958 -0
  191. algo_cli/resources/docs/algo-cli-algorithm-evidence-contract.md +60 -0
  192. algo_cli/resources/docs/algo-cli-execution-verification-contract.md +59 -0
  193. algo_cli/resources/docs/algo-cli-memory-lifecycle-contract.md +72 -0
  194. algo_cli/resources/docs/harness-extension-cleanup-recommendation.md +41 -0
  195. algo_cli/resources/docs/index-compute-lab-integration.md +32 -0
  196. algo_cli/resources/docs/inference-harness-loop-blueprint-2026-06.md +55 -0
  197. algo_cli/resources/docs/main-split-map.md +35 -0
  198. algo_cli/resources/docs/privacy-and-context.md +48 -0
  199. algo_cli/resources/docs/reflex-loop-v0.2.md +354 -0
  200. algo_cli/resources/skills/README.md +26 -0
  201. algo_cli/resources/skills/algo-cli.md +59 -0
  202. algo_cli/resources/skills/edit-file-precision.md +49 -0
  203. algo_cli/resources/skills/harness-search-first.md +47 -0
  204. algo_cli/resources/skills/memory-recall-ritual.md +51 -0
  205. algo_cli/resources/skills/qol-algorithms.md +224 -0
  206. algo_cli/resources/skills/smart-error-recovery.md +56 -0
  207. algo_cli/resources/skills/tool-selection-cheatsheet.md +65 -0
  208. algo_cli/retrieval_algorithms.py +127 -0
  209. algo_cli/runtime_qos.py +236 -0
  210. algo_cli/runtime_services.py +320 -0
  211. algo_cli/session_commands.py +95 -0
  212. algo_cli/session_mode.py +113 -0
  213. algo_cli/skills.py +430 -0
  214. algo_cli/slash_dispatch.py +1265 -0
  215. algo_cli/small_context.py +206 -0
  216. algo_cli/spawn_budget.py +89 -0
  217. algo_cli/task_ledger.py +84 -0
  218. algo_cli/task_router.py +197 -0
  219. algo_cli/tool_context.py +94 -0
  220. algo_cli/tool_contract.py +99 -0
  221. algo_cli/tool_policy.py +357 -0
  222. algo_cli/tool_runtime.py +647 -0
  223. algo_cli/tools.py +3056 -0
  224. algo_cli/url_scheme.py +174 -0
  225. algo_cli/verify.py +154 -0
  226. algo_cli/version_manifest.py +178 -0
  227. algo_cli/vision_screenshot_verify.py +76 -0
  228. algo_cli/workspace_resolver.py +68 -0
  229. algo_cli/x_account.py +209 -0
  230. algo_cli/xai_auth.py +374 -0
  231. algo_cli/xai_client.py +600 -0
  232. algo_cli_runtime-0.14.0.dist-info/METADATA +369 -0
  233. algo_cli_runtime-0.14.0.dist-info/RECORD +237 -0
  234. algo_cli_runtime-0.14.0.dist-info/WHEEL +4 -0
  235. algo_cli_runtime-0.14.0.dist-info/entry_points.txt +3 -0
  236. algo_cli_runtime-0.14.0.dist-info/licenses/LICENSE +21 -0
  237. ollama_cli/__init__.py +67 -0
algo_cli/identity.py ADDED
@@ -0,0 +1,557 @@
1
+ """Native identity layer for Algo CLI.
2
+
3
+ Four user-editable Markdown files in ~/.algo_cli/identity/ define the CLI's
4
+ persona, the user's profile, and accumulated lessons. Contents are cached by
5
+ mtime and prepended to the system prompt on every turn.
6
+
7
+ Design note: reads are stat-gated. On an unchanged turn the only cost is four
8
+ os.stat() calls (microseconds). On a changed file the cost is one read of that
9
+ file. Nothing is re-embedded or re-tokenized here; this module is the fast
10
+ always-inject layer.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import math
17
+ from dataclasses import dataclass
18
+ from datetime import datetime
19
+ from pathlib import Path
20
+ from typing import Any, Callable
21
+
22
+ try:
23
+ import numpy as _np
24
+ _NUMPY = True
25
+ except ImportError:
26
+ _np = None # type: ignore[assignment]
27
+ _NUMPY = False
28
+
29
+ from .cache_admission import WindowTinyLFUCache
30
+ from .config import CONFIG_DIR, _atomic_write_text, _exclusive_state_lock
31
+ from .retrieval_algorithms import stable_top_k
32
+
33
+
34
+ IDENTITY_DIR = CONFIG_DIR / "identity"
35
+ SOUL_PATH = IDENTITY_DIR / "SOUL.md"
36
+ IDENTITY_PATH = IDENTITY_DIR / "IDENTITY.md"
37
+ USER_PATH = IDENTITY_DIR / "USER.md"
38
+ LESSONS_PATH = IDENTITY_DIR / "lessons-learned.md"
39
+ LESSONS_INDEX_PATH = IDENTITY_DIR / "lessons_index.json"
40
+
41
+ ALL_PATHS: tuple[Path, ...] = (SOUL_PATH, IDENTITY_PATH, USER_PATH, LESSONS_PATH)
42
+
43
+ DEFAULT_EMBED_MODEL = "qwen3-embedding:latest"
44
+ QUERY_VEC_CACHE_SIZE = 32
45
+ LESSON_MIN_CHARS = 30
46
+ LESSONS_INDEX_VERSION = 2
47
+
48
+ EmbedFn = Callable[[list[str]], list[list[float]]]
49
+
50
+
51
+ DEFAULT_SOUL = """# Algo CLI - Soul
52
+
53
+ ## Voice
54
+ - Concise. Terminal-native. No filler greetings.
55
+ - Honest about uncertainty. Say "I don't know" when you don't.
56
+ - Show file paths and exact commands when relevant.
57
+
58
+ ## Operating values
59
+ - Local-first. Prefer local Ollama before cloud.
60
+ - Read before writing. Verify before claiming.
61
+ - Treat memory and wiki as navigation, not authority.
62
+ - Format code blocks with language tags.
63
+
64
+ ## Behavior
65
+ - Use tools when they materially help. Do not narrate them.
66
+ - Do not run destructive commands without explicit approval.
67
+ - After a tool call, summarize the result before the next step.
68
+ """
69
+
70
+ LEGACY_DEFAULT_IDENTITY = """# Algo CLI - Identity
71
+
72
+ You are Algo CLI: a local-first terminal coding assistant built on Ollama models.
73
+
74
+ You run on the user's machine. You have direct access to their filesystem through tools.
75
+ You can search a harness of skills, prompts, memories, and wiki pages across the user's
76
+ agent ecosystem (Codex, Claude, OpenClaw, Mercury, Pi, and shared .agents).
77
+
78
+ You are not a chatbot. You are a working partner for terminal tasks.
79
+ """
80
+
81
+ DEFAULT_IDENTITY = """# Algo CLI - Identity
82
+
83
+ You are Algo CLI: a local-first agent runtime for coding, research, and operational work.
84
+
85
+ You run on the user's machine and can use local or connected cloud inference through
86
+ Ollama, Ollama Cloud, xAI Grok, and ChatGPT/Codex. You plan, act with tools, verify
87
+ results, and retain useful context across sessions.
88
+
89
+ Your harness searches Algo CLI's built-in skills, documentation, memories, and
90
+ repository intelligence. Additional local agent stores (Codex, Claude, OpenClaw,
91
+ Mercury, Pi, and shared .agents) are available only after the user explicitly enables
92
+ them. You are a working partner, not a passive chat interface.
93
+ """
94
+
95
+ DEFAULT_USER = """# About the User
96
+
97
+ <!-- Edit this file to teach the CLI about yourself. The more specific, the better. -->
98
+
99
+ ## Who I am
100
+ (Your name, role, what you work on.)
101
+
102
+ ## How I work
103
+ (Editors, languages, shells, workflow preferences.)
104
+
105
+ ## What I want from this CLI
106
+ (Be terse? Verbose? Ask before acting? Prefer local models?)
107
+
108
+ ## Things to never do
109
+ (Hard nos, e.g. "never auto-push to main".)
110
+ """
111
+
112
+ DEFAULT_LESSONS = """# Lessons Learned
113
+
114
+ <!-- Append new lessons below. Each lesson is a paragraph or short list, separated by blank lines. -->
115
+ <!-- Add via /lesson <text> or by editing this file directly. -->
116
+ """
117
+
118
+
119
+ _DEFAULTS: dict[Path, str] = {
120
+ SOUL_PATH: DEFAULT_SOUL,
121
+ IDENTITY_PATH: DEFAULT_IDENTITY,
122
+ USER_PATH: DEFAULT_USER,
123
+ LESSONS_PATH: DEFAULT_LESSONS,
124
+ }
125
+
126
+
127
+ @dataclass
128
+ class CacheEntry:
129
+ mtime_ns: int
130
+ content: str
131
+
132
+
133
+ _CACHE: dict[Path, CacheEntry] = {}
134
+
135
+
136
+ def scaffold_if_needed() -> list[Path]:
137
+ """Create missing identity files and refresh the untouched legacy identity."""
138
+ created: list[Path] = []
139
+ IDENTITY_DIR.mkdir(parents=True, exist_ok=True)
140
+ for path, default in _DEFAULTS.items():
141
+ if not path.exists():
142
+ _atomic_write_text(path, default)
143
+ created.append(path)
144
+ elif path == IDENTITY_PATH:
145
+ try:
146
+ existing = path.read_text(encoding="utf-8")
147
+ except OSError:
148
+ continue
149
+ if existing == LEGACY_DEFAULT_IDENTITY:
150
+ _atomic_write_text(path, DEFAULT_IDENTITY)
151
+ _CACHE.pop(path, None)
152
+ return created
153
+
154
+
155
+ def read_cached(path: Path) -> str:
156
+ """Read file content, mtime-cached. Returns '' if missing or unreadable."""
157
+ if not path.exists():
158
+ return ""
159
+ try:
160
+ mtime_ns = path.stat().st_mtime_ns
161
+ except OSError:
162
+ return ""
163
+ entry = _CACHE.get(path)
164
+ if entry and entry.mtime_ns == mtime_ns:
165
+ return entry.content
166
+ try:
167
+ content = path.read_text(encoding="utf-8", errors="replace")
168
+ except OSError:
169
+ return ""
170
+ _CACHE[path] = CacheEntry(mtime_ns=mtime_ns, content=content)
171
+ return content
172
+
173
+
174
+ def detect_changes() -> list[Path]:
175
+ """List identity files whose mtime differs from the cache. Stat-only, no reads."""
176
+ changed: list[Path] = []
177
+ for path in ALL_PATHS:
178
+ if not path.exists():
179
+ continue
180
+ try:
181
+ mtime_ns = path.stat().st_mtime_ns
182
+ except OSError:
183
+ continue
184
+ entry = _CACHE.get(path)
185
+ if entry is None or entry.mtime_ns != mtime_ns:
186
+ changed.append(path)
187
+ return changed
188
+
189
+
190
+ def identity_mtime_key() -> tuple[int, ...]:
191
+ """Stable fingerprint of identity file mtimes for context-usage cache keys."""
192
+ key: list[int] = []
193
+ for path in ALL_PATHS:
194
+ if not path.exists():
195
+ key.append(0)
196
+ continue
197
+ try:
198
+ key.append(path.stat().st_mtime_ns)
199
+ except OSError:
200
+ key.append(0)
201
+ return tuple(key)
202
+
203
+
204
+ def build_identity_block(retrieved_lessons: list[str] | None = None) -> str:
205
+ """Assemble the identity prefix for the system prompt.
206
+
207
+ If retrieved_lessons is None: full lessons-learned.md is inlined (fallback path).
208
+ If retrieved_lessons is a list (possibly empty): use only those chunks.
209
+ """
210
+ identity_text = read_cached(IDENTITY_PATH).strip()
211
+ soul_text = read_cached(SOUL_PATH).strip()
212
+ user_text = read_cached(USER_PATH).strip()
213
+ parts: list[str] = []
214
+ if identity_text:
215
+ parts.append(f"## Identity\n{identity_text}")
216
+ if soul_text:
217
+ parts.append(f"## Soul\n{soul_text}")
218
+ if user_text:
219
+ parts.append(f"## About the User\n{user_text}")
220
+ if retrieved_lessons is not None:
221
+ if retrieved_lessons:
222
+ joined = "\n\n".join(retrieved_lessons)
223
+ parts.append(f"## Relevant Lessons\n{joined}")
224
+ else:
225
+ lessons_text = read_cached(LESSONS_PATH).strip()
226
+ if lessons_text:
227
+ parts.append(f"## Lessons Learned\n{lessons_text}")
228
+ return "\n\n".join(parts)
229
+
230
+
231
+ # ---------- Lessons embedding / RAG ----------
232
+
233
+ _LESSONS_INDEX: dict[str, Any] | None = None
234
+ _QUERY_VEC_CACHE: WindowTinyLFUCache[tuple[str, int, str], list[float]] = WindowTinyLFUCache(
235
+ max(1, QUERY_VEC_CACHE_SIZE)
236
+ )
237
+
238
+
239
+ def _chunk_lessons(text: str) -> list[str]:
240
+ """Split lessons-learned.md into chunks at `## ` headings. Drops top-level title."""
241
+ if not text.strip():
242
+ return []
243
+ chunks: list[str] = []
244
+ current: list[str] = []
245
+ for line in text.splitlines():
246
+ if line.startswith("## "):
247
+ if current:
248
+ joined = "\n".join(current).strip()
249
+ if joined:
250
+ chunks.append(joined)
251
+ current = [line]
252
+ elif current or line.strip():
253
+ current.append(line)
254
+ if current:
255
+ joined = "\n".join(current).strip()
256
+ if joined:
257
+ chunks.append(joined)
258
+ return [c for c in chunks if len(c) >= LESSON_MIN_CHARS and not c.startswith("# Lessons Learned")]
259
+
260
+
261
+ def _cosine(a: list[float], b: list[float]) -> float:
262
+ if not a or not b or len(a) != len(b):
263
+ return 0.0
264
+ dot = 0.0
265
+ na = 0.0
266
+ nb = 0.0
267
+ for x, y in zip(a, b):
268
+ dot += x * y
269
+ na += x * x
270
+ nb += y * y
271
+ if na == 0.0 or nb == 0.0:
272
+ return 0.0
273
+ return dot / (math.sqrt(na) * math.sqrt(nb))
274
+
275
+
276
+ def _load_lessons_index() -> dict[str, Any] | None:
277
+ global _LESSONS_INDEX
278
+ if _LESSONS_INDEX is not None:
279
+ return _LESSONS_INDEX
280
+ if not LESSONS_INDEX_PATH.exists():
281
+ return None
282
+ try:
283
+ _LESSONS_INDEX = json.loads(LESSONS_INDEX_PATH.read_text(encoding="utf-8"))
284
+ except (OSError, json.JSONDecodeError):
285
+ _LESSONS_INDEX = None
286
+ return _LESSONS_INDEX
287
+
288
+
289
+ def _save_lessons_index(idx: dict[str, Any]) -> None:
290
+ global _LESSONS_INDEX
291
+ IDENTITY_DIR.mkdir(parents=True, exist_ok=True)
292
+ _atomic_write_text(LESSONS_INDEX_PATH, json.dumps(idx, indent=2))
293
+ _LESSONS_INDEX = idx
294
+
295
+
296
+ def _index_embedding_model(idx: dict[str, Any]) -> str:
297
+ """Return the persisted embedding identity, including legacy index support."""
298
+ return str(idx.get("embedding_model") or idx.get("model") or "").strip()
299
+
300
+
301
+ def _normalise_vector(value: Any) -> list[float] | None:
302
+ if not isinstance(value, (list, tuple)) or not value:
303
+ return None
304
+ try:
305
+ vector = [float(component) for component in value]
306
+ except (TypeError, ValueError):
307
+ return None
308
+ if not all(math.isfinite(component) for component in vector):
309
+ return None
310
+ return vector
311
+
312
+
313
+ def _index_vector_dimensions(idx: dict[str, Any]) -> int | None:
314
+ """Return a validated vector width, 0 for an empty index, or None if corrupt."""
315
+ chunks = idx.get("chunks")
316
+ if not isinstance(chunks, list):
317
+ return None
318
+ if not chunks:
319
+ return 0
320
+
321
+ dimensions: int | None = None
322
+ for chunk in chunks:
323
+ if not isinstance(chunk, dict):
324
+ return None
325
+ vector = _normalise_vector(chunk.get("vector"))
326
+ if vector is None:
327
+ return None
328
+ if dimensions is None:
329
+ dimensions = len(vector)
330
+ elif len(vector) != dimensions:
331
+ return None
332
+
333
+ stored_dimensions = idx.get("vector_dimensions")
334
+ if stored_dimensions is not None:
335
+ try:
336
+ persisted = int(stored_dimensions)
337
+ except (TypeError, ValueError):
338
+ return None
339
+ if persisted != dimensions:
340
+ return None
341
+ return dimensions
342
+
343
+
344
+ def lessons_index_stale(
345
+ model: str | None = None,
346
+ dimensions: int | None = None,
347
+ ) -> bool:
348
+ """Return whether lesson text or the requested embedding space changed.
349
+
350
+ Older indexes remain readable, but an active model identity mismatch forces
351
+ a rebuild even when the source Markdown has not changed. A configured
352
+ vector width is checked when supplied.
353
+ """
354
+ if not LESSONS_PATH.exists():
355
+ return False
356
+ try:
357
+ file_mtime = LESSONS_PATH.stat().st_mtime_ns
358
+ except OSError:
359
+ return False
360
+ idx = _load_lessons_index()
361
+ if idx is None:
362
+ return True
363
+ try:
364
+ index_mtime = int(idx.get("mtime_ns", -1))
365
+ except (TypeError, ValueError):
366
+ return True
367
+ if index_mtime != file_mtime:
368
+ return True
369
+ if model is not None and _index_embedding_model(idx) != model.strip():
370
+ return True
371
+ index_dimensions = _index_vector_dimensions(idx)
372
+ if index_dimensions is None:
373
+ return True
374
+ if dimensions is not None and index_dimensions not in {0, int(dimensions)}:
375
+ return True
376
+ return False
377
+
378
+
379
+ def lessons_index_status() -> dict[str, Any]:
380
+ if not LESSONS_PATH.exists():
381
+ return {"file": False, "index": False, "chunk_count": 0, "model": None, "stale": False}
382
+ idx = _load_lessons_index()
383
+ if idx is None:
384
+ return {"file": True, "index": False, "chunk_count": 0, "model": None, "stale": True}
385
+ return {
386
+ "file": True,
387
+ "index": True,
388
+ "chunk_count": len(idx.get("chunks", [])),
389
+ "model": _index_embedding_model(idx) or None,
390
+ "dimensions": _index_vector_dimensions(idx),
391
+ "stale": lessons_index_stale(),
392
+ }
393
+
394
+
395
+ def rebuild_lessons_index(
396
+ embed_fn: EmbedFn,
397
+ model: str = DEFAULT_EMBED_MODEL,
398
+ *,
399
+ expected_dimensions: int | None = None,
400
+ ) -> dict[str, Any]:
401
+ """Embed all lesson chunks in one batched call. Persists to disk."""
402
+ if not LESSONS_PATH.exists():
403
+ return {"chunk_count": 0, "ready": False, "reason": "no_file"}
404
+ try:
405
+ text = LESSONS_PATH.read_text(encoding="utf-8", errors="replace")
406
+ mtime_ns = LESSONS_PATH.stat().st_mtime_ns
407
+ except OSError as exc:
408
+ return {"chunk_count": 0, "ready": False, "reason": f"read_error: {exc}"}
409
+ chunks = _chunk_lessons(text)
410
+ if not chunks:
411
+ idx = {
412
+ "version": LESSONS_INDEX_VERSION,
413
+ "mtime_ns": mtime_ns,
414
+ "model": model,
415
+ "embedding_model": model,
416
+ "vector_dimensions": 0,
417
+ "chunks": [],
418
+ }
419
+ _save_lessons_index(idx)
420
+ return {"chunk_count": 0, "dimensions": 0, "ready": True}
421
+ try:
422
+ vectors = embed_fn(chunks)
423
+ except Exception as exc:
424
+ return {"chunk_count": 0, "ready": False, "reason": f"embed_error: {exc}"}
425
+ if len(vectors) != len(chunks):
426
+ return {"chunk_count": 0, "ready": False, "reason": "embed_count_mismatch"}
427
+ normalised_vectors: list[list[float]] = []
428
+ vector_dimensions: int | None = None
429
+ for vector in vectors:
430
+ normalised = _normalise_vector(vector)
431
+ if normalised is None:
432
+ return {"chunk_count": 0, "ready": False, "reason": "invalid_embedding_vector"}
433
+ if vector_dimensions is None:
434
+ vector_dimensions = len(normalised)
435
+ elif len(normalised) != vector_dimensions:
436
+ return {"chunk_count": 0, "ready": False, "reason": "embed_dimension_mismatch"}
437
+ normalised_vectors.append(normalised)
438
+ assert vector_dimensions is not None
439
+ if expected_dimensions is not None and vector_dimensions != int(expected_dimensions):
440
+ return {
441
+ "chunk_count": 0,
442
+ "ready": False,
443
+ "reason": "embed_dimension_mismatch",
444
+ "expected_dimensions": int(expected_dimensions),
445
+ "actual_dimensions": vector_dimensions,
446
+ }
447
+ idx = {
448
+ "version": LESSONS_INDEX_VERSION,
449
+ "mtime_ns": mtime_ns,
450
+ "model": model,
451
+ "embedding_model": model,
452
+ "vector_dimensions": vector_dimensions,
453
+ "chunks": [
454
+ {"text": chunk, "vector": vector}
455
+ for chunk, vector in zip(chunks, normalised_vectors)
456
+ ],
457
+ }
458
+ _save_lessons_index(idx)
459
+ _QUERY_VEC_CACHE.clear()
460
+ return {"chunk_count": len(chunks), "dimensions": vector_dimensions, "ready": True}
461
+
462
+
463
+ def retrieve_lessons(query: str, embed_fn: EmbedFn, model: str = DEFAULT_EMBED_MODEL, k: int = 5) -> list[str]:
464
+ """Return top-K lesson chunks by cosine similarity. Empty list if index unavailable."""
465
+ idx = _load_lessons_index()
466
+ if idx is None or not idx.get("chunks"):
467
+ return []
468
+ if lessons_index_stale(model):
469
+ return []
470
+ vector_dimensions = _index_vector_dimensions(idx)
471
+ if vector_dimensions is None or vector_dimensions <= 0:
472
+ return []
473
+ query = (query or "").strip()
474
+ if not query:
475
+ return []
476
+ cache_key = (model, vector_dimensions, query)
477
+ _QUERY_VEC_CACHE.resize(max(1, QUERY_VEC_CACHE_SIZE))
478
+ qvec = _QUERY_VEC_CACHE.get(cache_key)
479
+ if qvec is None:
480
+ try:
481
+ vecs = embed_fn([query])
482
+ except Exception:
483
+ return []
484
+ if not vecs:
485
+ return []
486
+ qvec = _normalise_vector(vecs[0])
487
+ if qvec is None or len(qvec) != vector_dimensions:
488
+ return []
489
+ _QUERY_VEC_CACHE.put(cache_key, qvec)
490
+ elif len(qvec) != vector_dimensions:
491
+ return []
492
+
493
+ chunks = idx["chunks"]
494
+ vectors = [_normalise_vector(chunk.get("vector")) for chunk in chunks]
495
+ if any(vector is None or len(vector) != vector_dimensions for vector in vectors):
496
+ return []
497
+ safe_vectors = [vector for vector in vectors if vector is not None]
498
+ if _NUMPY and chunks:
499
+ import numpy as np
500
+ mat = np.array(safe_vectors, dtype=np.float32)
501
+ qv = np.array(qvec, dtype=np.float32)
502
+ mat_norms = np.linalg.norm(mat, axis=1)
503
+ q_norm = float(np.linalg.norm(qv))
504
+ if q_norm <= 0.0:
505
+ return []
506
+ denom = mat_norms * q_norm
507
+ sims = np.divide(mat @ qv, denom, out=np.zeros_like(mat_norms), where=denom > 0).tolist()
508
+ scored: list[tuple[float, str]] = [
509
+ (float(s), chunks[i]["text"]) for i, s in enumerate(sims) if s > 0.0
510
+ ]
511
+ else:
512
+ scored = []
513
+ for chunk, vector in zip(chunks, safe_vectors):
514
+ sim = _cosine(qvec, vector)
515
+ if sim > 0.0:
516
+ scored.append((sim, chunk["text"]))
517
+ top_scored = stable_top_k(scored, k, score=lambda pair: pair[0])
518
+ return [text for _, text in top_scored]
519
+
520
+
521
+ def status() -> list[dict[str, Any]]:
522
+ """Per-file metadata for the /identity command."""
523
+ rows: list[dict[str, Any]] = []
524
+ for path in ALL_PATHS:
525
+ row: dict[str, Any] = {"path": str(path), "name": path.name, "exists": path.exists()}
526
+ if path.exists():
527
+ try:
528
+ st = path.stat()
529
+ row["size"] = int(st.st_size)
530
+ row["modified"] = datetime.fromtimestamp(st.st_mtime).strftime("%Y-%m-%d %H:%M")
531
+ except OSError:
532
+ row["size"] = 0
533
+ row["modified"] = "?"
534
+ rows.append(row)
535
+ return rows
536
+
537
+
538
+ def append_lesson(text: str) -> Path:
539
+ """Append a timestamped lesson to lessons-learned.md and invalidate its cache."""
540
+ IDENTITY_DIR.mkdir(parents=True, exist_ok=True)
541
+ if not LESSONS_PATH.exists():
542
+ _atomic_write_text(LESSONS_PATH, DEFAULT_LESSONS)
543
+ stamp = datetime.now().strftime("%Y-%m-%d %H:%M")
544
+ entry = f"\n## {stamp}\n{text.strip()}\n"
545
+ with _exclusive_state_lock(LESSONS_PATH):
546
+ with LESSONS_PATH.open("a", encoding="utf-8") as handle:
547
+ handle.write(entry)
548
+ _CACHE.pop(LESSONS_PATH, None)
549
+ return LESSONS_PATH
550
+
551
+
552
+ def write_user_profile(content: str) -> Path:
553
+ """Overwrite USER.md with new content and invalidate its cache."""
554
+ IDENTITY_DIR.mkdir(parents=True, exist_ok=True)
555
+ _atomic_write_text(USER_PATH, content)
556
+ _CACHE.pop(USER_PATH, None)
557
+ return USER_PATH