algo-cli-runtime 0.14.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- algo_cli/__init__.py +3 -0
- algo_cli/__main__.py +7 -0
- algo_cli/_internal/__init__.py +12 -0
- algo_cli/_internal/policy_chain.py +259 -0
- algo_cli/action_registry.py +1047 -0
- algo_cli/agent_blocks.py +550 -0
- algo_cli/agent_pipeline.py +1457 -0
- algo_cli/agent_threads.py +308 -0
- algo_cli/animations.py +316 -0
- algo_cli/cache_admission.py +209 -0
- algo_cli/capability_mask.py +66 -0
- algo_cli/chat_protocol.py +116 -0
- algo_cli/chatgpt_auth.py +510 -0
- algo_cli/chatgpt_client.py +657 -0
- algo_cli/code_rag.py +479 -0
- algo_cli/config.py +651 -0
- algo_cli/context_budget.py +679 -0
- algo_cli/credential_helpers.py +315 -0
- algo_cli/deliberation.py +29 -0
- algo_cli/display.py +1470 -0
- algo_cli/evals/__init__.py +21 -0
- algo_cli/evals/algorithm_effectiveness.py +560 -0
- algo_cli/evals/competitive_harness_rating.py +702 -0
- algo_cli/evals/cot_quality.py +220 -0
- algo_cli/evals/harness_retrieval_benchmark.py +401 -0
- algo_cli/evals/performance_regression.py +136 -0
- algo_cli/evals/scorecard_grading.py +308 -0
- algo_cli/evals/session_distribution.py +84 -0
- algo_cli/execution_guardrails.py +806 -0
- algo_cli/extensions_manifest.py +84 -0
- algo_cli/git_evidence.py +227 -0
- algo_cli/google_workspace.py +407 -0
- algo_cli/google_workspace_auth.py +523 -0
- algo_cli/harness.py +2587 -0
- algo_cli/identity.py +557 -0
- algo_cli/index_compute_lab.py +228 -0
- algo_cli/inference_harness.py +70 -0
- algo_cli/intelligence/__init__.py +1103 -0
- algo_cli/intelligence/acrobat_config.py +307 -0
- algo_cli/intelligence/acrobat_manifests.py +338 -0
- algo_cli/intelligence/acrobat_models.py +195 -0
- algo_cli/intelligence/acrobat_pipeline.py +295 -0
- algo_cli/intelligence/acrobat_runtime.py +302 -0
- algo_cli/intelligence/acrobat_security.py +261 -0
- algo_cli/intelligence/acrobat_workflows.py +226 -0
- algo_cli/intelligence/actionability.py +165 -0
- algo_cli/intelligence/adversarial_audit.py +136 -0
- algo_cli/intelligence/agent_arena.py +92 -0
- algo_cli/intelligence/agent_benchmark.py +236 -0
- algo_cli/intelligence/agent_runtime.py +171 -0
- algo_cli/intelligence/agents_as_tools.py +70 -0
- algo_cli/intelligence/artifact_binding.py +80 -0
- algo_cli/intelligence/autonomous_engineer.py +1976 -0
- algo_cli/intelligence/backpressure.py +99 -0
- algo_cli/intelligence/bloom_filter.py +186 -0
- algo_cli/intelligence/bonferroni.py +66 -0
- algo_cli/intelligence/boundary_compaction.py +98 -0
- algo_cli/intelligence/catalog_verifier.py +172 -0
- algo_cli/intelligence/cavecrew.py +118 -0
- algo_cli/intelligence/changelog.py +176 -0
- algo_cli/intelligence/checkpoint_resume.py +92 -0
- algo_cli/intelligence/circuit_breaker.py +88 -0
- algo_cli/intelligence/clarification_gate.py +101 -0
- algo_cli/intelligence/code_graph.py +180 -0
- algo_cli/intelligence/coderank.py +97 -0
- algo_cli/intelligence/consistent_hash.py +150 -0
- algo_cli/intelligence/consortium_synthesis.py +139 -0
- algo_cli/intelligence/construction/__init__.py +241 -0
- algo_cli/intelligence/construction/common.py +273 -0
- algo_cli/intelligence/construction/documents.py +496 -0
- algo_cli/intelligence/construction/labor_units.py +1395 -0
- algo_cli/intelligence/construction/payments.py +470 -0
- algo_cli/intelligence/construction/risk.py +784 -0
- algo_cli/intelligence/content_extractor.py +132 -0
- algo_cli/intelligence/context_adaptive.py +102 -0
- algo_cli/intelligence/context_ops.py +95 -0
- algo_cli/intelligence/count_min.py +145 -0
- algo_cli/intelligence/cow_state.py +103 -0
- algo_cli/intelligence/critic_loop.py +119 -0
- algo_cli/intelligence/cross_source.py +113 -0
- algo_cli/intelligence/daemon_mode.py +99 -0
- algo_cli/intelligence/dag_orchestration.py +151 -0
- algo_cli/intelligence/deep_research.py +155 -0
- algo_cli/intelligence/degenerate_detector.py +78 -0
- algo_cli/intelligence/delta_report.py +92 -0
- algo_cli/intelligence/discovery_event_log.py +92 -0
- algo_cli/intelligence/document_ingest.py +298 -0
- algo_cli/intelligence/dual_layer_validate.py +151 -0
- algo_cli/intelligence/echo_fidelity.py +73 -0
- algo_cli/intelligence/ema_tuning.py +104 -0
- algo_cli/intelligence/event_log.py +92 -0
- algo_cli/intelligence/evidence_graph.py +114 -0
- algo_cli/intelligence/extension_host.py +162 -0
- algo_cli/intelligence/extension_manifest.py +115 -0
- algo_cli/intelligence/falsification_suite.py +178 -0
- algo_cli/intelligence/finance/__init__.py +169 -0
- algo_cli/intelligence/finance/anomalies.py +135 -0
- algo_cli/intelligence/finance/ap_ar.py +351 -0
- algo_cli/intelligence/finance/cash.py +162 -0
- algo_cli/intelligence/finance/close.py +332 -0
- algo_cli/intelligence/finance/common.py +244 -0
- algo_cli/intelligence/finance/construction.py +135 -0
- algo_cli/intelligence/finance/controls.py +172 -0
- algo_cli/intelligence/finance/evidence.py +119 -0
- algo_cli/intelligence/finance/exceptions.py +157 -0
- algo_cli/intelligence/finance/reconciliations.py +254 -0
- algo_cli/intelligence/finance/revenue.py +109 -0
- algo_cli/intelligence/finance/tax.py +74 -0
- algo_cli/intelligence/finance/workpapers.py +111 -0
- algo_cli/intelligence/finding_record.py +120 -0
- algo_cli/intelligence/flow_dag.py +267 -0
- algo_cli/intelligence/gatherer.py +223 -0
- algo_cli/intelligence/golden_master.py +98 -0
- algo_cli/intelligence/graph_rag.py +195 -0
- algo_cli/intelligence/group_chat.py +143 -0
- algo_cli/intelligence/hash_dedup.py +145 -0
- algo_cli/intelligence/hyperloglog.py +128 -0
- algo_cli/intelligence/incremental_index.py +316 -0
- algo_cli/intelligence/index_store.py +16 -0
- algo_cli/intelligence/iteration_plan.py +133 -0
- algo_cli/intelligence/kernel_plugins.py +167 -0
- algo_cli/intelligence/lesson_catalog.py +135 -0
- algo_cli/intelligence/llm_fallback.py +169 -0
- algo_cli/intelligence/log2_histogram.py +267 -0
- algo_cli/intelligence/lsp_integration.py +147 -0
- algo_cli/intelligence/memory_evolution.py +117 -0
- algo_cli/intelligence/minhash_lsh.py +182 -0
- algo_cli/intelligence/multi_model_score.py +174 -0
- algo_cli/intelligence/multi_tier_grade.py +211 -0
- algo_cli/intelligence/negative_controls.py +113 -0
- algo_cli/intelligence/numeric_clamp.py +63 -0
- algo_cli/intelligence/occ_editor.py +66 -0
- algo_cli/intelligence/output_normalize.py +112 -0
- algo_cli/intelligence/parallel_delegation.py +98 -0
- algo_cli/intelligence/parallel_fanout.py +104 -0
- algo_cli/intelligence/permission_modes.py +105 -0
- algo_cli/intelligence/pre_push_gate.py +68 -0
- algo_cli/intelligence/prefetch.py +171 -0
- algo_cli/intelligence/process_framework.py +217 -0
- algo_cli/intelligence/project_graph.py +387 -0
- algo_cli/intelligence/query_expansion.py +146 -0
- algo_cli/intelligence/ralph_loop.py +117 -0
- algo_cli/intelligence/rate_limiter.py +153 -0
- algo_cli/intelligence/refactor_transaction.py +94 -0
- algo_cli/intelligence/research_workspace.py +108 -0
- algo_cli/intelligence/retraction_ledger.py +72 -0
- algo_cli/intelligence/saga_pattern.py +88 -0
- algo_cli/intelligence/session_fork.py +100 -0
- algo_cli/intelligence/shadow_editor.py +67 -0
- algo_cli/intelligence/shell_session.py +213 -0
- algo_cli/intelligence/source_registry.py +143 -0
- algo_cli/intelligence/spawn_scales.py +99 -0
- algo_cli/intelligence/stat_stability.py +104 -0
- algo_cli/intelligence/structural_validator.py +148 -0
- algo_cli/intelligence/subagent_spawner.py +111 -0
- algo_cli/intelligence/symmetric_verify.py +70 -0
- algo_cli/intelligence/task_classifier.py +129 -0
- algo_cli/intelligence/team_execution.py +122 -0
- algo_cli/intelligence/tiered_access.py +121 -0
- algo_cli/intelligence/utility_registry.py +159 -0
- algo_cli/intuition_engine.py +560 -0
- algo_cli/intuition_injector.py +82 -0
- algo_cli/kernels/__init__.py +5 -0
- algo_cli/kernels/manifest.py +763 -0
- algo_cli/main.py +3903 -0
- algo_cli/memory_candidates.py +541 -0
- algo_cli/memory_echo_veil.py +394 -0
- algo_cli/memory_runtime.py +112 -0
- algo_cli/model_info.py +548 -0
- algo_cli/model_profile.py +160 -0
- algo_cli/model_routing.py +74 -0
- algo_cli/oneshot.py +331 -0
- algo_cli/perf_telemetry.py +389 -0
- algo_cli/plugins.py +245 -0
- algo_cli/private_event_store.py +654 -0
- algo_cli/quantization/__init__.py +24 -0
- algo_cli/quantization/lloyd_max.py +98 -0
- algo_cli/quantization/turbo_quant.py +308 -0
- algo_cli/reasoning/__init__.py +46 -0
- algo_cli/reasoning/combinatorial.py +356 -0
- algo_cli/reasoning/graph_of_thought.py +297 -0
- algo_cli/reasoning/mcts.py +220 -0
- algo_cli/reasoning/neuro_symbolic.py +250 -0
- algo_cli/reasoning/react.py +246 -0
- algo_cli/reasoning/reflexion.py +225 -0
- algo_cli/reasoning/tree_of_thought.py +241 -0
- algo_cli/reasoning_bridge.py +150 -0
- algo_cli/reconciliation.py +284 -0
- algo_cli/reflex.py +385 -0
- algo_cli/resources/docs/ALGO.md +13958 -0
- algo_cli/resources/docs/algo-cli-algorithm-evidence-contract.md +60 -0
- algo_cli/resources/docs/algo-cli-execution-verification-contract.md +59 -0
- algo_cli/resources/docs/algo-cli-memory-lifecycle-contract.md +72 -0
- algo_cli/resources/docs/harness-extension-cleanup-recommendation.md +41 -0
- algo_cli/resources/docs/index-compute-lab-integration.md +32 -0
- algo_cli/resources/docs/inference-harness-loop-blueprint-2026-06.md +55 -0
- algo_cli/resources/docs/main-split-map.md +35 -0
- algo_cli/resources/docs/privacy-and-context.md +48 -0
- algo_cli/resources/docs/reflex-loop-v0.2.md +354 -0
- algo_cli/resources/skills/README.md +26 -0
- algo_cli/resources/skills/algo-cli.md +59 -0
- algo_cli/resources/skills/edit-file-precision.md +49 -0
- algo_cli/resources/skills/harness-search-first.md +47 -0
- algo_cli/resources/skills/memory-recall-ritual.md +51 -0
- algo_cli/resources/skills/qol-algorithms.md +224 -0
- algo_cli/resources/skills/smart-error-recovery.md +56 -0
- algo_cli/resources/skills/tool-selection-cheatsheet.md +65 -0
- algo_cli/retrieval_algorithms.py +127 -0
- algo_cli/runtime_qos.py +236 -0
- algo_cli/runtime_services.py +320 -0
- algo_cli/session_commands.py +95 -0
- algo_cli/session_mode.py +113 -0
- algo_cli/skills.py +430 -0
- algo_cli/slash_dispatch.py +1265 -0
- algo_cli/small_context.py +206 -0
- algo_cli/spawn_budget.py +89 -0
- algo_cli/task_ledger.py +84 -0
- algo_cli/task_router.py +197 -0
- algo_cli/tool_context.py +94 -0
- algo_cli/tool_contract.py +99 -0
- algo_cli/tool_policy.py +357 -0
- algo_cli/tool_runtime.py +647 -0
- algo_cli/tools.py +3056 -0
- algo_cli/url_scheme.py +174 -0
- algo_cli/verify.py +154 -0
- algo_cli/version_manifest.py +178 -0
- algo_cli/vision_screenshot_verify.py +76 -0
- algo_cli/workspace_resolver.py +68 -0
- algo_cli/x_account.py +209 -0
- algo_cli/xai_auth.py +374 -0
- algo_cli/xai_client.py +600 -0
- algo_cli_runtime-0.14.0.dist-info/METADATA +369 -0
- algo_cli_runtime-0.14.0.dist-info/RECORD +237 -0
- algo_cli_runtime-0.14.0.dist-info/WHEEL +4 -0
- algo_cli_runtime-0.14.0.dist-info/entry_points.txt +3 -0
- algo_cli_runtime-0.14.0.dist-info/licenses/LICENSE +21 -0
- ollama_cli/__init__.py +67 -0
algo_cli/identity.py
ADDED
|
@@ -0,0 +1,557 @@
|
|
|
1
|
+
"""Native identity layer for Algo CLI.
|
|
2
|
+
|
|
3
|
+
Four user-editable Markdown files in ~/.algo_cli/identity/ define the CLI's
|
|
4
|
+
persona, the user's profile, and accumulated lessons. Contents are cached by
|
|
5
|
+
mtime and prepended to the system prompt on every turn.
|
|
6
|
+
|
|
7
|
+
Design note: reads are stat-gated. On an unchanged turn the only cost is four
|
|
8
|
+
os.stat() calls (microseconds). On a changed file the cost is one read of that
|
|
9
|
+
file. Nothing is re-embedded or re-tokenized here; this module is the fast
|
|
10
|
+
always-inject layer.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import math
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from datetime import datetime
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Any, Callable
|
|
21
|
+
|
|
22
|
+
try:
|
|
23
|
+
import numpy as _np
|
|
24
|
+
_NUMPY = True
|
|
25
|
+
except ImportError:
|
|
26
|
+
_np = None # type: ignore[assignment]
|
|
27
|
+
_NUMPY = False
|
|
28
|
+
|
|
29
|
+
from .cache_admission import WindowTinyLFUCache
|
|
30
|
+
from .config import CONFIG_DIR, _atomic_write_text, _exclusive_state_lock
|
|
31
|
+
from .retrieval_algorithms import stable_top_k
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
IDENTITY_DIR = CONFIG_DIR / "identity"
|
|
35
|
+
SOUL_PATH = IDENTITY_DIR / "SOUL.md"
|
|
36
|
+
IDENTITY_PATH = IDENTITY_DIR / "IDENTITY.md"
|
|
37
|
+
USER_PATH = IDENTITY_DIR / "USER.md"
|
|
38
|
+
LESSONS_PATH = IDENTITY_DIR / "lessons-learned.md"
|
|
39
|
+
LESSONS_INDEX_PATH = IDENTITY_DIR / "lessons_index.json"
|
|
40
|
+
|
|
41
|
+
ALL_PATHS: tuple[Path, ...] = (SOUL_PATH, IDENTITY_PATH, USER_PATH, LESSONS_PATH)
|
|
42
|
+
|
|
43
|
+
DEFAULT_EMBED_MODEL = "qwen3-embedding:latest"
|
|
44
|
+
QUERY_VEC_CACHE_SIZE = 32
|
|
45
|
+
LESSON_MIN_CHARS = 30
|
|
46
|
+
LESSONS_INDEX_VERSION = 2
|
|
47
|
+
|
|
48
|
+
EmbedFn = Callable[[list[str]], list[list[float]]]
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
DEFAULT_SOUL = """# Algo CLI - Soul
|
|
52
|
+
|
|
53
|
+
## Voice
|
|
54
|
+
- Concise. Terminal-native. No filler greetings.
|
|
55
|
+
- Honest about uncertainty. Say "I don't know" when you don't.
|
|
56
|
+
- Show file paths and exact commands when relevant.
|
|
57
|
+
|
|
58
|
+
## Operating values
|
|
59
|
+
- Local-first. Prefer local Ollama before cloud.
|
|
60
|
+
- Read before writing. Verify before claiming.
|
|
61
|
+
- Treat memory and wiki as navigation, not authority.
|
|
62
|
+
- Format code blocks with language tags.
|
|
63
|
+
|
|
64
|
+
## Behavior
|
|
65
|
+
- Use tools when they materially help. Do not narrate them.
|
|
66
|
+
- Do not run destructive commands without explicit approval.
|
|
67
|
+
- After a tool call, summarize the result before the next step.
|
|
68
|
+
"""
|
|
69
|
+
|
|
70
|
+
LEGACY_DEFAULT_IDENTITY = """# Algo CLI - Identity
|
|
71
|
+
|
|
72
|
+
You are Algo CLI: a local-first terminal coding assistant built on Ollama models.
|
|
73
|
+
|
|
74
|
+
You run on the user's machine. You have direct access to their filesystem through tools.
|
|
75
|
+
You can search a harness of skills, prompts, memories, and wiki pages across the user's
|
|
76
|
+
agent ecosystem (Codex, Claude, OpenClaw, Mercury, Pi, and shared .agents).
|
|
77
|
+
|
|
78
|
+
You are not a chatbot. You are a working partner for terminal tasks.
|
|
79
|
+
"""
|
|
80
|
+
|
|
81
|
+
DEFAULT_IDENTITY = """# Algo CLI - Identity
|
|
82
|
+
|
|
83
|
+
You are Algo CLI: a local-first agent runtime for coding, research, and operational work.
|
|
84
|
+
|
|
85
|
+
You run on the user's machine and can use local or connected cloud inference through
|
|
86
|
+
Ollama, Ollama Cloud, xAI Grok, and ChatGPT/Codex. You plan, act with tools, verify
|
|
87
|
+
results, and retain useful context across sessions.
|
|
88
|
+
|
|
89
|
+
Your harness searches Algo CLI's built-in skills, documentation, memories, and
|
|
90
|
+
repository intelligence. Additional local agent stores (Codex, Claude, OpenClaw,
|
|
91
|
+
Mercury, Pi, and shared .agents) are available only after the user explicitly enables
|
|
92
|
+
them. You are a working partner, not a passive chat interface.
|
|
93
|
+
"""
|
|
94
|
+
|
|
95
|
+
DEFAULT_USER = """# About the User
|
|
96
|
+
|
|
97
|
+
<!-- Edit this file to teach the CLI about yourself. The more specific, the better. -->
|
|
98
|
+
|
|
99
|
+
## Who I am
|
|
100
|
+
(Your name, role, what you work on.)
|
|
101
|
+
|
|
102
|
+
## How I work
|
|
103
|
+
(Editors, languages, shells, workflow preferences.)
|
|
104
|
+
|
|
105
|
+
## What I want from this CLI
|
|
106
|
+
(Be terse? Verbose? Ask before acting? Prefer local models?)
|
|
107
|
+
|
|
108
|
+
## Things to never do
|
|
109
|
+
(Hard nos, e.g. "never auto-push to main".)
|
|
110
|
+
"""
|
|
111
|
+
|
|
112
|
+
DEFAULT_LESSONS = """# Lessons Learned
|
|
113
|
+
|
|
114
|
+
<!-- Append new lessons below. Each lesson is a paragraph or short list, separated by blank lines. -->
|
|
115
|
+
<!-- Add via /lesson <text> or by editing this file directly. -->
|
|
116
|
+
"""
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
_DEFAULTS: dict[Path, str] = {
|
|
120
|
+
SOUL_PATH: DEFAULT_SOUL,
|
|
121
|
+
IDENTITY_PATH: DEFAULT_IDENTITY,
|
|
122
|
+
USER_PATH: DEFAULT_USER,
|
|
123
|
+
LESSONS_PATH: DEFAULT_LESSONS,
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
@dataclass
|
|
128
|
+
class CacheEntry:
|
|
129
|
+
mtime_ns: int
|
|
130
|
+
content: str
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
_CACHE: dict[Path, CacheEntry] = {}
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def scaffold_if_needed() -> list[Path]:
|
|
137
|
+
"""Create missing identity files and refresh the untouched legacy identity."""
|
|
138
|
+
created: list[Path] = []
|
|
139
|
+
IDENTITY_DIR.mkdir(parents=True, exist_ok=True)
|
|
140
|
+
for path, default in _DEFAULTS.items():
|
|
141
|
+
if not path.exists():
|
|
142
|
+
_atomic_write_text(path, default)
|
|
143
|
+
created.append(path)
|
|
144
|
+
elif path == IDENTITY_PATH:
|
|
145
|
+
try:
|
|
146
|
+
existing = path.read_text(encoding="utf-8")
|
|
147
|
+
except OSError:
|
|
148
|
+
continue
|
|
149
|
+
if existing == LEGACY_DEFAULT_IDENTITY:
|
|
150
|
+
_atomic_write_text(path, DEFAULT_IDENTITY)
|
|
151
|
+
_CACHE.pop(path, None)
|
|
152
|
+
return created
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def read_cached(path: Path) -> str:
|
|
156
|
+
"""Read file content, mtime-cached. Returns '' if missing or unreadable."""
|
|
157
|
+
if not path.exists():
|
|
158
|
+
return ""
|
|
159
|
+
try:
|
|
160
|
+
mtime_ns = path.stat().st_mtime_ns
|
|
161
|
+
except OSError:
|
|
162
|
+
return ""
|
|
163
|
+
entry = _CACHE.get(path)
|
|
164
|
+
if entry and entry.mtime_ns == mtime_ns:
|
|
165
|
+
return entry.content
|
|
166
|
+
try:
|
|
167
|
+
content = path.read_text(encoding="utf-8", errors="replace")
|
|
168
|
+
except OSError:
|
|
169
|
+
return ""
|
|
170
|
+
_CACHE[path] = CacheEntry(mtime_ns=mtime_ns, content=content)
|
|
171
|
+
return content
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def detect_changes() -> list[Path]:
|
|
175
|
+
"""List identity files whose mtime differs from the cache. Stat-only, no reads."""
|
|
176
|
+
changed: list[Path] = []
|
|
177
|
+
for path in ALL_PATHS:
|
|
178
|
+
if not path.exists():
|
|
179
|
+
continue
|
|
180
|
+
try:
|
|
181
|
+
mtime_ns = path.stat().st_mtime_ns
|
|
182
|
+
except OSError:
|
|
183
|
+
continue
|
|
184
|
+
entry = _CACHE.get(path)
|
|
185
|
+
if entry is None or entry.mtime_ns != mtime_ns:
|
|
186
|
+
changed.append(path)
|
|
187
|
+
return changed
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def identity_mtime_key() -> tuple[int, ...]:
|
|
191
|
+
"""Stable fingerprint of identity file mtimes for context-usage cache keys."""
|
|
192
|
+
key: list[int] = []
|
|
193
|
+
for path in ALL_PATHS:
|
|
194
|
+
if not path.exists():
|
|
195
|
+
key.append(0)
|
|
196
|
+
continue
|
|
197
|
+
try:
|
|
198
|
+
key.append(path.stat().st_mtime_ns)
|
|
199
|
+
except OSError:
|
|
200
|
+
key.append(0)
|
|
201
|
+
return tuple(key)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def build_identity_block(retrieved_lessons: list[str] | None = None) -> str:
|
|
205
|
+
"""Assemble the identity prefix for the system prompt.
|
|
206
|
+
|
|
207
|
+
If retrieved_lessons is None: full lessons-learned.md is inlined (fallback path).
|
|
208
|
+
If retrieved_lessons is a list (possibly empty): use only those chunks.
|
|
209
|
+
"""
|
|
210
|
+
identity_text = read_cached(IDENTITY_PATH).strip()
|
|
211
|
+
soul_text = read_cached(SOUL_PATH).strip()
|
|
212
|
+
user_text = read_cached(USER_PATH).strip()
|
|
213
|
+
parts: list[str] = []
|
|
214
|
+
if identity_text:
|
|
215
|
+
parts.append(f"## Identity\n{identity_text}")
|
|
216
|
+
if soul_text:
|
|
217
|
+
parts.append(f"## Soul\n{soul_text}")
|
|
218
|
+
if user_text:
|
|
219
|
+
parts.append(f"## About the User\n{user_text}")
|
|
220
|
+
if retrieved_lessons is not None:
|
|
221
|
+
if retrieved_lessons:
|
|
222
|
+
joined = "\n\n".join(retrieved_lessons)
|
|
223
|
+
parts.append(f"## Relevant Lessons\n{joined}")
|
|
224
|
+
else:
|
|
225
|
+
lessons_text = read_cached(LESSONS_PATH).strip()
|
|
226
|
+
if lessons_text:
|
|
227
|
+
parts.append(f"## Lessons Learned\n{lessons_text}")
|
|
228
|
+
return "\n\n".join(parts)
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
# ---------- Lessons embedding / RAG ----------
|
|
232
|
+
|
|
233
|
+
_LESSONS_INDEX: dict[str, Any] | None = None
|
|
234
|
+
_QUERY_VEC_CACHE: WindowTinyLFUCache[tuple[str, int, str], list[float]] = WindowTinyLFUCache(
|
|
235
|
+
max(1, QUERY_VEC_CACHE_SIZE)
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _chunk_lessons(text: str) -> list[str]:
|
|
240
|
+
"""Split lessons-learned.md into chunks at `## ` headings. Drops top-level title."""
|
|
241
|
+
if not text.strip():
|
|
242
|
+
return []
|
|
243
|
+
chunks: list[str] = []
|
|
244
|
+
current: list[str] = []
|
|
245
|
+
for line in text.splitlines():
|
|
246
|
+
if line.startswith("## "):
|
|
247
|
+
if current:
|
|
248
|
+
joined = "\n".join(current).strip()
|
|
249
|
+
if joined:
|
|
250
|
+
chunks.append(joined)
|
|
251
|
+
current = [line]
|
|
252
|
+
elif current or line.strip():
|
|
253
|
+
current.append(line)
|
|
254
|
+
if current:
|
|
255
|
+
joined = "\n".join(current).strip()
|
|
256
|
+
if joined:
|
|
257
|
+
chunks.append(joined)
|
|
258
|
+
return [c for c in chunks if len(c) >= LESSON_MIN_CHARS and not c.startswith("# Lessons Learned")]
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def _cosine(a: list[float], b: list[float]) -> float:
|
|
262
|
+
if not a or not b or len(a) != len(b):
|
|
263
|
+
return 0.0
|
|
264
|
+
dot = 0.0
|
|
265
|
+
na = 0.0
|
|
266
|
+
nb = 0.0
|
|
267
|
+
for x, y in zip(a, b):
|
|
268
|
+
dot += x * y
|
|
269
|
+
na += x * x
|
|
270
|
+
nb += y * y
|
|
271
|
+
if na == 0.0 or nb == 0.0:
|
|
272
|
+
return 0.0
|
|
273
|
+
return dot / (math.sqrt(na) * math.sqrt(nb))
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def _load_lessons_index() -> dict[str, Any] | None:
|
|
277
|
+
global _LESSONS_INDEX
|
|
278
|
+
if _LESSONS_INDEX is not None:
|
|
279
|
+
return _LESSONS_INDEX
|
|
280
|
+
if not LESSONS_INDEX_PATH.exists():
|
|
281
|
+
return None
|
|
282
|
+
try:
|
|
283
|
+
_LESSONS_INDEX = json.loads(LESSONS_INDEX_PATH.read_text(encoding="utf-8"))
|
|
284
|
+
except (OSError, json.JSONDecodeError):
|
|
285
|
+
_LESSONS_INDEX = None
|
|
286
|
+
return _LESSONS_INDEX
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def _save_lessons_index(idx: dict[str, Any]) -> None:
|
|
290
|
+
global _LESSONS_INDEX
|
|
291
|
+
IDENTITY_DIR.mkdir(parents=True, exist_ok=True)
|
|
292
|
+
_atomic_write_text(LESSONS_INDEX_PATH, json.dumps(idx, indent=2))
|
|
293
|
+
_LESSONS_INDEX = idx
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def _index_embedding_model(idx: dict[str, Any]) -> str:
|
|
297
|
+
"""Return the persisted embedding identity, including legacy index support."""
|
|
298
|
+
return str(idx.get("embedding_model") or idx.get("model") or "").strip()
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def _normalise_vector(value: Any) -> list[float] | None:
|
|
302
|
+
if not isinstance(value, (list, tuple)) or not value:
|
|
303
|
+
return None
|
|
304
|
+
try:
|
|
305
|
+
vector = [float(component) for component in value]
|
|
306
|
+
except (TypeError, ValueError):
|
|
307
|
+
return None
|
|
308
|
+
if not all(math.isfinite(component) for component in vector):
|
|
309
|
+
return None
|
|
310
|
+
return vector
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def _index_vector_dimensions(idx: dict[str, Any]) -> int | None:
|
|
314
|
+
"""Return a validated vector width, 0 for an empty index, or None if corrupt."""
|
|
315
|
+
chunks = idx.get("chunks")
|
|
316
|
+
if not isinstance(chunks, list):
|
|
317
|
+
return None
|
|
318
|
+
if not chunks:
|
|
319
|
+
return 0
|
|
320
|
+
|
|
321
|
+
dimensions: int | None = None
|
|
322
|
+
for chunk in chunks:
|
|
323
|
+
if not isinstance(chunk, dict):
|
|
324
|
+
return None
|
|
325
|
+
vector = _normalise_vector(chunk.get("vector"))
|
|
326
|
+
if vector is None:
|
|
327
|
+
return None
|
|
328
|
+
if dimensions is None:
|
|
329
|
+
dimensions = len(vector)
|
|
330
|
+
elif len(vector) != dimensions:
|
|
331
|
+
return None
|
|
332
|
+
|
|
333
|
+
stored_dimensions = idx.get("vector_dimensions")
|
|
334
|
+
if stored_dimensions is not None:
|
|
335
|
+
try:
|
|
336
|
+
persisted = int(stored_dimensions)
|
|
337
|
+
except (TypeError, ValueError):
|
|
338
|
+
return None
|
|
339
|
+
if persisted != dimensions:
|
|
340
|
+
return None
|
|
341
|
+
return dimensions
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
def lessons_index_stale(
|
|
345
|
+
model: str | None = None,
|
|
346
|
+
dimensions: int | None = None,
|
|
347
|
+
) -> bool:
|
|
348
|
+
"""Return whether lesson text or the requested embedding space changed.
|
|
349
|
+
|
|
350
|
+
Older indexes remain readable, but an active model identity mismatch forces
|
|
351
|
+
a rebuild even when the source Markdown has not changed. A configured
|
|
352
|
+
vector width is checked when supplied.
|
|
353
|
+
"""
|
|
354
|
+
if not LESSONS_PATH.exists():
|
|
355
|
+
return False
|
|
356
|
+
try:
|
|
357
|
+
file_mtime = LESSONS_PATH.stat().st_mtime_ns
|
|
358
|
+
except OSError:
|
|
359
|
+
return False
|
|
360
|
+
idx = _load_lessons_index()
|
|
361
|
+
if idx is None:
|
|
362
|
+
return True
|
|
363
|
+
try:
|
|
364
|
+
index_mtime = int(idx.get("mtime_ns", -1))
|
|
365
|
+
except (TypeError, ValueError):
|
|
366
|
+
return True
|
|
367
|
+
if index_mtime != file_mtime:
|
|
368
|
+
return True
|
|
369
|
+
if model is not None and _index_embedding_model(idx) != model.strip():
|
|
370
|
+
return True
|
|
371
|
+
index_dimensions = _index_vector_dimensions(idx)
|
|
372
|
+
if index_dimensions is None:
|
|
373
|
+
return True
|
|
374
|
+
if dimensions is not None and index_dimensions not in {0, int(dimensions)}:
|
|
375
|
+
return True
|
|
376
|
+
return False
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
def lessons_index_status() -> dict[str, Any]:
|
|
380
|
+
if not LESSONS_PATH.exists():
|
|
381
|
+
return {"file": False, "index": False, "chunk_count": 0, "model": None, "stale": False}
|
|
382
|
+
idx = _load_lessons_index()
|
|
383
|
+
if idx is None:
|
|
384
|
+
return {"file": True, "index": False, "chunk_count": 0, "model": None, "stale": True}
|
|
385
|
+
return {
|
|
386
|
+
"file": True,
|
|
387
|
+
"index": True,
|
|
388
|
+
"chunk_count": len(idx.get("chunks", [])),
|
|
389
|
+
"model": _index_embedding_model(idx) or None,
|
|
390
|
+
"dimensions": _index_vector_dimensions(idx),
|
|
391
|
+
"stale": lessons_index_stale(),
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def rebuild_lessons_index(
|
|
396
|
+
embed_fn: EmbedFn,
|
|
397
|
+
model: str = DEFAULT_EMBED_MODEL,
|
|
398
|
+
*,
|
|
399
|
+
expected_dimensions: int | None = None,
|
|
400
|
+
) -> dict[str, Any]:
|
|
401
|
+
"""Embed all lesson chunks in one batched call. Persists to disk."""
|
|
402
|
+
if not LESSONS_PATH.exists():
|
|
403
|
+
return {"chunk_count": 0, "ready": False, "reason": "no_file"}
|
|
404
|
+
try:
|
|
405
|
+
text = LESSONS_PATH.read_text(encoding="utf-8", errors="replace")
|
|
406
|
+
mtime_ns = LESSONS_PATH.stat().st_mtime_ns
|
|
407
|
+
except OSError as exc:
|
|
408
|
+
return {"chunk_count": 0, "ready": False, "reason": f"read_error: {exc}"}
|
|
409
|
+
chunks = _chunk_lessons(text)
|
|
410
|
+
if not chunks:
|
|
411
|
+
idx = {
|
|
412
|
+
"version": LESSONS_INDEX_VERSION,
|
|
413
|
+
"mtime_ns": mtime_ns,
|
|
414
|
+
"model": model,
|
|
415
|
+
"embedding_model": model,
|
|
416
|
+
"vector_dimensions": 0,
|
|
417
|
+
"chunks": [],
|
|
418
|
+
}
|
|
419
|
+
_save_lessons_index(idx)
|
|
420
|
+
return {"chunk_count": 0, "dimensions": 0, "ready": True}
|
|
421
|
+
try:
|
|
422
|
+
vectors = embed_fn(chunks)
|
|
423
|
+
except Exception as exc:
|
|
424
|
+
return {"chunk_count": 0, "ready": False, "reason": f"embed_error: {exc}"}
|
|
425
|
+
if len(vectors) != len(chunks):
|
|
426
|
+
return {"chunk_count": 0, "ready": False, "reason": "embed_count_mismatch"}
|
|
427
|
+
normalised_vectors: list[list[float]] = []
|
|
428
|
+
vector_dimensions: int | None = None
|
|
429
|
+
for vector in vectors:
|
|
430
|
+
normalised = _normalise_vector(vector)
|
|
431
|
+
if normalised is None:
|
|
432
|
+
return {"chunk_count": 0, "ready": False, "reason": "invalid_embedding_vector"}
|
|
433
|
+
if vector_dimensions is None:
|
|
434
|
+
vector_dimensions = len(normalised)
|
|
435
|
+
elif len(normalised) != vector_dimensions:
|
|
436
|
+
return {"chunk_count": 0, "ready": False, "reason": "embed_dimension_mismatch"}
|
|
437
|
+
normalised_vectors.append(normalised)
|
|
438
|
+
assert vector_dimensions is not None
|
|
439
|
+
if expected_dimensions is not None and vector_dimensions != int(expected_dimensions):
|
|
440
|
+
return {
|
|
441
|
+
"chunk_count": 0,
|
|
442
|
+
"ready": False,
|
|
443
|
+
"reason": "embed_dimension_mismatch",
|
|
444
|
+
"expected_dimensions": int(expected_dimensions),
|
|
445
|
+
"actual_dimensions": vector_dimensions,
|
|
446
|
+
}
|
|
447
|
+
idx = {
|
|
448
|
+
"version": LESSONS_INDEX_VERSION,
|
|
449
|
+
"mtime_ns": mtime_ns,
|
|
450
|
+
"model": model,
|
|
451
|
+
"embedding_model": model,
|
|
452
|
+
"vector_dimensions": vector_dimensions,
|
|
453
|
+
"chunks": [
|
|
454
|
+
{"text": chunk, "vector": vector}
|
|
455
|
+
for chunk, vector in zip(chunks, normalised_vectors)
|
|
456
|
+
],
|
|
457
|
+
}
|
|
458
|
+
_save_lessons_index(idx)
|
|
459
|
+
_QUERY_VEC_CACHE.clear()
|
|
460
|
+
return {"chunk_count": len(chunks), "dimensions": vector_dimensions, "ready": True}
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
def retrieve_lessons(query: str, embed_fn: EmbedFn, model: str = DEFAULT_EMBED_MODEL, k: int = 5) -> list[str]:
|
|
464
|
+
"""Return top-K lesson chunks by cosine similarity. Empty list if index unavailable."""
|
|
465
|
+
idx = _load_lessons_index()
|
|
466
|
+
if idx is None or not idx.get("chunks"):
|
|
467
|
+
return []
|
|
468
|
+
if lessons_index_stale(model):
|
|
469
|
+
return []
|
|
470
|
+
vector_dimensions = _index_vector_dimensions(idx)
|
|
471
|
+
if vector_dimensions is None or vector_dimensions <= 0:
|
|
472
|
+
return []
|
|
473
|
+
query = (query or "").strip()
|
|
474
|
+
if not query:
|
|
475
|
+
return []
|
|
476
|
+
cache_key = (model, vector_dimensions, query)
|
|
477
|
+
_QUERY_VEC_CACHE.resize(max(1, QUERY_VEC_CACHE_SIZE))
|
|
478
|
+
qvec = _QUERY_VEC_CACHE.get(cache_key)
|
|
479
|
+
if qvec is None:
|
|
480
|
+
try:
|
|
481
|
+
vecs = embed_fn([query])
|
|
482
|
+
except Exception:
|
|
483
|
+
return []
|
|
484
|
+
if not vecs:
|
|
485
|
+
return []
|
|
486
|
+
qvec = _normalise_vector(vecs[0])
|
|
487
|
+
if qvec is None or len(qvec) != vector_dimensions:
|
|
488
|
+
return []
|
|
489
|
+
_QUERY_VEC_CACHE.put(cache_key, qvec)
|
|
490
|
+
elif len(qvec) != vector_dimensions:
|
|
491
|
+
return []
|
|
492
|
+
|
|
493
|
+
chunks = idx["chunks"]
|
|
494
|
+
vectors = [_normalise_vector(chunk.get("vector")) for chunk in chunks]
|
|
495
|
+
if any(vector is None or len(vector) != vector_dimensions for vector in vectors):
|
|
496
|
+
return []
|
|
497
|
+
safe_vectors = [vector for vector in vectors if vector is not None]
|
|
498
|
+
if _NUMPY and chunks:
|
|
499
|
+
import numpy as np
|
|
500
|
+
mat = np.array(safe_vectors, dtype=np.float32)
|
|
501
|
+
qv = np.array(qvec, dtype=np.float32)
|
|
502
|
+
mat_norms = np.linalg.norm(mat, axis=1)
|
|
503
|
+
q_norm = float(np.linalg.norm(qv))
|
|
504
|
+
if q_norm <= 0.0:
|
|
505
|
+
return []
|
|
506
|
+
denom = mat_norms * q_norm
|
|
507
|
+
sims = np.divide(mat @ qv, denom, out=np.zeros_like(mat_norms), where=denom > 0).tolist()
|
|
508
|
+
scored: list[tuple[float, str]] = [
|
|
509
|
+
(float(s), chunks[i]["text"]) for i, s in enumerate(sims) if s > 0.0
|
|
510
|
+
]
|
|
511
|
+
else:
|
|
512
|
+
scored = []
|
|
513
|
+
for chunk, vector in zip(chunks, safe_vectors):
|
|
514
|
+
sim = _cosine(qvec, vector)
|
|
515
|
+
if sim > 0.0:
|
|
516
|
+
scored.append((sim, chunk["text"]))
|
|
517
|
+
top_scored = stable_top_k(scored, k, score=lambda pair: pair[0])
|
|
518
|
+
return [text for _, text in top_scored]
|
|
519
|
+
|
|
520
|
+
|
|
521
|
+
def status() -> list[dict[str, Any]]:
|
|
522
|
+
"""Per-file metadata for the /identity command."""
|
|
523
|
+
rows: list[dict[str, Any]] = []
|
|
524
|
+
for path in ALL_PATHS:
|
|
525
|
+
row: dict[str, Any] = {"path": str(path), "name": path.name, "exists": path.exists()}
|
|
526
|
+
if path.exists():
|
|
527
|
+
try:
|
|
528
|
+
st = path.stat()
|
|
529
|
+
row["size"] = int(st.st_size)
|
|
530
|
+
row["modified"] = datetime.fromtimestamp(st.st_mtime).strftime("%Y-%m-%d %H:%M")
|
|
531
|
+
except OSError:
|
|
532
|
+
row["size"] = 0
|
|
533
|
+
row["modified"] = "?"
|
|
534
|
+
rows.append(row)
|
|
535
|
+
return rows
|
|
536
|
+
|
|
537
|
+
|
|
538
|
+
def append_lesson(text: str) -> Path:
|
|
539
|
+
"""Append a timestamped lesson to lessons-learned.md and invalidate its cache."""
|
|
540
|
+
IDENTITY_DIR.mkdir(parents=True, exist_ok=True)
|
|
541
|
+
if not LESSONS_PATH.exists():
|
|
542
|
+
_atomic_write_text(LESSONS_PATH, DEFAULT_LESSONS)
|
|
543
|
+
stamp = datetime.now().strftime("%Y-%m-%d %H:%M")
|
|
544
|
+
entry = f"\n## {stamp}\n{text.strip()}\n"
|
|
545
|
+
with _exclusive_state_lock(LESSONS_PATH):
|
|
546
|
+
with LESSONS_PATH.open("a", encoding="utf-8") as handle:
|
|
547
|
+
handle.write(entry)
|
|
548
|
+
_CACHE.pop(LESSONS_PATH, None)
|
|
549
|
+
return LESSONS_PATH
|
|
550
|
+
|
|
551
|
+
|
|
552
|
+
def write_user_profile(content: str) -> Path:
|
|
553
|
+
"""Overwrite USER.md with new content and invalidate its cache."""
|
|
554
|
+
IDENTITY_DIR.mkdir(parents=True, exist_ok=True)
|
|
555
|
+
_atomic_write_text(USER_PATH, content)
|
|
556
|
+
_CACHE.pop(USER_PATH, None)
|
|
557
|
+
return USER_PATH
|