fidelis-memory 0.0.93__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fidelis/__init__.py +22 -0
- fidelis/augment.py +111 -0
- fidelis/calibrate.py +214 -0
- fidelis/cli.py +353 -0
- fidelis/config.py +187 -0
- fidelis/degrade.py +263 -0
- fidelis/ingest_claude_sessions.py +355 -0
- fidelis/init_cmd.py +365 -0
- fidelis/lpci.py +252 -0
- fidelis/mcp_cmd.py +122 -0
- fidelis/mcp_server.py +204 -0
- fidelis/recall.py +296 -0
- fidelis/recall_b.py +346 -0
- fidelis/recall_hybrid.py +653 -0
- fidelis/recall_sessions.py +266 -0
- fidelis/scaffold/__init__.py +45 -0
- fidelis/scaffold/_core.py +156 -0
- fidelis/scaffold/preflight.py +153 -0
- fidelis/scaffold_server.py +193 -0
- fidelis/seed.py +373 -0
- fidelis/server.py +440 -0
- fidelis/snapshot.py +225 -0
- fidelis/telemetry.py +86 -0
- fidelis/watch_cmd.py +257 -0
- fidelis_memory-0.0.93.dist-info/METADATA +316 -0
- fidelis_memory-0.0.93.dist-info/RECORD +29 -0
- fidelis_memory-0.0.93.dist-info/WHEEL +4 -0
- fidelis_memory-0.0.93.dist-info/entry_points.txt +4 -0
- fidelis_memory-0.0.93.dist-info/licenses/LICENSE +21 -0
fidelis/__init__.py
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""
|
|
2
|
+
fidelis — faithful memory retrieval for AI agents.
|
|
3
|
+
|
|
4
|
+
Default path (zero-LLM): BM25 + dense + RRF fusion, 83.2% R@1 on
|
|
5
|
+
LongMemEval_S, $0/query, ~90ms, fully local.
|
|
6
|
+
|
|
7
|
+
Optional LLM tiers (experimental): ``filter`` and ``flagship``. The filter
|
|
8
|
+
LLM outputs only integer indices (e.g. [3, 7, 12]) — never memory text —
|
|
9
|
+
so it cannot corrupt or hallucinate into the content returned to the agent.
|
|
10
|
+
Fidelity is structural, not a prompting convention.
|
|
11
|
+
|
|
12
|
+
fidelis was previously published as ``cogito-ergo`` (0.0.8 and 0.3.0 on PyPI).
|
|
13
|
+
Data paths and env var names retain the ``cogito`` prefix for continuity
|
|
14
|
+
with existing deployments.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
__version__ = "0.0.93"
|
|
18
|
+
|
|
19
|
+
from fidelis.recall import recall # noqa: F401
|
|
20
|
+
|
|
21
|
+
# Note: avoid shadowing the ``fidelis.recall_hybrid`` submodule. Users who want
|
|
22
|
+
# the function directly can do ``from fidelis.recall_hybrid import recall_hybrid``.
|
fidelis/augment.py
ADDED
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""fidelis.augment — the one-line caller helper.
|
|
2
|
+
|
|
3
|
+
For users who want "give me memory + scaffold + answer" in one call instead of
|
|
4
|
+
wiring three separate functions. Pure wrapper; uses the running fidelis-server
|
|
5
|
+
for retrieval and the user's own LLM client for generation.
|
|
6
|
+
|
|
7
|
+
Example:
|
|
8
|
+
|
|
9
|
+
from fidelis.augment import augment
|
|
10
|
+
from anthropic import Anthropic
|
|
11
|
+
|
|
12
|
+
client = Anthropic()
|
|
13
|
+
response = augment(
|
|
14
|
+
question="What did I say about Sarah?",
|
|
15
|
+
qtype="single-session-user",
|
|
16
|
+
llm_call=lambda system, user: client.messages.create(
|
|
17
|
+
model="claude-opus-4-7", # use any current Claude Messages API model
|
|
18
|
+
system=system,
|
|
19
|
+
messages=[{"role": "user", "content": user}],
|
|
20
|
+
max_tokens=512,
|
|
21
|
+
).content[0].text,
|
|
22
|
+
)
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import json
|
|
28
|
+
import os
|
|
29
|
+
import urllib.error
|
|
30
|
+
import urllib.request
|
|
31
|
+
from typing import Callable
|
|
32
|
+
|
|
33
|
+
from fidelis.scaffold import wrap_system_prompt
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _server_url() -> str:
|
|
37
|
+
port = os.environ.get("FIDELIS_PORT", os.environ.get("COGITO_PORT", "19420"))
|
|
38
|
+
return f"http://127.0.0.1:{port}"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _recall(query: str, limit: int = 5) -> tuple[str, float | None]:
|
|
42
|
+
"""Retrieve memories from fidelis-server. Returns (formatted_context, top_score)."""
|
|
43
|
+
body = json.dumps({"text": query, "limit": limit}).encode()
|
|
44
|
+
req = urllib.request.Request(
|
|
45
|
+
f"{_server_url()}/recall",
|
|
46
|
+
data=body,
|
|
47
|
+
headers={"Content-Type": "application/json"},
|
|
48
|
+
method="POST",
|
|
49
|
+
)
|
|
50
|
+
try:
|
|
51
|
+
with urllib.request.urlopen(req, timeout=30) as resp:
|
|
52
|
+
data = json.loads(resp.read())
|
|
53
|
+
except urllib.error.URLError as e:
|
|
54
|
+
raise RuntimeError(
|
|
55
|
+
f"fidelis-server unreachable at {_server_url()}. "
|
|
56
|
+
f"Run `fidelis init` to install + start the service. ({e})"
|
|
57
|
+
) from e
|
|
58
|
+
|
|
59
|
+
memories = data.get("memories", [])
|
|
60
|
+
if not memories:
|
|
61
|
+
return "(no memories retrieved)", None
|
|
62
|
+
|
|
63
|
+
lines = []
|
|
64
|
+
top_score: float | None = None
|
|
65
|
+
for m in memories:
|
|
66
|
+
text = m.get("text", "").strip()
|
|
67
|
+
score = m.get("score")
|
|
68
|
+
if score is not None and top_score is None:
|
|
69
|
+
try:
|
|
70
|
+
top_score = float(score)
|
|
71
|
+
# mem0/cogito returns "lower is better" similarity; normalize roughly to [0, 1]
|
|
72
|
+
if top_score > 1.0:
|
|
73
|
+
# heuristic: distance scores in [0, ~500] → invert to [0, 1]
|
|
74
|
+
top_score = max(0.0, min(1.0, 1.0 - (top_score / 500.0)))
|
|
75
|
+
except (ValueError, TypeError):
|
|
76
|
+
top_score = None
|
|
77
|
+
lines.append(text)
|
|
78
|
+
|
|
79
|
+
return "\n\n---\n\n".join(lines), top_score
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def augment(
|
|
83
|
+
question: str,
|
|
84
|
+
qtype: str = "single-session-user",
|
|
85
|
+
*,
|
|
86
|
+
llm_call: Callable[[str, str], str],
|
|
87
|
+
limit: int = 5,
|
|
88
|
+
) -> str:
|
|
89
|
+
"""Retrieve memory + wrap with scaffold + invoke caller's LLM.
|
|
90
|
+
|
|
91
|
+
Args:
|
|
92
|
+
question: the user's natural-language question
|
|
93
|
+
qtype: one of single-session-user / single-session-assistant /
|
|
94
|
+
single-session-preference / knowledge-update / multi-session /
|
|
95
|
+
temporal-reasoning. Caller chooses based on their question type.
|
|
96
|
+
Default: single-session-user.
|
|
97
|
+
llm_call: a callable taking (system_prompt, user_message) and returning
|
|
98
|
+
the LLM's response text. The user supplies this so fidelis stays
|
|
99
|
+
agnostic to LLM SDK choice.
|
|
100
|
+
limit: max memories to retrieve (default 5)
|
|
101
|
+
|
|
102
|
+
Returns:
|
|
103
|
+
The LLM's response text.
|
|
104
|
+
|
|
105
|
+
Raises:
|
|
106
|
+
RuntimeError: if fidelis-server is unreachable.
|
|
107
|
+
"""
|
|
108
|
+
context, top_score = _recall(question, limit=limit)
|
|
109
|
+
system = wrap_system_prompt(qtype, top_score=top_score)
|
|
110
|
+
user_message = f"Conversation memory:\n{context}\n\nQuestion: {question}"
|
|
111
|
+
return llm_call(system, user_message)
|
fidelis/calibrate.py
ADDED
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
"""
|
|
2
|
+
cogito calibrate — one-time vocabulary bridge extraction.
|
|
3
|
+
|
|
4
|
+
Samples memories from the store, asks the filter LLM to identify vocabulary
|
|
5
|
+
gaps between natural language queries and stored technical facts, and writes
|
|
6
|
+
a vocab_map to .cogito.json.
|
|
7
|
+
|
|
8
|
+
At query time, recall_b uses the vocab_map for zero-LLM expansion:
|
|
9
|
+
"freeze" → ["timeout", "cascade", "ollama"]
|
|
10
|
+
"adoption" → ["downloads", "PyPI", "installs"]
|
|
11
|
+
|
|
12
|
+
Run once after initial seeding, and optionally after large corpus updates.
|
|
13
|
+
|
|
14
|
+
Usage:
|
|
15
|
+
cogito calibrate
|
|
16
|
+
cogito calibrate --sample 300 --dry-run
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import json
|
|
22
|
+
import random
|
|
23
|
+
import urllib.error
|
|
24
|
+
import urllib.request
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
from typing import Any
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
_CALIBRATE_SYSTEM = (
|
|
30
|
+
"You are building a vocabulary bridge for a technical memory retrieval system. "
|
|
31
|
+
"Users ask questions in plain English. Stored facts use technical jargon, "
|
|
32
|
+
"proper nouns, acronyms, and domain-specific terms.\n\n"
|
|
33
|
+
"Your job: identify pairs where a plain-English query word would NOT appear "
|
|
34
|
+
"in the stored fact, but the stored fact IS the right answer.\n\n"
|
|
35
|
+
"Output ONLY a valid JSON object. "
|
|
36
|
+
"Keys: plain-English words a user might type in a query. "
|
|
37
|
+
"Values: arrays of SHORT TECHNICAL TERMS (1-5 words each, NEVER full sentences) "
|
|
38
|
+
"that appear verbatim in the stored facts and carry the same meaning as the key. "
|
|
39
|
+
"Do NOT include full sentences as values. "
|
|
40
|
+
"Do NOT include generic stop words. "
|
|
41
|
+
"Max 60 pairs. Focus on genuine vocabulary gap cases only.\n\n"
|
|
42
|
+
'Example: {"freeze": ["timeout", "cascade", "blocked"], '
|
|
43
|
+
'"adoption": ["downloads", "PyPI", "installs"], '
|
|
44
|
+
'"broken": ["flat scores", "metadata missing", "zero results"], '
|
|
45
|
+
'"memory not found": ["flat score", "user_id", "chromadb"], '
|
|
46
|
+
'"agent stuck": ["timeout", "seed", "ollama cascade"]}'
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _sample_memories(memory: Any, user_id: str, n: int) -> list[str]:
|
|
51
|
+
"""Fetch all memories and return up to n randomly sampled texts."""
|
|
52
|
+
try:
|
|
53
|
+
raw = memory.get_all(filters={"user_id": user_id}, top_k=10000) # type: ignore
|
|
54
|
+
results = raw.get("results", [])
|
|
55
|
+
texts = [r.get("memory", "") for r in results if r.get("memory")]
|
|
56
|
+
if len(texts) > n:
|
|
57
|
+
texts = random.sample(texts, n)
|
|
58
|
+
return texts
|
|
59
|
+
except Exception as e:
|
|
60
|
+
raise RuntimeError(f"Failed to fetch memories: {e}") from e
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _build_vocab_map(
|
|
64
|
+
memories: list[str],
|
|
65
|
+
endpoint: str,
|
|
66
|
+
token: str,
|
|
67
|
+
model: str,
|
|
68
|
+
timeout: float,
|
|
69
|
+
) -> dict[str, list[str]]:
|
|
70
|
+
"""Single LLM call to extract vocabulary bridge from sampled memories."""
|
|
71
|
+
lines = [f"[{i+1}] {m[:120].replace(chr(10), ' ')}" for i, m in enumerate(memories)]
|
|
72
|
+
memories_block = "\n".join(lines)
|
|
73
|
+
|
|
74
|
+
payload = json.dumps({
|
|
75
|
+
"model": model,
|
|
76
|
+
"messages": [
|
|
77
|
+
{"role": "system", "content": _CALIBRATE_SYSTEM},
|
|
78
|
+
{"role": "user", "content": f"Memory facts:\n{memories_block}\n\nOutput the vocabulary bridge JSON:"},
|
|
79
|
+
],
|
|
80
|
+
"max_tokens": 2000,
|
|
81
|
+
"temperature": 0,
|
|
82
|
+
}).encode()
|
|
83
|
+
|
|
84
|
+
req = urllib.request.Request(
|
|
85
|
+
f"{endpoint}/v1/chat/completions",
|
|
86
|
+
data=payload,
|
|
87
|
+
headers={
|
|
88
|
+
"Content-Type": "application/json",
|
|
89
|
+
"Authorization": f"Bearer {token}",
|
|
90
|
+
},
|
|
91
|
+
method="POST",
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
try:
|
|
95
|
+
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
96
|
+
result = json.loads(resp.read())
|
|
97
|
+
raw = result["choices"][0]["message"]["content"].strip()
|
|
98
|
+
except urllib.error.URLError as e:
|
|
99
|
+
raise RuntimeError(f"LLM call failed: {e}") from e
|
|
100
|
+
except Exception as e:
|
|
101
|
+
raise RuntimeError(f"Unexpected error: {e}") from e
|
|
102
|
+
|
|
103
|
+
# Strip thinking tokens (<think>...</think>)
|
|
104
|
+
# Some models (qwen3, deepseek-r1) put reasoning in <think> and answer after.
|
|
105
|
+
# If nothing comes after </think>, fall back to looking inside the block.
|
|
106
|
+
if "<think>" in raw:
|
|
107
|
+
end = raw.rfind("</think>")
|
|
108
|
+
if end >= 0:
|
|
109
|
+
after = raw[end + 8:].strip()
|
|
110
|
+
raw = after if after else raw # keep full raw if nothing after </think>
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
# Extract JSON object
|
|
114
|
+
start = raw.find("{")
|
|
115
|
+
end = raw.rfind("}") + 1
|
|
116
|
+
if start < 0 or end <= start:
|
|
117
|
+
raise RuntimeError(f"No JSON object in LLM output: {raw[:200]}")
|
|
118
|
+
|
|
119
|
+
try:
|
|
120
|
+
vocab_map = json.loads(raw[start:end])
|
|
121
|
+
except json.JSONDecodeError as e:
|
|
122
|
+
raise RuntimeError(f"JSON parse failed: {e}\nRaw: {raw[:300]}") from e
|
|
123
|
+
|
|
124
|
+
if not isinstance(vocab_map, dict):
|
|
125
|
+
raise RuntimeError(f"Expected dict, got {type(vocab_map)}")
|
|
126
|
+
|
|
127
|
+
# Normalise: lowercase keys, flatten values to list[str]
|
|
128
|
+
clean: dict[str, list[str]] = {}
|
|
129
|
+
for k, v in vocab_map.items():
|
|
130
|
+
if not isinstance(k, str):
|
|
131
|
+
continue
|
|
132
|
+
if isinstance(v, list):
|
|
133
|
+
clean[k.lower().strip()] = [str(t) for t in v if t]
|
|
134
|
+
elif isinstance(v, str) and v:
|
|
135
|
+
clean[k.lower().strip()] = [v]
|
|
136
|
+
|
|
137
|
+
return clean
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _write_vocab_map(vocab_map: dict[str, list[str]], cfg: dict[str, Any]) -> Path:
|
|
141
|
+
"""Merge vocab_map into the config file. Returns the path written."""
|
|
142
|
+
config_path_str = cfg.get("_config_file", "")
|
|
143
|
+
|
|
144
|
+
if config_path_str:
|
|
145
|
+
config_path = Path(config_path_str)
|
|
146
|
+
else:
|
|
147
|
+
# No file loaded — write to default location
|
|
148
|
+
config_path = Path.home() / ".cogito" / "config.json"
|
|
149
|
+
config_path.parent.mkdir(parents=True, exist_ok=True)
|
|
150
|
+
|
|
151
|
+
# Read existing config (preserve all other keys)
|
|
152
|
+
existing: dict[str, Any] = {}
|
|
153
|
+
if config_path.exists():
|
|
154
|
+
try:
|
|
155
|
+
existing = json.loads(config_path.read_text())
|
|
156
|
+
except Exception: # noqa: silent — corrupted config → write fresh on save
|
|
157
|
+
pass
|
|
158
|
+
|
|
159
|
+
existing["vocab_map"] = vocab_map
|
|
160
|
+
|
|
161
|
+
# Remove internal keys before writing
|
|
162
|
+
existing.pop("_config_file", None)
|
|
163
|
+
|
|
164
|
+
config_path.write_text(json.dumps(existing, indent=2))
|
|
165
|
+
return config_path
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def calibrate(
|
|
169
|
+
memory: Any,
|
|
170
|
+
cfg: dict[str, Any],
|
|
171
|
+
n: int = 200,
|
|
172
|
+
dry_run: bool = False,
|
|
173
|
+
) -> dict[str, list[str]]:
|
|
174
|
+
"""
|
|
175
|
+
Run calibration. Returns the vocab_map produced.
|
|
176
|
+
|
|
177
|
+
Writes to config file unless dry_run=True.
|
|
178
|
+
Raises RuntimeError on LLM call failure (does not write partial results).
|
|
179
|
+
"""
|
|
180
|
+
from fidelis.recall import _resolve_filter_endpoint # avoid circular at module level
|
|
181
|
+
|
|
182
|
+
endpoint, token = _resolve_filter_endpoint(cfg)
|
|
183
|
+
if not endpoint:
|
|
184
|
+
raise RuntimeError(
|
|
185
|
+
"No filter endpoint configured. Set COGITO_FILTER_ENDPOINT + "
|
|
186
|
+
"COGITO_FILTER_TOKEN, or ANTHROPIC_API_KEY."
|
|
187
|
+
)
|
|
188
|
+
|
|
189
|
+
# calibrate_model overrides filter_model — calibration benefits from a larger model
|
|
190
|
+
model = cfg.get("calibrate_model", cfg.get("filter_model", "anthropic/claude-haiku-4-5"))
|
|
191
|
+
# Calibration runs once and has a large prompt — always use at least 90s
|
|
192
|
+
timeout = max(cfg.get("filter_timeout_ms", 30000), 90000) / 1000
|
|
193
|
+
user_id = cfg.get("user_id", "agent")
|
|
194
|
+
|
|
195
|
+
print(f"[cogito calibrate] Sampling memories (n={n})...")
|
|
196
|
+
memories = _sample_memories(memory, user_id, n)
|
|
197
|
+
if not memories:
|
|
198
|
+
raise RuntimeError("No memories in store. Run `cogito seed` first.")
|
|
199
|
+
print(f"[cogito calibrate] Sampled {len(memories)} memories. Calling {model}...")
|
|
200
|
+
|
|
201
|
+
vocab_map = _build_vocab_map(memories, endpoint, token, model, timeout)
|
|
202
|
+
print(f"[cogito calibrate] {len(vocab_map)} vocab mappings extracted.")
|
|
203
|
+
|
|
204
|
+
if dry_run:
|
|
205
|
+
print("[cogito calibrate] DRY RUN — not writing config.")
|
|
206
|
+
for k, v in list(vocab_map.items())[:20]:
|
|
207
|
+
print(f" {k!r:30s} → {v}")
|
|
208
|
+
if len(vocab_map) > 20:
|
|
209
|
+
print(f" ... ({len(vocab_map) - 20} more)")
|
|
210
|
+
return vocab_map
|
|
211
|
+
|
|
212
|
+
path = _write_vocab_map(vocab_map, cfg)
|
|
213
|
+
print(f"[cogito calibrate] Written to {path}")
|
|
214
|
+
return vocab_map
|