fidelis-memory 0.0.93__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
fidelis/__init__.py ADDED
@@ -0,0 +1,22 @@
1
+ """
2
+ fidelis — faithful memory retrieval for AI agents.
3
+
4
+ Default path (zero-LLM): BM25 + dense + RRF fusion, 83.2% R@1 on
5
+ LongMemEval_S, $0/query, ~90ms, fully local.
6
+
7
+ Optional LLM tiers (experimental): ``filter`` and ``flagship``. The filter
8
+ LLM outputs only integer indices (e.g. [3, 7, 12]) — never memory text —
9
+ so it cannot corrupt or hallucinate into the content returned to the agent.
10
+ Fidelity is structural, not a prompting convention.
11
+
12
+ fidelis was previously published as ``cogito-ergo`` (0.0.8 and 0.3.0 on PyPI).
13
+ Data paths and env var names retain the ``cogito`` prefix for continuity
14
+ with existing deployments.
15
+ """
16
+
17
+ __version__ = "0.0.93"
18
+
19
+ from fidelis.recall import recall # noqa: F401
20
+
21
+ # Note: avoid shadowing the ``fidelis.recall_hybrid`` submodule. Users who want
22
+ # the function directly can do ``from fidelis.recall_hybrid import recall_hybrid``.
fidelis/augment.py ADDED
@@ -0,0 +1,111 @@
1
+ """fidelis.augment — the one-line caller helper.
2
+
3
+ For users who want "give me memory + scaffold + answer" in one call instead of
4
+ wiring three separate functions. Pure wrapper; uses the running fidelis-server
5
+ for retrieval and the user's own LLM client for generation.
6
+
7
+ Example:
8
+
9
+ from fidelis.augment import augment
10
+ from anthropic import Anthropic
11
+
12
+ client = Anthropic()
13
+ response = augment(
14
+ question="What did I say about Sarah?",
15
+ qtype="single-session-user",
16
+ llm_call=lambda system, user: client.messages.create(
17
+ model="claude-opus-4-7", # use any current Claude Messages API model
18
+ system=system,
19
+ messages=[{"role": "user", "content": user}],
20
+ max_tokens=512,
21
+ ).content[0].text,
22
+ )
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ import json
28
+ import os
29
+ import urllib.error
30
+ import urllib.request
31
+ from typing import Callable
32
+
33
+ from fidelis.scaffold import wrap_system_prompt
34
+
35
+
36
+ def _server_url() -> str:
37
+ port = os.environ.get("FIDELIS_PORT", os.environ.get("COGITO_PORT", "19420"))
38
+ return f"http://127.0.0.1:{port}"
39
+
40
+
41
+ def _recall(query: str, limit: int = 5) -> tuple[str, float | None]:
42
+ """Retrieve memories from fidelis-server. Returns (formatted_context, top_score)."""
43
+ body = json.dumps({"text": query, "limit": limit}).encode()
44
+ req = urllib.request.Request(
45
+ f"{_server_url()}/recall",
46
+ data=body,
47
+ headers={"Content-Type": "application/json"},
48
+ method="POST",
49
+ )
50
+ try:
51
+ with urllib.request.urlopen(req, timeout=30) as resp:
52
+ data = json.loads(resp.read())
53
+ except urllib.error.URLError as e:
54
+ raise RuntimeError(
55
+ f"fidelis-server unreachable at {_server_url()}. "
56
+ f"Run `fidelis init` to install + start the service. ({e})"
57
+ ) from e
58
+
59
+ memories = data.get("memories", [])
60
+ if not memories:
61
+ return "(no memories retrieved)", None
62
+
63
+ lines = []
64
+ top_score: float | None = None
65
+ for m in memories:
66
+ text = m.get("text", "").strip()
67
+ score = m.get("score")
68
+ if score is not None and top_score is None:
69
+ try:
70
+ top_score = float(score)
71
+ # mem0/cogito returns "lower is better" similarity; normalize roughly to [0, 1]
72
+ if top_score > 1.0:
73
+ # heuristic: distance scores in [0, ~500] → invert to [0, 1]
74
+ top_score = max(0.0, min(1.0, 1.0 - (top_score / 500.0)))
75
+ except (ValueError, TypeError):
76
+ top_score = None
77
+ lines.append(text)
78
+
79
+ return "\n\n---\n\n".join(lines), top_score
80
+
81
+
82
+ def augment(
83
+ question: str,
84
+ qtype: str = "single-session-user",
85
+ *,
86
+ llm_call: Callable[[str, str], str],
87
+ limit: int = 5,
88
+ ) -> str:
89
+ """Retrieve memory + wrap with scaffold + invoke caller's LLM.
90
+
91
+ Args:
92
+ question: the user's natural-language question
93
+ qtype: one of single-session-user / single-session-assistant /
94
+ single-session-preference / knowledge-update / multi-session /
95
+ temporal-reasoning. Caller chooses based on their question type.
96
+ Default: single-session-user.
97
+ llm_call: a callable taking (system_prompt, user_message) and returning
98
+ the LLM's response text. The user supplies this so fidelis stays
99
+ agnostic to LLM SDK choice.
100
+ limit: max memories to retrieve (default 5)
101
+
102
+ Returns:
103
+ The LLM's response text.
104
+
105
+ Raises:
106
+ RuntimeError: if fidelis-server is unreachable.
107
+ """
108
+ context, top_score = _recall(question, limit=limit)
109
+ system = wrap_system_prompt(qtype, top_score=top_score)
110
+ user_message = f"Conversation memory:\n{context}\n\nQuestion: {question}"
111
+ return llm_call(system, user_message)
fidelis/calibrate.py ADDED
@@ -0,0 +1,214 @@
1
+ """
2
+ cogito calibrate — one-time vocabulary bridge extraction.
3
+
4
+ Samples memories from the store, asks the filter LLM to identify vocabulary
5
+ gaps between natural language queries and stored technical facts, and writes
6
+ a vocab_map to .cogito.json.
7
+
8
+ At query time, recall_b uses the vocab_map for zero-LLM expansion:
9
+ "freeze" → ["timeout", "cascade", "ollama"]
10
+ "adoption" → ["downloads", "PyPI", "installs"]
11
+
12
+ Run once after initial seeding, and optionally after large corpus updates.
13
+
14
+ Usage:
15
+ cogito calibrate
16
+ cogito calibrate --sample 300 --dry-run
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import json
22
+ import random
23
+ import urllib.error
24
+ import urllib.request
25
+ from pathlib import Path
26
+ from typing import Any
27
+
28
+
29
+ _CALIBRATE_SYSTEM = (
30
+ "You are building a vocabulary bridge for a technical memory retrieval system. "
31
+ "Users ask questions in plain English. Stored facts use technical jargon, "
32
+ "proper nouns, acronyms, and domain-specific terms.\n\n"
33
+ "Your job: identify pairs where a plain-English query word would NOT appear "
34
+ "in the stored fact, but the stored fact IS the right answer.\n\n"
35
+ "Output ONLY a valid JSON object. "
36
+ "Keys: plain-English words a user might type in a query. "
37
+ "Values: arrays of SHORT TECHNICAL TERMS (1-5 words each, NEVER full sentences) "
38
+ "that appear verbatim in the stored facts and carry the same meaning as the key. "
39
+ "Do NOT include full sentences as values. "
40
+ "Do NOT include generic stop words. "
41
+ "Max 60 pairs. Focus on genuine vocabulary gap cases only.\n\n"
42
+ 'Example: {"freeze": ["timeout", "cascade", "blocked"], '
43
+ '"adoption": ["downloads", "PyPI", "installs"], '
44
+ '"broken": ["flat scores", "metadata missing", "zero results"], '
45
+ '"memory not found": ["flat score", "user_id", "chromadb"], '
46
+ '"agent stuck": ["timeout", "seed", "ollama cascade"]}'
47
+ )
48
+
49
+
50
+ def _sample_memories(memory: Any, user_id: str, n: int) -> list[str]:
51
+ """Fetch all memories and return up to n randomly sampled texts."""
52
+ try:
53
+ raw = memory.get_all(filters={"user_id": user_id}, top_k=10000) # type: ignore
54
+ results = raw.get("results", [])
55
+ texts = [r.get("memory", "") for r in results if r.get("memory")]
56
+ if len(texts) > n:
57
+ texts = random.sample(texts, n)
58
+ return texts
59
+ except Exception as e:
60
+ raise RuntimeError(f"Failed to fetch memories: {e}") from e
61
+
62
+
63
+ def _build_vocab_map(
64
+ memories: list[str],
65
+ endpoint: str,
66
+ token: str,
67
+ model: str,
68
+ timeout: float,
69
+ ) -> dict[str, list[str]]:
70
+ """Single LLM call to extract vocabulary bridge from sampled memories."""
71
+ lines = [f"[{i+1}] {m[:120].replace(chr(10), ' ')}" for i, m in enumerate(memories)]
72
+ memories_block = "\n".join(lines)
73
+
74
+ payload = json.dumps({
75
+ "model": model,
76
+ "messages": [
77
+ {"role": "system", "content": _CALIBRATE_SYSTEM},
78
+ {"role": "user", "content": f"Memory facts:\n{memories_block}\n\nOutput the vocabulary bridge JSON:"},
79
+ ],
80
+ "max_tokens": 2000,
81
+ "temperature": 0,
82
+ }).encode()
83
+
84
+ req = urllib.request.Request(
85
+ f"{endpoint}/v1/chat/completions",
86
+ data=payload,
87
+ headers={
88
+ "Content-Type": "application/json",
89
+ "Authorization": f"Bearer {token}",
90
+ },
91
+ method="POST",
92
+ )
93
+
94
+ try:
95
+ with urllib.request.urlopen(req, timeout=timeout) as resp:
96
+ result = json.loads(resp.read())
97
+ raw = result["choices"][0]["message"]["content"].strip()
98
+ except urllib.error.URLError as e:
99
+ raise RuntimeError(f"LLM call failed: {e}") from e
100
+ except Exception as e:
101
+ raise RuntimeError(f"Unexpected error: {e}") from e
102
+
103
+ # Strip thinking tokens (<think>...</think>)
104
+ # Some models (qwen3, deepseek-r1) put reasoning in <think> and answer after.
105
+ # If nothing comes after </think>, fall back to looking inside the block.
106
+ if "<think>" in raw:
107
+ end = raw.rfind("</think>")
108
+ if end >= 0:
109
+ after = raw[end + 8:].strip()
110
+ raw = after if after else raw # keep full raw if nothing after </think>
111
+
112
+
113
+ # Extract JSON object
114
+ start = raw.find("{")
115
+ end = raw.rfind("}") + 1
116
+ if start < 0 or end <= start:
117
+ raise RuntimeError(f"No JSON object in LLM output: {raw[:200]}")
118
+
119
+ try:
120
+ vocab_map = json.loads(raw[start:end])
121
+ except json.JSONDecodeError as e:
122
+ raise RuntimeError(f"JSON parse failed: {e}\nRaw: {raw[:300]}") from e
123
+
124
+ if not isinstance(vocab_map, dict):
125
+ raise RuntimeError(f"Expected dict, got {type(vocab_map)}")
126
+
127
+ # Normalise: lowercase keys, flatten values to list[str]
128
+ clean: dict[str, list[str]] = {}
129
+ for k, v in vocab_map.items():
130
+ if not isinstance(k, str):
131
+ continue
132
+ if isinstance(v, list):
133
+ clean[k.lower().strip()] = [str(t) for t in v if t]
134
+ elif isinstance(v, str) and v:
135
+ clean[k.lower().strip()] = [v]
136
+
137
+ return clean
138
+
139
+
140
+ def _write_vocab_map(vocab_map: dict[str, list[str]], cfg: dict[str, Any]) -> Path:
141
+ """Merge vocab_map into the config file. Returns the path written."""
142
+ config_path_str = cfg.get("_config_file", "")
143
+
144
+ if config_path_str:
145
+ config_path = Path(config_path_str)
146
+ else:
147
+ # No file loaded — write to default location
148
+ config_path = Path.home() / ".cogito" / "config.json"
149
+ config_path.parent.mkdir(parents=True, exist_ok=True)
150
+
151
+ # Read existing config (preserve all other keys)
152
+ existing: dict[str, Any] = {}
153
+ if config_path.exists():
154
+ try:
155
+ existing = json.loads(config_path.read_text())
156
+ except Exception: # noqa: silent — corrupted config → write fresh on save
157
+ pass
158
+
159
+ existing["vocab_map"] = vocab_map
160
+
161
+ # Remove internal keys before writing
162
+ existing.pop("_config_file", None)
163
+
164
+ config_path.write_text(json.dumps(existing, indent=2))
165
+ return config_path
166
+
167
+
168
+ def calibrate(
169
+ memory: Any,
170
+ cfg: dict[str, Any],
171
+ n: int = 200,
172
+ dry_run: bool = False,
173
+ ) -> dict[str, list[str]]:
174
+ """
175
+ Run calibration. Returns the vocab_map produced.
176
+
177
+ Writes to config file unless dry_run=True.
178
+ Raises RuntimeError on LLM call failure (does not write partial results).
179
+ """
180
+ from fidelis.recall import _resolve_filter_endpoint # avoid circular at module level
181
+
182
+ endpoint, token = _resolve_filter_endpoint(cfg)
183
+ if not endpoint:
184
+ raise RuntimeError(
185
+ "No filter endpoint configured. Set COGITO_FILTER_ENDPOINT + "
186
+ "COGITO_FILTER_TOKEN, or ANTHROPIC_API_KEY."
187
+ )
188
+
189
+ # calibrate_model overrides filter_model — calibration benefits from a larger model
190
+ model = cfg.get("calibrate_model", cfg.get("filter_model", "anthropic/claude-haiku-4-5"))
191
+ # Calibration runs once and has a large prompt — always use at least 90s
192
+ timeout = max(cfg.get("filter_timeout_ms", 30000), 90000) / 1000
193
+ user_id = cfg.get("user_id", "agent")
194
+
195
+ print(f"[cogito calibrate] Sampling memories (n={n})...")
196
+ memories = _sample_memories(memory, user_id, n)
197
+ if not memories:
198
+ raise RuntimeError("No memories in store. Run `cogito seed` first.")
199
+ print(f"[cogito calibrate] Sampled {len(memories)} memories. Calling {model}...")
200
+
201
+ vocab_map = _build_vocab_map(memories, endpoint, token, model, timeout)
202
+ print(f"[cogito calibrate] {len(vocab_map)} vocab mappings extracted.")
203
+
204
+ if dry_run:
205
+ print("[cogito calibrate] DRY RUN — not writing config.")
206
+ for k, v in list(vocab_map.items())[:20]:
207
+ print(f" {k!r:30s} → {v}")
208
+ if len(vocab_map) > 20:
209
+ print(f" ... ({len(vocab_map) - 20} more)")
210
+ return vocab_map
211
+
212
+ path = _write_vocab_map(vocab_map, cfg)
213
+ print(f"[cogito calibrate] Written to {path}")
214
+ return vocab_map