contexara 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contexara/__init__.py +11 -0
- contexara/__main__.py +7 -0
- contexara/cli.py +236 -0
- contexara/embedder.py +39 -0
- contexara/enhance.py +30 -0
- contexara/extract.py +262 -0
- contexara/ingest.py +43 -0
- contexara/llm.py +143 -0
- contexara/main.py +40 -0
- contexara/retrieve.py +155 -0
- contexara/store.py +674 -0
- contexara-0.2.1.dist-info/METADATA +68 -0
- contexara-0.2.1.dist-info/RECORD +16 -0
- contexara-0.2.1.dist-info/WHEEL +5 -0
- contexara-0.2.1.dist-info/entry_points.txt +2 -0
- contexara-0.2.1.dist-info/top_level.txt +1 -0
contexara/__init__.py
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
"""Contexara public package API."""
|
|
2
|
+
|
|
3
|
+
from .enhance import enhance
|
|
4
|
+
from .ingest import ingest_turn
|
|
5
|
+
from .main import ask
|
|
6
|
+
from .retrieve import retrieve
|
|
7
|
+
from .store import clear_all, compress_old_memories, consolidate, delete_memory, get_stats, list_all, search_memories, store
|
|
8
|
+
|
|
9
|
+
__all__ = ["ask", "store", "list_all", "clear_all", "delete_memory", "search_memories", "get_stats", "consolidate", "compress_old_memories", "retrieve", "enhance", "ingest_turn"]
|
|
10
|
+
|
|
11
|
+
__version__ = "0.2.1"
|
contexara/__main__.py
ADDED
contexara/cli.py
ADDED
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
"""Command-line interface for Contexara."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
from typing import Iterable
|
|
7
|
+
|
|
8
|
+
from . import __version__
|
|
9
|
+
from .main import ask
|
|
10
|
+
from .store import clear_all, consolidate, delete_memory, get_stats, list_all, search_memories, store
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _build_parser() -> argparse.ArgumentParser:
|
|
14
|
+
parser = argparse.ArgumentParser(
|
|
15
|
+
prog="contexara",
|
|
16
|
+
description="CLI-first memory engine for LLM context management.",
|
|
17
|
+
)
|
|
18
|
+
parser.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
|
|
19
|
+
|
|
20
|
+
subparsers = parser.add_subparsers(dest="command")
|
|
21
|
+
|
|
22
|
+
subparsers.add_parser("chat", help="Start interactive chat mode.").set_defaults(func=_cmd_chat)
|
|
23
|
+
|
|
24
|
+
ask_parser = subparsers.add_parser("ask", help="Ask a memory-aware question.")
|
|
25
|
+
ask_parser.add_argument("query", nargs="+")
|
|
26
|
+
ask_parser.set_defaults(func=_cmd_ask)
|
|
27
|
+
|
|
28
|
+
store_parser = subparsers.add_parser("store", help="Store a memory string.")
|
|
29
|
+
store_parser.add_argument("content", nargs="+")
|
|
30
|
+
store_parser.set_defaults(func=_cmd_store)
|
|
31
|
+
|
|
32
|
+
subparsers.add_parser("list", help="List all stored memories.").set_defaults(func=_cmd_list)
|
|
33
|
+
|
|
34
|
+
delete_parser = subparsers.add_parser("delete", help="Delete a memory by ID.")
|
|
35
|
+
delete_parser.add_argument("id", type=int, help="Memory ID (from contexara list).")
|
|
36
|
+
delete_parser.set_defaults(func=_cmd_delete)
|
|
37
|
+
|
|
38
|
+
search_parser = subparsers.add_parser("search", help="Semantic search over memories.")
|
|
39
|
+
search_parser.add_argument("query", nargs="+")
|
|
40
|
+
search_parser.add_argument("--top", type=int, default=5, help="Number of results (default 5).")
|
|
41
|
+
search_parser.set_defaults(func=_cmd_search)
|
|
42
|
+
|
|
43
|
+
subparsers.add_parser("stats", help="Show memory statistics.").set_defaults(func=_cmd_stats)
|
|
44
|
+
|
|
45
|
+
subparsers.add_parser("consolidate", help="LLM-merge overlapping memories.").set_defaults(func=_cmd_consolidate)
|
|
46
|
+
|
|
47
|
+
clear_parser = subparsers.add_parser("clear", help="Delete all stored memories.")
|
|
48
|
+
clear_parser.add_argument("--yes", action="store_true", help="Skip confirmation.")
|
|
49
|
+
clear_parser.set_defaults(func=_cmd_clear)
|
|
50
|
+
|
|
51
|
+
return parser
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _join_parts(parts: Iterable[str]) -> str:
|
|
55
|
+
return " ".join(parts).strip()
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _print_memories_table(memories: list[dict]) -> None:
|
|
59
|
+
if not memories:
|
|
60
|
+
print("(no memories)")
|
|
61
|
+
return
|
|
62
|
+
|
|
63
|
+
id_w = max(len("ID"), max(len(str(m["id"])) for m in memories))
|
|
64
|
+
kind_w = max(len("KIND"), max(len(str(m.get("kind", ""))) for m in memories))
|
|
65
|
+
imp_w = max(len("IMP"), max(len(str(m.get("importance", ""))) for m in memories))
|
|
66
|
+
src_w = max(len("SRC"), max(len(str(m.get("source", "raw"))) for m in memories))
|
|
67
|
+
content_w = 60
|
|
68
|
+
|
|
69
|
+
header = f"{'ID':<{id_w}} {'KIND':<{kind_w}} {'IMP':<{imp_w}} {'SRC':<{src_w}} {'CONTENT':<{content_w}}"
|
|
70
|
+
print(header)
|
|
71
|
+
print("-" * len(header))
|
|
72
|
+
for m in memories:
|
|
73
|
+
content = m["content"]
|
|
74
|
+
if len(content) > content_w:
|
|
75
|
+
content = content[: content_w - 3] + "..."
|
|
76
|
+
print(f"{str(m['id']):<{id_w}} {str(m.get('kind','')):<{kind_w}} {str(m.get('importance','')):<{imp_w}} {str(m.get('source','raw')):<{src_w}} {content}")
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _cmd_chat(_args: argparse.Namespace) -> int:
|
|
80
|
+
print("\nContexara Chat (type 'exit' to quit)")
|
|
81
|
+
print("Commands: /store <text>, /list, /show <id>, /delete <id>, /search <query>, /stats, /consolidate, /clear\n")
|
|
82
|
+
|
|
83
|
+
while True:
|
|
84
|
+
user_input = input("You: ").strip()
|
|
85
|
+
if not user_input:
|
|
86
|
+
continue
|
|
87
|
+
|
|
88
|
+
if user_input.lower() == "exit":
|
|
89
|
+
print("Goodbye.")
|
|
90
|
+
return 0
|
|
91
|
+
|
|
92
|
+
if user_input.startswith("/store "):
|
|
93
|
+
store(user_input[7:].strip())
|
|
94
|
+
print("Memory stored.")
|
|
95
|
+
continue
|
|
96
|
+
|
|
97
|
+
if user_input == "/list":
|
|
98
|
+
_print_memories_table(list_all())
|
|
99
|
+
continue
|
|
100
|
+
|
|
101
|
+
if user_input.startswith("/delete "):
|
|
102
|
+
try:
|
|
103
|
+
mid = int(user_input[8:].strip())
|
|
104
|
+
if delete_memory(mid):
|
|
105
|
+
print(f"Memory {mid} deleted.")
|
|
106
|
+
else:
|
|
107
|
+
print(f"No memory with ID {mid}.")
|
|
108
|
+
except ValueError:
|
|
109
|
+
print("Usage: /delete <id>")
|
|
110
|
+
continue
|
|
111
|
+
|
|
112
|
+
if user_input.startswith("/search "):
|
|
113
|
+
query = user_input[8:].strip()
|
|
114
|
+
results = search_memories(query)
|
|
115
|
+
_print_memories_table(results)
|
|
116
|
+
continue
|
|
117
|
+
|
|
118
|
+
if user_input.startswith("/show "):
|
|
119
|
+
try:
|
|
120
|
+
mid = int(user_input[6:].strip())
|
|
121
|
+
match = next((m for m in list_all() if m["id"] == mid), None)
|
|
122
|
+
if match:
|
|
123
|
+
print(f"\nID : {match['id']}")
|
|
124
|
+
print(f"Kind : {match['kind']}")
|
|
125
|
+
print(f"Importance: {match['importance']}")
|
|
126
|
+
print(f"Source : {match.get('source', 'raw')}")
|
|
127
|
+
print(f"Level : {match.get('compression_level', 0)}")
|
|
128
|
+
print(f"Created : {match['created_at']}")
|
|
129
|
+
print(f"Content : {match['content']}\n")
|
|
130
|
+
else:
|
|
131
|
+
print(f"No memory with ID {mid}.")
|
|
132
|
+
except ValueError:
|
|
133
|
+
print("Usage: /show <id>")
|
|
134
|
+
continue
|
|
135
|
+
|
|
136
|
+
if user_input == "/consolidate":
|
|
137
|
+
print("Consolidating memories...")
|
|
138
|
+
result = consolidate()
|
|
139
|
+
print(f"Done. {result['before']} memories → {result['after']} memories.")
|
|
140
|
+
continue
|
|
141
|
+
|
|
142
|
+
if user_input == "/stats":
|
|
143
|
+
_print_stats(get_stats())
|
|
144
|
+
continue
|
|
145
|
+
|
|
146
|
+
if user_input == "/clear":
|
|
147
|
+
clear_all()
|
|
148
|
+
print("All memories cleared.")
|
|
149
|
+
continue
|
|
150
|
+
|
|
151
|
+
response = ask(user_input)
|
|
152
|
+
print(f"\nBot: {response}\n")
|
|
153
|
+
|
|
154
|
+
return 0
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _print_stats(stats: dict) -> None:
|
|
158
|
+
print(f"\nTotal memories : {stats['total']}")
|
|
159
|
+
print(f"Oldest : {stats['oldest'] or 'N/A'}")
|
|
160
|
+
print(f"Newest : {stats['newest'] or 'N/A'}")
|
|
161
|
+
print("\nBy kind:")
|
|
162
|
+
for kind, count in stats["by_kind"].items():
|
|
163
|
+
print(f" {kind:<12} {count}")
|
|
164
|
+
print("\nBy source:")
|
|
165
|
+
for source, count in stats.get("by_source", {}).items():
|
|
166
|
+
ss = stats.get("source_stats", {}).get(source, {})
|
|
167
|
+
avg_imp = ss.get("avg_importance", "")
|
|
168
|
+
avg_age = ss.get("avg_age_days", "")
|
|
169
|
+
print(f" {source:<12} {count:<6} avg_importance={avg_imp} avg_age={avg_age}d")
|
|
170
|
+
print()
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _cmd_ask(args: argparse.Namespace) -> int:
|
|
174
|
+
print(ask(_join_parts(args.query)))
|
|
175
|
+
return 0
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _cmd_store(args: argparse.Namespace) -> int:
|
|
179
|
+
store(_join_parts(args.content))
|
|
180
|
+
print("Memory stored.")
|
|
181
|
+
return 0
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _cmd_list(_args: argparse.Namespace) -> int:
|
|
185
|
+
_print_memories_table(list_all())
|
|
186
|
+
return 0
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _cmd_delete(args: argparse.Namespace) -> int:
|
|
190
|
+
if delete_memory(args.id):
|
|
191
|
+
print(f"Memory {args.id} deleted.")
|
|
192
|
+
else:
|
|
193
|
+
print(f"No memory with ID {args.id}.")
|
|
194
|
+
return 0
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _cmd_search(args: argparse.Namespace) -> int:
|
|
198
|
+
results = search_memories(_join_parts(args.query), top_k=args.top)
|
|
199
|
+
_print_memories_table(results)
|
|
200
|
+
return 0
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def _cmd_stats(_args: argparse.Namespace) -> int:
|
|
204
|
+
_print_stats(get_stats())
|
|
205
|
+
return 0
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _cmd_consolidate(_args: argparse.Namespace) -> int:
|
|
209
|
+
print("Consolidating memories...")
|
|
210
|
+
result = consolidate()
|
|
211
|
+
print(f"Done. {result['before']} memories → {result['after']} memories.")
|
|
212
|
+
return 0
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def _cmd_clear(args: argparse.Namespace) -> int:
|
|
216
|
+
if not args.yes:
|
|
217
|
+
confirm = input("Delete all memories? [y/N]: ").strip().lower()
|
|
218
|
+
if confirm not in {"y", "yes"}:
|
|
219
|
+
print("Cancelled.")
|
|
220
|
+
return 0
|
|
221
|
+
clear_all()
|
|
222
|
+
print("All memories cleared.")
|
|
223
|
+
return 0
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def main() -> int:
|
|
227
|
+
parser = _build_parser()
|
|
228
|
+
args = parser.parse_args()
|
|
229
|
+
if not args.command:
|
|
230
|
+
parser.print_help()
|
|
231
|
+
return 0
|
|
232
|
+
return args.func(args)
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
if __name__ == "__main__":
|
|
236
|
+
raise SystemExit(main())
|
contexara/embedder.py
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import io
|
|
4
|
+
from functools import lru_cache
|
|
5
|
+
|
|
6
|
+
import numpy as np
|
|
7
|
+
|
|
8
|
+
_MODEL_NAME = "all-MiniLM-L6-v2"
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@lru_cache(maxsize=1)
|
|
12
|
+
def _get_model():
|
|
13
|
+
import logging
|
|
14
|
+
from sentence_transformers import SentenceTransformer
|
|
15
|
+
logging.getLogger("transformers.modeling_utils").setLevel(logging.ERROR)
|
|
16
|
+
return SentenceTransformer(_MODEL_NAME)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def embed(text: str) -> np.ndarray:
|
|
20
|
+
"""Return a normalized float32 embedding vector for text."""
|
|
21
|
+
model = _get_model()
|
|
22
|
+
vector = model.encode(text, normalize_embeddings=True)
|
|
23
|
+
return vector.astype(np.float32)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def serialize(vector: np.ndarray) -> bytes:
|
|
27
|
+
buf = io.BytesIO()
|
|
28
|
+
np.save(buf, vector)
|
|
29
|
+
return buf.getvalue()
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def deserialize(blob: bytes) -> np.ndarray:
|
|
33
|
+
buf = io.BytesIO(blob)
|
|
34
|
+
return np.load(buf)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def cosine_similarity(a: np.ndarray, b: np.ndarray) -> float:
|
|
38
|
+
"""Cosine similarity for pre-normalized vectors (dot product suffices)."""
|
|
39
|
+
return float(np.dot(a, b))
|
contexara/enhance.py
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
def enhance(query: str, memories: list[str]) -> str:
|
|
2
|
+
"""
|
|
3
|
+
Build an enhanced prompt by injecting relevant memories.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
if not query or not query.strip():
|
|
7
|
+
raise ValueError("Query cannot be empty")
|
|
8
|
+
|
|
9
|
+
query = query.strip()
|
|
10
|
+
|
|
11
|
+
# Format memory block
|
|
12
|
+
if memories:
|
|
13
|
+
memory_lines = "\n".join(f"- {m}" for m in memories)
|
|
14
|
+
|
|
15
|
+
context_block = f"""You are an AI assistant with memory.
|
|
16
|
+
|
|
17
|
+
Relevant context:
|
|
18
|
+
{memory_lines}
|
|
19
|
+
|
|
20
|
+
"""
|
|
21
|
+
else:
|
|
22
|
+
context_block = "You are an AI assistant.\n\n"
|
|
23
|
+
|
|
24
|
+
# Final prompt
|
|
25
|
+
prompt = f"""{context_block}User:
|
|
26
|
+
{query}
|
|
27
|
+
|
|
28
|
+
Assistant:"""
|
|
29
|
+
|
|
30
|
+
return prompt
|
contexara/extract.py
ADDED
|
@@ -0,0 +1,262 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import re
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
MEMORY_IMPORTANCE = {
|
|
8
|
+
"profile": 5,
|
|
9
|
+
"preference": 4,
|
|
10
|
+
"constraint": 4,
|
|
11
|
+
"task": 3,
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
VALID_KINDS = set(MEMORY_IMPORTANCE.keys()) | {"note"}
|
|
15
|
+
|
|
16
|
+
LOW_SIGNAL_PREFIXES = (
|
|
17
|
+
"hi",
|
|
18
|
+
"hello",
|
|
19
|
+
"thanks",
|
|
20
|
+
"thank you",
|
|
21
|
+
"ok",
|
|
22
|
+
"okay",
|
|
23
|
+
"cool",
|
|
24
|
+
"great",
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
_EXTRACTION_SYSTEM_PROMPT = """\
|
|
28
|
+
You are a memory extraction engine. Output ONLY a raw JSON array — no markdown, no prose, no explanation.
|
|
29
|
+
|
|
30
|
+
Extract the most important, distinct, long-term facts about the user from the conversation turn.
|
|
31
|
+
Be selective — prefer 1-4 high-quality memories over many overlapping ones.
|
|
32
|
+
|
|
33
|
+
Rules:
|
|
34
|
+
- Include: name, role, things built/completed/tested/shipped, stated preferences, hard constraints, current tasks.
|
|
35
|
+
- Skip: vague plans, filler, questions, anything already covered by the existing memories listed below.
|
|
36
|
+
- Do NOT paraphrase an existing memory — if the fact is already stored, skip it entirely.
|
|
37
|
+
- Third-person statements only (e.g. "User built hybrid search with RRF").
|
|
38
|
+
- kind: profile | preference | task | constraint | note
|
|
39
|
+
- importance: integer 1-5 (profile=5, preference/constraint=4, task=3, note=1)
|
|
40
|
+
|
|
41
|
+
Output format (strictly):
|
|
42
|
+
[{"content": "...", "kind": "...", "importance": N}, ...]
|
|
43
|
+
|
|
44
|
+
If nothing new is worth remembering output exactly: []
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _call_llm_for_extraction(
|
|
49
|
+
user_text: str,
|
|
50
|
+
assistant_text: str,
|
|
51
|
+
existing_memories: list[str] | None = None,
|
|
52
|
+
) -> list[dict[str, Any]]:
|
|
53
|
+
"""Call Bedrock to extract memories from a conversation turn. Returns [] on any failure."""
|
|
54
|
+
try:
|
|
55
|
+
from contexara.llm import BedrockConfig, _get_client
|
|
56
|
+
|
|
57
|
+
import json as _json
|
|
58
|
+
|
|
59
|
+
existing_block = ""
|
|
60
|
+
if existing_memories:
|
|
61
|
+
items = "\n".join(f"- {m}" for m in existing_memories[:20])
|
|
62
|
+
existing_block = f"\n\nAlready stored memories (do NOT re-extract these):\n{items}\n"
|
|
63
|
+
|
|
64
|
+
turn = f"{existing_block}\nConversation turn:\nUser: {user_text.strip()}\nAssistant: {assistant_text.strip()}"
|
|
65
|
+
payload = {
|
|
66
|
+
"anthropic_version": "bedrock-2023-05-31",
|
|
67
|
+
"system": _EXTRACTION_SYSTEM_PROMPT,
|
|
68
|
+
"messages": [{"role": "user", "content": [{"type": "text", "text": turn}]}],
|
|
69
|
+
"max_tokens": 512,
|
|
70
|
+
"temperature": 0.0,
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
config = BedrockConfig.from_env()
|
|
74
|
+
client = _get_client(config.region)
|
|
75
|
+
response = client.invoke_model(
|
|
76
|
+
modelId=config.model_id,
|
|
77
|
+
body=_json.dumps(payload),
|
|
78
|
+
contentType="application/json",
|
|
79
|
+
accept="application/json",
|
|
80
|
+
)
|
|
81
|
+
result = _json.loads(response["body"].read())
|
|
82
|
+
raw_text = result["content"][0]["text"].strip()
|
|
83
|
+
|
|
84
|
+
# Strip markdown code fences if model wraps output in ```json ... ```
|
|
85
|
+
raw_text = re.sub(r"^```(?:json)?\s*", "", raw_text)
|
|
86
|
+
raw_text = re.sub(r"\s*```$", "", raw_text).strip()
|
|
87
|
+
|
|
88
|
+
candidates = json.loads(raw_text)
|
|
89
|
+
if not isinstance(candidates, list):
|
|
90
|
+
return []
|
|
91
|
+
|
|
92
|
+
memories = []
|
|
93
|
+
for item in candidates:
|
|
94
|
+
if not isinstance(item, dict):
|
|
95
|
+
continue
|
|
96
|
+
content = str(item.get("content", "")).strip()
|
|
97
|
+
kind = str(item.get("kind", "note")).strip().lower()
|
|
98
|
+
importance = item.get("importance", 1)
|
|
99
|
+
|
|
100
|
+
if not content:
|
|
101
|
+
continue
|
|
102
|
+
if kind not in VALID_KINDS:
|
|
103
|
+
kind = "note"
|
|
104
|
+
try:
|
|
105
|
+
importance = max(1, min(5, int(importance)))
|
|
106
|
+
except (TypeError, ValueError):
|
|
107
|
+
importance = MEMORY_IMPORTANCE.get(kind, 1)
|
|
108
|
+
|
|
109
|
+
memories.append({"content": content, "kind": kind, "importance": importance, "ttl_days": None})
|
|
110
|
+
|
|
111
|
+
return memories
|
|
112
|
+
|
|
113
|
+
except Exception:
|
|
114
|
+
return []
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
# ---------------------------------------------------------------------------
|
|
118
|
+
# Regex fallback (V1 logic — used when LLM extraction fails)
|
|
119
|
+
# ---------------------------------------------------------------------------
|
|
120
|
+
|
|
121
|
+
def _clean_text(text: str) -> str:
|
|
122
|
+
return " ".join(text.strip().split())
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _split_into_segments(text: str) -> list[str]:
|
|
126
|
+
if not text:
|
|
127
|
+
return []
|
|
128
|
+
normalized = _clean_text(text)
|
|
129
|
+
punctuation_parts = re.split(r"[.;]\s+|\n+", normalized)
|
|
130
|
+
segments: list[str] = []
|
|
131
|
+
for part in punctuation_parts:
|
|
132
|
+
chunk = part.strip()
|
|
133
|
+
if not chunk:
|
|
134
|
+
continue
|
|
135
|
+
chunk = re.sub(
|
|
136
|
+
r"\s+and\s+(?=(i\s+(?:am|m|will)|my\s+|first\s+task|tomorrow\b))",
|
|
137
|
+
" ||| ",
|
|
138
|
+
chunk,
|
|
139
|
+
flags=re.IGNORECASE,
|
|
140
|
+
)
|
|
141
|
+
sub_parts = [p.strip() for p in chunk.split("|||") if p.strip()]
|
|
142
|
+
segments.extend(sub_parts)
|
|
143
|
+
return segments
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _extract_profile(text: str) -> list[dict[str, Any]]:
|
|
147
|
+
items: list[dict[str, Any]] = []
|
|
148
|
+
name_match = re.search(r"\bmy name is\s+([A-Za-z][A-Za-z'\-\s]{1,50})", text, re.IGNORECASE)
|
|
149
|
+
if name_match:
|
|
150
|
+
name = _clean_text(name_match.group(1)).rstrip(".,!?")
|
|
151
|
+
items.append({"content": f"User name is {name}", "kind": "profile", "importance": 5, "ttl_days": None})
|
|
152
|
+
role_match = re.search(r"\bi am\s+(a|an)\s+([A-Za-z][A-Za-z'\-\s]{1,80})", text, re.IGNORECASE)
|
|
153
|
+
if role_match:
|
|
154
|
+
role = _clean_text(role_match.group(2)).rstrip(".,!?")
|
|
155
|
+
if len(role.split()) <= 8:
|
|
156
|
+
items.append({"content": f"User role is {role}", "kind": "profile", "importance": 5, "ttl_days": None})
|
|
157
|
+
return items
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _extract_preferences(text: str) -> list[dict[str, Any]]:
|
|
161
|
+
items: list[dict[str, Any]] = []
|
|
162
|
+
patterns = [
|
|
163
|
+
r"\bi prefer\s+(.+?)(?=$|\bbut\b|\bhowever\b)",
|
|
164
|
+
r"\bi like\s+(.+?)(?=$|\bbut\b|\bhowever\b)",
|
|
165
|
+
r"\bi love\s+(.+?)(?=$|\bbut\b|\bhowever\b)",
|
|
166
|
+
r"\bi dislike\s+(.+?)(?=$|\bbut\b|\bhowever\b)",
|
|
167
|
+
r"\bi hate\s+(.+?)(?=$|\bbut\b|\bhowever\b)",
|
|
168
|
+
]
|
|
169
|
+
for pattern in patterns:
|
|
170
|
+
match = re.search(pattern, text, re.IGNORECASE)
|
|
171
|
+
if not match:
|
|
172
|
+
continue
|
|
173
|
+
detail = _clean_text(match.group(1)).rstrip(".,!?")
|
|
174
|
+
if not detail or detail.lower().startswith(LOW_SIGNAL_PREFIXES):
|
|
175
|
+
continue
|
|
176
|
+
items.append({"content": f"User preference: {detail}", "kind": "preference", "importance": 4, "ttl_days": None})
|
|
177
|
+
return items
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _extract_tasks(text: str) -> list[dict[str, Any]]:
|
|
181
|
+
items: list[dict[str, Any]] = []
|
|
182
|
+
task_patterns = [
|
|
183
|
+
r"\bi am building\s+(.+)",
|
|
184
|
+
r"\bi'?m building\s+(.+)",
|
|
185
|
+
r"\bi am working on\s+(.+)",
|
|
186
|
+
r"\bi'?m working on\s+(.+)",
|
|
187
|
+
r"\bi am creating\s+(.+)",
|
|
188
|
+
r"\bi'?m creating\s+(.+)",
|
|
189
|
+
r"\bi (?:just\s+)?built\s+(.+)",
|
|
190
|
+
r"\bi (?:just\s+)?finished\s+(.+)",
|
|
191
|
+
r"\bi (?:just\s+)?completed\s+(.+)",
|
|
192
|
+
r"\bi (?:just\s+)?shipped\s+(.+)",
|
|
193
|
+
r"\bi(?:'?m|'m| am) (?:now\s+)?testing\s+(.+)",
|
|
194
|
+
r"\bi will (?:start\s+)?(?:work on|build|implement|create)\s+(.+)",
|
|
195
|
+
r"\btomorrow(?:\s+i)?\s+will\s+(?:start\s+)?(?:work on|build|implement|create)\s+(.+)",
|
|
196
|
+
r"\bfirst task(?:\s+tomorrow)?\s+will\s+be\s+(?:implementing|building|creating|working on)\s+(.+)",
|
|
197
|
+
]
|
|
198
|
+
for pattern in task_patterns:
|
|
199
|
+
match = re.search(pattern, text, re.IGNORECASE)
|
|
200
|
+
if not match:
|
|
201
|
+
continue
|
|
202
|
+
task = _clean_text(match.group(1)).rstrip(".,!?")
|
|
203
|
+
if task:
|
|
204
|
+
items.append({"content": f"User is working on {task}", "kind": "task", "importance": 3, "ttl_days": 30})
|
|
205
|
+
return items
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _extract_constraints(text: str) -> list[dict[str, Any]]:
|
|
209
|
+
items: list[dict[str, Any]] = []
|
|
210
|
+
if re.search(r"\b(i need|must|should|do not|don't|cannot|can't)\b", text, re.IGNORECASE):
|
|
211
|
+
cleaned = _clean_text(text).rstrip(".,!?")
|
|
212
|
+
if cleaned and not cleaned.endswith("?"):
|
|
213
|
+
items.append({"content": f"User constraint: {cleaned}", "kind": "constraint", "importance": 4, "ttl_days": 14})
|
|
214
|
+
return items
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _regex_extract(user_text: str) -> list[dict[str, Any]]:
|
|
218
|
+
extracted: list[dict[str, Any]] = []
|
|
219
|
+
for segment in _split_into_segments(user_text):
|
|
220
|
+
cleaned = _clean_text(segment)
|
|
221
|
+
if not cleaned or cleaned.endswith("?"):
|
|
222
|
+
continue
|
|
223
|
+
extracted.extend(_extract_profile(cleaned))
|
|
224
|
+
extracted.extend(_extract_preferences(cleaned))
|
|
225
|
+
extracted.extend(_extract_tasks(cleaned))
|
|
226
|
+
extracted.extend(_extract_constraints(cleaned))
|
|
227
|
+
return extracted
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
# ---------------------------------------------------------------------------
|
|
231
|
+
# Public API
|
|
232
|
+
# ---------------------------------------------------------------------------
|
|
233
|
+
|
|
234
|
+
def extract_memories(
|
|
235
|
+
user_text: str,
|
|
236
|
+
assistant_text: str = "",
|
|
237
|
+
existing_memories: list[str] | None = None,
|
|
238
|
+
) -> list[dict[str, Any]]:
|
|
239
|
+
"""
|
|
240
|
+
Extract high-signal memories from a conversation turn.
|
|
241
|
+
|
|
242
|
+
Strategy: LLM extraction (Bedrock) analysing both user and assistant text,
|
|
243
|
+
with existing memories passed as context so the LLM skips already-known facts.
|
|
244
|
+
Falls back to deterministic regex on user text if LLM call fails.
|
|
245
|
+
"""
|
|
246
|
+
llm_results = _call_llm_for_extraction(user_text or "", assistant_text or "", existing_memories)
|
|
247
|
+
|
|
248
|
+
if llm_results:
|
|
249
|
+
candidates = llm_results
|
|
250
|
+
else:
|
|
251
|
+
candidates = _regex_extract(user_text or "")
|
|
252
|
+
|
|
253
|
+
# Deduplicate within this extraction call
|
|
254
|
+
deduped: list[dict[str, Any]] = []
|
|
255
|
+
seen: set[tuple[str, str]] = set()
|
|
256
|
+
for item in candidates:
|
|
257
|
+
key = (item["kind"], item["content"].lower())
|
|
258
|
+
if key not in seen:
|
|
259
|
+
seen.add(key)
|
|
260
|
+
deduped.append(item)
|
|
261
|
+
|
|
262
|
+
return deduped
|
contexara/ingest.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from .extract import extract_memories
|
|
4
|
+
from .store import list_all, upsert_memory
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def ingest_turn(user_text: str, assistant_text: str) -> dict[str, int]:
|
|
8
|
+
"""
|
|
9
|
+
Ingest one conversation turn and persist extracted high-signal memories.
|
|
10
|
+
|
|
11
|
+
Passes existing memory contents to the extractor so the LLM skips
|
|
12
|
+
facts that are already stored (context-aware extraction).
|
|
13
|
+
|
|
14
|
+
Returns ingestion stats for observability.
|
|
15
|
+
"""
|
|
16
|
+
stats = {
|
|
17
|
+
"extracted": 0,
|
|
18
|
+
"inserted": 0,
|
|
19
|
+
"updated": 0,
|
|
20
|
+
"skipped": 0,
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
existing = [m["content"] for m in list_all()]
|
|
24
|
+
candidates = extract_memories(
|
|
25
|
+
user_text=user_text,
|
|
26
|
+
assistant_text=assistant_text,
|
|
27
|
+
existing_memories=existing,
|
|
28
|
+
)
|
|
29
|
+
stats["extracted"] = len(candidates)
|
|
30
|
+
|
|
31
|
+
for memory in candidates:
|
|
32
|
+
try:
|
|
33
|
+
action = upsert_memory(
|
|
34
|
+
content=memory["content"],
|
|
35
|
+
kind=memory["kind"],
|
|
36
|
+
importance=memory["importance"],
|
|
37
|
+
ttl_days=memory.get("ttl_days"),
|
|
38
|
+
)
|
|
39
|
+
stats[action] += 1
|
|
40
|
+
except Exception:
|
|
41
|
+
stats["skipped"] += 1
|
|
42
|
+
|
|
43
|
+
return stats
|