memor-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- memor/__init__.py +0 -0
- memor/cli.py +463 -0
- memor/daemon.py +294 -0
- memor/dashboard/__init__.py +0 -0
- memor/dashboard/server.py +153 -0
- memor/dashboard/static/index.html +688 -0
- memor/distill/__init__.py +0 -0
- memor/distill/distiller.py +112 -0
- memor/distill/extractive.py +161 -0
- memor/embed/__init__.py +0 -0
- memor/embed/api.py +15 -0
- memor/embed/fake.py +16 -0
- memor/embed/local.py +16 -0
- memor/eval/__init__.py +0 -0
- memor/eval/baselines/__init__.py +5 -0
- memor/eval/baselines/base.py +15 -0
- memor/eval/baselines/claude_mem.py +19 -0
- memor/eval/baselines/graphiti.py +25 -0
- memor/eval/dataset.py +48 -0
- memor/eval/embed_benchmark.py +67 -0
- memor/eval/judge.py +137 -0
- memor/eval/metrics.py +13 -0
- memor/eval/runner.py +78 -0
- memor/feedback.py +96 -0
- memor/hook_server.py +144 -0
- memor/ingest/__init__.py +0 -0
- memor/ingest/claude_code.py +135 -0
- memor/ingest/documents.py +28 -0
- memor/interfaces.py +20 -0
- memor/llm/__init__.py +0 -0
- memor/llm/anthropic.py +14 -0
- memor/llm/base.py +7 -0
- memor/llm/openai_compat.py +20 -0
- memor/project.py +69 -0
- memor/recall.py +115 -0
- memor/redact.py +129 -0
- memor/retrieve/__init__.py +0 -0
- memor/retrieve/retriever.py +78 -0
- memor/store/__init__.py +0 -0
- memor/store/sqlite_store.py +336 -0
- memor/tokencount.py +9 -0
- memor/types.py +45 -0
- memor_cli-0.1.0.dist-info/METADATA +273 -0
- memor_cli-0.1.0.dist-info/RECORD +48 -0
- memor_cli-0.1.0.dist-info/WHEEL +5 -0
- memor_cli-0.1.0.dist-info/entry_points.txt +2 -0
- memor_cli-0.1.0.dist-info/licenses/LICENSE +21 -0
- memor_cli-0.1.0.dist-info/top_level.txt +1 -0
memor/eval/runner.py
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
from memor.types import Scope
|
|
3
|
+
from memor.retrieve.retriever import Retriever
|
|
4
|
+
from memor.eval.dataset import EvalCase, CaseResult
|
|
5
|
+
from memor.eval.metrics import recall_at_k, ndcg_at_k
|
|
6
|
+
|
|
7
|
+
BASELINES = ["no-memory", "last-N", "naive-RAG", "memory"]
|
|
8
|
+
|
|
9
|
+
def _result(ranked_ids, hits_tokens, latency, case, k):
|
|
10
|
+
return CaseResult(
|
|
11
|
+
recall=recall_at_k(ranked_ids, case.relevant_ids, k),
|
|
12
|
+
ndcg=ndcg_at_k(ranked_ids, case.relevant_ids, k),
|
|
13
|
+
tokens_sent=hits_tokens, latency_ms=latency)
|
|
14
|
+
|
|
15
|
+
def run_case(case: EvalCase, *, store, embedder, k: int = 8) -> dict[str, CaseResult]:
|
|
16
|
+
scope = Scope(project=case.scope_project)
|
|
17
|
+
out: dict[str, CaseResult] = {}
|
|
18
|
+
|
|
19
|
+
# no-memory: retrieve nothing
|
|
20
|
+
out["no-memory"] = _result([], 0, 0.0, case, k)
|
|
21
|
+
|
|
22
|
+
# last-N: most recent k chunks, no vector search
|
|
23
|
+
recent = store.recent(scope, k) if hasattr(store, 'recent') else []
|
|
24
|
+
out["last-N"] = _result([a.id for a in recent],
|
|
25
|
+
sum(a.token_count for a in recent),
|
|
26
|
+
0.0, case, k)
|
|
27
|
+
|
|
28
|
+
# naive-RAG: similarity only, no edges, no distill
|
|
29
|
+
naive = Retriever(store, embedder, k=k, recency_weight=0.0, edge_expand=False)
|
|
30
|
+
tr = naive.query(case.query, scope)
|
|
31
|
+
out["naive-RAG"] = _result([h.artifact.id for h in tr.hits],
|
|
32
|
+
sum(h.artifact.token_count for h in tr.hits),
|
|
33
|
+
tr.latency_ms, case, k)
|
|
34
|
+
|
|
35
|
+
# memory: full retriever (recency + edge expansion over distilled+raw artifacts)
|
|
36
|
+
mem = Retriever(store, embedder, k=k, recency_weight=0.2, edge_expand=True)
|
|
37
|
+
tr = mem.query(case.query, scope)
|
|
38
|
+
out["memory"] = _result([h.artifact.id for h in tr.hits],
|
|
39
|
+
sum(h.artifact.token_count for h in tr.hits),
|
|
40
|
+
tr.latency_ms, case, k)
|
|
41
|
+
return out
|
|
42
|
+
|
|
43
|
+
def run_suite(cases: list[EvalCase], *, store, embedder, k: int = 8) -> dict[str, dict]:
|
|
44
|
+
"""Aggregate mean metrics per strategy + Δ vs naive-RAG and full-history tokens."""
|
|
45
|
+
agg: dict[str, list[CaseResult]] = {}
|
|
46
|
+
for c in cases:
|
|
47
|
+
for strat, res in run_case(c, store=store, embedder=embedder, k=k).items():
|
|
48
|
+
agg.setdefault(strat, []).append(res)
|
|
49
|
+
summary = {}
|
|
50
|
+
n = len(cases) or 1
|
|
51
|
+
full_tokens = sum(c.baseline_full_tokens for c in cases) / n
|
|
52
|
+
for strat, results in agg.items():
|
|
53
|
+
mean_tokens = sum(r.tokens_sent for r in results) / n
|
|
54
|
+
latencies = sorted(r.latency_ms for r in results)
|
|
55
|
+
summary[strat] = {
|
|
56
|
+
"recall@k": sum(r.recall for r in results) / n,
|
|
57
|
+
"ndcg@k": sum(r.ndcg for r in results) / n,
|
|
58
|
+
"tokens_sent": mean_tokens,
|
|
59
|
+
"token_savings_vs_full": 1.0 - (mean_tokens / full_tokens) if full_tokens else 0.0,
|
|
60
|
+
"latency_ms_p50": latencies[len(latencies)//2],
|
|
61
|
+
"latency_ms_p95": latencies[int(len(latencies)*0.95)],
|
|
62
|
+
}
|
|
63
|
+
return summary
|
|
64
|
+
|
|
65
|
+
def run_ablation(*, query, project, relevant_ids, store, embedder, k=8):
|
|
66
|
+
out = {}
|
|
67
|
+
for name, expand in (("similarity-only", False), ("similarity+edges", True)):
|
|
68
|
+
r = Retriever(store, embedder, k=k, recency_weight=0.0, edge_expand=expand)
|
|
69
|
+
tr = r.query(query, Scope(project=project))
|
|
70
|
+
ids = [h.artifact.id for h in tr.hits]
|
|
71
|
+
out[name] = {"recall@k": recall_at_k(ids, relevant_ids, k),
|
|
72
|
+
"ndcg@k": ndcg_at_k(ids, relevant_ids, k)}
|
|
73
|
+
return out
|
|
74
|
+
|
|
75
|
+
def run_contradiction_eval(*, query, project, stale_id, current_id, store, embedder, k=8):
|
|
76
|
+
r = Retriever(store, embedder, k=k, edge_expand=True)
|
|
77
|
+
ids = [h.artifact.id for h in r.query(query, Scope(project=project)).hits]
|
|
78
|
+
return (current_id in ids) and (stale_id not in ids)
|
memor/feedback.py
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
"""Feedback analyzer — detects whether recalled memories were used by the agent.
|
|
2
|
+
|
|
3
|
+
After a session ends, cross-references recall_log with the transcript to see
|
|
4
|
+
if the agent's responses referenced recalled content. Updates memory_quality
|
|
5
|
+
scores accordingly."""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
import json
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from memor.store.sqlite_store import SqliteStore
|
|
10
|
+
|
|
11
|
+
MIN_OVERLAP_TOKENS = 5
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _extract_assistant_texts(transcript_path: Path) -> list[str]:
|
|
15
|
+
texts = []
|
|
16
|
+
for line in transcript_path.read_text().splitlines():
|
|
17
|
+
line = line.strip()
|
|
18
|
+
if not line:
|
|
19
|
+
continue
|
|
20
|
+
try:
|
|
21
|
+
rec = json.loads(line)
|
|
22
|
+
except json.JSONDecodeError:
|
|
23
|
+
continue
|
|
24
|
+
if rec.get("type") != "assistant":
|
|
25
|
+
continue
|
|
26
|
+
msg = rec.get("message", {})
|
|
27
|
+
content = msg.get("content", "")
|
|
28
|
+
if isinstance(content, str):
|
|
29
|
+
texts.append(content.lower())
|
|
30
|
+
elif isinstance(content, list):
|
|
31
|
+
for block in content:
|
|
32
|
+
if isinstance(block, dict) and block.get("type") == "text":
|
|
33
|
+
texts.append(block.get("text", "").lower())
|
|
34
|
+
return texts
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _text_was_used(memory_text: str, assistant_texts: list[str]) -> bool:
|
|
38
|
+
words = memory_text.lower().split()
|
|
39
|
+
if len(words) < MIN_OVERLAP_TOKENS:
|
|
40
|
+
return False
|
|
41
|
+
key_phrases = []
|
|
42
|
+
for i in range(0, len(words) - 4):
|
|
43
|
+
key_phrases.append(" ".join(words[i:i+5]))
|
|
44
|
+
if not key_phrases:
|
|
45
|
+
return False
|
|
46
|
+
matches = 0
|
|
47
|
+
for phrase in key_phrases:
|
|
48
|
+
for text in assistant_texts:
|
|
49
|
+
if phrase in text:
|
|
50
|
+
matches += 1
|
|
51
|
+
break
|
|
52
|
+
return matches >= max(2, len(key_phrases) // 5)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def analyze_session_feedback(
|
|
56
|
+
store: SqliteStore, session_id: str, transcript_path: Path
|
|
57
|
+
) -> int:
|
|
58
|
+
recalls = store.db.execute(
|
|
59
|
+
"SELECT * FROM recall_log WHERE session_id=? AND hits_count > 0",
|
|
60
|
+
(session_id,)
|
|
61
|
+
).fetchall()
|
|
62
|
+
if not recalls:
|
|
63
|
+
return 0
|
|
64
|
+
|
|
65
|
+
recalled_ids = set()
|
|
66
|
+
for r in recalls:
|
|
67
|
+
log_id = r["id"]
|
|
68
|
+
rl_time = r["timestamp"]
|
|
69
|
+
nearby = store.db.execute("""
|
|
70
|
+
SELECT q.artifact_id FROM memory_quality q
|
|
71
|
+
JOIN artifacts a ON a.id = q.artifact_id
|
|
72
|
+
WHERE q.last_recalled BETWEEN ? - 2 AND ? + 2
|
|
73
|
+
AND a.active = 1
|
|
74
|
+
""", (rl_time, rl_time)).fetchall()
|
|
75
|
+
for row in nearby:
|
|
76
|
+
recalled_ids.add(row["artifact_id"])
|
|
77
|
+
|
|
78
|
+
if not recalled_ids:
|
|
79
|
+
return 0
|
|
80
|
+
|
|
81
|
+
assistant_texts = _extract_assistant_texts(transcript_path)
|
|
82
|
+
if not assistant_texts:
|
|
83
|
+
return 0
|
|
84
|
+
|
|
85
|
+
used_ids = []
|
|
86
|
+
for aid in recalled_ids:
|
|
87
|
+
art = store.db.execute(
|
|
88
|
+
"SELECT text FROM artifacts WHERE id=?", (aid,)
|
|
89
|
+
).fetchone()
|
|
90
|
+
if art and _text_was_used(art["text"], assistant_texts):
|
|
91
|
+
used_ids.append(aid)
|
|
92
|
+
|
|
93
|
+
if used_ids:
|
|
94
|
+
store.record_usage(used_ids)
|
|
95
|
+
|
|
96
|
+
return len(used_ids)
|
memor/hook_server.py
ADDED
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
import asyncio
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
import signal
|
|
6
|
+
import sys
|
|
7
|
+
import time
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
SOCK_PATH = Path.home() / ".memor" / "hook.sock"
|
|
11
|
+
PID_PATH = Path.home() / ".memor" / "hook.pid"
|
|
12
|
+
DEFAULT_DB = str(Path.home() / ".memor" / "memor.db")
|
|
13
|
+
IDLE_TIMEOUT_S = 600
|
|
14
|
+
|
|
15
|
+
_embedder = None
|
|
16
|
+
_last_activity = 0.0
|
|
17
|
+
|
|
18
|
+
_UNSET = object() # sentinel for "auto-discover embedder"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _get_embedder():
|
|
22
|
+
global _embedder
|
|
23
|
+
if _embedder is not None:
|
|
24
|
+
return _embedder
|
|
25
|
+
from memor.embed.local import LocalEmbedder
|
|
26
|
+
_embedder = LocalEmbedder()
|
|
27
|
+
return _embedder
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def handle_request(req: dict, *, db_path: str = DEFAULT_DB,
|
|
31
|
+
embedder=_UNSET) -> dict:
|
|
32
|
+
"""Process a recall request and return the hook JSON response.
|
|
33
|
+
This is the core logic, used by both the socket server and inline fallback.
|
|
34
|
+
|
|
35
|
+
Pass embedder=None to explicitly indicate no embedder is available (returns
|
|
36
|
+
the no_embedder status message). Omit embedder (default) to auto-discover."""
|
|
37
|
+
from memor.recall import recall, _status_message
|
|
38
|
+
from memor.project import resolve_project
|
|
39
|
+
|
|
40
|
+
cwd = req.get("cwd", "")
|
|
41
|
+
project = resolve_project(cwd) if cwd else "unknown"
|
|
42
|
+
query = req.get("prompt", "")
|
|
43
|
+
session_id = req.get("session_id", "")
|
|
44
|
+
|
|
45
|
+
if embedder is _UNSET:
|
|
46
|
+
embedder = _get_embedder()
|
|
47
|
+
if embedder is None:
|
|
48
|
+
msg = _status_message("no_embedder", project, 0, 0, 0.0)
|
|
49
|
+
return {
|
|
50
|
+
"hookSpecificOutput": {
|
|
51
|
+
"hookEventName": "UserPromptSubmit",
|
|
52
|
+
"additionalContext": f"---\n{msg}",
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
result = recall(query, project, db_path, embedder=embedder, k=8, threshold=0.15,
|
|
57
|
+
session_id=session_id)
|
|
58
|
+
|
|
59
|
+
if Path(db_path).exists():
|
|
60
|
+
try:
|
|
61
|
+
from memor.store.sqlite_store import SqliteStore
|
|
62
|
+
store = SqliteStore(db_path, dim=embedder.dim)
|
|
63
|
+
store.log_recall(
|
|
64
|
+
project=project, query_preview=query[:100],
|
|
65
|
+
hits_count=result.hits_count, top_score=result.top_score,
|
|
66
|
+
tokens_injected=result.tokens_injected, latency_ms=result.latency_ms,
|
|
67
|
+
status=result.status, session_id=session_id)
|
|
68
|
+
if result.hit_ids:
|
|
69
|
+
store.record_recall(result.hit_ids)
|
|
70
|
+
except Exception:
|
|
71
|
+
pass
|
|
72
|
+
|
|
73
|
+
additional_context = result.formatted_context or f"---\n{result.status_message}"
|
|
74
|
+
return {
|
|
75
|
+
"hookSpecificOutput": {
|
|
76
|
+
"hookEventName": "UserPromptSubmit",
|
|
77
|
+
"additionalContext": additional_context,
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
async def _handle_client(reader: asyncio.StreamReader,
|
|
83
|
+
writer: asyncio.StreamWriter) -> None:
|
|
84
|
+
global _last_activity
|
|
85
|
+
_last_activity = time.time()
|
|
86
|
+
try:
|
|
87
|
+
data = await asyncio.wait_for(reader.read(1_000_000), timeout=10)
|
|
88
|
+
req = json.loads(data.decode())
|
|
89
|
+
resp = handle_request(req)
|
|
90
|
+
writer.write(json.dumps(resp).encode())
|
|
91
|
+
await writer.drain()
|
|
92
|
+
except Exception as e:
|
|
93
|
+
err = json.dumps({"error": str(e)})
|
|
94
|
+
writer.write(err.encode())
|
|
95
|
+
await writer.drain()
|
|
96
|
+
finally:
|
|
97
|
+
writer.close()
|
|
98
|
+
await writer.wait_closed()
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
async def _idle_watchdog():
|
|
102
|
+
global _last_activity
|
|
103
|
+
while True:
|
|
104
|
+
await asyncio.sleep(60)
|
|
105
|
+
if time.time() - _last_activity > IDLE_TIMEOUT_S:
|
|
106
|
+
_cleanup()
|
|
107
|
+
os._exit(0)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _cleanup():
|
|
111
|
+
if SOCK_PATH.exists():
|
|
112
|
+
SOCK_PATH.unlink()
|
|
113
|
+
if PID_PATH.exists():
|
|
114
|
+
PID_PATH.unlink()
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
async def serve(sock_path: str = str(SOCK_PATH)) -> None:
|
|
118
|
+
global _last_activity
|
|
119
|
+
_last_activity = time.time()
|
|
120
|
+
p = Path(sock_path)
|
|
121
|
+
if p.exists():
|
|
122
|
+
p.unlink()
|
|
123
|
+
p.parent.mkdir(parents=True, exist_ok=True)
|
|
124
|
+
|
|
125
|
+
_get_embedder()
|
|
126
|
+
|
|
127
|
+
server = await asyncio.start_unix_server(_handle_client, path=sock_path)
|
|
128
|
+
PID_PATH.write_text(str(os.getpid()))
|
|
129
|
+
asyncio.create_task(_idle_watchdog())
|
|
130
|
+
|
|
131
|
+
async with server:
|
|
132
|
+
await server.serve_forever()
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def main():
|
|
136
|
+
signal.signal(signal.SIGTERM, lambda *_: (_cleanup(), sys.exit(0)))
|
|
137
|
+
try:
|
|
138
|
+
asyncio.run(serve())
|
|
139
|
+
except KeyboardInterrupt:
|
|
140
|
+
_cleanup()
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
if __name__ == "__main__":
|
|
144
|
+
main()
|
memor/ingest/__init__.py
ADDED
|
File without changes
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
import hashlib, json, re
|
|
3
|
+
from datetime import datetime
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from memor.types import Artifact
|
|
6
|
+
from memor.tokencount import count_tokens
|
|
7
|
+
from memor.redact import redact_text
|
|
8
|
+
|
|
9
|
+
def _epoch(ts: str) -> float:
|
|
10
|
+
return datetime.fromisoformat(ts.replace("Z", "+00:00")).timestamp()
|
|
11
|
+
|
|
12
|
+
def _text_of(message: dict) -> str:
|
|
13
|
+
c = message.get("content", "")
|
|
14
|
+
if isinstance(c, str):
|
|
15
|
+
return c
|
|
16
|
+
parts = []
|
|
17
|
+
for block in c:
|
|
18
|
+
if isinstance(block, dict) and block.get("type") == "text":
|
|
19
|
+
parts.append(block.get("text", ""))
|
|
20
|
+
return "\n".join(parts)
|
|
21
|
+
|
|
22
|
+
_FILLER_STARTS = re.compile(
|
|
23
|
+
r"^(Let me |Now let me |Now I|I'll |I will |Good[,.]|Great[,.]|Perfect[!,.]|Done[!,.]"
|
|
24
|
+
r"|Alright|Sure[,.]|OK[,.]|Moving to |Looking at |Checking )", re.I)
|
|
25
|
+
_SKILL_BOILERPLATE = "Base directory for this skill"
|
|
26
|
+
|
|
27
|
+
_DECISION_RE = re.compile(
|
|
28
|
+
r"(we decided|the approach is|instead of|switched to|chose .+ over|"
|
|
29
|
+
r"trade-?off|architecture:|design decision)", re.I)
|
|
30
|
+
_BUGFIX_RE = re.compile(
|
|
31
|
+
r"(the fix is|root cause|the bug was|the issue was|caused by|"
|
|
32
|
+
r"the problem is|this fails because|the error occurs)", re.I)
|
|
33
|
+
_LESSON_RE = re.compile(
|
|
34
|
+
r"(always use|never use|never do|important:|note:|pattern:|"
|
|
35
|
+
r"best practice|lesson learned|rule of thumb|should always|should never)", re.I)
|
|
36
|
+
|
|
37
|
+
_BASE64_RE = re.compile(r"[A-Za-z0-9+/=]{200,}")
|
|
38
|
+
_PERMISSION_RE = re.compile(r"^(Allow |Permission |Do you want to allow |Grant )", re.I)
|
|
39
|
+
_PATH_LINE_RE = re.compile(r"^[/~][\w./\-]+$")
|
|
40
|
+
_SYSTEM_REMINDER_RE = re.compile(r"<system-reminder>.*?</system-reminder>", re.DOTALL)
|
|
41
|
+
|
|
42
|
+
MIN_SIGNAL_TOKENS = 8
|
|
43
|
+
MIN_FILLER_TOKENS = 30
|
|
44
|
+
MIN_USER_TOKENS = 6
|
|
45
|
+
MIN_CODE_TOKENS = 40
|
|
46
|
+
MIN_LONG_TOKENS = 100
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _strip_system_reminders(text: str) -> str:
|
|
50
|
+
"""Strip <system-reminder>...</system-reminder> tags and their content."""
|
|
51
|
+
return _SYSTEM_REMINDER_RE.sub("", text)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _is_file_listing(text: str) -> bool:
|
|
55
|
+
"""Return True if text is a pure file listing (multiple lines of just paths)."""
|
|
56
|
+
lines = [l for l in text.splitlines() if l.strip()]
|
|
57
|
+
if len(lines) < 2:
|
|
58
|
+
return False
|
|
59
|
+
return all(_PATH_LINE_RE.match(l.strip()) for l in lines)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _signal_score(text: str, role: str, token_count: int) -> float:
|
|
63
|
+
"""Return a score > 0 for signal content, 0 for noise."""
|
|
64
|
+
# Hard noise filters
|
|
65
|
+
if token_count < MIN_SIGNAL_TOKENS:
|
|
66
|
+
return 0.0
|
|
67
|
+
if _SKILL_BOILERPLATE in text:
|
|
68
|
+
return 0.0
|
|
69
|
+
if _BASE64_RE.search(text):
|
|
70
|
+
return 0.0
|
|
71
|
+
if _PERMISSION_RE.match(text):
|
|
72
|
+
return 0.0
|
|
73
|
+
if _is_file_listing(text):
|
|
74
|
+
return 0.0
|
|
75
|
+
if _FILLER_STARTS.match(text) and token_count < MIN_FILLER_TOKENS:
|
|
76
|
+
return 0.0
|
|
77
|
+
|
|
78
|
+
score = 0.0
|
|
79
|
+
|
|
80
|
+
if role == "user" and token_count >= MIN_USER_TOKENS:
|
|
81
|
+
score += 1.0
|
|
82
|
+
elif role == "assistant":
|
|
83
|
+
if _DECISION_RE.search(text):
|
|
84
|
+
score += 2.0
|
|
85
|
+
if _BUGFIX_RE.search(text):
|
|
86
|
+
score += 2.0
|
|
87
|
+
if _LESSON_RE.search(text):
|
|
88
|
+
score += 2.0
|
|
89
|
+
if "```" in text and token_count >= MIN_CODE_TOKENS:
|
|
90
|
+
score += 1.0
|
|
91
|
+
if score == 0.0 and token_count >= MIN_LONG_TOKENS:
|
|
92
|
+
score += 0.5
|
|
93
|
+
|
|
94
|
+
return score
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def parse_transcript(path: Path, project: str, *, filter_noise: bool = True) -> list[Artifact]:
|
|
98
|
+
session_id = path.stem
|
|
99
|
+
arts: list[Artifact] = []
|
|
100
|
+
seen_hashes: set[str] = set()
|
|
101
|
+
for i, line in enumerate(path.read_text().splitlines()):
|
|
102
|
+
line = line.strip()
|
|
103
|
+
if not line:
|
|
104
|
+
continue
|
|
105
|
+
rec = json.loads(line)
|
|
106
|
+
if rec.get("type") not in ("user", "assistant"):
|
|
107
|
+
continue
|
|
108
|
+
msg = rec.get("message", {})
|
|
109
|
+
text = _text_of(msg).strip()
|
|
110
|
+
if not text:
|
|
111
|
+
continue
|
|
112
|
+
text = _strip_system_reminders(text).strip()
|
|
113
|
+
if not text:
|
|
114
|
+
continue
|
|
115
|
+
text, _ = redact_text(text)
|
|
116
|
+
if not text.strip():
|
|
117
|
+
continue
|
|
118
|
+
# Deduplicate by MD5 hash within session
|
|
119
|
+
text_hash = hashlib.md5(text.encode()).hexdigest()
|
|
120
|
+
if text_hash in seen_hashes:
|
|
121
|
+
continue
|
|
122
|
+
seen_hashes.add(text_hash)
|
|
123
|
+
token_count = max(1, count_tokens(text))
|
|
124
|
+
role = msg.get("role", "")
|
|
125
|
+
if filter_noise and _signal_score(text, role, token_count) == 0:
|
|
126
|
+
continue
|
|
127
|
+
arts.append(Artifact(
|
|
128
|
+
id=f"{session_id}:{i}", kind="session_chunk", project=project,
|
|
129
|
+
source="claude_code", text=text, token_count=token_count,
|
|
130
|
+
created_at=_epoch(rec["timestamp"]),
|
|
131
|
+
meta={"session_id": session_id, "role": role, "ord": i}))
|
|
132
|
+
return arts
|
|
133
|
+
|
|
134
|
+
def discover_project(transcript_path: Path) -> str:
|
|
135
|
+
return transcript_path.parent.name
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
import hashlib, re
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from memor.types import Artifact
|
|
5
|
+
|
|
6
|
+
def parse_document(path: Path, project: str, kind: str = "note",
|
|
7
|
+
created_at: float = 0.0) -> list[Artifact]:
|
|
8
|
+
text = path.read_text()
|
|
9
|
+
# split on markdown headings; keep the heading with its body
|
|
10
|
+
parts = re.split(r"(?m)^(#{1,6}\s.*)$", text)
|
|
11
|
+
chunks: list[str] = []
|
|
12
|
+
buf = ""
|
|
13
|
+
for seg in parts:
|
|
14
|
+
if re.match(r"^#{1,6}\s", seg or ""):
|
|
15
|
+
if buf.strip():
|
|
16
|
+
chunks.append(buf.strip())
|
|
17
|
+
buf = seg + "\n"
|
|
18
|
+
else:
|
|
19
|
+
buf += seg
|
|
20
|
+
if buf.strip():
|
|
21
|
+
chunks.append(buf.strip())
|
|
22
|
+
arts = []
|
|
23
|
+
for i, c in enumerate(chunks):
|
|
24
|
+
cid = f"{kind}:{path.stem}:{hashlib.sha1(c.encode()).hexdigest()[:8]}"
|
|
25
|
+
arts.append(Artifact(id=cid, kind=kind, project=project, source=str(path),
|
|
26
|
+
text=c, token_count=max(1, len(c)//4),
|
|
27
|
+
created_at=created_at, meta={"path": str(path), "ord": i}))
|
|
28
|
+
return arts
|
memor/interfaces.py
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
from typing import Protocol, runtime_checkable
|
|
3
|
+
from memor.types import Artifact, Scope
|
|
4
|
+
|
|
5
|
+
@runtime_checkable
|
|
6
|
+
class Embedder(Protocol):
|
|
7
|
+
dim: int
|
|
8
|
+
def embed(self, texts: list[str]) -> list[list[float]]: ...
|
|
9
|
+
|
|
10
|
+
@runtime_checkable
|
|
11
|
+
class LLM(Protocol):
|
|
12
|
+
def complete(self, prompt: str, *, max_tokens: int = 1024) -> str: ...
|
|
13
|
+
|
|
14
|
+
@runtime_checkable
|
|
15
|
+
class MemoryStore(Protocol):
|
|
16
|
+
def add_artifacts(self, artifacts: list[Artifact], vectors: list[list[float]]) -> None: ...
|
|
17
|
+
def add_edge(self, src_id: str, dst_id: str, type: str) -> None: ...
|
|
18
|
+
def search(self, vector: list[float], scope: Scope, k: int) -> list[tuple[Artifact, float]]: ...
|
|
19
|
+
def neighbors(self, ids: list[str], types: list[str], hops: int = 1) -> list[Artifact]: ...
|
|
20
|
+
def deactivate(self, artifact_id: str, superseded_by: str) -> None: ...
|
memor/llm/__init__.py
ADDED
|
File without changes
|
memor/llm/anthropic.py
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
class AnthropicLLM:
|
|
2
|
+
def __init__(self, model: str = "claude-opus-4-8", api_key: str | None = None):
|
|
3
|
+
import anthropic
|
|
4
|
+
|
|
5
|
+
self.client = anthropic.Anthropic(api_key=api_key)
|
|
6
|
+
self.model = model
|
|
7
|
+
|
|
8
|
+
def complete(self, prompt: str, *, max_tokens: int = 1024) -> str:
|
|
9
|
+
msg = self.client.messages.create(
|
|
10
|
+
model=self.model,
|
|
11
|
+
max_tokens=max_tokens,
|
|
12
|
+
messages=[{"role": "user", "content": prompt}],
|
|
13
|
+
)
|
|
14
|
+
return "".join(b.text for b in msg.content if b.type == "text")
|
memor/llm/base.py
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
DISTILL_PROMPT = """You are distilling a coding session into reusable memory.
|
|
2
|
+
Return STRICT JSON: {{"memories":[{{"type":"decision|lesson|snippet|bugfix",
|
|
3
|
+
"text":"<one concise reusable fact>", "supersedes_text":"<prior fact this reverses, or omit>"}}]}}
|
|
4
|
+
Only include durable, reusable facts. Session text:
|
|
5
|
+
---
|
|
6
|
+
{session_text}
|
|
7
|
+
---"""
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
import httpx
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class OpenAICompatLLM:
|
|
5
|
+
def __init__(self, base_url: str, api_key: str, model: str):
|
|
6
|
+
self.base_url, self.api_key, self.model = base_url, api_key, model
|
|
7
|
+
|
|
8
|
+
def complete(self, prompt: str, *, max_tokens: int = 1024) -> str:
|
|
9
|
+
r = httpx.post(
|
|
10
|
+
f"{self.base_url}/chat/completions",
|
|
11
|
+
headers={"Authorization": f"Bearer {self.api_key}"},
|
|
12
|
+
json={
|
|
13
|
+
"model": self.model,
|
|
14
|
+
"max_tokens": max_tokens,
|
|
15
|
+
"messages": [{"role": "user", "content": prompt}],
|
|
16
|
+
},
|
|
17
|
+
timeout=60,
|
|
18
|
+
)
|
|
19
|
+
r.raise_for_status()
|
|
20
|
+
return r.json()["choices"][0]["message"]["content"]
|
memor/project.py
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def resolve_project(cwd: str) -> str:
|
|
6
|
+
"""Resolve project name from a working directory by finding the git root."""
|
|
7
|
+
p = Path(cwd).resolve()
|
|
8
|
+
git_root = _find_git_root(p)
|
|
9
|
+
if git_root:
|
|
10
|
+
return git_root.name
|
|
11
|
+
return p.name
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _find_git_root(path: Path) -> Path | None:
|
|
15
|
+
current = path
|
|
16
|
+
while current != current.parent:
|
|
17
|
+
if (current / ".git").exists():
|
|
18
|
+
return current
|
|
19
|
+
current = current.parent
|
|
20
|
+
return None
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def decode_claude_dir(dirname: str) -> str:
|
|
24
|
+
"""Decode a Claude projects directory name back to a filesystem path.
|
|
25
|
+
'-Users-nimit-Documents-Projects-plirin' -> '/Users/nimit/Documents/Projects/plirin'
|
|
26
|
+
|
|
27
|
+
Dashes in the encoded name are ambiguous — they may represent either a path
|
|
28
|
+
separator or a literal dash in a directory name. This function walks the
|
|
29
|
+
real filesystem to resolve ambiguity greedily (longest matching segment wins).
|
|
30
|
+
Falls back to naive split-on-dash if the filesystem walk yields no result.
|
|
31
|
+
"""
|
|
32
|
+
stripped = dirname.lstrip("-")
|
|
33
|
+
tokens = stripped.split("-")
|
|
34
|
+
|
|
35
|
+
result = _fs_decode(tokens)
|
|
36
|
+
if result is not None:
|
|
37
|
+
return result
|
|
38
|
+
|
|
39
|
+
# Naive fallback: treat every dash as a separator
|
|
40
|
+
return "/" + "/".join(tokens)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _fs_decode(tokens: list[str]) -> str | None:
|
|
44
|
+
"""Greedily match tokens against the real filesystem, joining dashes as needed."""
|
|
45
|
+
path = Path("/")
|
|
46
|
+
remaining = list(tokens)
|
|
47
|
+
|
|
48
|
+
while remaining:
|
|
49
|
+
# Try to match a segment by joining increasing numbers of tokens with dashes
|
|
50
|
+
matched = False
|
|
51
|
+
for end in range(len(remaining), 0, -1):
|
|
52
|
+
candidate = "-".join(remaining[:end])
|
|
53
|
+
candidate_path = path / candidate
|
|
54
|
+
if candidate_path.exists():
|
|
55
|
+
path = candidate_path
|
|
56
|
+
remaining = remaining[end:]
|
|
57
|
+
matched = True
|
|
58
|
+
break
|
|
59
|
+
if not matched:
|
|
60
|
+
# No filesystem match found; give up and return None to trigger fallback
|
|
61
|
+
return None
|
|
62
|
+
|
|
63
|
+
return str(path)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def resolve_project_from_claude_dir(dirname: str) -> str:
|
|
67
|
+
"""Resolve project name from a Claude projects directory name."""
|
|
68
|
+
decoded = decode_claude_dir(dirname)
|
|
69
|
+
return resolve_project(decoded)
|