remcycle 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
dream/__init__.py ADDED
File without changes
dream/archive.py ADDED
@@ -0,0 +1,406 @@
1
+ """The archive: verbatim session turns in SQLite, found by ranked full-text search."""
2
+
3
+ import json
4
+ import re
5
+ import sqlite3
6
+ from collections.abc import Iterable, Sequence
7
+ from dataclasses import dataclass, field
8
+ from pathlib import Path
9
+ from typing import Self
10
+
11
+ from dream.redact import redact
12
+ from dream.transcript import Author, Kind, Session, Turn, parse_transcript
13
+
14
+ _RRF_K = 60
15
+
16
+ RULES_VERSION = 1
17
+ """Raise this whenever ingest would derive something different from the same transcript
18
+ (what the parser keeps, what is redacted, how a project is identified). Sessions archived
19
+ under an older value are derived again, for as long as their transcripts still exist."""
20
+
21
+ _SCHEMA = """
22
+ CREATE TABLE IF NOT EXISTS sessions (
23
+ session_id TEXT PRIMARY KEY,
24
+ project TEXT,
25
+ title TEXT,
26
+ cwd TEXT,
27
+ git_branch TEXT,
28
+ entrypoint TEXT,
29
+ started_at TEXT,
30
+ ended_at TEXT,
31
+ source_size INTEGER NOT NULL,
32
+ source_mtime_ns INTEGER NOT NULL,
33
+ rules INTEGER NOT NULL
34
+ );
35
+ CREATE TABLE IF NOT EXISTS turns (
36
+ id INTEGER PRIMARY KEY,
37
+ session_id TEXT NOT NULL REFERENCES sessions(session_id),
38
+ seq INTEGER NOT NULL,
39
+ author TEXT NOT NULL,
40
+ kind TEXT NOT NULL,
41
+ text TEXT NOT NULL,
42
+ uuid TEXT,
43
+ timestamp TEXT,
44
+ UNIQUE (session_id, seq)
45
+ );
46
+ CREATE TABLE IF NOT EXISTS dreams (
47
+ session_id TEXT PRIMARY KEY REFERENCES sessions(session_id),
48
+ turn_count INTEGER NOT NULL
49
+ );
50
+ CREATE TABLE IF NOT EXISTS digests (
51
+ session_id TEXT PRIMARY KEY REFERENCES sessions(session_id),
52
+ turn_count INTEGER NOT NULL,
53
+ body TEXT NOT NULL
54
+ );
55
+ CREATE VIRTUAL TABLE IF NOT EXISTS turns_fts USING fts5(
56
+ text, content='turns', content_rowid='id', tokenize='porter unicode61'
57
+ );
58
+ CREATE TRIGGER IF NOT EXISTS turns_ai AFTER INSERT ON turns BEGIN
59
+ INSERT INTO turns_fts(rowid, text) VALUES (new.id, new.text);
60
+ END;
61
+ CREATE TRIGGER IF NOT EXISTS turns_ad AFTER DELETE ON turns BEGIN
62
+ INSERT INTO turns_fts(turns_fts, rowid, text) VALUES ('delete', old.id, old.text);
63
+ END;
64
+ """
65
+
66
+
67
+ _CLAUDE_WORKTREE = re.compile(r"/\.claude/worktrees/.*")
68
+ _LINKED_WORKTREE = re.compile(r"gitdir:\s*(.*)/\.git/worktrees/[^/]+\s*$")
69
+
70
+
71
+ NO_PROJECT = "(no project)"
72
+ """The project of every session started without a folder."""
73
+
74
+
75
+ def project_of(cwd: str | None) -> str:
76
+ """The project a working directory belongs to: its git repository, or itself outside one.
77
+
78
+ A linked worktree counts as the repository it was made from. That is read from the
79
+ worktree while it still exists; once it is gone, only Claude's own worktree layout
80
+ can still be recognised, from the path.
81
+ """
82
+ if not cwd or "/scratch-workspaces/" in cwd: # where the desktop app puts a session with no folder
83
+ return NO_PROJECT
84
+ for folder in (Path(cwd), *Path(cwd).parents):
85
+ marker = folder / ".git"
86
+ if marker.is_file() and (linked := _LINKED_WORKTREE.match(marker.read_text())):
87
+ return linked[1]
88
+ if marker.exists():
89
+ return str(folder)
90
+ return _CLAUDE_WORKTREE.sub("", cwd)
91
+
92
+
93
+ def _within(project: str | None, parents: Iterable[str]) -> bool:
94
+ """Whether the project is one of `parents` or lies inside one."""
95
+ return any(project == parent or (project or "").startswith(parent.rstrip("/") + "/") for parent in parents)
96
+
97
+
98
+ @dataclass(frozen=True)
99
+ class Kept:
100
+ """An archived session, as much of it as deciding whether to keep it takes."""
101
+
102
+ session_id: str
103
+ project: str
104
+ title: str | None
105
+ turns: int
106
+
107
+
108
+ @dataclass(frozen=True)
109
+ class Hit:
110
+ session_id: str
111
+ seq: int
112
+ author: Author
113
+ kind: Kind
114
+ timestamp: str | None
115
+ snippet: str
116
+ title: str | None
117
+ project: str | None
118
+ matched_all: bool
119
+ """False when no turn held every word and this one holds only some of them."""
120
+
121
+
122
+ @dataclass(frozen=True)
123
+ class Recap:
124
+ """One archived session in brief."""
125
+
126
+ session_id: str
127
+ title: str | None
128
+ ended_at: str | None
129
+ turns: int
130
+ summary: str | None
131
+ """What the dream made of the session. None until it has read the session as it now stands."""
132
+ opening: str | None
133
+ """The first thing the person typed."""
134
+
135
+
136
+ @dataclass(frozen=True)
137
+ class Undreamt:
138
+ """A session the dream has not read, or has not read all of."""
139
+
140
+ session_id: str
141
+ project: str
142
+ title: str | None
143
+ ended_at: str | None
144
+
145
+
146
+ @dataclass
147
+ class IngestReport:
148
+ added: int = 0
149
+ updated: int = 0
150
+ unchanged: int = 0
151
+ kept_longer: list[str] = field(default_factory=list)
152
+ redacted: int = 0
153
+ """Secrets replaced in the sessions stored by this run."""
154
+ excluded: int = 0
155
+
156
+
157
+ class Archive:
158
+ def __init__(self, path: Path) -> None:
159
+ # The archive holds what was typed into every session, so only its owner may read it.
160
+ path.parent.mkdir(mode=0o700, parents=True, exist_ok=True)
161
+ self._db = sqlite3.connect(path)
162
+ path.chmod(0o600)
163
+ self._db.executescript(_SCHEMA)
164
+
165
+ def __enter__(self) -> Self:
166
+ return self
167
+
168
+ def __exit__(self, *exc: object) -> None:
169
+ self._db.close()
170
+
171
+ def ingest(self, root: Path, exclude: Iterable[str] = ()) -> IngestReport:
172
+ """Bring the archive up to date with the transcripts under root.
173
+
174
+ Sessions belonging to a project in `exclude`, or to one inside it, are not archived.
175
+ """
176
+ report = IngestReport()
177
+ # One level down only: deeper .jsonl files are subagent transcripts, not sessions.
178
+ for path in sorted(root.glob("*/*.jsonl")):
179
+ stat = path.stat()
180
+ source = (stat.st_size, stat.st_mtime_ns)
181
+ stored = self._db.execute(
182
+ "SELECT source_size, source_mtime_ns, rules FROM sessions WHERE session_id = ?", (path.stem,)
183
+ ).fetchone()
184
+ if stored == (*source, RULES_VERSION):
185
+ report.unchanged += 1
186
+ continue
187
+ session = parse_transcript(path)
188
+ project = project_of(session.cwd)
189
+ if _within(project, exclude):
190
+ report.excluded += 1
191
+ continue
192
+ if len(session.turns) < self._turn_count(session.session_id):
193
+ # The archive outlives its sources: a transcript that shrank never replaces a fuller copy.
194
+ report.kept_longer.append(session.session_id)
195
+ continue
196
+ report.redacted += self._store(session, project, source)
197
+ if stored:
198
+ report.updated += 1
199
+ else:
200
+ report.added += 1
201
+ return report
202
+
203
+ def under(self, parents: Iterable[str]) -> list[Kept]:
204
+ """Archived sessions belonging to one of these projects, or to a project inside one."""
205
+ parents = list(parents)
206
+ rows = self._db.execute(
207
+ """
208
+ SELECT s.session_id, s.project, s.title, (SELECT count(*) FROM turns t WHERE t.session_id = s.session_id)
209
+ FROM sessions s ORDER BY s.started_at, s.session_id
210
+ """
211
+ )
212
+ return [Kept(*row) for row in rows if _within(row[1], parents)]
213
+
214
+ def remove(self, session_ids: Iterable[str]) -> int:
215
+ """Delete these sessions and everything kept about them, and rewrite the file without their text."""
216
+ removed = 0
217
+ with self._db:
218
+ for session_id in session_ids:
219
+ for table in ("turns", "dreams", "digests"):
220
+ self._db.execute(f"DELETE FROM {table} WHERE session_id = ?", (session_id,))
221
+ removed += self._db.execute("DELETE FROM sessions WHERE session_id = ?", (session_id,)).rowcount
222
+ if removed:
223
+ # Deleted rows stay in the file, and deleted words in the search index, until both are rebuilt.
224
+ self._db.execute("INSERT INTO turns_fts(turns_fts) VALUES ('rebuild')")
225
+ self._db.commit()
226
+ self._db.execute("VACUUM")
227
+ return removed
228
+
229
+ def awaiting_dream(self, since: str | None = None) -> list[Undreamt]:
230
+ """Sessions with turns the dream has not read, oldest first.
231
+
232
+ With `since`, an ISO time in UTC, only those that ended at or after it.
233
+ """
234
+ rows = self._db.execute(
235
+ """
236
+ SELECT s.session_id, s.project, s.title, s.ended_at
237
+ FROM sessions s LEFT JOIN dreams d ON d.session_id = s.session_id
238
+ WHERE (d.turn_count IS NULL
239
+ OR d.turn_count != (SELECT count(*) FROM turns t WHERE t.session_id = s.session_id))
240
+ -- Both sides are UTC, so the first 19 characters compare whatever each writes after the seconds.
241
+ AND (:since IS NULL OR substr(s.ended_at, 1, 19) >= substr(:since, 1, 19))
242
+ ORDER BY s.ended_at
243
+ """,
244
+ {"since": since},
245
+ )
246
+ return [Undreamt(*row) for row in rows]
247
+
248
+ def record_dream(self, session_id: str) -> None:
249
+ """Note that the dream has read the session as it now stands."""
250
+ with self._db:
251
+ self._db.execute(
252
+ "INSERT OR REPLACE INTO dreams VALUES (?, ?)", (session_id, self._turn_count(session_id))
253
+ )
254
+
255
+ def keep_digest(self, session_id: str, digest: dict) -> None:
256
+ """Keep what the model made of the session as it now stands, so it is never asked twice."""
257
+ with self._db:
258
+ self._db.execute(
259
+ "INSERT OR REPLACE INTO digests VALUES (?, ?, ?)",
260
+ (session_id, self._turn_count(session_id), json.dumps(digest)),
261
+ )
262
+
263
+ def digest(self, session_id: str) -> dict | None:
264
+ """The session's digest, unless the session has grown since it was made."""
265
+ row = self._db.execute(
266
+ "SELECT body FROM digests WHERE session_id = ? AND turn_count = ?",
267
+ (session_id, self._turn_count(session_id)),
268
+ ).fetchone()
269
+ return json.loads(row[0]) if row else None
270
+
271
+ def _turn_count(self, session_id: str) -> int:
272
+ (count,) = self._db.execute("SELECT count(*) FROM turns WHERE session_id = ?", (session_id,)).fetchone()
273
+ return count
274
+
275
+ def search(
276
+ self,
277
+ query: str | Sequence[str],
278
+ *,
279
+ project: str | None,
280
+ since: str | None = None,
281
+ until: str | None = None,
282
+ tools: bool = False,
283
+ reports: bool = False,
284
+ limit: int = 10,
285
+ ) -> list[Hit]:
286
+ """Best-matching turns first. `project=None` searches every project.
287
+
288
+ A query is read as plain words. Turns holding all of them win; if none
289
+ does, turns holding any of them are returned instead. Several queries are
290
+ several phrasings of one question: each is searched and the rankings are
291
+ fused, so a turn that more phrasings find ranks higher. Among turns that
292
+ match alike, what the person typed comes before what the assistant wrote.
293
+
294
+ `since` and `until` are inclusive YYYY-MM-DD dates, compared against the
295
+ turn's UTC timestamp. Tool calls outnumber prose and would crowd it out, so
296
+ they are searched only when `tools` is set. Subagent reports are what an
297
+ agent observed, not what was said in the session, and are searched only
298
+ when `reports` is set.
299
+ """
300
+ phrasings = [query] if isinstance(query, str) else list(query)
301
+ scope = {"project": project, "since": since, "until": until, "tools": tools, "reports": reports}
302
+ fused: dict[tuple[str, int], float] = {}
303
+ best: dict[tuple[str, int], Hit] = {}
304
+ for phrasing in phrasings:
305
+ terms = [f'"{term}"' for term in re.findall(r"\w+", phrasing)]
306
+ if not terms:
307
+ continue
308
+ pool = {**scope, "limit": max(limit * 3, 30)}
309
+ hits = self._matching(" ".join(terms), pool, True) or self._matching(" OR ".join(terms), pool, False)
310
+ for rank, hit in enumerate(hits):
311
+ key = (hit.session_id, hit.seq)
312
+ # Reciprocal rank fusion: each phrasing votes by where it placed the turn.
313
+ fused[key] = fused.get(key, 0.0) + 1.0 / (_RRF_K + rank)
314
+ if key not in best or (hit.matched_all and not best[key].matched_all):
315
+ best[key] = hit
316
+ if any(hit.matched_all for hit in best.values()):
317
+ best = {key: hit for key, hit in best.items() if hit.matched_all}
318
+ return [best[key] for key in sorted(best, key=lambda key: -fused[key])][:limit]
319
+
320
+ def _matching(self, match: str, scope: dict[str, str | int | None], matched_all: bool) -> list[Hit]:
321
+ rows = self._db.execute(
322
+ """
323
+ SELECT s.session_id, t.seq, t.author, t.kind, t.timestamp,
324
+ snippet(turns_fts, 0, '', '', '…', 32), s.title, s.project
325
+ FROM turns_fts
326
+ JOIN turns t ON t.id = turns_fts.rowid
327
+ JOIN sessions s ON s.session_id = t.session_id
328
+ WHERE turns_fts MATCH :match
329
+ AND (:project IS NULL OR s.project = :project
330
+ OR substr(s.project, 1, length(:project) + 1) = :project || '/')
331
+ AND (:since IS NULL OR substr(t.timestamp, 1, 10) >= :since)
332
+ AND (:until IS NULL OR substr(t.timestamp, 1, 10) <= :until)
333
+ AND (:tools OR t.kind != 'tool')
334
+ AND (:reports OR t.kind != 'report')
335
+ -- bm25 is negative and lower is better, so a weight above 1 moves a turn up.
336
+ ORDER BY bm25(turns_fts) * CASE t.author WHEN 'human' THEN 1.3 WHEN 'assistant' THEN 1.0 ELSE 0.8 END
337
+ LIMIT :limit
338
+ """,
339
+ {"match": match, **scope},
340
+ )
341
+ return [
342
+ Hit(sid, seq, Author(author), Kind(kind), *rest, matched_all) for sid, seq, author, kind, *rest in rows
343
+ ]
344
+
345
+ def recap(self, session: str) -> Recap:
346
+ """One session in brief. `session` is a session id or any prefix that picks out a single session."""
347
+ session_id = self._one(session)
348
+ title, ended_at = self._db.execute("SELECT title, ended_at FROM sessions WHERE session_id = ?", (session_id,)).fetchone()
349
+ opening = self._db.execute(
350
+ "SELECT text FROM turns WHERE session_id = ? AND author = 'human' ORDER BY seq LIMIT 1", (session_id,)
351
+ ).fetchone()
352
+ summary = (self.digest(session_id) or {}).get("summary") or None
353
+ return Recap(session_id, title, ended_at, self._turn_count(session_id), summary, opening[0] if opening else None)
354
+
355
+ def _one(self, session: str) -> str:
356
+ matches = self._db.execute(
357
+ "SELECT session_id FROM sessions WHERE substr(session_id, 1, length(:prefix)) = :prefix", {"prefix": session}
358
+ ).fetchall()
359
+ if not matches:
360
+ raise LookupError(f"no archived session starts with {session!r}")
361
+ if len(matches) > 1:
362
+ raise LookupError(f"{len(matches)} archived sessions start with {session!r}; give more of the id")
363
+ ((session_id,),) = matches
364
+ return session_id
365
+
366
+ def show(self, session: str, *, first: int = 0, last: int | None = None) -> list[Turn]:
367
+ """One session's turns as archived, from `first` to `last` inclusive.
368
+
369
+ `session` is a session id or any prefix of one that picks out a single session.
370
+ """
371
+ session_id = self._one(session)
372
+ rows = self._db.execute(
373
+ """
374
+ SELECT seq, author, kind, text, uuid, timestamp FROM turns
375
+ WHERE session_id = :session AND seq >= :first AND (:last IS NULL OR seq <= :last)
376
+ ORDER BY seq
377
+ """,
378
+ {"session": session_id, "first": first, "last": last},
379
+ )
380
+ return [Turn(seq, Author(author), Kind(kind), *rest) for seq, author, kind, *rest in rows]
381
+
382
+ def _store(self, session: Session, project: str, source: tuple[int, int]) -> int:
383
+ """Replace the session's archived copy; returns how many secrets were redacted from it."""
384
+ cleaned = [(turn, *redact(turn.text)) for turn in session.turns]
385
+ with self._db:
386
+ self._db.execute("DELETE FROM turns WHERE session_id = ?", (session.session_id,))
387
+ self._db.execute(
388
+ "INSERT OR REPLACE INTO sessions VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
389
+ (
390
+ session.session_id,
391
+ project,
392
+ session.title,
393
+ session.cwd,
394
+ session.git_branch,
395
+ session.entrypoint,
396
+ session.started_at,
397
+ session.ended_at,
398
+ *source,
399
+ RULES_VERSION,
400
+ ),
401
+ )
402
+ self._db.executemany(
403
+ "INSERT INTO turns (session_id, seq, author, kind, text, uuid, timestamp) VALUES (?, ?, ?, ?, ?, ?, ?)",
404
+ [(session.session_id, t.seq, t.author, t.kind, text, t.uuid, t.timestamp) for t, text, _ in cleaned],
405
+ )
406
+ return sum(count for _, _, count in cleaned)
dream/claims.py ADDED
@@ -0,0 +1,102 @@
1
+ """What the dream believes: claims heard in sessions, and the entries memory holds."""
2
+
3
+ from dataclasses import asdict, dataclass
4
+ from enum import StrEnum
5
+
6
+
7
+ class ClaimType(StrEnum):
8
+ DECISION = "decision"
9
+ PREFERENCE = "preference"
10
+ FACT = "fact"
11
+ LESSON = "lesson"
12
+
13
+
14
+ class Scope(StrEnum):
15
+ PROJECT = "project"
16
+ GLOBAL = "global"
17
+
18
+
19
+ class Provenance(StrEnum):
20
+ """Where a claim came from, strongest first."""
21
+
22
+ HUMAN = "human"
23
+ ACCEPTED = "accepted"
24
+ INFERRED = "inferred"
25
+ OBSERVED = "observed"
26
+
27
+
28
+ class Status(StrEnum):
29
+ ACTIVE = "active"
30
+ CONTESTED = "contested"
31
+ """Withheld from sessions until a person rules on a disagreement."""
32
+ STALE = "stale"
33
+ """Withheld because what it is about could not be found, until a person rules."""
34
+ RETIRED = "retired"
35
+ """Taken out of sessions on the person's ruling. Kept on file."""
36
+
37
+
38
+ @dataclass(frozen=True)
39
+ class Evidence:
40
+ """The turns of one session that a claim rests on."""
41
+
42
+ session_id: str
43
+ first_turn: int
44
+ last_turn: int
45
+
46
+ @property
47
+ def command(self) -> str:
48
+ """The command that prints these turns."""
49
+ return f"dream show {self.session_id[:8]} --first {self.first_turn} --last {self.last_turn}"
50
+
51
+
52
+ @dataclass(frozen=True)
53
+ class Claim:
54
+ slot: str
55
+ """What the claim is about. Two claims with one slot are about the same thing."""
56
+ type: ClaimType
57
+ scope: Scope
58
+ statement: str
59
+ why: str
60
+ provenance: Provenance
61
+ evidence: Evidence
62
+ said_at: str
63
+ anchor: str | None = None
64
+ """For a fact about the repository: the path it can be checked against."""
65
+ asks: str = ""
66
+ """A question a later session might have that this claim answers. The gate asks it of the index."""
67
+ topic: str = ""
68
+ """The area the claim belongs to. A long index is grouped by topic."""
69
+
70
+
71
+ @dataclass(frozen=True)
72
+ class Entry:
73
+ """What memory currently holds for one slot."""
74
+
75
+ slot: str
76
+ statement: str
77
+ status: Status = Status.ACTIVE
78
+ type: ClaimType | None = None
79
+ provenance: Provenance | None = None
80
+ """None for a memory written before remcycle kept provenance."""
81
+ evidence: tuple[Evidence, ...] = ()
82
+ said_at: str | None = None
83
+ why: str = ""
84
+ anchor: str | None = None
85
+ aliases: tuple[str, ...] = ()
86
+ """Other names sessions have used for this slot."""
87
+
88
+
89
+ def claim_to_json(claim: Claim) -> dict:
90
+ return asdict(claim)
91
+
92
+
93
+ def claim_from_json(data: dict) -> Claim:
94
+ return Claim(
95
+ **{
96
+ **data,
97
+ "type": ClaimType(data["type"]),
98
+ "scope": Scope(data["scope"]),
99
+ "provenance": Provenance(data["provenance"]),
100
+ "evidence": Evidence(**data["evidence"]),
101
+ }
102
+ )