memdebug 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,731 @@
1
+ """Read-only adapter for agent memory kept as markdown files in a git repository.
2
+
3
+ One markdown file is one memory; its id is its path relative to the repository top.
4
+
5
+ * live memories -> the files in the working tree (what the agent actually reads)
6
+ * history -> `git log` of the checked-out branch, first-parent, newest last
7
+ * "outside the API" -> an uncommitted edit, addition or deletion in the working tree
8
+
9
+ The repository is UNTRUSTED. A repo carries its own config, which can make git run programs,
10
+ and its tree can hold symlinks aimed at your secrets. So:
11
+
12
+ * git runs with a scrubbed environment and the repo's risky settings overridden on the command
13
+ line (no fsmonitor, no hooks, no signature checking, no pager, no external diff, no textconv,
14
+ no replace-refs, no global or system config);
15
+ * only read-only git commands are used, with fixed argument lists and no shell;
16
+ * every git process has a wall-clock limit, an output cap and a stderr cap;
17
+ * history is read from git objects by id, never from files on disk;
18
+ * the working tree is read with symlinks refused and non-regular files skipped;
19
+ * paths with control characters, quotes, backslashes, `..` or `.git` parts are skipped.
20
+
21
+ This adapter never writes to the repository. (Restoring is done by restore.py, which is a separate, explicit step.)
22
+ """
23
+ from __future__ import annotations
24
+
25
+ import hashlib
26
+ import os
27
+ import re
28
+ import signal
29
+ import stat
30
+ import subprocess
31
+ import tempfile
32
+ import threading
33
+ from contextlib import ExitStack, contextmanager
34
+ from dataclasses import dataclass
35
+ from datetime import datetime, timedelta, timezone
36
+ from pathlib import Path
37
+ from typing import IO, Callable, Iterator
38
+
39
+ from ..errors import AdapterError
40
+ from ..models import MAX_ID_CHARS, Memory, MemoryEvent, Op, Source
41
+ from ..textsafe import bound_text, has_unsafe_chars, safe_text
42
+ from .base import HistoryRead, LiveMemories
43
+ from .common import Warnings, clean_id, parse_ts
44
+
45
+ MAX_FILE_BYTES = 1_048_576
46
+ MAX_WALK_DEPTH = 32
47
+ _MAX_LINE = 65_536
48
+ _MAX_LOG_BYTES = 256 * 1024 * 1024
49
+ _SMALL_OUTPUT = 65_536
50
+ _SHA = r"(?:[0-9a-f]{40}|[0-9a-f]{64})"
51
+ _SHA_RE = re.compile(rf"^{_SHA}\Z")
52
+ _RAW_RE = re.compile(rf"^:(\d{{6}}) (\d{{6}}) ({_SHA}) ({_SHA}) ([A-Z])\t(.+)\Z")
53
+ _REGULAR_MODES = {"100644", "100755"}
54
+ _STATUS_OPS = {"A": Op.ADD, "M": Op.UPDATE, "D": Op.DELETE}
55
+ _MIN_GIT = (2, 31) # --diff-merges=first-parent
56
+ _NO_WINDOW = getattr(subprocess, "CREATE_NO_WINDOW", 0)
57
+
58
+
59
+ def find_git() -> str | None:
60
+ """Find git on PATH, ignoring the current folder. Windows (and an empty or '.' PATH entry on
61
+ POSIX) would otherwise run a git.exe planted in whatever folder you happen to be in."""
62
+ names = ["git.exe"] if os.name == "nt" else ["git"]
63
+ for directory in os.environ.get("PATH", "").split(os.pathsep):
64
+ directory = directory.strip().strip('"')
65
+ if not directory or not os.path.isabs(directory):
66
+ continue
67
+ for name in names:
68
+ candidate = os.path.join(directory, name)
69
+ if os.path.isfile(candidate) and os.access(candidate, os.X_OK):
70
+ return candidate
71
+ return None
72
+
73
+
74
+ def _too_large(size: int) -> str:
75
+ return f"[file too large to read: {size} bytes]"
76
+
77
+
78
+ def _text_from_bytes(data: bytes) -> str:
79
+ # Line endings are normalised so a checkout with CRLF does not look like an edit.
80
+ return bound_text(data.decode("utf-8", "replace").replace("\r\n", "\n"))
81
+
82
+
83
+ _RESERVED_NAMES = {"con", "prn", "aux", "nul", "conin$", "conout$"} | {
84
+ f"{base}{n}" for base in ("com", "lpt") for n in range(1, 10)
85
+ }
86
+ _FORBIDDEN_CHARS = set('"\\:*?<>|') # quotes git adds, NTFS streams and drive letters, wildcards
87
+
88
+
89
+ def _bad_component(part: str) -> bool:
90
+ """A path component that is dangerous on some platform. Applied everywhere, so a repository
91
+ looks the same from every operating system."""
92
+ if part in ("", ".", ".."):
93
+ return True
94
+ folded = part.casefold()
95
+ if folded != folded.rstrip(" ."): # Windows silently drops trailing dots and spaces
96
+ return True
97
+ if folded == ".git" or folded.startswith("git~"): # the git folder, and its NTFS short name
98
+ return True
99
+ return folded.split(".")[0] in _RESERVED_NAMES # CON, NUL.md, com1.txt are devices on Windows
100
+
101
+
102
+ def _is_reparse_point(info: os.stat_result) -> bool:
103
+ """Windows junctions and other reparse points; they are not reported as symlinks."""
104
+ return bool(getattr(info, "st_file_attributes", 0) & 0x400)
105
+
106
+
107
+ def _valid_relpath(path: str, suffixes: tuple[str, ...]) -> str | None:
108
+ """A repository-relative path that is safe to show, store and compare; else None.
109
+ git quotes names with a double quote or backslash, so refusing them everywhere keeps the
110
+ history and the working-tree listing in agreement."""
111
+ if not path or len(path) > MAX_ID_CHARS or path[0] == "/" or has_unsafe_chars(path):
112
+ return None
113
+ if any(ch in _FORBIDDEN_CHARS for ch in path):
114
+ return None
115
+ if any(_bad_component(part) for part in path.split("/")):
116
+ return None
117
+ if not path.lower().endswith(suffixes):
118
+ return None
119
+ return path
120
+
121
+
122
+ def _clean_env() -> dict[str, str]:
123
+ env = {
124
+ "GIT_CONFIG_NOSYSTEM": "1",
125
+ "GIT_CONFIG_GLOBAL": os.devnull,
126
+ "GIT_TERMINAL_PROMPT": "0",
127
+ "GIT_OPTIONAL_LOCKS": "0",
128
+ "GIT_NO_REPLACE_OBJECTS": "1",
129
+ "GIT_ASKPASS": "false",
130
+ "LC_ALL": "C",
131
+ }
132
+ for key in ("PATH", "SYSTEMROOT", "TEMP", "TMP"):
133
+ if key in os.environ:
134
+ env[key] = os.environ[key]
135
+ return env
136
+
137
+
138
+ def _spawn_flags() -> dict:
139
+ """Own process group on POSIX, no console window on Windows."""
140
+ if os.name == "posix":
141
+ return {"start_new_session": True}
142
+ return {"creationflags": _NO_WINDOW}
143
+
144
+
145
+ class _Run:
146
+ """One git child process with a hard time limit."""
147
+
148
+ def __init__(self, proc: subprocess.Popen, errfile, timeout: float):
149
+ self.proc = proc
150
+ self._errfile = errfile
151
+ self.timed_out = False
152
+ self.stopped = False # we ended it on purpose (a limit was reached)
153
+ self._timer = threading.Timer(timeout, self._expire)
154
+ self._timer.daemon = True
155
+ self._timer.start()
156
+
157
+ @property
158
+ def out(self) -> IO[bytes]:
159
+ """The child's output pipe. It is always started with one; this makes that checkable and typed."""
160
+ stream = self.proc.stdout
161
+ if stream is None:
162
+ raise AdapterError("git has no output stream")
163
+ return stream
164
+
165
+ @property
166
+ def inp(self) -> IO[bytes]:
167
+ stream = self.proc.stdin
168
+ if stream is None:
169
+ raise AdapterError("git was started without an input stream")
170
+ return stream
171
+
172
+ def _expire(self) -> None:
173
+ self.timed_out = True
174
+ self.stop()
175
+
176
+ def stop(self) -> None:
177
+ """Kill git and anything it started (a helper holding the pipe would block us)."""
178
+ self.stopped = True
179
+ if self.proc.returncode is not None:
180
+ return
181
+ try:
182
+ if os.name == "posix":
183
+ os.killpg(self.proc.pid, signal.SIGKILL)
184
+ else:
185
+ # taskkill /T ends the whole tree. Called by absolute path, never via PATH.
186
+ taskkill = os.path.join(os.environ.get("SystemRoot", r"C:\Windows"), "System32", "taskkill.exe")
187
+ try:
188
+ subprocess.run([taskkill, "/F", "/T", "/PID", str(self.proc.pid)],
189
+ capture_output=True, timeout=10, creationflags=_NO_WINDOW)
190
+ except (OSError, subprocess.SubprocessError):
191
+ pass
192
+ self.proc.kill()
193
+ except OSError:
194
+ pass
195
+
196
+ def stderr_text(self) -> str:
197
+ try:
198
+ self._errfile.seek(0)
199
+ return self._errfile.read(2048).decode("utf-8", "replace")
200
+ except (OSError, ValueError):
201
+ return ""
202
+
203
+ def close(self) -> None:
204
+ self._timer.cancel()
205
+ if self.proc.returncode is None:
206
+ self.stop()
207
+ for pipe in (self.proc.stdin, self.proc.stdout):
208
+ try:
209
+ if pipe:
210
+ pipe.close()
211
+ except OSError:
212
+ pass
213
+ self.proc.wait()
214
+ self._errfile.close()
215
+
216
+
217
+ class _Git:
218
+ def __init__(self, git_path: str, root: Path, timeout: float):
219
+ self.git_path = git_path
220
+ self.root = root
221
+ self.timeout = timeout
222
+
223
+ def _command(self, args: list[str]) -> list[str]:
224
+ null = os.devnull
225
+ return [
226
+ self.git_path, "--no-pager", "--literal-pathspecs",
227
+ "-c", "core.quotepath=false", "-c", "core.fsmonitor=false",
228
+ "-c", f"core.hooksPath={null}", "-c", f"core.attributesFile={null}",
229
+ "-c", "log.showSignature=false", "-c", "gpg.program=false",
230
+ *args,
231
+ ]
232
+
233
+ @contextmanager
234
+ def spawn(self, args: list[str], *, interactive: bool = False, plain: bool = False,
235
+ env: dict[str, str] | None = None) -> Iterator[_Run]:
236
+ errfile = tempfile.TemporaryFile()
237
+ command = [self.git_path, *args] if plain else self._command(args)
238
+ try:
239
+ proc = subprocess.Popen(
240
+ command,
241
+ stdin=subprocess.PIPE if interactive else subprocess.DEVNULL,
242
+ stdout=subprocess.PIPE, stderr=errfile,
243
+ env={**_clean_env(), **(env or {})}, cwd=str(self.root), shell=False,
244
+ **_spawn_flags(),
245
+ )
246
+ except OSError as exc:
247
+ errfile.close()
248
+ raise AdapterError(f"cannot run git: {exc.strerror}") from exc
249
+ run = _Run(proc, errfile, self.timeout)
250
+ try:
251
+ yield run
252
+ finally:
253
+ run.close()
254
+
255
+ def run_small(self, args: list[str], *, plain: bool = False) -> tuple[int, bytes]:
256
+ with self.spawn(args, plain=plain) as run:
257
+ data = run.out.read(_SMALL_OUTPUT + 1)
258
+ if len(data) > _SMALL_OUTPUT:
259
+ run.stop()
260
+ raise AdapterError("git produced unexpectedly large output")
261
+ code = run.proc.wait()
262
+ if run.timed_out:
263
+ raise AdapterError(f"git timed out after {self.timeout:g} seconds")
264
+ if code != 0:
265
+ return code, run.stderr_text().encode()
266
+ return code, data
267
+
268
+ def finish(self, run: _Run) -> None:
269
+ """Call after reading a stream to the end; raises if git failed or timed out."""
270
+ code = run.proc.wait()
271
+ if run.timed_out:
272
+ raise AdapterError(f"git timed out after {self.timeout:g} seconds")
273
+ if code != 0 and not run.stopped:
274
+ raise AdapterError(f"git failed: {safe_text(run.stderr_text(), 200)}")
275
+
276
+
277
+ def _read_lines(stream, run: _Run) -> Iterator[bytes | None]:
278
+ """Lines from git, with bounded line length and total size. None marks an overlong line."""
279
+ total = 0
280
+ while True:
281
+ chunk = stream.readline(_MAX_LINE)
282
+ if not chunk:
283
+ return
284
+ total += len(chunk)
285
+ if total > _MAX_LOG_BYTES:
286
+ run.stop()
287
+ return
288
+ if len(chunk) == _MAX_LINE and not chunk.endswith(b"\n"):
289
+ while True: # drain the rest of the overlong line
290
+ more = stream.readline(_MAX_LINE)
291
+ total += len(more)
292
+ if not more or more.endswith(b"\n"):
293
+ break
294
+ yield None
295
+ continue
296
+ yield chunk.rstrip(b"\n")
297
+
298
+
299
+ class _BlobReader:
300
+ """Reads blobs by object id through two long-lived `git cat-file` processes."""
301
+
302
+ def __init__(self, git: _Git):
303
+ self._git = git
304
+ self._cache: dict[str, str | None] = {}
305
+ self._stack = ExitStack()
306
+
307
+ def __enter__(self) -> "_BlobReader":
308
+ self._check = self._stack.enter_context(self._git.spawn(["cat-file", "--batch-check"], interactive=True))
309
+ self._batch = self._stack.enter_context(self._git.spawn(["cat-file", "--batch"], interactive=True))
310
+ return self
311
+
312
+ def __exit__(self, *exc) -> None:
313
+ self._stack.close()
314
+
315
+ @property
316
+ def timed_out(self) -> bool:
317
+ return self._check.timed_out or self._batch.timed_out
318
+
319
+ @staticmethod
320
+ def _ask(run: _Run, sha: str) -> bytes:
321
+ run.inp.write(f"{sha}\n".encode()) # sha was validated as pure hex
322
+ run.inp.flush()
323
+ return run.out.readline(256)
324
+
325
+ def raw(self, sha: str) -> bytes | None:
326
+ """The exact bytes of a blob, or None if it cannot be read or is larger than MAX_FILE_BYTES."""
327
+ if not (_SHA_RE.match(sha) and set(sha) != {"0"}):
328
+ return None
329
+ try:
330
+ parts = self._ask(self._check, sha).decode("ascii", "replace").split()
331
+ if len(parts) == 3 and parts[1] == "blob" and parts[2].isdigit():
332
+ size = int(parts[2])
333
+ if size > MAX_FILE_BYTES:
334
+ return None
335
+ header = self._ask(self._batch, sha).decode("ascii", "replace").split()
336
+ if len(header) == 3 and header[1] == "blob" and header[2] == str(size):
337
+ data = self._batch.out.read(size)
338
+ if len(data) == size and self._batch.out.read(1) == b"\n":
339
+ return data
340
+ except (OSError, ValueError):
341
+ return None
342
+ return None
343
+
344
+ def lookup(self, commit: str, path: str) -> str | None:
345
+ """The object id of the file `path` as it was in `commit`, or None if it did not exist there."""
346
+ if not _SHA_RE.match(commit) or not path or any(ch in path for ch in "\n\r\0"):
347
+ return None
348
+ try:
349
+ parts = self._ask(self._check, f"{commit}:{path}").decode("utf-8", "replace").split()
350
+ except (OSError, ValueError):
351
+ return None
352
+ return parts[0] if len(parts) == 3 and parts[1] == "blob" and _SHA_RE.match(parts[0]) else None
353
+
354
+ def text(self, sha: str) -> str | None:
355
+ """Text of the blob, a size marker if it is too big, or None if it cannot be read."""
356
+ if sha in self._cache:
357
+ return self._cache[sha]
358
+ result: str | None = None
359
+ if _SHA_RE.match(sha) and set(sha) != {"0"}:
360
+ try:
361
+ parts = self._ask(self._check, sha).decode("ascii", "replace").split()
362
+ if len(parts) == 3 and parts[1] == "blob" and parts[2].isdigit():
363
+ size = int(parts[2])
364
+ if size > MAX_FILE_BYTES:
365
+ result = _too_large(size)
366
+ else:
367
+ header = self._ask(self._batch, sha).decode("ascii", "replace").split()
368
+ if len(header) == 3 and header[1] == "blob" and header[2] == str(size):
369
+ data = self._batch.out.read(size)
370
+ if len(data) == size and self._batch.out.read(1) == b"\n":
371
+ result = _text_from_bytes(data)
372
+ except (OSError, ValueError):
373
+ result = None
374
+ self._cache[sha] = result
375
+ return result
376
+
377
+
378
+ @dataclass(frozen=True)
379
+ class _Change:
380
+ commit: str
381
+ ts: datetime
382
+ observed: bool
383
+ author: str | None
384
+ path: str
385
+ status: str
386
+ old_sha: str
387
+ new_sha: str
388
+
389
+ @property
390
+ def ref(self) -> str:
391
+ return hashlib.sha256(f"{self.commit}\x00{self.path}".encode()).hexdigest()
392
+
393
+
394
+ class MarkdownGitAdapter:
395
+ name = "markdown-git"
396
+ capabilities = {"history", "global_feed"}
397
+
398
+ def __init__(
399
+ self,
400
+ root: str | Path,
401
+ *,
402
+ store: str | None = None,
403
+ subdir: str | None = None,
404
+ suffixes: tuple[str, ...] = (".md",),
405
+ max_files: int = 20_000,
406
+ git_timeout: float = 120.0,
407
+ max_total_chars: int = 100_000_000,
408
+ git_path: str | None = None,
409
+ clock: Callable[[], datetime] | None = None,
410
+ ):
411
+ try:
412
+ self._root = Path(root).resolve(strict=True)
413
+ except OSError as exc:
414
+ raise AdapterError(f"cannot access repository path: {exc.strerror}") from exc
415
+ if not self._root.is_dir():
416
+ raise AdapterError("repository path is not a directory")
417
+ if not (isinstance(max_files, int) and max_files > 0 and git_timeout > 0 and max_total_chars > 0):
418
+ raise AdapterError("max_files, git_timeout and max_total_chars must be positive")
419
+ if not suffixes or any(not s.startswith(".") or s != s.lower() for s in suffixes):
420
+ raise AdapterError("suffixes must be lowercase and start with a dot, such as '.md'")
421
+ self._suffixes = tuple(suffixes)
422
+ self._store = clean_id(store) or clean_id(self._root.name) or "memory"
423
+ self._max_files = max_files
424
+ self._max_total_chars = max_total_chars
425
+ self._clock = clock or (lambda: datetime.now(timezone.utc))
426
+
427
+ found = git_path or find_git()
428
+ if not found:
429
+ raise AdapterError("git was not found; install git 2.31 or newer")
430
+ if not os.path.isabs(found) or not os.path.isfile(found):
431
+ raise AdapterError("the git path must be an absolute path to an executable file")
432
+ if os.name == "nt" and not found.lower().endswith(".exe"):
433
+ raise AdapterError("on Windows the git path must be a real git.exe, not a script or shim")
434
+ self._git = _Git(found, self._root, git_timeout)
435
+ self._check_git_version()
436
+ self._subdir = self._validate_subdir(subdir)
437
+ self._check_repository()
438
+
439
+ @property
440
+ def store(self) -> str:
441
+ return self._store
442
+
443
+ # Read-only views for restore.py, which lives beside this adapter and shares its hardening.
444
+ @property
445
+ def root(self) -> Path:
446
+ return self._root
447
+
448
+ @property
449
+ def git(self) -> "_Git":
450
+ return self._git
451
+
452
+ @property
453
+ def subdir(self) -> str | None:
454
+ return self._subdir
455
+
456
+ @property
457
+ def suffixes(self) -> tuple[str, ...]:
458
+ return self._suffixes
459
+
460
+ # -- setup checks -------------------------------------------------------------------------
461
+
462
+ def _check_git_version(self) -> None:
463
+ _, out = self._git.run_small(["--version"], plain=True)
464
+ match = re.search(rb"git version (\d+)\.(\d+)", out)
465
+ if not match or (int(match.group(1)), int(match.group(2))) < _MIN_GIT:
466
+ raise AdapterError("git 2.31 or newer is required")
467
+
468
+ def _validate_subdir(self, subdir: str | None) -> str | None:
469
+ if subdir is None:
470
+ return None
471
+ value = subdir.strip("/")
472
+ parts = value.split("/")
473
+ if (not value or len(value) > MAX_ID_CHARS or has_unsafe_chars(value)
474
+ or any(ch in _FORBIDDEN_CHARS for ch in value) or any(_bad_component(p) for p in parts)):
475
+ raise AdapterError("subdir must be a plain relative path inside the repository")
476
+ target = self._root.joinpath(*parts)
477
+ if target.is_symlink() or not target.is_dir():
478
+ raise AdapterError("subdir does not exist or is not a plain directory")
479
+ return value
480
+
481
+ def _check_repository(self) -> None:
482
+ code, out = self._git.run_small(["rev-parse", "--show-toplevel"])
483
+ if code != 0:
484
+ detail = safe_text(out.decode("utf-8", "replace"), 200)
485
+ hint = " (git refuses repositories owned by another user)" if "dubious" in detail else ""
486
+ raise AdapterError(f"not a usable git repository: {detail}{hint}")
487
+ try:
488
+ top = Path(out.decode("utf-8", "replace").strip()).resolve()
489
+ except OSError as exc:
490
+ raise AdapterError("cannot resolve the repository top") from exc
491
+ if top != self._root:
492
+ raise AdapterError("the path must be the top of the git repository, not a folder inside it")
493
+
494
+ # -- history ------------------------------------------------------------------------------
495
+
496
+ def read_history(self, max_rows: int) -> HistoryRead:
497
+ if not (isinstance(max_rows, int) and max_rows > 0):
498
+ raise AdapterError("max_rows must be a positive integer")
499
+ warnings = Warnings()
500
+ now = self._clock()
501
+ code, _ = self._git.run_small(["rev-parse", "--verify", "-q", "HEAD"])
502
+ if code != 0: # a repository with no commits yet
503
+ return HistoryRead(events=[], refs=set(), truncated=False)
504
+
505
+ args = [
506
+ "log", "--topo-order", "--reverse", "--first-parent", "--diff-merges=first-parent",
507
+ "--no-renames", "--raw", "--no-abbrev", "--no-color", "--no-ext-diff", "--no-textconv",
508
+ "--format=%x01%H%x02%cI%x02%an", "HEAD", "--",
509
+ ]
510
+ if self._subdir:
511
+ args.append(self._subdir)
512
+
513
+ changes: list[_Change] = []
514
+ refs: set[str] = set()
515
+ truncated = False
516
+ current: tuple[str, datetime, bool, str | None] | None = None
517
+ with self._git.spawn(args) as run:
518
+ for line in _read_lines(run.out, run):
519
+ if line is None:
520
+ warnings.add("git log produced an overlong line; skipped")
521
+ continue
522
+ text = line.decode("utf-8", "replace")
523
+ if text.startswith("\x01"):
524
+ current = self._parse_header(text, now, warnings)
525
+ elif text.startswith(":"):
526
+ if current is None:
527
+ continue
528
+ change = self._parse_change(text, current, warnings)
529
+ if change is None:
530
+ continue
531
+ if len(changes) >= max_rows:
532
+ truncated = True
533
+ run.stop()
534
+ break
535
+ changes.append(change)
536
+ refs.add(change.ref)
537
+ elif text:
538
+ warnings.add("unexpected output from git log; line skipped")
539
+ if run.timed_out:
540
+ raise AdapterError(f"git timed out after {self._git.timeout:g} seconds")
541
+ if not truncated:
542
+ self._git.finish(run)
543
+
544
+ events, skipped, over_budget = self._events_from(changes, warnings)
545
+ if over_budget:
546
+ truncated = True
547
+ if truncated:
548
+ warnings.add("history was only partly read (row limit or memory budget reached)")
549
+ return HistoryRead(events=events, refs=refs, truncated=truncated, skipped=skipped, warnings=warnings.as_list())
550
+
551
+ def _parse_header(self, text: str, now: datetime, warnings: Warnings):
552
+ parts = text[1:].split("\x02")
553
+ if len(parts) != 3 or not _SHA_RE.match(parts[0]):
554
+ warnings.add("unreadable commit header; its changes were skipped")
555
+ return None
556
+ ts, observed = parse_ts(parts[1]), False
557
+ if ts is None:
558
+ ts, observed = now, True
559
+ warnings.add(f"commit {parts[0][:12]}: no usable date; using the time it was read")
560
+ elif ts > now + timedelta(days=1):
561
+ warnings.add(f"commit {parts[0][:12]}: date is in the future")
562
+ return (parts[0], ts, observed, clean_id(parts[2]))
563
+
564
+ def _parse_change(self, text: str, header, warnings: Warnings) -> _Change | None:
565
+ match = _RAW_RE.match(text)
566
+ if not match:
567
+ warnings.add("unreadable change line from git; skipped")
568
+ return None
569
+ src_mode, dst_mode, old_sha, new_sha, status, path = match.groups()
570
+ quoted = path.startswith('"')
571
+ if not path.rstrip('"').lower().endswith(self._suffixes):
572
+ return None # not a memory file at all; nothing to say
573
+ relpath = None if quoted else _valid_relpath(path, self._suffixes)
574
+ if relpath is None:
575
+ warnings.add(f"path with unsafe or unsupported characters skipped: {safe_text(path, 60)}")
576
+ return None
577
+ if status not in _STATUS_OPS:
578
+ warnings.add(f"unsupported change type {status} for {safe_text(relpath, 60)}; skipped")
579
+ return None
580
+ modes = {"A": [dst_mode], "D": [src_mode], "M": [src_mode, dst_mode]}[status]
581
+ if any(m not in _REGULAR_MODES for m in modes):
582
+ warnings.add(f"{safe_text(relpath, 60)} is a symlink or submodule; not followed")
583
+ return None
584
+ commit, ts, observed, author = header
585
+ return _Change(commit, ts, observed, author, relpath, status, old_sha, new_sha)
586
+
587
+ def _events_from(self, changes: list[_Change], warnings: Warnings):
588
+ events: list[MemoryEvent] = []
589
+ skipped = 0
590
+ total = 0
591
+ over_budget = False
592
+ scope = {"store": self._store}
593
+ with _BlobReader(self._git) as blobs:
594
+ for change in changes:
595
+ before = after = None
596
+ if change.status in ("M", "D"):
597
+ before = blobs.text(change.old_sha)
598
+ if before is None:
599
+ skipped += 1
600
+ warnings.add(f"{safe_text(change.path, 60)}: earlier version is not available; skipped")
601
+ continue
602
+ if change.status in ("A", "M"):
603
+ after = blobs.text(change.new_sha)
604
+ if after is None:
605
+ skipped += 1
606
+ warnings.add(f"{safe_text(change.path, 60)}: version is not available; skipped")
607
+ continue
608
+ if change.status == "M" and before == after:
609
+ continue # only the file mode changed
610
+ total += len(before or "") + len(after or "")
611
+ if total > self._max_total_chars:
612
+ over_budget = True
613
+ break
614
+ try:
615
+ events.append(MemoryEvent(
616
+ backend=self.name, memory_id=change.path, op=_STATUS_OPS[change.status],
617
+ ts=change.ts, ts_observed=change.observed, backend_ref=change.ref, scope=scope,
618
+ before=before, after=after,
619
+ source=Source(actor_id=change.author) if change.author else None,
620
+ ))
621
+ except ValueError:
622
+ skipped += 1
623
+ warnings.add(f"{safe_text(change.path, 60)}: failed validation; skipped")
624
+ if blobs.timed_out:
625
+ raise AdapterError(f"git timed out after {self._git.timeout:g} seconds")
626
+ return events, skipped, over_budget
627
+
628
+ # -- live listing (the working tree) -----------------------------------------------------------------
629
+
630
+ def list_memories(self, scope: dict[str, str]) -> LiveMemories:
631
+ if scope != {"store": self._store}:
632
+ raise AdapterError(f"scope must be {{'store': {safe_text(self._store, 40)!r}}} for this repository")
633
+ warnings = Warnings()
634
+ memories: list[Memory] = []
635
+ state = {"complete": True}
636
+
637
+ def walk_error(exc: OSError) -> None:
638
+ state["complete"] = False # an unreadable folder could hide memories
639
+ warnings.add(f"cannot read a folder: {exc.strerror}")
640
+
641
+ top = self._root.joinpath(*self._subdir.split("/")) if self._subdir else self._root
642
+ count = 0
643
+ stop = False
644
+ for dirpath, dirnames, filenames in os.walk(top, topdown=True, followlinks=False, onerror=walk_error):
645
+ kept = []
646
+ for d in sorted(dirnames):
647
+ if d == ".git": # the real git folder; look-alikes are reported below
648
+ continue
649
+ if _bad_component(d):
650
+ state["complete"] = False
651
+ warnings.add(f"folder with an unsafe name skipped: {safe_text(d, 60)}")
652
+ continue
653
+ try:
654
+ folder_info = os.lstat(os.path.join(dirpath, d))
655
+ except OSError as exc:
656
+ state["complete"] = False
657
+ warnings.add(f"cannot inspect a folder: {exc.strerror}")
658
+ continue
659
+ if stat.S_ISLNK(folder_info.st_mode) or _is_reparse_point(folder_info):
660
+ state["complete"] = False
661
+ warnings.add(f"folder {safe_text(d, 60)} is a link or junction; not followed")
662
+ continue
663
+ kept.append(d)
664
+ dirnames[:] = kept
665
+ depth = len(Path(dirpath).relative_to(self._root).parts)
666
+ if depth >= MAX_WALK_DEPTH:
667
+ if dirnames:
668
+ state["complete"] = False
669
+ warnings.add(f"folders deeper than {MAX_WALK_DEPTH} levels were not read")
670
+ dirnames[:] = []
671
+ for name in sorted(filenames):
672
+ if not name.lower().endswith(self._suffixes):
673
+ continue
674
+ full = os.path.join(dirpath, name)
675
+ relative = Path(full).relative_to(self._root).as_posix()
676
+ relpath = _valid_relpath(relative, self._suffixes)
677
+ if relpath is None:
678
+ state["complete"] = False
679
+ warnings.add(f"path with unsafe characters skipped: {safe_text(relative, 60)}")
680
+ continue
681
+ count += 1
682
+ if count > self._max_files:
683
+ state["complete"] = False
684
+ warnings.add(f"more than {self._max_files} memory files; the rest were not read")
685
+ stop = True
686
+ break
687
+ text = self._read_working_file(full, relpath, warnings)
688
+ if text is None:
689
+ state["complete"] = False
690
+ continue
691
+ memories.append(Memory(id=relpath, text=text, scope={"store": self._store}))
692
+ if stop:
693
+ break
694
+ return LiveMemories(memories=memories, complete=state["complete"], warnings=warnings.as_list())
695
+
696
+ @staticmethod
697
+ def _read_working_file(full: str, relpath: str, warnings: Warnings) -> str | None:
698
+ label = safe_text(relpath, 60)
699
+ try:
700
+ info = os.lstat(full)
701
+ if stat.S_ISLNK(info.st_mode):
702
+ warnings.add(f"{label} is a symlink; not followed")
703
+ return None
704
+ if not stat.S_ISREG(info.st_mode) or _is_reparse_point(info):
705
+ warnings.add(f"{label} is not a plain regular file; skipped")
706
+ return None
707
+ flags = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_NONBLOCK", 0)
708
+ fd = os.open(full, flags)
709
+ except OSError as exc:
710
+ warnings.add(f"{label} could not be opened: {exc.strerror}")
711
+ return None
712
+ try:
713
+ fstat = os.fstat(fd)
714
+ if not stat.S_ISREG(fstat.st_mode):
715
+ warnings.add(f"{label} is not a regular file; skipped")
716
+ return None
717
+ with os.fdopen(fd, "rb", closefd=False) as handle:
718
+ data = handle.read(MAX_FILE_BYTES + 1)
719
+ except OSError as exc:
720
+ warnings.add(f"{label} could not be read: {exc.strerror}")
721
+ return None
722
+ finally:
723
+ os.close(fd)
724
+ if len(data) > MAX_FILE_BYTES:
725
+ return _too_large(fstat.st_size)
726
+ return _text_from_bytes(data)
727
+
728
+ def history(self, memory_id: str) -> list[MemoryEvent]:
729
+ if clean_id(memory_id) is None:
730
+ raise AdapterError(f"memory_id must be a non-empty string of at most {MAX_ID_CHARS} characters")
731
+ return [e for e in self.read_history(100_000).events if e.memory_id == memory_id]