memdebug 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
memdebug/stores.py ADDED
@@ -0,0 +1,276 @@
1
+ """The stores memdebug watches: a small settings file, finding likely stores, and opening them.
2
+
3
+ The settings file is read back into the program, so it is validated strictly (names, kinds, option types, size) and a
4
+ damaged or hostile file produces a clear message, never a crash or a surprise. Finding likely stores only looks at
5
+ folder names and whether they hold markdown files; it never reads their contents.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ import os
11
+ import re
12
+ import secrets
13
+ import sqlite3
14
+ import stat
15
+ from contextlib import closing
16
+ from dataclasses import asdict, dataclass, field
17
+ from pathlib import Path
18
+
19
+ from .adapters.base import MemoryAdapter
20
+ from .adapters.folder import FolderAdapter
21
+ from .adapters.markdown_git import MarkdownGitAdapter, _is_reparse_point, _valid_relpath
22
+ from .adapters.mem0 import Mem0Adapter, build_mem0_memory, validate_scope
23
+ from .adapters.openwebui import OpenWebUIAdapter
24
+ from .docker_source import Docker, copy_database, list_open_webui, valid_container
25
+ from .errors import MemdebugError
26
+ from .textsafe import has_unsafe_chars, safe_text
27
+
28
+ KINDS = ("markdown", "folder", "mem0", "openwebui")
29
+ KIND_NAMES = {"markdown": "markdown notes in a git repository", "folder": "a folder of markdown notes",
30
+ "mem0": "self-hosted Mem0", "openwebui": "Open WebUI's memory"}
31
+ MAX_SETTINGS_BYTES = 1_000_000
32
+ MAX_STORES = 200
33
+ _NAME = re.compile(r"^[a-z0-9][a-z0-9._-]{0,39}\Z")
34
+ _OPTIONS = ("subdir", "user_id", "agent_id", "run_id", "docker", "files")
35
+
36
+
37
+ class SettingsError(MemdebugError):
38
+ """The settings file is unusable or a store's settings are wrong."""
39
+
40
+
41
+ @dataclass(frozen=True)
42
+ class StoreConfig:
43
+ name: str
44
+ kind: str
45
+ path: str
46
+ subdir: str | None = None
47
+ user_id: str | None = None
48
+ agent_id: str | None = None
49
+ run_id: str | None = None
50
+ docker: str | None = None # Open WebUI only: the container its database is copied from, before every look
51
+ files: str | None = None # folder only: watch just these files in it (comma-separated names), nothing else in the folder
52
+
53
+ def describe(self) -> str:
54
+ return f"{self.name} ({KIND_NAMES[self.kind]})"
55
+
56
+
57
+ @dataclass
58
+ class Registry:
59
+ stores: list[StoreConfig] = field(default_factory=list)
60
+ witness: str | None = None
61
+
62
+ def get(self, name: str) -> StoreConfig | None:
63
+ return next((s for s in self.stores if s.name == name), None)
64
+
65
+ def add(self, store: StoreConfig) -> None:
66
+ if self.get(store.name) is not None:
67
+ raise SettingsError(f"there is already a store called {safe_text(store.name, 40)}; choose another name or remove it first")
68
+ if len(self.stores) >= MAX_STORES:
69
+ raise SettingsError("too many stores")
70
+ self.stores.append(store)
71
+
72
+ def remove(self, name: str) -> StoreConfig:
73
+ store = self.get(name)
74
+ if store is None:
75
+ raise SettingsError(f"no store is called {safe_text(name, 40)}")
76
+ self.stores.remove(store)
77
+ return store
78
+
79
+
80
+ def valid_name(name: str) -> bool:
81
+ return bool(_NAME.match(name))
82
+
83
+
84
+ def _text(value: object, what: str, *, limit: int = 1000) -> str:
85
+ if not isinstance(value, str) or not value or len(value) > limit or has_unsafe_chars(value) or "\0" in value:
86
+ raise SettingsError(f"the settings file has an invalid {what}")
87
+ return value
88
+
89
+
90
+ def validate_store(data: object) -> StoreConfig:
91
+ if not isinstance(data, dict) or not {"name", "kind", "path"} <= set(data) or set(data) - {"name", "kind", "path", *_OPTIONS}:
92
+ raise SettingsError("the settings file has a store entry with unexpected fields")
93
+ name, kind = data["name"], data["kind"]
94
+ if not isinstance(name, str) or not valid_name(name):
95
+ raise SettingsError("a store name must be 1 to 40 lowercase letters, digits, dots, dashes or underscores")
96
+ if kind not in KINDS:
97
+ raise SettingsError(f"unknown store type for {safe_text(name, 40)}")
98
+ options = {key: (None if data.get(key) is None else _text(data[key], key, limit=256)) for key in _OPTIONS}
99
+ if kind == "mem0" and options["user_id"] is None:
100
+ raise SettingsError(f"the Mem0 store {name} needs a user id")
101
+ if options["files"] is not None:
102
+ names = options["files"].split(",")
103
+ if kind != "folder" or len(names) > 10 or any("/" in n or _valid_relpath(n, (".md",)) is None for n in names):
104
+ raise SettingsError(f"{safe_text(name, 40)}: 'files' must be 1 to 10 plain markdown file names, for a folder store")
105
+ if options["docker"] is not None and (kind != "openwebui" or not valid_container(options["docker"])):
106
+ raise SettingsError(f"{safe_text(name, 40)}: a container can only be named for an Open WebUI store, with a plain container name")
107
+ return StoreConfig(name=name, kind=kind, path=_text(data["path"], "path"), **options)
108
+
109
+
110
+ def load_registry(path: Path) -> Registry:
111
+ """The registry from its settings file; an empty one if there is no file yet."""
112
+ try:
113
+ info = os.lstat(path)
114
+ except FileNotFoundError:
115
+ return Registry()
116
+ except OSError as exc:
117
+ raise SettingsError(f"cannot read the settings file ({exc.strerror})") from exc
118
+ if stat.S_ISLNK(info.st_mode) or not stat.S_ISREG(info.st_mode) or info.st_size > MAX_SETTINGS_BYTES:
119
+ raise SettingsError("the settings file is not a plain file of reasonable size")
120
+ try:
121
+ data = json.loads(path.read_text(encoding="utf-8"))
122
+ except (OSError, ValueError) as exc:
123
+ raise SettingsError(f"the settings file is damaged ({safe_text(exc, 80)}); fix or delete {safe_text(path, 120)}") from exc
124
+ if not isinstance(data, dict) or data.get("version") != 1 or set(data) - {"version", "stores", "witness"}:
125
+ raise SettingsError("the settings file is not in a format this version understands")
126
+ raw_stores = data.get("stores", [])
127
+ if not isinstance(raw_stores, list) or len(raw_stores) > MAX_STORES:
128
+ raise SettingsError("the settings file lists too many stores")
129
+ registry = Registry()
130
+ for item in raw_stores:
131
+ registry.add(validate_store(item))
132
+ witness = data.get("witness")
133
+ registry.witness = None if witness is None else _text(witness, "witness path")
134
+ return registry
135
+
136
+
137
+ def save_registry(registry: Registry, path: Path) -> None:
138
+ """Write the settings atomically and privately."""
139
+ document = {"version": 1, "stores": [{k: v for k, v in asdict(s).items() if v is not None} for s in registry.stores]}
140
+ if registry.witness:
141
+ document["witness"] = registry.witness
142
+ try:
143
+ path.parent.mkdir(parents=True, exist_ok=True, mode=0o700)
144
+ temporary = path.with_name(f".{path.name}.{secrets.token_hex(6)}.tmp")
145
+ fd = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_BINARY", 0) | getattr(os, "O_NOFOLLOW", 0), 0o600)
146
+ try:
147
+ with os.fdopen(fd, "wb") as handle:
148
+ handle.write((json.dumps(document, indent=2, ensure_ascii=True) + "\n").encode("utf-8"))
149
+ handle.flush()
150
+ os.fsync(handle.fileno())
151
+ os.replace(temporary, path)
152
+ except BaseException:
153
+ try:
154
+ os.unlink(temporary)
155
+ except OSError:
156
+ pass
157
+ raise
158
+ except OSError as exc:
159
+ raise SettingsError(f"cannot save the settings file ({exc.strerror})") from exc
160
+
161
+
162
+ # -- opening a store ----------------------------------------------------------------------------------------------------
163
+
164
+ @dataclass
165
+ class OpenedStore:
166
+ adapter: MemoryAdapter
167
+ scope: dict[str, str]
168
+ notes: list[str] = field(default_factory=list)
169
+
170
+
171
+ def open_store(cfg: StoreConfig, *, refresh: bool = True) -> OpenedStore:
172
+ """Open a store for reading. An Open WebUI store that lives in Docker gets a fresh copy of its database first
173
+ (`refresh=False` reads the copy that is already there, for commands that only look)."""
174
+ if cfg.docker and refresh:
175
+ copy_database(Docker(), cfg.docker, Path(cfg.path))
176
+ if cfg.kind == "markdown":
177
+ adapter = MarkdownGitAdapter(cfg.path, store=cfg.name, subdir=cfg.subdir)
178
+ return OpenedStore(adapter, {"store": adapter.store})
179
+ if cfg.kind == "folder":
180
+ folder = FolderAdapter(cfg.path, store=cfg.name, subdir=cfg.subdir, only=tuple(cfg.files.split(",")) if cfg.files else None)
181
+ return OpenedStore(folder, {"store": folder.store})
182
+ if cfg.kind == "openwebui":
183
+ webui = OpenWebUIAdapter(cfg.path, user_id=cfg.user_id)
184
+ return OpenedStore(webui, webui.scope)
185
+ scope = validate_scope({k: v for k, v in (("user_id", cfg.user_id), ("agent_id", cfg.agent_id), ("run_id", cfg.run_id)) if v is not None})
186
+ memory, notes = build_mem0_memory(None)
187
+ return OpenedStore(Mem0Adapter(memory, Path(cfg.path)), scope, notes)
188
+
189
+
190
+ # -- telling what a path is -------------------------------------------------------------------------------------------
191
+
192
+ def detect_kind(path: Path) -> str:
193
+ """markdown, folder, openwebui or mem0, from what the path is. Reads structure only, never contents."""
194
+ try:
195
+ info = os.stat(path)
196
+ except OSError as exc:
197
+ raise SettingsError(f"cannot find that path ({exc.strerror})") from exc
198
+ if stat.S_ISDIR(info.st_mode):
199
+ return "markdown" if (path / ".git").exists() else "folder"
200
+ if not stat.S_ISREG(info.st_mode):
201
+ raise SettingsError("that is neither a folder nor a file")
202
+ try:
203
+ with closing(sqlite3.connect(f"{path.resolve().as_uri()}?mode=ro", uri=True, timeout=3)) as db:
204
+ tables = {row[0] for row in db.execute("SELECT name FROM sqlite_master WHERE type='table'")}
205
+ if "memory" in tables:
206
+ columns = {row[1] for row in db.execute('PRAGMA table_info("memory")')}
207
+ if {"id", "user_id", "content"} <= columns:
208
+ return "openwebui"
209
+ if "history" in tables:
210
+ return "mem0"
211
+ except sqlite3.Error:
212
+ pass
213
+ raise SettingsError("that file is not a database memdebug recognises (Open WebUI's webui.db or Mem0's history.db)")
214
+
215
+
216
+ # -- finding likely stores ---------------------------------------------------------------------------------------------
217
+
218
+ @dataclass(frozen=True)
219
+ class Candidate:
220
+ kind: str
221
+ name: str
222
+ path: Path
223
+ why: str
224
+ docker: str | None = None
225
+ files: tuple[str, ...] | None = None
226
+
227
+
228
+ def _slug(text: str) -> str:
229
+ return re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-")
230
+
231
+
232
+ def discover(home: Path | None = None) -> list[Candidate]:
233
+ """Stores that are probably on this computer. Looks only at folder names and for markdown files, never inside them."""
234
+ base = (home or Path.home()) / ".claude" / "projects"
235
+ found: list[Candidate] = []
236
+ taken: set[str] = set()
237
+ try:
238
+ entries = sorted(os.scandir(base), key=lambda e: e.name)[:300]
239
+ except OSError:
240
+ return []
241
+ for entry in entries:
242
+ try:
243
+ if not entry.is_dir(follow_symlinks=False) or has_unsafe_chars(entry.name):
244
+ continue
245
+ memory = Path(entry.path) / "memory"
246
+ info = os.lstat(memory)
247
+ if stat.S_ISLNK(info.st_mode) or _is_reparse_point(info) or not stat.S_ISDIR(info.st_mode):
248
+ continue
249
+ names = os.listdir(memory)[:2000]
250
+ except OSError:
251
+ continue
252
+ if not any(n.lower().endswith(".md") for n in names):
253
+ continue
254
+ slug = "claude-" + (_slug(entry.name)[-24:].strip("-") or "project")
255
+ name, n = slug[:40], 2
256
+ while name in taken:
257
+ name, n = f"{slug[:36]}-{n}", n + 1
258
+ taken.add(name)
259
+ found.append(Candidate("folder", name, memory, "Claude Code's memory notes for one project"))
260
+ if len(found) >= 50:
261
+ break
262
+ return found
263
+
264
+
265
+ def discover_docker(copies_dir: Path, docker: Docker | None = None) -> list[Candidate]:
266
+ """Open WebUI containers that are running right now. Quietly nothing if Docker is missing or not running: it is only a convenience."""
267
+ try:
268
+ client = docker or Docker()
269
+ names = list_open_webui(client)
270
+ except MemdebugError:
271
+ return []
272
+ found = []
273
+ for container in names:
274
+ slug = _slug(container)[:40] or "open-webui"
275
+ found.append(Candidate("openwebui", slug, copies_dir / slug / "webui.db", f"Open WebUI's memory, running in Docker as '{container}'", container))
276
+ return found
memdebug/sync.py ADDED
@@ -0,0 +1,177 @@
1
+ """One sync: copy new backend history into the ledger, then look for changes that bypassed it.
2
+
3
+ Safety rules baked in:
4
+ * history rows are deduplicated by the backend's own row id, so repeating a sync is harmless;
5
+ * a change outside the API is only recorded if it is seen in two passes, so a write that is
6
+ halfway through (store updated, history row not yet written) is not a false alarm;
7
+ * no deletions are claimed from a listing that may be cut short, and no changes at all are
8
+ claimed from a history that was only partly read;
9
+ * history rows that were recorded earlier but have since disappeared are reported;
10
+ * everything is appended in one transaction.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import time
15
+ from dataclasses import dataclass, field
16
+ from datetime import datetime, timezone
17
+ from typing import Callable
18
+
19
+ from .adapters.base import HistoryRead, LiveMemories, MemoryAdapter
20
+ from .errors import LedgerConflictError
21
+ from .ledger import Ledger
22
+ from .models import MemoryEvent, Op, Source, derive_trust
23
+ from .reconcile import find_external_changes, replay_events
24
+ from .textsafe import safe_text
25
+
26
+
27
+ @dataclass
28
+ class SyncReport:
29
+ history_events: int = 0
30
+ external_events: int = 0
31
+ observed_events: int = 0 # changes seen in a store that keeps no history
32
+ unconfirmed: int = 0
33
+ skipped_rows: int = 0
34
+ reconcile_skipped: bool = False
35
+ warnings: list[str] = field(default_factory=list)
36
+ live: LiveMemories | None = None # the listing the final pass looked at
37
+
38
+
39
+ def _key(event: MemoryEvent) -> tuple[str, str | None, str | None]:
40
+ return (event.memory_id, event.before, event.after)
41
+
42
+
43
+ ROLLBACK_ACTOR = "memdebug rollback"
44
+
45
+
46
+ def _with_source(event: MemoryEvent, memory) -> MemoryEvent:
47
+ """A change that adds or alters a memory carries what the backend says about where that memory came from."""
48
+ if memory is None or memory.source is None or event.after is None or event.source is not None:
49
+ return event
50
+ return event.model_copy(update={"source": memory.source, "trust": derive_trust(memory.source)})
51
+
52
+
53
+ def _observed(event: MemoryEvent) -> MemoryEvent:
54
+ """A change seen in a store that keeps no history: an ordinary ADD, UPDATE or DELETE, time-stamped when it was noticed."""
55
+ op = Op.DELETE if event.after is None else Op.ADD if event.before is None else Op.UPDATE
56
+ return event.model_copy(update={"op": op})
57
+
58
+
59
+ def _acknowledged(event: MemoryEvent, written: dict[str, str | None]) -> MemoryEvent:
60
+ """An outside change that is exactly what a rollback just wrote becomes an ordinary change with a named actor."""
61
+ if event.memory_id not in written or written[event.memory_id] != event.after:
62
+ return event
63
+ op = Op.DELETE if event.after is None else Op.ADD if event.before is None else Op.UPDATE
64
+ return event.model_copy(update={"op": op, "source": Source(actor_id=ROLLBACK_ACTOR)})
65
+
66
+
67
+ def sync(
68
+ adapter: MemoryAdapter,
69
+ ledger: Ledger,
70
+ scope: dict[str, str],
71
+ *,
72
+ now: datetime | None = None,
73
+ settle_seconds: float = 1.0,
74
+ sleep: Callable[[float], None] = time.sleep,
75
+ adopt_existing: bool = False,
76
+ max_history_rows: int = 100_000,
77
+ acknowledge: dict[str, str | None] | None = None,
78
+ ) -> SyncReport:
79
+ """adopt_existing: on the very first sync of a store, record memories that have no history
80
+ as observed ADDs instead of as changes made outside the API.
81
+
82
+ acknowledge: memory id -> the exact text (or None for removed) that this tool itself just wrote. A change that
83
+ matches is recorded as an ordinary ADD, UPDATE or DELETE by "memdebug rollback", not as a change made outside
84
+ the history, so a rollback does not raise a false alarm. Anything that differs is still reported."""
85
+ last_error: LedgerConflictError | None = None
86
+ for _ in range(3): # another process may have written to the ledger meanwhile
87
+ try:
88
+ return _sync_once(
89
+ adapter, ledger, scope, now or datetime.now(timezone.utc),
90
+ settle_seconds, sleep, adopt_existing, max_history_rows, acknowledge or {},
91
+ )
92
+ except LedgerConflictError as exc:
93
+ last_error = exc
94
+ assert last_error is not None
95
+ raise last_error
96
+
97
+
98
+ def _sync_once(adapter, ledger, scope, now, settle_seconds, sleep, adopt_existing, max_rows, acknowledge) -> SyncReport:
99
+ known_refs = ledger.known_refs(adapter.name)
100
+ had_events = ledger.has_events(adapter.name)
101
+
102
+ def observe() -> tuple[list[MemoryEvent], list[MemoryEvent], HistoryRead, LiveMemories]:
103
+ history = adapter.read_history(max_rows) # history first, then the live listing
104
+ live = adapter.list_memories(scope)
105
+ scopes = {m.id: m.scope for m in live.memories}
106
+ new: list[MemoryEvent] = []
107
+ seen: set[str] = set()
108
+ for event in history.events:
109
+ ref = event.backend_ref
110
+ if ref is None or ref in known_refs or ref in seen:
111
+ continue
112
+ seen.add(ref)
113
+ if not event.scope and event.memory_id in scopes:
114
+ event = event.model_copy(update={"scope": scopes[event.memory_id]})
115
+ new.append(event)
116
+ if history.truncated:
117
+ candidates: list[MemoryEvent] = [] # a partial history cannot support any claim
118
+ else:
119
+ prior = [e.event for e in ledger.entries() if e.event.backend == adapter.name]
120
+ state = replay_events(prior + new)
121
+ candidates = find_external_changes(
122
+ state, live.memories, adapter.name, now, scope=scope, complete=live.complete
123
+ )
124
+ return new, candidates, history, live
125
+
126
+ report = SyncReport()
127
+ new, candidates, history, live = observe()
128
+ confirmed = candidates
129
+ if candidates:
130
+ sleep(settle_seconds)
131
+ new, second, history, live = observe()
132
+ wanted = {_key(c) for c in candidates}
133
+ confirmed = [c for c in second if _key(c) in wanted]
134
+ report.unconfirmed = len(candidates) - len(confirmed)
135
+ if report.unconfirmed:
136
+ report.warnings.append(
137
+ f"{report.unconfirmed} change(s) were not seen twice and were not recorded; "
138
+ "they will be checked again on the next sync"
139
+ )
140
+
141
+ if adopt_existing and not had_events:
142
+ confirmed = [
143
+ c.model_copy(update={"op": Op.ADD}) if c.before is None and c.after is not None else c
144
+ for c in confirmed
145
+ ]
146
+
147
+ if acknowledge:
148
+ confirmed = [_acknowledged(c, acknowledge) for c in confirmed]
149
+
150
+ observed_only = "history" not in adapter.capabilities
151
+ if observed_only: # no history to bypass: a difference is simply a change seen between two looks
152
+ confirmed = [_observed(c) for c in confirmed]
153
+ by_id = {m.id: m for m in live.memories}
154
+ confirmed = [_with_source(c, by_id.get(c.memory_id)) for c in confirmed]
155
+
156
+ if history.truncated:
157
+ report.reconcile_skipped = True
158
+ report.warnings.append("history was only partly read, so changes outside the API were not checked")
159
+ else:
160
+ missing = known_refs - history.refs
161
+ if missing:
162
+ sample = ", ".join(safe_text(ref, 40) for ref in sorted(missing)[:3])
163
+ report.warnings.append(
164
+ f"{len(missing)} history row(s) recorded earlier are no longer in the backend "
165
+ f"history (removed or scrubbed); for example: {sample}"
166
+ )
167
+ report.warnings += history.warnings + live.warnings
168
+ report.skipped_rows = history.skipped
169
+ report.live = live
170
+
171
+ ledger.append_many(new + confirmed)
172
+ report.history_events = len(new)
173
+ if observed_only:
174
+ report.observed_events = len(confirmed)
175
+ else:
176
+ report.external_events = len(confirmed)
177
+ return report
memdebug/textsafe.py ADDED
@@ -0,0 +1,73 @@
1
+ """Helpers for handling text that came from a memory store.
2
+
3
+ Memory text is untrusted. It may be planted by an attacker, for example through an email an
4
+ agent read. Two rules follow:
5
+ * bound its size before storing it (bound_text), and
6
+ * never print it raw (safe_text), so it cannot move the cursor, recolour the terminal,
7
+ forge extra timeline lines or reorder what the reader sees.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import hashlib
12
+ import unicodedata
13
+
14
+ MAX_TEXT_CHARS = 100_000
15
+
16
+ _NAMED = {"\n": "\\n", "\r": "\\r", "\t": "\\t"}
17
+ _BIDI = set("\u200e\u200f\u061c\u202a\u202b\u202c\u202d\u202e\u2066\u2067\u2068\u2069")
18
+ _UNSAFE_CATEGORIES = {"Cc", "Cf", "Cs", "Co", "Cn", "Zl", "Zp"}
19
+
20
+
21
+ def bound_text(text: str, max_chars: int = MAX_TEXT_CHARS) -> str:
22
+ """Cap length. A cut text carries a hash of the full text, so two different long texts
23
+ that share a prefix still compare as different."""
24
+ if len(text) <= max_chars:
25
+ return text
26
+ digest = hashlib.sha256(text.encode("utf-8", "surrogatepass")).hexdigest()[:16]
27
+ return f"{text[:max_chars]}...[cut: {len(text)} chars, sha256 {digest}]"
28
+
29
+
30
+ def _escape(ch: str) -> str:
31
+ code = ord(ch)
32
+ if code < 0x100:
33
+ return f"\\x{code:02x}"
34
+ if code < 0x10000:
35
+ return f"\\u{code:04x}"
36
+ return f"\\U{code:08x}"
37
+
38
+
39
+ def safe_text(value: object, limit: int | None = 200) -> str:
40
+ """Single-line, printable rendering of untrusted text. Control, format, bidi and
41
+ line-separator characters are shown as visible escapes."""
42
+ pieces: list[str] = []
43
+ length = 0
44
+ for ch in str(value):
45
+ if ch in _NAMED:
46
+ piece = _NAMED[ch]
47
+ elif ch in _BIDI or unicodedata.category(ch) in _UNSAFE_CATEGORIES:
48
+ piece = _escape(ch)
49
+ else:
50
+ piece = ch
51
+ if limit is not None and length + len(piece) > limit:
52
+ pieces.append("...")
53
+ break
54
+ pieces.append(piece)
55
+ length += len(piece)
56
+ return "".join(pieces)
57
+
58
+
59
+ def has_unsafe_chars(text: str) -> bool:
60
+ """True if the text contains anything safe_text would have to escape."""
61
+ return any(ch in _BIDI or unicodedata.category(ch) in _UNSAFE_CATEGORIES for ch in text)
62
+
63
+
64
+ def console_safe(text: str, encoding: str | None = None) -> str:
65
+ """Make text printable on a console whose encoding cannot show every character (a Windows
66
+ code page, for example). Unencodable characters become visible escapes instead of raising."""
67
+ import sys
68
+
69
+ target = encoding or getattr(sys.stdout, "encoding", None) or "utf-8"
70
+ try:
71
+ return text.encode(target, "backslashreplace").decode(target)
72
+ except LookupError:
73
+ return text.encode("ascii", "backslashreplace").decode("ascii")
@@ -0,0 +1 @@
1
+ """A read-only local web viewer for the ledger. See server.py for the security model."""
@@ -0,0 +1,136 @@
1
+ """Safe-by-construction HTML.
2
+
3
+ Memory text is untrusted: an attacker can plant `<script>` or markup in a memory. Instead of
4
+ remembering to escape at every use, pages are built only through `el()`:
5
+
6
+ * a plain string child is ALWAYS escaped (and control, bidi and line-separator characters are
7
+ made visible), so forgetting to escape is not possible;
8
+ * only `Markup` (created by this module) is inserted as-is;
9
+ * tags and attributes come from fixed allow-lists: no script, style, iframe, image, object or
10
+ event-handler attributes can be produced at all;
11
+ * `href` and `action` accept only internal paths that start with a single `/`.
12
+
13
+ The pages carry no JavaScript and the server adds a Content-Security-Policy that forbids it, so even
14
+ a bug here could not run code.
15
+ """
16
+ from __future__ import annotations
17
+
18
+ import html as _html
19
+ import re
20
+ import unicodedata
21
+ from typing import Iterable
22
+
23
+ from ..textsafe import _BIDI, _UNSAFE_CATEGORIES, safe_text
24
+
25
+ __all__ = ["Markup", "el", "inline", "block", "raw_doctype"]
26
+
27
+
28
+ class Markup(str):
29
+ """HTML that is already safe. Only functions in this module create it."""
30
+
31
+ __slots__ = ()
32
+
33
+
34
+ _TAGS = frozenset({
35
+ "html", "head", "meta", "title", "link", "body", "header", "nav", "main", "aside", "section", "footer",
36
+ "h1", "h2", "h3", "p", "div", "span", "a", "table", "thead", "tbody", "tr", "th", "td", "caption",
37
+ "ul", "ol", "li", "pre", "code", "details", "summary", "form", "label", "select", "option", "button",
38
+ "input", "strong", "em", "br", "small", "time", "dl", "dt", "dd",
39
+ })
40
+ _VOID = frozenset({"meta", "link", "br", "input"})
41
+ _ATTRS = frozenset({
42
+ "class", "href", "id", "lang", "charset", "name", "content", "rel", "type", "for", "value", "selected",
43
+ "checked", "scope", "colspan", "aria-label", "aria-current", "method", "action", "datetime", "title",
44
+ "role", "open", "disabled", "http-equiv", "media",
45
+ })
46
+ _URL_ATTRS = frozenset({"href", "action"})
47
+ _INTERNAL_URL = re.compile(r"^/(?!/)[A-Za-z0-9_\-./?=&%:~+,]*\Z")
48
+
49
+
50
+ def raw_doctype() -> Markup:
51
+ return Markup("<!doctype html>")
52
+
53
+
54
+ def _attr_name(name: str) -> str:
55
+ return name.rstrip("_").replace("_", "-")
56
+
57
+
58
+ def _render_attrs(attrs: dict) -> str:
59
+ parts: list[str] = []
60
+ for key, value in attrs.items():
61
+ name = _attr_name(key)
62
+ if name not in _ATTRS:
63
+ raise ValueError(f"attribute not allowed: {name!r}")
64
+ if value is None or value is False:
65
+ continue
66
+ if value is True:
67
+ parts.append(name)
68
+ continue
69
+ text = str(value)
70
+ if name in _URL_ATTRS and not _INTERNAL_URL.match(text):
71
+ raise ValueError(f"only internal paths are allowed in {name}: {text!r}")
72
+ parts.append(f'{name}="{_html.escape(safe_text(text, None), quote=True)}"')
73
+ return (" " + " ".join(parts)) if parts else ""
74
+
75
+
76
+ def _render_children(children: Iterable) -> str:
77
+ out: list[str] = []
78
+ for child in children:
79
+ if child is None or child is False:
80
+ continue
81
+ if isinstance(child, Markup):
82
+ out.append(str(child))
83
+ elif isinstance(child, (list, tuple)):
84
+ out.append(_render_children(child))
85
+ elif isinstance(child, (int, float)) and not isinstance(child, bool):
86
+ out.append(_html.escape(str(child)))
87
+ else: # any ordinary string is untrusted by default
88
+ out.append(_html.escape(safe_text(child, None), quote=True))
89
+ return "".join(out)
90
+
91
+
92
+ def el(tag: str, *children, **attrs) -> Markup:
93
+ if tag not in _TAGS:
94
+ raise ValueError(f"tag not allowed: {tag!r}")
95
+ opening = f"<{tag}{_render_attrs(attrs)}>"
96
+ if tag in _VOID:
97
+ if children:
98
+ raise ValueError(f"<{tag}> cannot have children")
99
+ return Markup(opening)
100
+ return Markup(f"{opening}{_render_children(children)}</{tag}>")
101
+
102
+
103
+ def inline(text: object, limit: int | None = 200) -> Markup:
104
+ """Untrusted text on one line, escaped and shortened."""
105
+ return Markup(_html.escape(safe_text(text, limit), quote=True))
106
+
107
+
108
+ def block(text: object, limit: int = 20_000) -> Markup:
109
+ """Untrusted multi-line text for a <pre>: line breaks and tabs are kept, every other control,
110
+ format, bidi or line-separator character is shown as a visible escape."""
111
+ value = str(text if text is not None else "")
112
+ total = len(value)
113
+ shown = value[:limit]
114
+ pieces: list[str] = []
115
+ skip_next = False
116
+ for index, ch in enumerate(shown):
117
+ if skip_next:
118
+ skip_next = False
119
+ continue
120
+ if ch == "\r":
121
+ if shown[index + 1:index + 2] == "\n":
122
+ skip_next = True
123
+ pieces.append("\n")
124
+ continue
125
+ pieces.append("\\r")
126
+ elif ch in ("\n", "\t"):
127
+ pieces.append(ch)
128
+ elif ch in _BIDI or unicodedata.category(ch) in _UNSAFE_CATEGORIES:
129
+ code = ord(ch)
130
+ pieces.append(f"\\x{code:02x}" if code < 0x100 else f"\\u{code:04x}" if code < 0x10000 else f"\\U{code:08x}")
131
+ else:
132
+ pieces.append(ch)
133
+ out = _html.escape("".join(pieces), quote=True)
134
+ if total > limit:
135
+ out += _html.escape(f"\n... (shortened: showing the first {limit} of {total} characters)")
136
+ return Markup(out)