underwrit 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
underwrit/cedar.py ADDED
@@ -0,0 +1,150 @@
1
+ """A Cedar subset, imported into Underwrit's policy document. Honest about what it covers.
2
+
3
+ Underwrit's decision engine is not a general authorisation language: it holds or denies *tool calls*
4
+ based on provenance, consequence, ceilings and allowlists. Teams that write Cedar can bring the
5
+ parts of their policy that map onto those knobs:
6
+
7
+ forbid (principal, action == Action::"send_email", resource);
8
+ → deniedTools += send_email
9
+ forbid (principal, action in [Action::"a", Action::"b"], resource);
10
+ → deniedTools += a, b
11
+ permit (principal, action == Action::"send_email", resource)
12
+ when { resource.to like "*@acme.com" };
13
+ → allowlist.to += "*@acme.com" (== "x" is the exact pattern "x")
14
+ forbid (principal, action == Action::"transfer", resource)
15
+ when { resource.amount > 10000 };
16
+ → ceilings.amount = 10000 (>= N becomes N - 1 for integers)
17
+
18
+ Everything else — principals other than the wildcard, `unless`, `context.*`, entity hierarchies,
19
+ `has`, `&&` chains — is refused with the line it sits on. The importer never guesses: an
20
+ unsupported policy fails the whole import rather than producing a document that silently
21
+ drops a rule. What is imported is the *conservative* projection: a permit can only add an
22
+ allowlist pattern (narrowing what passes), and a forbid can only deny or cap.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ import fnmatch
28
+ import math
29
+ import re
30
+
31
+ ACTION_ONE = re.compile(r'action\s*==\s*Action::"([^"]+)"')
32
+ ACTION_MANY = re.compile(r'action\s+in\s*\[([^\]]*)\]')
33
+ ACTION_ITEM = re.compile(r'Action::"([^"]+)"')
34
+ HEAD = re.compile(r'^(permit|forbid)\s*\(\s*principal\s*,\s*(.+?)\s*,\s*resource\s*\)\s*(.*?)\s*;\s*$', re.S)
35
+ WHEN = re.compile(r'^when\s*\{\s*(.+?)\s*\}$', re.S)
36
+ LIKE = re.compile(r'^resource\.([A-Za-z_][A-Za-z0-9_]*)\s*(like|==)\s*"([^"]*)"$')
37
+ CEIL = re.compile(r'^resource\.([A-Za-z_][A-Za-z0-9_]*)\s*(>=|>)\s*(-?\d+(?:\.\d+)?)$')
38
+
39
+
40
+ class CedarError(ValueError):
41
+ pass
42
+
43
+
44
+ def _strip_comment(raw: str) -> str:
45
+ """Drop a `//` comment, but not a `//` inside a string ("https://…")."""
46
+ in_str = False
47
+ i = 0
48
+ while i < len(raw):
49
+ ch = raw[i]
50
+ if ch == '"' and (i == 0 or raw[i - 1] != "\\"):
51
+ in_str = not in_str
52
+ elif not in_str and raw.startswith("//", i):
53
+ return raw[:i]
54
+ i += 1
55
+ return raw
56
+
57
+
58
+ def _statements(text: str) -> list[tuple[int, str]]:
59
+ """(line number, statement) pairs, comments stripped, split on `;`."""
60
+ out, buf, start = [], [], None
61
+ for n, raw in enumerate(text.splitlines(), 1):
62
+ line = _strip_comment(raw)
63
+ if not line.strip():
64
+ continue
65
+ if start is None:
66
+ start = n
67
+ buf.append(line.strip())
68
+ if line.rstrip().endswith(";"):
69
+ out.append((start, " ".join(buf)))
70
+ buf, start = [], None
71
+ if buf:
72
+ raise CedarError(f"line {start}: statement does not end with ';'")
73
+ return out
74
+
75
+
76
+ def _actions(scope: str, line: int) -> list[str]:
77
+ m = ACTION_ONE.fullmatch(scope.strip())
78
+ if m:
79
+ return [m.group(1)]
80
+ m = ACTION_MANY.fullmatch(scope.strip())
81
+ if m:
82
+ items = ACTION_ITEM.findall(m.group(1))
83
+ if not items:
84
+ raise CedarError(f"line {line}: empty action list")
85
+ return items
86
+ raise CedarError(f"line {line}: only `action == Action::\"x\"` and `action in [...]` are supported "
87
+ f"(got {scope.strip()!r})")
88
+
89
+
90
+ def to_policy(text: str) -> dict:
91
+ """The policy-document fragment a Cedar text maps to: deniedTools, allowlist, ceilings."""
92
+ denied: list[str] = []
93
+ allowlist: dict[str, list[str]] = {}
94
+ ceilings: dict[str, float] = {}
95
+ for line, stmt in _statements(text):
96
+ m = HEAD.match(stmt)
97
+ if not m:
98
+ raise CedarError(f"line {line}: expected `permit|forbid (principal, action ..., resource) [when {{...}}];`")
99
+ effect, scope, tail = m.groups()
100
+ actions = _actions(scope, line)
101
+ if not tail:
102
+ if effect == "forbid":
103
+ denied.extend(a for a in actions if a not in denied)
104
+ continue
105
+ raise CedarError(f"line {line}: an unconditional permit does not narrow anything Underwrit enforces; "
106
+ f"Underwrit allows by default and records — remove it or add a `when` on a resource field")
107
+ w = WHEN.match(tail)
108
+ if not w:
109
+ raise CedarError(f"line {line}: only a single `when {{ ... }}` clause is supported (no `unless`)")
110
+ cond = w.group(1).strip()
111
+ if effect == "permit":
112
+ c = LIKE.match(cond)
113
+ if not c:
114
+ raise CedarError(f"line {line}: a permit's condition must be `resource.<field> like|== \"pattern\"`")
115
+ fld, op, pat = c.groups()
116
+ # Cedar `like` is a glob; Underwrit's allowlist speaks literals and anchored `re:` patterns.
117
+ # Translate rather than store verbatim, or the rule would never match anything.
118
+ rule = f"re:{fnmatch.translate(pat)}" if op == "like" and any(ch in pat for ch in "*?[") else pat
119
+ allowlist.setdefault(fld, [])
120
+ if rule not in allowlist[fld]:
121
+ allowlist[fld].append(rule)
122
+ else:
123
+ c = CEIL.match(cond)
124
+ if not c:
125
+ raise CedarError(f"line {line}: a forbid's condition must be `resource.<field> > N` or `>= N`")
126
+ fld, op, num = c.groups()
127
+ val = float(num)
128
+ if op == ">=":
129
+ # Underwrit holds strictly above a ceiling; "forbid >= N" means N itself is forbidden.
130
+ val = val - 1 if val == int(val) else math.nextafter(val, -math.inf)
131
+ ceilings[fld] = min(val, ceilings[fld]) if fld in ceilings else val
132
+ return {"deniedTools": denied, "allowlist": allowlist, "ceilings": ceilings,
133
+ "note": "Cedar subset: forbid→deniedTools, permit-when-like→allowlist, forbid-when-greater→ceilings. "
134
+ "Principals, contexts, unless-clauses and entity hierarchies are not imported."}
135
+
136
+
137
+ def merge(current: dict, fragment: dict) -> dict:
138
+ """The current policy document with the fragment folded in (lists union, ceilings tightest)."""
139
+ out = dict(current)
140
+ out["deniedTools"] = sorted(set(current.get("deniedTools") or []) | set(fragment["deniedTools"]))
141
+ al = {k: list(v) for k, v in (current.get("allowlist") or {}).items()}
142
+ for k, pats in fragment["allowlist"].items():
143
+ al.setdefault(k, [])
144
+ al[k] += [p for p in pats if p not in al[k]]
145
+ out["allowlist"] = al
146
+ ce = dict(current.get("ceilings") or {})
147
+ for k, v in fragment["ceilings"].items():
148
+ ce[k] = min(v, ce[k]) if k in ce else v
149
+ out["ceilings"] = ce
150
+ return out
underwrit/chain.py ADDED
@@ -0,0 +1,519 @@
1
+ """The audit log: what happened, who caused it, and whether the record has been altered since.
2
+
3
+ Separate from the run log. A run log is an operator's view of one execution and is written for
4
+ debugging; this is evidence, written for someone reconstructing events afterwards who does not trust
5
+ the system that produced them. The two have different readers and different retention, so they are
6
+ different tables.
7
+
8
+ **Append-only, and tamper-evident.** Every entry stores the hash of the entry before it, so the log
9
+ is a chain: altering or deleting any row breaks every hash after it, and `verify_chain` finds the
10
+ first break. This does not make the log immutable — anyone with the database file can rewrite it —
11
+ but it makes a rewrite *detectable*, which is the achievable property for a log that lives beside
12
+ the application. Genuine immutability means shipping these entries to somewhere the application
13
+ cannot reach, and `export_jsonl` exists for exactly that.
14
+
15
+ What is deliberately not recorded: request bodies, tokens, credentials, or anything from a
16
+ `secretRef`. An audit log is a high-value target precisely because it is trusted and retained, and a
17
+ log that accumulates secrets is a breach waiting for a reader.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import hashlib
23
+ import os
24
+ import json
25
+ import pathlib
26
+ import sqlite3
27
+ import threading
28
+
29
+ from . import db
30
+ import time
31
+
32
+ GENESIS = "0" * 64
33
+
34
+
35
+ def entry_hash(previous: str, payload: dict) -> str:
36
+ """Hash one entry together with its predecessor.
37
+
38
+ `sort_keys` matters: the same entry must hash identically on every machine and every Python
39
+ version, or verification fails on a log nobody touched. Separators are pinned for the same
40
+ reason — json.dumps' default spacing has changed before.
41
+ """
42
+ body = json.dumps(payload, sort_keys=True, separators=(",", ":"))
43
+ return hashlib.sha256(f"{previous}{body}".encode()).hexdigest()
44
+
45
+
46
+ def init(conn) -> None:
47
+ conn.execute(
48
+ db.ddl("""
49
+ CREATE TABLE IF NOT EXISTS audit_log (
50
+ seq INTEGER PRIMARY KEY AUTOINCREMENT,
51
+ at REAL NOT NULL,
52
+ actor TEXT NOT NULL DEFAULT '',
53
+ actor_id TEXT NOT NULL DEFAULT '',
54
+ action TEXT NOT NULL,
55
+ environment TEXT NOT NULL DEFAULT '',
56
+ target TEXT NOT NULL DEFAULT '',
57
+ outcome TEXT NOT NULL DEFAULT '',
58
+ reason TEXT NOT NULL DEFAULT '',
59
+ detail TEXT NOT NULL DEFAULT '{}',
60
+ source_ip TEXT NOT NULL DEFAULT '',
61
+ prev_hash TEXT NOT NULL DEFAULT '',
62
+ hash TEXT NOT NULL DEFAULT ''
63
+ );
64
+ """)
65
+ )
66
+ # Added after the table shipped. A redacted entry keeps its row and its hash — see `redact`.
67
+ if "redacted_at" not in db.table_columns(conn, "audit_log"):
68
+ conn.execute(db.ddl("ALTER TABLE audit_log ADD COLUMN redacted_at REAL"))
69
+ # Created here rather than by `seal()`, which is where it used to appear the first time a chain
70
+ # was sealed. A lazily-created table means two installs of the same version have different
71
+ # schemas, and that is not theoretical: restoring a backup taken from a *sealed* database into
72
+ # a fresh one failed on "no such table: audit_log_archive" — so the sealed chain, the hardest
73
+ # evidence here to reconstruct, was the one thing that could not be recovered. An empty archive
74
+ # table costs nothing and makes the schema the same everywhere.
75
+ conn.execute(db.ddl("""
76
+ CREATE TABLE IF NOT EXISTS audit_log_archive (
77
+ seq INTEGER,
78
+ at REAL,
79
+ actor TEXT,
80
+ actor_id TEXT,
81
+ action TEXT,
82
+ environment TEXT,
83
+ target TEXT,
84
+ outcome TEXT,
85
+ reason TEXT,
86
+ detail TEXT,
87
+ source_ip TEXT,
88
+ prev_hash TEXT,
89
+ hash TEXT,
90
+ sealed_at REAL
91
+ );
92
+ """))
93
+ conn.execute("CREATE INDEX IF NOT EXISTS audit_at ON audit_log(at)")
94
+ conn.execute("CREATE INDEX IF NOT EXISTS audit_actor ON audit_log(actor)")
95
+
96
+ # "No two entries claim the same predecessor" *is* the no-fork invariant, so a unique index on
97
+ # prev_hash states it to the database rather than trusting every writer to be careful. Where it
98
+ # applies, a second process racing the first gets an integrity error instead of a silent fork,
99
+ # and `record` retries against the fresh tail.
100
+ #
101
+ # It cannot be created over a log that already forked, and rewriting that history to make room
102
+ # is exactly the tampering the chain exists to detect. So the failure is reported rather than
103
+ # swallowed: a guarantee that quietly did not apply is the "looks configured, does nothing"
104
+ # failure this codebase keeps finding.
105
+ conn.commit() # what came before must survive the rollback below on a forked log
106
+ try:
107
+ conn.execute("CREATE UNIQUE INDEX IF NOT EXISTS audit_prev_unique ON audit_log(prev_hash)")
108
+ except Exception as exc: # noqa: BLE001 — any engine's integrity error means pre-existing forks
109
+ conn.rollback()
110
+ print(
111
+ "Audit: could not add the unique index on prev_hash — this log already contains forked "
112
+ f"entries, so cross-process protection is not active ({type(exc).__name__}). Appends are "
113
+ "still serialised within this process. Run audit.verify_chain to see where it forked.",
114
+ flush=True,
115
+ )
116
+
117
+
118
+ def record(
119
+ conn,
120
+ *,
121
+ actor: str,
122
+ actor_id: str = "",
123
+ action: str,
124
+ outcome: str,
125
+ environment: str = "",
126
+ target: str = "",
127
+ reason: str = "",
128
+ detail: dict | None = None,
129
+ source_ip: str = "",
130
+ at: float | None = None,
131
+ ) -> dict:
132
+ """Append one entry. Never raises for a caller — an audit failure must not break the request.
133
+
134
+ `outcome` is `allowed`, `denied`, `succeeded` or `failed`. Denials are recorded as carefully as
135
+ successes: a log containing only what worked cannot show someone probing for what does not.
136
+ """
137
+ # Bounded, not `with _APPEND_LOCK:`. The lock serialises appends so two threads cannot read
138
+ # the same chain tail; held forever by a thread stuck on a dead database socket, it wedged
139
+ # *every* audit write — and since signing in writes one, nobody could authenticate while
140
+ # `/api/health` answered normally, because health touches no database. A lock that can only be
141
+ # waited on indefinitely turns one stuck query into a total outage with no error anywhere.
142
+ if not _APPEND_LOCK.acquire(timeout=_APPEND_LOCK_TIMEOUT):
143
+ raise TimeoutError(
144
+ f"audit append lock held for more than {_APPEND_LOCK_TIMEOUT}s — a previous write is stuck"
145
+ )
146
+ try:
147
+ last = None
148
+ for _ in range(_MAX_APPEND_RETRIES):
149
+ try:
150
+ return _append(conn, actor=actor, actor_id=actor_id, action=action, outcome=outcome,
151
+ environment=environment, target=target, reason=reason,
152
+ detail=detail, source_ip=source_ip, at=at)
153
+ except Exception as exc: # noqa: BLE001 — engines raise different integrity errors
154
+ # Another process appended between this one's read and write. The unique index on
155
+ # prev_hash turned what would have been a silent fork into this error, so re-read
156
+ # the tail and try again rather than writing a second claim on the same predecessor.
157
+ last = exc
158
+ conn.rollback()
159
+ raise last # type: ignore[misc]
160
+ finally:
161
+ _APPEND_LOCK.release()
162
+
163
+
164
+ # Appending is serialised. Reading the tail and writing the next entry is a read-modify-write, and
165
+ # `server.py` is a ThreadingHTTPServer, so without this two requests read the same tail and both
166
+ # write claiming the same predecessor. That is not a hypothetical: a real log reached 13 forks
167
+ # across 614 entries, the first pair 0.4ms apart, and `verify_chain` reported the result exactly as
168
+ # it reports tampering. A control that cannot tell a burst of traffic from a rewrite is worse than
169
+ # no control, because it is still believed.
170
+ #
171
+ # The lock spans the commit, not just the insert. Releasing it after the INSERT and before the
172
+ # COMMIT would fix nothing: the next thread's SELECT runs on its own connection and cannot see an
173
+ # uncommitted row, so it would read the same tail again and fork anyway.
174
+ #
175
+ # **This serialises one process.** The server is single-process today, so the chain is correct as
176
+ # deployed — but a second replica appends on its own lock and the two do not see each other. The
177
+ # structural guarantee is the unique index in `init()`; see the note there for when it applies.
178
+ _APPEND_LOCK = threading.Lock()
179
+ # Seconds to wait for the append lock. Long enough that ordinary contention never trips it, short
180
+ # enough that a stuck writer surfaces as an error rather than as authentication hanging forever.
181
+ _APPEND_LOCK_TIMEOUT = float(os.environ.get("AUDIT_LOCK_TIMEOUT", "15") or 15)
182
+
183
+
184
+ def _append(conn, *, actor, actor_id, action, outcome, environment, target, reason, detail, source_ip, at=None):
185
+ row = conn.execute("SELECT hash FROM audit_log ORDER BY seq DESC LIMIT 1").fetchone()
186
+ previous = (row[0] if row else "") or GENESIS
187
+ payload = {
188
+ "at": float(time.time() if at is None else at),
189
+ "actor": str(actor or ""),
190
+ # Coerced to the type the COLUMN holds, before hashing. The hash is computed over the values
191
+ # going in and re-computed over the values coming back out, so any coercion the database
192
+ # performs between those two moments breaks verification permanently.
193
+ #
194
+ # `actor_id` is TEXT. A caller passing the integer 7 hashed `7` and stored `"7"`, and the
195
+ # entry could never verify afterwards — indistinguishable from a row somebody edited, which
196
+ # is the single worst way for this to fail. Found by an int slipping into a test.
197
+ #
198
+ # Safe to apply to an existing chain: for a caller that already passed a string this is the
199
+ # identity, so no entry that verifies today changes. It can only turn entries that were
200
+ # already unverifiable into correct ones.
201
+ "actor_id": str(actor_id or ""),
202
+ "action": str(action or ""),
203
+ "environment": str(environment or ""),
204
+ "target": str(target or ""),
205
+ "outcome": str(outcome or ""),
206
+ "reason": str(reason or ""),
207
+ "detail": detail or {},
208
+ "source_ip": str(source_ip or ""),
209
+ }
210
+ digest = entry_hash(previous, payload)
211
+ cur = conn.execute(
212
+ "INSERT INTO audit_log (at, actor, actor_id, action, environment, target, outcome, reason, detail, "
213
+ "source_ip, prev_hash, hash) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)"
214
+ + (" RETURNING seq" if db.is_postgres() else ""),
215
+ # Every value comes from `payload` — the same object that was hashed. Passing the raw
216
+ # parameters here instead let the stored form drift from the hashed form, which is the
217
+ # defect above: coercing one without the other only moves where the mismatch happens.
218
+ (
219
+ payload["at"], payload["actor"], payload["actor_id"], payload["action"],
220
+ payload["environment"], payload["target"], payload["outcome"], payload["reason"],
221
+ json.dumps(payload["detail"], sort_keys=True), payload["source_ip"], previous, digest,
222
+ ),
223
+ )
224
+ # Committed inside the lock so the next appender's SELECT sees this row. The one caller opens a
225
+ # connection solely for this write, so nothing else is committed by it.
226
+ seq = cur.fetchone()["seq"] if db.is_postgres() else getattr(cur, "lastrowid", None)
227
+ conn.commit()
228
+ return {**payload, "hash": digest, "prevHash": previous, "seq": seq}
229
+
230
+
231
+ # How many times an append will re-read the tail and try again. Only reachable when another
232
+ # *process* appended between this one's read and write, which the unique index turns into an error
233
+ # rather than a fork. One retry covers the race; a persistent failure is a real problem and should
234
+ # surface rather than spin.
235
+ _MAX_APPEND_RETRIES = 3
236
+
237
+
238
+ def _payload_of(row) -> dict:
239
+ return {
240
+ "at": row["at"],
241
+ "actor": row["actor"],
242
+ "actor_id": row["actor_id"],
243
+ "action": row["action"],
244
+ "environment": row["environment"],
245
+ "target": row["target"],
246
+ "outcome": row["outcome"],
247
+ "reason": row["reason"],
248
+ "detail": json.loads(row["detail"] or "{}"),
249
+ "source_ip": row["source_ip"],
250
+ }
251
+
252
+
253
+ def explained_redactions(conn) -> set:
254
+ """Sequence numbers that have a matching `audit.redact` entry in the chain.
255
+
256
+ The check that makes redaction safe. `verify_rows` accepts a redacted row's recorded hash
257
+ without recomputing it, so on its own a redaction flag is a way to make an edited row verify
258
+ clean — set `redacted_at` in the database and the content check is skipped. What an attacker
259
+ cannot do is append the corresponding `audit.redact` entry without it being in the chain,
260
+ hashed, and visible. So a redaction with no explanation is reported, and it is a stronger
261
+ signal than a broken hash: a broken hash means somebody edited a row, this means somebody
262
+ edited a row and tried to make it look authorised.
263
+ """
264
+ return {int(r["target"]) for r in conn.execute(
265
+ "SELECT target FROM audit_log WHERE action = ? AND target != ?", ("audit.redact", ""))
266
+ if str(r["target"]).isdigit()}
267
+
268
+
269
+ def verify_chain(conn) -> dict:
270
+ """Recompute every hash and report the first entry that does not match.
271
+
272
+ Returns {"ok", "checked", "brokenAt", "detail"}. `brokenAt` is the sequence number of the first
273
+ entry whose recorded hash disagrees with its contents — everything before it is intact, which is
274
+ what makes this useful rather than merely alarming.
275
+ """
276
+ result = verify_rows(conn.execute("SELECT * FROM audit_log ORDER BY seq ASC"))
277
+ if not result.get("redacted"):
278
+ return result
279
+ explained = explained_redactions(conn)
280
+ unexplained = sorted(
281
+ r["seq"] for r in conn.execute("SELECT seq FROM audit_log WHERE redacted_at IS NOT NULL")
282
+ if r["seq"] not in explained)
283
+ if unexplained:
284
+ return {**result, "ok": False, "brokenAt": unexplained[0], "unexplained": unexplained,
285
+ "detail": f"{len(unexplained)} entr"
286
+ f"{'y is' if len(unexplained) == 1 else 'ies are'} marked redacted with "
287
+ "no `audit.redact` entry authorising it. A redaction skips the content "
288
+ "check, so an unexplained one is how an edited row would be made to "
289
+ f"verify clean. First: {unexplained[0]}."}
290
+ return {**result, "unexplained": []}
291
+
292
+
293
+ def verify_rows(rows) -> dict:
294
+ """The verification itself, over anything that yields entries in sequence order.
295
+
296
+ Split out so a *backup* is checked by the same definition as the live table (`backup.verify`
297
+ reads the dumped rows through here). Two implementations of "does this chain hold" is two
298
+ answers waiting to disagree, on the one question where disagreement is the whole point. Rows
299
+ may be sqlite3.Row, psycopg rows or plain dicts — all three index by column name.
300
+ """
301
+ previous = GENESIS
302
+ checked = 0
303
+ redacted = 0
304
+ for row in rows:
305
+ if row["prev_hash"] != previous:
306
+ return {"ok": False, "checked": checked, "redacted": redacted, "brokenAt": row["seq"],
307
+ "detail": "entry does not follow the one before it — a row was removed or reordered"}
308
+ if is_redacted(row):
309
+ # A redacted entry's *content* is gone, so its hash cannot be recomputed — and that is
310
+ # the honest report, not a silent pass. What survives is the property that matters:
311
+ # every later entry's hash covers this row's recorded hash, so the value accepted here
312
+ # cannot be changed without breaking the whole remainder of the chain. The entry is
313
+ # still proof that something happened at this position; it is no longer proof of what.
314
+ #
315
+ # This is why `redact()` writes its own audit entry. Redaction would otherwise be a way
316
+ # to make tampering verify clean: the erasure is itself in the chain, naming who did it
317
+ # and why, so a redaction used to hide something leaves the record of the hiding.
318
+ redacted += 1
319
+ elif entry_hash(previous, _payload_of(row)) != row["hash"]:
320
+ return {"ok": False, "checked": checked, "redacted": redacted, "brokenAt": row["seq"],
321
+ "detail": "entry contents do not match its hash — a row was edited"}
322
+ previous = row["hash"]
323
+ checked += 1
324
+ return {"ok": True, "checked": checked, "redacted": redacted, "brokenAt": None,
325
+ "detail": "" if not redacted else
326
+ f"{redacted} of {checked} entries are redacted: the chain is intact and each "
327
+ "redaction is itself recorded in it, but the content of those entries cannot "
328
+ "be verified because it no longer exists."}
329
+
330
+
331
+ # Fields blanked by `redact`. `at`, `action`, `outcome` and `environment` deliberately survive:
332
+ # erasure must remove what identifies a person, not the fact that an action occurred — a log that
333
+ # forgets an apply happened is not a retention policy, it is a gap in the record.
334
+ REDACTED_FIELDS = ("actor", "actor_id", "target", "reason", "detail", "source_ip")
335
+ REDACTION_MARK = "[redacted]"
336
+
337
+
338
+ def is_redacted(row) -> bool:
339
+ keys = row.keys() if hasattr(row, "keys") else row
340
+ return "redacted_at" in keys and bool(row["redacted_at"])
341
+
342
+
343
+ def redact(conn, seq: int, *, actor: str, reason: str) -> dict:
344
+ """Erase one entry's identifying content, keeping its place in the chain.
345
+
346
+ Used for a right-to-erasure request. It is not deletion: the row stays, so `prev_hash` still
347
+ links its neighbours and nothing about the sequence changes. See `verify_rows` for what
348
+ survives and what does not.
349
+
350
+ A reason is required, and the redaction is appended to the log as its own entry before the
351
+ content goes — if that append fails, nothing is erased.
352
+ """
353
+ row = conn.execute("SELECT * FROM audit_log WHERE seq = ?", (seq,)).fetchone()
354
+ if row is None:
355
+ return {"ok": False, "detail": f"no audit entry {seq}"}
356
+ if is_redacted(row):
357
+ return {"ok": False, "detail": f"entry {seq} is already redacted"}
358
+ if not reason:
359
+ return {"ok": False, "detail": "a redaction must say why — an unexplained erasure in an "
360
+ "audit log is indistinguishable from tampering"}
361
+ record(conn, actor=actor, action="audit.redact", outcome="succeeded", target=str(seq),
362
+ reason=reason, detail={"originalAction": row["action"], "at": row["at"]})
363
+ conn.execute(
364
+ "UPDATE audit_log SET actor = ?, actor_id = ?, target = ?, reason = ?, detail = ?, "
365
+ "source_ip = ?, redacted_at = ? WHERE seq = ?",
366
+ (REDACTION_MARK, "", REDACTION_MARK, REDACTION_MARK, "{}", "", time.time(), seq),
367
+ )
368
+ conn.commit()
369
+ return {"ok": True, "seq": seq,
370
+ "detail": f"entry {seq} redacted; its hash remains bound into every later entry"}
371
+
372
+
373
+ def verify_segment(entries) -> dict:
374
+ """Verify a *subset* of the log — each entry against its own recorded predecessor hash.
375
+
376
+ Different from `verify_rows`, and the difference is not a detail. `verify_rows` walks from
377
+ GENESIS and requires each entry to follow the last, which is the right check for a whole chain
378
+ and is *guaranteed to fail* on a subset: a run's entries are interleaved with everyone else's,
379
+ so the first one's `prev_hash` is some other actor's entry, not the genesis. Used on a segment
380
+ it reports "broken" for every honest bundle, which is worse than no check — it would teach
381
+ people that the verdict means nothing.
382
+
383
+ What is checkable about a subset: each entry's hash is the SHA-256 of its recorded predecessor
384
+ hash plus its own contents, so any edit to any field breaks that entry. What is *not* checkable
385
+ is completeness — nothing here can show an entry was removed, because the neighbours that would
386
+ have revealed the gap are not in the subset. The bundle says so rather than implying otherwise.
387
+ """
388
+ broken = []
389
+ for row in entries:
390
+ if entry_hash(row["prev_hash"] or GENESIS, _payload_of(row)) != row["hash"]:
391
+ broken.append(row["seq"])
392
+ return {
393
+ "ok": not broken,
394
+ "checked": len(entries),
395
+ "brokenAt": broken[0] if broken else None,
396
+ "detail": "" if not broken else
397
+ f"{len(broken)} entr{'y' if len(broken) == 1 else 'ies'} do not match their hash: "
398
+ + ", ".join(str(s) for s in broken[:5]),
399
+ }
400
+
401
+
402
+ def read(conn, limit: int = 200, actor: str = "", action: str = "") -> list[dict]:
403
+ sql = "SELECT * FROM audit_log WHERE 1=1"
404
+ params: list[object] = []
405
+ if actor:
406
+ sql += " AND actor = ?"
407
+ params.append(actor)
408
+ if action:
409
+ sql += " AND action LIKE ?"
410
+ params.append(f"{action}%")
411
+ sql += " ORDER BY seq DESC LIMIT ?"
412
+ params.append(max(1, min(limit, 1000)))
413
+ return [{**_payload_of(r), "seq": r["seq"], "hash": r["hash"]} for r in conn.execute(sql, params)]
414
+
415
+
416
+ def read_for_target(conn, target: str, environment: str = "", limit: int = 100) -> list[dict]:
417
+ """Audit entries about one thing — a workflow, an incident, an account.
418
+
419
+ Separate from `read()` because the question is different: `read()` answers "what has been
420
+ happening", filtered by who or what kind; this answers "what has happened *to this*", which is
421
+ what an activity timeline is. An entry recorded before the environment column carried a value
422
+ keeps showing, the same tolerance `list_runs` already has for runs that predate it.
423
+ """
424
+ sql = "SELECT * FROM audit_log WHERE target = ?"
425
+ params: list[object] = [target]
426
+ if environment:
427
+ sql += " AND (environment = ? OR environment = '' OR environment IS NULL)"
428
+ params.append(environment)
429
+ sql += " ORDER BY seq DESC LIMIT ?"
430
+ params.append(max(1, min(limit, 500)))
431
+ return [{**_payload_of(r), "seq": r["seq"]} for r in conn.execute(sql, params)]
432
+
433
+
434
+ def seal(conn, archive_dir="audit-archive") -> dict:
435
+ """Close the current chain, preserve it, and start a clean one.
436
+
437
+ Needed because a chain that has already forked cannot be repaired. Rewriting the broken segment
438
+ to make it verify is precisely the tampering the chain exists to detect, so the only honest
439
+ moves are to leave it broken or to draw a line and start again. In production the first is not
440
+ a choice: an audit log that permanently answers "ok: false" fails its own check, and nobody can
441
+ tell a historical concurrency bug from a live intrusion by looking at it.
442
+
443
+ So: export everything, move it to `audit_log_archive`, and begin a new chain whose first entry
444
+ records the seal — including the verdict at the moment of sealing, so the break is on the record
445
+ rather than quietly disappearing. Nothing is deleted.
446
+
447
+ Returns the manifest. The archive file is the evidence; `export_jsonl` already writes the format
448
+ a SIEM or WORM store wants, and getting entries out of reach of the process that writes them is
449
+ the only thing that makes them truly immutable.
450
+ """
451
+ before = verify_chain(conn)
452
+ rows = conn.execute("SELECT count(*) AS n FROM audit_log").fetchone()
453
+ total = rows["n"] if rows else 0
454
+
455
+ directory = pathlib.Path(archive_dir)
456
+ directory.mkdir(parents=True, exist_ok=True)
457
+ stamp = time.strftime("%Y%m%dT%H%M%SZ", time.gmtime())
458
+ path = directory / f"audit-{stamp}.jsonl"
459
+ body = export_jsonl(conn)
460
+ path.write_text(body + ("\n" if body else ""))
461
+
462
+ manifest = {
463
+ "sealedAt": time.time(),
464
+ "entries": total,
465
+ "archive": str(path),
466
+ "archiveSha256": hashlib.sha256(body.encode()).hexdigest(),
467
+ "verdictAtSeal": {k: before.get(k) for k in ("ok", "checked", "brokenAt", "detail")},
468
+ }
469
+ (directory / f"audit-{stamp}.manifest.json").write_text(json.dumps(manifest, indent=2, sort_keys=True) + "\n")
470
+
471
+ conn.execute(
472
+ "INSERT INTO audit_log_archive (seq, at, actor, actor_id, action, environment, target, outcome, "
473
+ "reason, detail, source_ip, prev_hash, hash, sealed_at) "
474
+ "SELECT seq, at, actor, actor_id, action, environment, target, outcome, reason, detail, "
475
+ f"source_ip, prev_hash, hash, {manifest['sealedAt']} FROM audit_log"
476
+ )
477
+ conn.execute("DELETE FROM audit_log")
478
+ conn.commit()
479
+
480
+ # Only now can the invariant be stated to the database: the table is empty, so there is nothing
481
+ # left for a unique index on prev_hash to collide with.
482
+ try:
483
+ conn.execute("CREATE UNIQUE INDEX IF NOT EXISTS audit_prev_unique ON audit_log(prev_hash)")
484
+ conn.commit()
485
+ manifest["uniqueIndex"] = True
486
+ except Exception as exc: # noqa: BLE001
487
+ conn.rollback()
488
+ manifest["uniqueIndex"] = False
489
+ manifest["uniqueIndexError"] = type(exc).__name__
490
+
491
+ # The new chain's first entry says why it starts here. A log that simply began one day, with no
492
+ # explanation, is the shape a cover-up takes.
493
+ record(
494
+ conn,
495
+ actor="system",
496
+ action="audit.seal",
497
+ outcome="succeeded",
498
+ reason="previous chain archived and a new one started",
499
+ detail={
500
+ "archivedEntries": total,
501
+ "archive": str(path),
502
+ "archiveSha256": manifest["archiveSha256"],
503
+ "verdictAtSeal": manifest["verdictAtSeal"],
504
+ },
505
+ )
506
+ return manifest
507
+
508
+
509
+ def export_jsonl(conn) -> str:
510
+ """The whole chain as JSON Lines, for shipping somewhere the application cannot rewrite.
511
+
512
+ Tamper-evidence proves a local log was altered; it cannot stop it. Getting entries out of reach
513
+ of the process that writes them is the only thing that does, and this is the seam for that —
514
+ a SIEM, an append-only bucket, a WORM store.
515
+ """
516
+ return "\n".join(
517
+ json.dumps({**_payload_of(r), "seq": r["seq"], "prev_hash": r["prev_hash"], "hash": r["hash"]}, sort_keys=True)
518
+ for r in conn.execute("SELECT * FROM audit_log ORDER BY seq ASC")
519
+ )