strata-agent-memory 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,12 @@
1
+ """agent-memory: portable, auditable, cross-vendor memory for LLM agents.
2
+
3
+ The unit of memory is a *claim* made by an identified author (human or agent,
4
+ with vendor lineage) inside a scope, bound to *evidence* that can be re-run.
5
+ Context handed to any agent is a *compiled view* over that ledger, and every
6
+ compilation is itself recorded so you can always answer: what did this agent
7
+ see, what did it claim, and can the proof be re-run.
8
+
9
+ The core is dependency-free (Python 3.10+ standard library only).
10
+ """
11
+
12
+ __version__ = "0.2.0"
@@ -0,0 +1,4 @@
1
+ from .cli import main
2
+ import sys
3
+
4
+ sys.exit(main())
agent_memory/audit.py ADDED
@@ -0,0 +1,345 @@
1
+ """Audit: the questions a human asks when something went wrong, answered from the ledger.
2
+
3
+ - What is verified, and by whom? Which verifications are same-vendor only, or come
4
+ from the claim's own author?
5
+ - Do cross-lineage verifiers refute more than same-lineage ones? (lineage_matrix)
6
+ - Which outcome claims have receipts nobody has re-run?
7
+ - Which outcome claims cite a receipt whose command never succeeded?
8
+ - Which outcome claims rest on static evidence only (lint clean, never executed)?
9
+ - Which receipts declare an assurance class stronger than their content backs?
10
+ - What has been refuted, and what was built on top of it (taint chains)?
11
+ - Which claims are stale because the world moved (files changed, commits gone)?
12
+ - How often do claims like this one turn out wrong, and how long does each class of
13
+ binding keep matching? (reliability, same_lineage_agreement, staleness,
14
+ judgment_calibration; docs/PHASE3-RELIABILITY.md)
15
+ - Which records carry a signature, and does every signature check out? (signatures;
16
+ a signature says who filed a record, never that it is true)
17
+ - How much of memory is human-authored vs agent-authored?
18
+ - Which contracts are open, and do their claims meet the bar?
19
+ - What did each agent session actually get shown (compilation receipts)?
20
+ """
21
+ from __future__ import annotations
22
+
23
+ import re
24
+ from collections import Counter, defaultdict
25
+ from typing import Dict, List, Optional
26
+
27
+ from . import evidence as ev
28
+ from . import reliability as rel
29
+ from . import signing
30
+ from .contracts import status_of
31
+ from .records import is_redacted
32
+ from .trust import TIERS, assess, lineage_differs, summarize
33
+
34
+ # Anything a compiled view would have to render on its own line. The renderer collapses
35
+ # these (review S5); the audit names the records that carried them, because a statement
36
+ # shaped like a document is a statement written to be read as one.
37
+ _CONTROL_RE = re.compile(r"[\x00-\x1f\x7f]")
38
+ _FORMATTED_FIELDS = ("statement", "rationale", "example")
39
+
40
+
41
+ def _formatting_problem(text: str) -> Optional[str]:
42
+ """The first line of ``text`` that could forge structure in a compiled view, or None."""
43
+ lines = text.splitlines()
44
+ if not _CONTROL_RE.search(text) and not any(line.lstrip().startswith("#") for line in lines):
45
+ return None
46
+ for line in lines:
47
+ if line.lstrip().startswith("#") or _CONTROL_RE.search(line):
48
+ return line.strip()[:120]
49
+ filled = [line for line in lines if line.strip()]
50
+ return (filled[-1] if filled else text).strip()[:120]
51
+
52
+
53
+ def _same_session_other_harness(claim_author: dict, verifier: dict) -> bool:
54
+ """True when a "cross-vendor" verification came out of the claim's own session.
55
+
56
+ Harness and vendor are self-asserted outside ``serve``'s token path (review S9), so
57
+ one shell that exports a different AGENT_MEMORY_HARNESS produces a verification that
58
+ counts as independent. Same user, same session, different harness is what that looks
59
+ like from the ledger, and it is free to compute.
60
+ """
61
+ session = claim_author.get("session")
62
+ return bool(
63
+ session and session != "unknown"
64
+ and verifier.get("session") == session
65
+ and verifier.get("user") == claim_author.get("user")
66
+ and verifier.get("harness") != claim_author.get("harness")
67
+ )
68
+
69
+
70
+ def report(ledger, recheck: bool = False, cwd: Optional[str] = None, timeout: Optional[float] = 300,
71
+ path_map=None) -> dict:
72
+ assessment = assess(ledger)
73
+ claims = ledger.claims()
74
+ by_id = {c["id"]: c for c in claims}
75
+ rep: Dict = {
76
+ "project": ledger.config.get("project"),
77
+ # whether this machine trusts the repository's config.json, and so whether its
78
+ # policy keys decided anything (review F-3, F-6)
79
+ "trusted": bool(getattr(ledger, "trusted", True)),
80
+ "records": Counter(r["kind"] for r in ledger.all()),
81
+ "tiers": summarize(assessment),
82
+ "authors": Counter(),
83
+ "human_vs_agent": Counter(),
84
+ "vendors": Counter(),
85
+ "unrechecked_outcomes": [],
86
+ "failing_receipt_outcomes": [],
87
+ "static_only_outcomes": [],
88
+ "assurance_overridden": [],
89
+ "same_vendor_only": [],
90
+ "self_verified_by_author": [],
91
+ "same_session_different_harness": [],
92
+ "formatted_statements": [],
93
+ "redactions": {"count": 0, "ids": []},
94
+ # absent and empty mean opposite things (an absent key allows every command, an
95
+ # empty list allows none), so the report carries both facts
96
+ "recheck_allow_set": "recheck_allow" in ledger.config,
97
+ "recheck_allow": list(ledger.config.get("recheck_allow") or []),
98
+ "lineage_matrix": {},
99
+ "same_lineage_refutation_rate": None,
100
+ "cross_lineage_refutation_rate": None,
101
+ "refuted": [],
102
+ "taint_chains": [],
103
+ "stale": [],
104
+ "asserted_facts": [],
105
+ "contracts": [],
106
+ "compilations": [],
107
+ "signatures": {},
108
+ "recheck": [],
109
+ }
110
+ # measured history, computed from the ledger alone: counts, intervals, and the word
111
+ # "insufficient history" wherever there are too few of them
112
+ measured = rel.report(ledger, assessment)
113
+ rep["reliability"] = measured["table"]
114
+ rep["same_lineage_agreement"] = measured["same_lineage_agreement"]
115
+ rep["staleness"] = measured["staleness"]
116
+ rep["judgment_calibration"] = measured["judgment_calibration"]
117
+ for c in claims:
118
+ a = c["author"]
119
+ rep["authors"][f"{a.get('harness')}:{a.get('user')}"] += 1
120
+ rep["human_vs_agent"][a.get("kind")] += 1
121
+ rep["vendors"][a.get("vendor")] += 1
122
+ info = assessment[c["id"]]
123
+ b = c["body"]
124
+ entry = {"id": c["id"], "type": b["type"], "statement": b["statement"][:160], "by": a.get("harness"), "vendor": a.get("vendor"), "ts": c["ts"]}
125
+ if b["type"] == "outcome" and info["tier"] == "evidenced":
126
+ rep["unrechecked_outcomes"].append(entry)
127
+ if b["type"] == "outcome":
128
+ # legacy ledgers can hold these; the filing rule rejects new ones (§4)
129
+ unmet = ev.unmet_expectations(ledger, c)
130
+ if unmet:
131
+ rep["failing_receipt_outcomes"].append({**entry, "receipts": unmet})
132
+ # a declared class above the computed default backs nothing, so this reads the
133
+ # receipts rather than the declaration
134
+ backed = ev.max_assurance(ev.effective_assurance(eb) for _, eb in ev.resolve_evidence(ledger, c))
135
+ if backed in ("static", "attested"):
136
+ # a clean static gate is not evidence that the thing runs
137
+ rep["static_only_outcomes"].append({**entry, "assurance": backed})
138
+ if info["tier"] == "self-verified":
139
+ rep["same_vendor_only"].append({**entry, "verified_by": info["verified_by"]})
140
+ if info.get("self_verified_by_author"):
141
+ rep["self_verified_by_author"].append({**entry, "verified_by": info["verified_by"]})
142
+ if info["tier"] == "refuted":
143
+ # sorted: dependents_of returns a set, and an audit two runs cannot reproduce
144
+ # byte for byte is not the audit this ledger promises
145
+ deps = sorted(d for d in ledger.dependents_of(c["id"], through_kinds=("claim",)) if d in by_id)
146
+ rep["refuted"].append({**entry, "reasons": info["reasons"], "dependents": len(deps)})
147
+ if deps:
148
+ rep["taint_chains"].append({"root": c["id"], "tainted": [{"id": d, "statement": by_id[d]["body"]["statement"][:120]} for d in deps]})
149
+ if info["tier"] == "stale":
150
+ rep["stale"].append({**entry, "reasons": info["reasons"]})
151
+ if info["tier"] == "asserted" and b["type"] == "fact":
152
+ rep["asserted_facts"].append(entry)
153
+ for field in _FORMATTED_FIELDS:
154
+ value = b.get(field)
155
+ problem = _formatting_problem(value) if isinstance(value, str) else None
156
+ if problem:
157
+ rep["formatted_statements"].append({**entry, "field": field, "line": problem})
158
+ break
159
+ # assurance is self-declared, so say which receipts declared more than they can back
160
+ cited_by: Dict[str, List[str]] = defaultdict(list)
161
+ for c in claims:
162
+ for item in c["body"].get("evidence", []):
163
+ if isinstance(item, str):
164
+ cited_by[item].append(c["id"])
165
+ for rec in ledger.by_kind("evidence"):
166
+ b = rec["body"]
167
+ if not ev.assurance_is_declared(b):
168
+ continue
169
+ rep["assurance_overridden"].append({
170
+ "id": rec["id"], "kind": b.get("kind"), "declared": b.get("assurance"),
171
+ "computed": ev.default_assurance(b), "by": rec["author"].get("harness"),
172
+ "what": str(b.get("cmd") or b.get("path") or b.get("url") or b.get("sha") or "")[:120],
173
+ "claims": sorted(cited_by.get(rec["id"], [])),
174
+ })
175
+ # verifier lineage: the refutation rate of same-lineage vs cross-lineage review is
176
+ # the number that decides whether the cross-vendor rule earns its keep
177
+ same_counts = Counter()
178
+ cross_counts = Counter()
179
+ for v in ledger.by_kind("verification"):
180
+ claim = by_id.get(v["body"].get("claim"))
181
+ if claim is None:
182
+ continue
183
+ verdict = v["body"]["verdict"]
184
+ key = f"{claim['author'].get('vendor')}->{v['author'].get('vendor')}"
185
+ cell = rep["lineage_matrix"].setdefault(key, {"verified": 0, "refuted": 0, "inconclusive": 0})
186
+ cell[verdict] += 1
187
+ (cross_counts if lineage_differs(claim["author"], v["author"]) else same_counts)[verdict] += 1
188
+ if _same_session_other_harness(claim["author"], v["author"]):
189
+ cb = claim["body"]
190
+ rep["same_session_different_harness"].append({
191
+ "id": claim["id"], "verification": v["id"], "verdict": verdict,
192
+ "statement": cb["statement"][:160], "by": claim["author"].get("harness"),
193
+ "verified_by": v["author"].get("harness"), "user": v["author"].get("user"),
194
+ "session": v["author"].get("session"),
195
+ })
196
+ for name, counts in (("same_lineage_refutation_rate", same_counts), ("cross_lineage_refutation_rate", cross_counts)):
197
+ decisive = counts["verified"] + counts["refuted"] # inconclusive is not a verdict
198
+ rep[name] = (counts["refuted"] / decisive) if decisive else None
199
+ for rec in ledger.by_kind("contract"):
200
+ st = status_of(ledger, rec["id"], assessment)
201
+ tiers = Counter(c["tier"] for c in st["claims"])
202
+ rep["contracts"].append({"id": st["id"], "title": st["title"], "status": st["status"], "claims": len(st["claims"]), "tiers": dict(tiers)})
203
+ for rec in ledger.by_kind("compilation"):
204
+ b = rec["body"]
205
+ rep["compilations"].append({"id": rec["id"], "ts": rec["ts"], "role": b["role"], "mode": b["mode"], "reader": (b.get("reader") or {}).get("harness"),
206
+ "included": len(b["included"]), "excluded": len(b["excluded"]), "tokens": b.get("used_tokens")})
207
+ # what was erased. The tombstones are still here, which is the point: an audit can
208
+ # say that something was removed, by whom and why (review C1).
209
+ redacted = [r for r in ledger.all() if is_redacted(r)]
210
+ rep["redactions"] = {
211
+ "count": len(redacted),
212
+ "ids": [r["id"] for r in redacted],
213
+ "items": [{"id": r["id"], "kind": r["kind"], "reason": str(r["body"].get("reason", ""))[:120],
214
+ "by": (r["body"].get("redacted_by") or {}).get("user"),
215
+ "at": r["body"].get("redacted_at")} for r in redacted],
216
+ }
217
+ # who signed what. Verification needs the optional extra, so on a machine without it
218
+ # every attestation is reported as unverifiable rather than as good or bad, and a
219
+ # signature never moves a claim between tiers either way (§5).
220
+ sig = signing.report(ledger)
221
+ attested_by_author = Counter()
222
+ for record_id in sig["by_record"]:
223
+ signed = ledger.get(record_id)
224
+ if signed is not None:
225
+ a = signed["author"]
226
+ attested_by_author[f"{a.get('harness')}:{a.get('user')}"] += 1
227
+ rep["signatures"] = {
228
+ "attested": sig["attested"], "valid": sig["valid"], "invalid": sig["invalid"],
229
+ "unknown": sig["unknown"], "unavailable": sig["unavailable"],
230
+ "invalid_items": [i for i in sig["items"] if i["status"] != "valid"],
231
+ "by_author": dict(attested_by_author),
232
+ }
233
+ if recheck:
234
+ for c in claims:
235
+ info = assessment[c["id"]]
236
+ if info["tier"] in ("refuted", "tainted"):
237
+ continue
238
+ unmet = set(ev.unmet_expectations(ledger, c))
239
+ for r in ev.recheck_claim(ledger, c, cwd=cwd, timeout=timeout, path_map=path_map):
240
+ if r["outcome"] == "skipped":
241
+ continue
242
+ entry = {"claim": c["id"], "evidence": r["evidence"], "outcome": r["outcome"], "detail": r["detail"][:200]}
243
+ if r.get("role"): # an environment mismatch is staleness, not a refutation
244
+ entry["role"] = r["role"]
245
+ if r["outcome"] == "match" and r["evidence"] in unmet:
246
+ entry["note"] = ev.FAILING_RECEIPT_NOTE # reproducing a failure verifies nothing
247
+ rep["recheck"].append(entry)
248
+ return rep
249
+
250
+
251
+ def format_report(rep: dict) -> str:
252
+ lines = [f"# Audit: {rep['project']}", ""]
253
+ lines.append("records: " + ", ".join(f"{k}={v}" for k, v in sorted(rep["records"].items())))
254
+ lines.append("tiers: " + ", ".join(f"{t}={rep['tiers'].get(t, 0)}" for t in TIERS))
255
+ lines.append("authors: " + ", ".join(f"{k}={v}" for k, v in rep["authors"].most_common()))
256
+ lines.append("kind: " + ", ".join(f"{k}={v}" for k, v in rep["human_vs_agent"].items()))
257
+ lines.append("vendors: " + ", ".join(f"{k}={v}" for k, v in rep["vendors"].most_common()))
258
+ if not rep.get("trusted", True):
259
+ lines.append("repo: UNTRUSTED -- config.json's policy keys (recheck_allow, "
260
+ "recheck_allow_shell, redact, path_map, org_path, signers, vendor_map, "
261
+ "cross_vendor_required) were ignored; run `agent-memory trust .` to honour them")
262
+ if not rep.get("recheck_allow_set"):
263
+ lines.append("warning: config.recheck_allow is not set: the re-check lane will run whatever "
264
+ "command a receipt records; run `agent-memory init` to write the starter list")
265
+ elif not rep.get("recheck_allow"):
266
+ # deliberate, and the opposite of the absent case: nothing is re-run here
267
+ lines.append("note: recheck_allow is empty: no receipt will be re-run on this machine")
268
+
269
+ def section(title: str, rows: List[dict], fmt):
270
+ lines.append("")
271
+ lines.append(f"## {title} ({len(rows)})")
272
+ for r in rows[:50]:
273
+ lines.append("- " + fmt(r))
274
+ if len(rows) > 50:
275
+ lines.append(f"- ... {len(rows) - 50} more")
276
+
277
+ section("Outcome claims with receipts nobody re-ran", rep["unrechecked_outcomes"],
278
+ lambda r: f"{r['id']} [{r['by']}] {r['statement']}")
279
+ section("Outcome claims whose receipts did not succeed", rep["failing_receipt_outcomes"],
280
+ lambda r: f"{r['id']} [{r['by']}] {r['statement']} -- receipts: {', '.join(r['receipts'])}")
281
+ section("Outcome claims backed only by static evidence (static-pass/dynamic-fail risk)",
282
+ rep["static_only_outcomes"], lambda r: f"{r['id']} [{r['by']}, {r['assurance']}] {r['statement']}")
283
+ section("Receipts whose assurance was declared above the computed default", rep["assurance_overridden"],
284
+ lambda r: f"{r['id']} [{r['by']}] declared {r['declared']}, computed {r['computed']}: {r['what']}"
285
+ + (f" (cited by {', '.join(r['claims'])})" if r["claims"] else " (cited by no claim)"))
286
+ section("Verified only by the same vendor (collusion risk)", rep["same_vendor_only"],
287
+ lambda r: f"{r['id']} [{r['vendor']} verified by {', '.join(r['verified_by'])}] {r['statement']}")
288
+ section("Verified only by the claim's own author", rep["self_verified_by_author"],
289
+ lambda r: f"{r['id']} [{r['by']}] {r['statement']}")
290
+ section("Verified from the claim's own session under another harness (collusion risk)",
291
+ rep.get("same_session_different_harness", []),
292
+ lambda r: f"{r['id']} [{r['by']} -> {r['verified_by']}, {r['user']} session {r['session']}] "
293
+ f"{r['statement']}")
294
+ section("Statements shaped like a document (injection risk)", rep.get("formatted_statements", []),
295
+ lambda r: f"{r['id']} [{r['by']}] {r['field']} starts a line with: {r['line']}")
296
+ section("Refuted", rep["refuted"], lambda r: f"{r['id']} {r['statement']} -- {'; '.join(r['reasons'])} (dependents: {r['dependents']})")
297
+ section("Taint chains", rep["taint_chains"], lambda r: f"{r['root']} -> " + ", ".join(t['id'] for t in r['tainted']))
298
+ section("Stale", rep["stale"], lambda r: f"{r['id']} {r['statement']} -- {'; '.join(r['reasons'])}")
299
+ section("Facts with no evidence (asserted)", rep["asserted_facts"], lambda r: f"{r['id']} [{r['by']}] {r['statement']}")
300
+ section("Contracts", rep["contracts"], lambda r: f"{r['id']} {r['status']}: {r['title']} ({r['claims']} claims {r['tiers']})")
301
+ section("Compilation receipts", rep["compilations"],
302
+ lambda r: f"{r['id']} {r['ts']} {r['role']}/{r['mode']} for {r['reader']}: {r['included']} in, {r['excluded']} out, {r['tokens']} tok")
303
+ red = rep.get("redactions") or {}
304
+ if red.get("count"):
305
+ lines.append("")
306
+ lines.append(f"## Redacted records ({red['count']})")
307
+ for item in (red.get("items") or [{"id": i} for i in red.get("ids", [])])[:50]:
308
+ lines.append(f"- {item['id']}" + (f" [{item.get('kind')}] by {item.get('by')} "
309
+ f"at {item.get('at')}: {item.get('reason')}"
310
+ if item.get("reason") is not None else ""))
311
+ sig = rep.get("signatures") or {}
312
+ if sig.get("attested"):
313
+ lines.append("")
314
+ lines.append(f"## Signatures ({sig['attested']})")
315
+ lines.append(f"- valid={sig.get('valid', 0)} invalid={sig.get('invalid', 0)} "
316
+ f"unknown_key={sig.get('unknown', 0)} unverifiable={sig.get('unavailable', 0)}")
317
+ if sig.get("by_author"):
318
+ lines.append("- attested records by author: "
319
+ + ", ".join(f"{k}={v}" for k, v in sorted(sig["by_author"].items())))
320
+ for item in sig.get("invalid_items", [])[:50]:
321
+ lines.append(f"- {item['id']} over {item['record']} [{item['key_id']}]: {item['status']} -- "
322
+ f"{item['detail']}")
323
+ lines.append("")
324
+ lines.append(f"## Verifier lineage (refutation rates) ({len(rep['lineage_matrix'])})")
325
+ for key, cell in sorted(rep["lineage_matrix"].items()):
326
+ total = cell["verified"] + cell["refuted"]
327
+ rate = f"{cell['refuted'] / total:.2f}" if total else "n/a"
328
+ lines.append(f"- {key:<32} verified={cell['verified']} refuted={cell['refuted']} "
329
+ f"inconclusive={cell['inconclusive']} refutation_rate={rate}")
330
+ def _rate(value):
331
+ return "n/a" if value is None else f"{value:.2f}"
332
+ lines.append(f"- same-lineage refutation rate: {_rate(rep['same_lineage_refutation_rate'])} | "
333
+ f"cross-lineage refutation rate: {_rate(rep['cross_lineage_refutation_rate'])}")
334
+ if "reliability" in rep:
335
+ lines.append("")
336
+ lines.append(rel.format_report({"table": rep["reliability"],
337
+ "same_lineage_agreement": rep["same_lineage_agreement"],
338
+ "staleness": rep["staleness"],
339
+ "judgment_calibration": rep["judgment_calibration"]}).rstrip("\n"))
340
+ if rep["recheck"]:
341
+ section("Evidence re-check", rep["recheck"],
342
+ lambda r: f"{r['claim']} {r['evidence']}: {r['outcome']} -- {r['detail']}"
343
+ + (" [environment binding: staleness, not refutation]" if r.get("role") == "environment" else "")
344
+ + (f" [{r['note']}]" if r.get("note") else ""))
345
+ return "\n".join(lines) + "\n"