agentmetry 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. agentmetry/__init__.py +12 -0
  2. agentmetry/api/__init__.py +0 -0
  3. agentmetry/api/main.py +232 -0
  4. agentmetry/api/routes/__init__.py +0 -0
  5. agentmetry/api/routes/audit.py +396 -0
  6. agentmetry/api/websocket.py +53 -0
  7. agentmetry/api/ws_bridge.py +34 -0
  8. agentmetry/cli/__init__.py +941 -0
  9. agentmetry/cli/__main__.py +5 -0
  10. agentmetry/core/__init__.py +0 -0
  11. agentmetry/core/audit/__init__.py +1 -0
  12. agentmetry/core/audit/adapters/__init__.py +0 -0
  13. agentmetry/core/audit/adapters/agt.py +304 -0
  14. agentmetry/core/audit/adapters/cloudevents.py +159 -0
  15. agentmetry/core/audit/adapters/ecs.py +103 -0
  16. agentmetry/core/audit/adapters/splunk.py +40 -0
  17. agentmetry/core/audit/alerts.py +56 -0
  18. agentmetry/core/audit/canonical.py +150 -0
  19. agentmetry/core/audit/compliance_digest.py +299 -0
  20. agentmetry/core/audit/detection/__init__.py +9 -0
  21. agentmetry/core/audit/detection/benchmark.py +194 -0
  22. agentmetry/core/audit/detection/corpus/attack_approval_denied_then_executed.jsonl +3 -0
  23. agentmetry/core/audit/detection/corpus/attack_arbitrary_host_stage_execute.jsonl +3 -0
  24. agentmetry/core/audit/detection/corpus/attack_autonomous_unapproved_write.jsonl +3 -0
  25. agentmetry/core/audit/detection/corpus/attack_credential_exfil.jsonl +2 -0
  26. agentmetry/core/audit/detection/corpus/attack_credential_then_cloud_api.jsonl +2 -0
  27. agentmetry/core/audit/detection/corpus/attack_destructive_delete_burst.jsonl +6 -0
  28. agentmetry/core/audit/detection/corpus/attack_discovery_then_collect.jsonl +5 -0
  29. agentmetry/core/audit/detection/corpus/attack_dotfile_then_git_push.jsonl +2 -0
  30. agentmetry/core/audit/detection/corpus/attack_encoded_command_download.jsonl +1 -0
  31. agentmetry/core/audit/detection/corpus/attack_env_credential_exfil.jsonl +3 -0
  32. agentmetry/core/audit/detection/corpus/attack_hashed_only_no_command.jsonl +2 -0
  33. agentmetry/core/audit/detection/corpus/attack_interpreter_egress.jsonl +2 -0
  34. agentmetry/core/audit/detection/corpus/attack_pr_merged_without_review.jsonl +2 -0
  35. agentmetry/core/audit/detection/corpus/attack_proc_substitution_cradle.jsonl +2 -0
  36. agentmetry/core/audit/detection/corpus/attack_remote_pipe_to_shell.jsonl +1 -0
  37. agentmetry/core/audit/detection/corpus/attack_remote_staging_then_execute.jsonl +2 -0
  38. agentmetry/core/audit/detection/corpus/attack_session_tool_burst.jsonl +42 -0
  39. agentmetry/core/audit/detection/corpus/attack_single_command_exfil.jsonl +2 -0
  40. agentmetry/core/audit/detection/corpus/attack_ssh_directory_exfil.jsonl +3 -0
  41. agentmetry/core/audit/detection/corpus/attack_subagent_swarm.jsonl +6 -0
  42. agentmetry/core/audit/detection/corpus/attack_timestamp_collision.jsonl +2 -0
  43. agentmetry/core/audit/detection/corpus/attack_untrusted_input_then_action.jsonl +3 -0
  44. agentmetry/core/audit/detection/corpus/benign_authoring_merge_fixtures.jsonl +3 -0
  45. agentmetry/core/audit/detection/corpus/benign_autonomous_after_approval.jsonl +4 -0
  46. agentmetry/core/audit/detection/corpus/benign_build_artifact_cleanup.jsonl +5 -0
  47. agentmetry/core/audit/detection/corpus/benign_ci_artifact_download.jsonl +3 -0
  48. agentmetry/core/audit/detection/corpus/benign_database_migration.jsonl +5 -0
  49. agentmetry/core/audit/detection/corpus/benign_dependency_install_and_build.jsonl +5 -0
  50. agentmetry/core/audit/detection/corpus/benign_download_release_archive.jsonl +4 -0
  51. agentmetry/core/audit/detection/corpus/benign_fetch_data_then_run_repo_script.jsonl +3 -0
  52. agentmetry/core/audit/detection/corpus/benign_fetch_dataset_then_analyse.jsonl +3 -0
  53. agentmetry/core/audit/detection/corpus/benign_fetch_lockfile_then_install.jsonl +3 -0
  54. agentmetry/core/audit/detection/corpus/benign_git_review_and_push.jsonl +6 -0
  55. agentmetry/core/audit/detection/corpus/benign_human_driven_deletes.jsonl +6 -0
  56. agentmetry/core/audit/detection/corpus/benign_local_api_probing.jsonl +5 -0
  57. agentmetry/core/audit/detection/corpus/benign_long_but_calm_session.jsonl +30 -0
  58. agentmetry/core/audit/detection/corpus/benign_loopback_is_not_egress.jsonl +2 -0
  59. agentmetry/core/audit/detection/corpus/benign_loopback_pipe_to_interpreter.jsonl +3 -0
  60. agentmetry/core/audit/detection/corpus/benign_ordinary_development.jsonl +5 -0
  61. agentmetry/core/audit/detection/corpus/benign_package_manager_after_fetch.jsonl +2 -0
  62. agentmetry/core/audit/detection/corpus/benign_reading_config_that_is_not_secret.jsonl +5 -0
  63. agentmetry/core/audit/detection/corpus/benign_remote_api_call_no_credentials.jsonl +4 -0
  64. agentmetry/core/audit/detection/corpus/benign_research_then_docs.jsonl +5 -0
  65. agentmetry/core/audit/detection/corpus/benign_reversed_order_is_not_exfil.jsonl +2 -0
  66. agentmetry/core/audit/detection/corpus/benign_test_and_fix_loop.jsonl +6 -0
  67. agentmetry/core/audit/detection/corpus/benign_writing_about_credentials.jsonl +6 -0
  68. agentmetry/core/audit/detection/corpus/corpus.yaml +443 -0
  69. agentmetry/core/audit/detection/disposition.py +651 -0
  70. agentmetry/core/audit/detection/engine.py +78 -0
  71. agentmetry/core/audit/detection/live.py +127 -0
  72. agentmetry/core/audit/detection/live_store.py +355 -0
  73. agentmetry/core/audit/detection/models.py +53 -0
  74. agentmetry/core/audit/detection/rules.py +1314 -0
  75. agentmetry/core/audit/detection/traits.py +648 -0
  76. agentmetry/core/audit/detection/yaml_config.py +91 -0
  77. agentmetry/core/audit/detection/yaml_rules.py +83 -0
  78. agentmetry/core/audit/dlp/__init__.py +4 -0
  79. agentmetry/core/audit/dlp/loader.py +29 -0
  80. agentmetry/core/audit/dlp/models.py +29 -0
  81. agentmetry/core/audit/dlp/scanner.py +96 -0
  82. agentmetry/core/audit/dogfood.py +398 -0
  83. agentmetry/core/audit/evidence_pack.py +500 -0
  84. agentmetry/core/audit/external.py +213 -0
  85. agentmetry/core/audit/hashing.py +21 -0
  86. agentmetry/core/audit/hook_bootstrap.py +451 -0
  87. agentmetry/core/audit/identity.py +39 -0
  88. agentmetry/core/audit/ingest.py +242 -0
  89. agentmetry/core/audit/migrate.py +73 -0
  90. agentmetry/core/audit/mitre.py +244 -0
  91. agentmetry/core/audit/policy.py +99 -0
  92. agentmetry/core/audit/redaction.py +50 -0
  93. agentmetry/core/audit/replay.py +54 -0
  94. agentmetry/core/audit/run_context.py +129 -0
  95. agentmetry/core/audit/sinks.py +235 -0
  96. agentmetry/core/audit/spool.py +394 -0
  97. agentmetry/core/audit/tool_policy/__init__.py +4 -0
  98. agentmetry/core/audit/tool_policy/evaluator.py +198 -0
  99. agentmetry/core/audit/tool_policy/loader.py +44 -0
  100. agentmetry/core/audit/tool_policy/models.py +25 -0
  101. agentmetry/core/audit/trail_chain.py +300 -0
  102. agentmetry/core/audit/trail_db.py +491 -0
  103. agentmetry/core/audit/trail_merkle.py +332 -0
  104. agentmetry/core/auth.py +54 -0
  105. agentmetry/core/bus/__init__.py +5 -0
  106. agentmetry/core/bus/audit_exporter.py +107 -0
  107. agentmetry/core/bus/bridges.py +26 -0
  108. agentmetry/core/bus/bus.py +102 -0
  109. agentmetry/core/bus/events.py +50 -0
  110. agentmetry/core/bus/outbox.py +124 -0
  111. agentmetry/core/config.py +177 -0
  112. agentmetry/core/diagnostics/__init__.py +0 -0
  113. agentmetry/core/diagnostics/autostart.py +563 -0
  114. agentmetry/core/diagnostics/doctor.py +535 -0
  115. agentmetry/core/diagnostics/driver_paths.py +156 -0
  116. agentmetry/core/diagnostics/env_file.py +45 -0
  117. agentmetry/core/drivers/__init__.py +4 -0
  118. agentmetry/core/drivers/host.py +263 -0
  119. agentmetry/core/drivers/permissions.py +37 -0
  120. agentmetry/core/drivers/spec.py +118 -0
  121. agentmetry/core/extensions.py +107 -0
  122. agentmetry/core/health.py +26 -0
  123. agentmetry/core/version.py +13 -0
  124. agentmetry/policies/detection/manifest.yaml +43 -0
  125. agentmetry/policies/dlp/manifest.yaml +161 -0
  126. agentmetry/policies/opa/agent_rules.rego +33 -0
  127. agentmetry/policies/tool/manifest.yaml +117 -0
  128. agentmetry-0.4.0.dist-info/METADATA +86 -0
  129. agentmetry-0.4.0.dist-info/RECORD +131 -0
  130. agentmetry-0.4.0.dist-info/WHEEL +4 -0
  131. agentmetry-0.4.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,491 @@
1
+ """SQLite WAL-backed audit event store.
2
+
3
+ Replaces the read-the-whole-JSONL-file-into-RAM pattern with indexed queries.
4
+ The JSONL sinks remain for SIEM forwarding; this module handles *queries*.
5
+
6
+ WAL mode gives concurrent readers + one writer without blocking — perfect for
7
+ the "one ingest writer, many dashboard readers" access pattern.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import json
13
+ import logging
14
+ import sqlite3
15
+ import threading
16
+ from datetime import datetime, timedelta, timezone
17
+ from pathlib import Path
18
+ from typing import Any, Literal
19
+
20
+ logger = logging.getLogger(__name__)
21
+
22
+ # Legacy source names that should be normalized on read.
23
+ _LEGACY_SOURCE_APPS = frozenset({"blackbox"})
24
+
25
+ _RUN_ACTION_TYPES = frozenset({
26
+ "session_start",
27
+ "session_end",
28
+ "tool_called",
29
+ "approval_request",
30
+ "approval_response",
31
+ "detection",
32
+ # The human's answer to a detection belongs in the same feed as the
33
+ # detection, or the decision is invisible where it matters most.
34
+ "detection_disposition",
35
+ })
36
+
37
+ _SCHEMA_SQL = """
38
+ CREATE TABLE IF NOT EXISTS audit_events (
39
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
40
+ event_id TEXT UNIQUE NOT NULL,
41
+ correlation_id TEXT NOT NULL DEFAULT '',
42
+ session_id TEXT NOT NULL DEFAULT '',
43
+ timestamp_utc TEXT NOT NULL DEFAULT '',
44
+ action_type TEXT NOT NULL DEFAULT '',
45
+ action_outcome TEXT NOT NULL DEFAULT '',
46
+ source_app TEXT NOT NULL DEFAULT '',
47
+ tool_qualified TEXT NOT NULL DEFAULT '',
48
+ event_json TEXT NOT NULL
49
+ );
50
+
51
+ CREATE INDEX IF NOT EXISTS idx_events_ts ON audit_events(timestamp_utc);
52
+ CREATE INDEX IF NOT EXISTS idx_events_corr ON audit_events(correlation_id);
53
+ CREATE INDEX IF NOT EXISTS idx_events_source ON audit_events(source_app);
54
+ CREATE INDEX IF NOT EXISTS idx_events_action ON audit_events(action_type);
55
+ CREATE INDEX IF NOT EXISTS idx_events_session ON audit_events(session_id);
56
+ """
57
+
58
+
59
+ def _normalize_source_app(name: str) -> str:
60
+ return "agentmetry" if name in _LEGACY_SOURCE_APPS else name
61
+
62
+
63
+ def _extract_source_app(event: dict[str, Any]) -> str:
64
+ source = event.get("source")
65
+ if isinstance(source, dict) and source.get("app"):
66
+ return _normalize_source_app(str(source["app"]).lower())
67
+ agent = event.get("agent")
68
+ if isinstance(agent, dict) and agent.get("name"):
69
+ name = _normalize_source_app(str(agent["name"]).lower())
70
+ if name != "agentmetry":
71
+ return name
72
+ return "agentmetry"
73
+
74
+
75
+ class AuditTrailDB:
76
+ """Thread-safe SQLite WAL audit store."""
77
+
78
+ def __init__(self, db_path: Path | str) -> None:
79
+ self._db_path = str(db_path)
80
+ self._local = threading.local()
81
+ # Ensure the directory exists.
82
+ Path(self._db_path).parent.mkdir(parents=True, exist_ok=True)
83
+ # Initialize schema on the calling thread.
84
+ self._init_schema()
85
+
86
+ def _get_conn(self) -> sqlite3.Connection:
87
+ conn = getattr(self._local, "conn", None)
88
+ if conn is None:
89
+ conn = sqlite3.connect(self._db_path, timeout=10)
90
+ conn.execute("PRAGMA journal_mode=WAL")
91
+ conn.execute("PRAGMA synchronous=NORMAL")
92
+ conn.execute("PRAGMA busy_timeout=5000")
93
+ conn.row_factory = sqlite3.Row
94
+ self._local.conn = conn
95
+ return conn
96
+
97
+ def _init_schema(self) -> None:
98
+ conn = self._get_conn()
99
+ conn.executescript(_SCHEMA_SQL)
100
+ conn.commit()
101
+
102
+ # ------------------------------------------------------------------
103
+ # Write
104
+ # ------------------------------------------------------------------
105
+
106
+ def insert(self, event: dict[str, Any]) -> None:
107
+ """Insert a canonical event. Silently skips duplicates (UNIQUE on event_id)."""
108
+ import uuid
109
+ event_id = event.get("event_id", "")
110
+ if not event_id:
111
+ event_id = str(uuid.uuid4())
112
+ event["event_id"] = event_id
113
+
114
+ action = event.get("action") or {}
115
+ tool = event.get("tool") or {}
116
+
117
+ try:
118
+ self._get_conn().execute(
119
+ """INSERT OR IGNORE INTO audit_events
120
+ (event_id, correlation_id, session_id, timestamp_utc,
121
+ action_type, action_outcome, source_app, tool_qualified, event_json)
122
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)""",
123
+ (
124
+ event_id,
125
+ str(event.get("correlation_id") or ""),
126
+ str(event.get("session_id") or ""),
127
+ str(event.get("timestamp_utc") or ""),
128
+ str(action.get("type") or ""),
129
+ str(action.get("outcome") or ""),
130
+ _extract_source_app(event),
131
+ str(tool.get("qualified") or ""),
132
+ json.dumps(event, default=str),
133
+ ),
134
+ )
135
+ self._get_conn().commit()
136
+ except sqlite3.Error as exc:
137
+ logger.warning("trail_db insert failed: %s", exc)
138
+
139
+ def insert_batch(self, events: list[dict[str, Any]]) -> int:
140
+ """Bulk insert for migration. Returns rows ACTUALLY inserted.
141
+
142
+ Counted via total_changes rather than attempts: INSERT OR IGNORE skips
143
+ duplicates, so counting loop iterations overstates the result and made
144
+ the backfill log claim work it had not done.
145
+ """
146
+ import uuid
147
+ conn = self._get_conn()
148
+ before = conn.total_changes
149
+ for event in events:
150
+ event_id = event.get("event_id", "")
151
+ if not event_id:
152
+ event_id = str(uuid.uuid4())
153
+ event["event_id"] = event_id
154
+ action = event.get("action") or {}
155
+ tool = event.get("tool") or {}
156
+ try:
157
+ conn.execute(
158
+ """INSERT OR IGNORE INTO audit_events
159
+ (event_id, correlation_id, session_id, timestamp_utc,
160
+ action_type, action_outcome, source_app, tool_qualified, event_json)
161
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)""",
162
+ (
163
+ event_id,
164
+ str(event.get("correlation_id") or ""),
165
+ str(event.get("session_id") or ""),
166
+ str(event.get("timestamp_utc") or ""),
167
+ str(action.get("type") or ""),
168
+ str(action.get("outcome") or ""),
169
+ _extract_source_app(event),
170
+ str(tool.get("qualified") or ""),
171
+ json.dumps(event, default=str),
172
+ ),
173
+ )
174
+ except sqlite3.Error:
175
+ continue
176
+ conn.commit()
177
+ return conn.total_changes - before
178
+
179
+ # ------------------------------------------------------------------
180
+ # Read
181
+ # ------------------------------------------------------------------
182
+
183
+ def tail(
184
+ self,
185
+ *,
186
+ limit: int = 50,
187
+ scope: Literal["runs", "all"] = "runs",
188
+ sources: set[str] | None = None,
189
+ session_id: str | None = None,
190
+ since_minutes: int | None = None,
191
+ before_utc: str | None = None,
192
+ after_utc: str | None = None,
193
+ focus: Literal["denied", "dlp", "policy", "detection"] | None = None,
194
+ ) -> tuple[list[dict[str, Any]], dict[str, Any]]:
195
+ """Query the tail of the audit trail with filters and pagination."""
196
+ clauses: list[str] = []
197
+ params: list[Any] = []
198
+
199
+ if scope == "runs":
200
+ placeholders = ",".join("?" for _ in _RUN_ACTION_TYPES)
201
+ clauses.append(f"action_type IN ({placeholders})")
202
+ params.extend(_RUN_ACTION_TYPES)
203
+
204
+ if sources:
205
+ placeholders = ",".join("?" for _ in sources)
206
+ clauses.append(f"source_app IN ({placeholders})")
207
+ params.extend(sources)
208
+
209
+ if session_id:
210
+ clauses.append("(session_id = ? OR source_app != 'agentmetry')")
211
+ params.append(session_id)
212
+
213
+ if since_minutes is not None and since_minutes > 0:
214
+ cutoff = (datetime.now(timezone.utc) - timedelta(minutes=since_minutes)).isoformat()
215
+ clauses.append("timestamp_utc >= ?")
216
+ params.append(cutoff)
217
+
218
+ # Analytics dogfood counts are over the whole window; a client that only
219
+ # loads the newest N rows will miss older DLP/policy hits. Focus pushes
220
+ # the same predicates the stats query uses down into SQL.
221
+ if focus == "denied":
222
+ clauses.append("action_outcome = 'denied'")
223
+ elif focus == "dlp":
224
+ clauses.append("json_extract(event_json, '$.dlp.rule_id') IS NOT NULL")
225
+ elif focus == "policy":
226
+ clauses.append("json_extract(event_json, '$.tool_policy.blocked') = 1")
227
+ elif focus == "detection":
228
+ clauses.append("action_type = 'detection'")
229
+ elif focus is not None:
230
+ raise ValueError(f"Unknown focus: {focus}")
231
+
232
+ if before_utc and after_utc:
233
+ raise ValueError("Use only one of before_utc or after_utc")
234
+
235
+ where = f"WHERE {' AND '.join(clauses)}" if clauses else ""
236
+
237
+ conn = self._get_conn()
238
+
239
+ if before_utc:
240
+ sql = f"""SELECT event_json FROM audit_events {where}
241
+ {"AND" if clauses else "WHERE"} timestamp_utc < ?
242
+ ORDER BY timestamp_utc DESC, id DESC LIMIT ?"""
243
+ params.extend([before_utc, limit + 1])
244
+ rows = conn.execute(sql, params).fetchall()
245
+ has_older = len(rows) > limit
246
+ page_rows = list(reversed(rows[:limit]))
247
+ has_newer = True
248
+ elif after_utc:
249
+ sql = f"""SELECT event_json FROM audit_events {where}
250
+ {"AND" if clauses else "WHERE"} timestamp_utc > ?
251
+ ORDER BY timestamp_utc ASC, id ASC LIMIT ?"""
252
+ params.extend([after_utc, limit + 1])
253
+ rows = conn.execute(sql, params).fetchall()
254
+ has_newer = len(rows) > limit
255
+ page_rows = rows[:limit]
256
+ has_older = True
257
+ else:
258
+ # Latest page: get last N rows
259
+ sql = f"""SELECT event_json FROM audit_events {where}
260
+ ORDER BY timestamp_utc DESC, id DESC LIMIT ?"""
261
+ params.append(limit + 1)
262
+ rows = conn.execute(sql, params).fetchall()
263
+ has_older = len(rows) > limit
264
+ page_rows = list(reversed(rows[:limit]))
265
+ has_newer = False
266
+
267
+ events = []
268
+ for row in page_rows:
269
+ try:
270
+ events.append(json.loads(row["event_json"]))
271
+ except (json.JSONDecodeError, KeyError):
272
+ continue
273
+
274
+ pagination = {
275
+ "has_older": has_older,
276
+ "has_newer": has_newer,
277
+ "oldest_utc": events[0].get("timestamp_utc") if events else None,
278
+ "newest_utc": events[-1].get("timestamp_utc") if events else None,
279
+ "count": len(events),
280
+ }
281
+ return events, pagination
282
+
283
+ def read_between(
284
+ self, start_utc: str, end_utc: str, *, limit: int = 200_000
285
+ ) -> list[dict[str, Any]]:
286
+ """Return canonical events in [start_utc, end_utc], time-ordered.
287
+
288
+ The range query behind evidence exports. Timestamps are ISO-8601 UTC
289
+ strings, which sort lexicographically, so a plain BETWEEN is correct.
290
+ """
291
+ conn = self._get_conn()
292
+ rows = conn.execute(
293
+ """SELECT event_json FROM audit_events
294
+ WHERE timestamp_utc >= ? AND timestamp_utc <= ?
295
+ ORDER BY timestamp_utc ASC, id ASC
296
+ LIMIT ?""",
297
+ (start_utc, end_utc, limit),
298
+ ).fetchall()
299
+ events: list[dict[str, Any]] = []
300
+ for row in rows:
301
+ try:
302
+ events.append(json.loads(row["event_json"]))
303
+ except (json.JSONDecodeError, KeyError):
304
+ continue
305
+ return events
306
+
307
+ def session(self, correlation_id: str, limit: int = 2000) -> list[dict[str, Any]]:
308
+ """Return all events for one correlation_id, time-ordered."""
309
+ conn = self._get_conn()
310
+ rows = conn.execute(
311
+ """SELECT event_json FROM audit_events
312
+ WHERE correlation_id = ?
313
+ ORDER BY timestamp_utc ASC, id ASC
314
+ LIMIT ?""",
315
+ (correlation_id, limit),
316
+ ).fetchall()
317
+ events = []
318
+ for row in rows:
319
+ try:
320
+ events.append(json.loads(row["event_json"]))
321
+ except (json.JSONDecodeError, KeyError):
322
+ continue
323
+ return events
324
+
325
+ def events_for_detection(self, correlation_id: str) -> list[dict[str, Any]]:
326
+ """Return events for detection correlation — no limit, full session."""
327
+ return self.session(correlation_id, limit=10000)
328
+
329
+ def events_by_action_type(
330
+ self, action_type: str, limit: int = 100_000
331
+ ) -> list[dict[str, Any]]:
332
+ """All events of one action type, oldest first.
333
+
334
+ Used to replay `detection_disposition` events, where the ordering is
335
+ the point: the last decision recorded for a detection is the one in
336
+ force.
337
+ """
338
+ rows = self._get_conn().execute(
339
+ """SELECT event_json FROM audit_events
340
+ WHERE action_type = ?
341
+ ORDER BY timestamp_utc ASC, id ASC
342
+ LIMIT ?""",
343
+ (action_type, limit),
344
+ ).fetchall()
345
+ events = []
346
+ for row in rows:
347
+ try:
348
+ events.append(json.loads(row["event_json"]))
349
+ except (json.JSONDecodeError, KeyError):
350
+ continue
351
+ return events
352
+
353
+ def status(self) -> dict[str, Any]:
354
+ """Freshness + per-source counts for the dashboard badge."""
355
+ conn = self._get_conn()
356
+
357
+ # Per-source counts from recent events (last 500 by id)
358
+ rows = conn.execute(
359
+ """SELECT source_app, COUNT(*) as cnt
360
+ FROM (SELECT source_app FROM audit_events ORDER BY id DESC LIMIT 500)
361
+ GROUP BY source_app"""
362
+ ).fetchall()
363
+ by_source = {row["source_app"]: row["cnt"] for row in rows}
364
+
365
+ # Last event timestamp
366
+ row = conn.execute(
367
+ "SELECT MAX(timestamp_utc) as last_ts FROM audit_events"
368
+ ).fetchone()
369
+ last_ts = row["last_ts"] if row else None
370
+
371
+ # Total recent count
372
+ row2 = conn.execute(
373
+ "SELECT COUNT(*) as cnt FROM (SELECT 1 FROM audit_events ORDER BY id DESC LIMIT 500)"
374
+ ).fetchone()
375
+ recent = row2["cnt"] if row2 else 0
376
+
377
+ return {
378
+ "last_event_utc": last_ts,
379
+ "recent": recent,
380
+ "by_source": by_source,
381
+ }
382
+
383
+ def stats(self, window_days: int = 7) -> dict[str, Any]:
384
+ """Aggregate audit metrics for dogfood / operator dashboards."""
385
+ days = max(1, min(window_days, 90))
386
+ cutoff = (datetime.now(timezone.utc) - timedelta(days=days)).isoformat()
387
+ conn = self._get_conn()
388
+
389
+ def _scalar(sql: str, *params: Any) -> int:
390
+ row = conn.execute(sql, params).fetchone()
391
+ return int(row[0]) if row and row[0] is not None else 0
392
+
393
+ total = _scalar(
394
+ "SELECT COUNT(*) FROM audit_events WHERE timestamp_utc >= ?",
395
+ cutoff,
396
+ )
397
+ sessions = _scalar(
398
+ """SELECT COUNT(DISTINCT correlation_id) FROM audit_events
399
+ WHERE timestamp_utc >= ? AND correlation_id != ''""",
400
+ cutoff,
401
+ )
402
+ detections = _scalar(
403
+ """SELECT COUNT(*) FROM audit_events
404
+ WHERE timestamp_utc >= ? AND action_type = 'detection'""",
405
+ cutoff,
406
+ )
407
+ denied = _scalar(
408
+ """SELECT COUNT(*) FROM audit_events
409
+ WHERE timestamp_utc >= ? AND action_outcome = 'denied'""",
410
+ cutoff,
411
+ )
412
+ dlp_matches = _scalar(
413
+ """SELECT COUNT(*) FROM audit_events
414
+ WHERE timestamp_utc >= ?
415
+ AND json_extract(event_json, '$.dlp.rule_id') IS NOT NULL""",
416
+ cutoff,
417
+ )
418
+ tool_policy_hits = _scalar(
419
+ """SELECT COUNT(*) FROM audit_events
420
+ WHERE timestamp_utc >= ?
421
+ AND json_extract(event_json, '$.tool_policy.rule_id') IS NOT NULL""",
422
+ cutoff,
423
+ )
424
+ tool_policy_blocks = _scalar(
425
+ """SELECT COUNT(*) FROM audit_events
426
+ WHERE timestamp_utc >= ?
427
+ AND json_extract(event_json, '$.tool_policy.blocked') = 1""",
428
+ cutoff,
429
+ )
430
+
431
+ rows = conn.execute(
432
+ """SELECT source_app, COUNT(*) as cnt FROM audit_events
433
+ WHERE timestamp_utc >= ?
434
+ GROUP BY source_app
435
+ ORDER BY cnt DESC""",
436
+ (cutoff,),
437
+ ).fetchall()
438
+ by_source = {row["source_app"]: row["cnt"] for row in rows}
439
+
440
+ last_row = conn.execute(
441
+ "SELECT MAX(timestamp_utc) as last_ts FROM audit_events WHERE timestamp_utc >= ?",
442
+ (cutoff,),
443
+ ).fetchone()
444
+ last_ts = last_row["last_ts"] if last_row else None
445
+
446
+ return {
447
+ "window_days": days,
448
+ "total_events": total,
449
+ "sessions": sessions,
450
+ "detections": detections,
451
+ "denied": denied,
452
+ "dlp_matches": dlp_matches,
453
+ "tool_policy_hits": tool_policy_hits,
454
+ "tool_policy_blocks": tool_policy_blocks,
455
+ "by_source": by_source,
456
+ "last_event_utc": last_ts,
457
+ }
458
+
459
+ def count(self) -> int:
460
+ """Total event count."""
461
+ row = self._get_conn().execute("SELECT COUNT(*) as cnt FROM audit_events").fetchone()
462
+ return row["cnt"] if row else 0
463
+
464
+
465
+ # ---------------------------------------------------------------------------
466
+ # Module-level singleton — lazily created from settings.
467
+ # ---------------------------------------------------------------------------
468
+
469
+ _db: AuditTrailDB | None = None
470
+ # Guards singleton creation: check-then-set on a global is a race, and the
471
+ # backfill runs in a worker thread while requests are already being served.
472
+ _db_lock = threading.Lock()
473
+
474
+
475
+ def get_trail_db() -> AuditTrailDB:
476
+ """Return the module-level trail DB singleton, creating it on first call."""
477
+ global _db
478
+ if _db is None:
479
+ with _db_lock:
480
+ if _db is None: # re-check: another thread may have won the race
481
+ from agentmetry.core.config import settings
482
+
483
+ _db = AuditTrailDB(settings.audit_db_path)
484
+ return _db
485
+
486
+
487
+ def reset_trail_db() -> None:
488
+ """Test helper — clear the singleton."""
489
+ global _db
490
+ with _db_lock:
491
+ _db = None