agentmetry 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. agentmetry/__init__.py +12 -0
  2. agentmetry/api/__init__.py +0 -0
  3. agentmetry/api/main.py +232 -0
  4. agentmetry/api/routes/__init__.py +0 -0
  5. agentmetry/api/routes/audit.py +396 -0
  6. agentmetry/api/websocket.py +53 -0
  7. agentmetry/api/ws_bridge.py +34 -0
  8. agentmetry/cli/__init__.py +941 -0
  9. agentmetry/cli/__main__.py +5 -0
  10. agentmetry/core/__init__.py +0 -0
  11. agentmetry/core/audit/__init__.py +1 -0
  12. agentmetry/core/audit/adapters/__init__.py +0 -0
  13. agentmetry/core/audit/adapters/agt.py +304 -0
  14. agentmetry/core/audit/adapters/cloudevents.py +159 -0
  15. agentmetry/core/audit/adapters/ecs.py +103 -0
  16. agentmetry/core/audit/adapters/splunk.py +40 -0
  17. agentmetry/core/audit/alerts.py +56 -0
  18. agentmetry/core/audit/canonical.py +150 -0
  19. agentmetry/core/audit/compliance_digest.py +299 -0
  20. agentmetry/core/audit/detection/__init__.py +9 -0
  21. agentmetry/core/audit/detection/benchmark.py +194 -0
  22. agentmetry/core/audit/detection/corpus/attack_approval_denied_then_executed.jsonl +3 -0
  23. agentmetry/core/audit/detection/corpus/attack_arbitrary_host_stage_execute.jsonl +3 -0
  24. agentmetry/core/audit/detection/corpus/attack_autonomous_unapproved_write.jsonl +3 -0
  25. agentmetry/core/audit/detection/corpus/attack_credential_exfil.jsonl +2 -0
  26. agentmetry/core/audit/detection/corpus/attack_credential_then_cloud_api.jsonl +2 -0
  27. agentmetry/core/audit/detection/corpus/attack_destructive_delete_burst.jsonl +6 -0
  28. agentmetry/core/audit/detection/corpus/attack_discovery_then_collect.jsonl +5 -0
  29. agentmetry/core/audit/detection/corpus/attack_dotfile_then_git_push.jsonl +2 -0
  30. agentmetry/core/audit/detection/corpus/attack_encoded_command_download.jsonl +1 -0
  31. agentmetry/core/audit/detection/corpus/attack_env_credential_exfil.jsonl +3 -0
  32. agentmetry/core/audit/detection/corpus/attack_hashed_only_no_command.jsonl +2 -0
  33. agentmetry/core/audit/detection/corpus/attack_interpreter_egress.jsonl +2 -0
  34. agentmetry/core/audit/detection/corpus/attack_pr_merged_without_review.jsonl +2 -0
  35. agentmetry/core/audit/detection/corpus/attack_proc_substitution_cradle.jsonl +2 -0
  36. agentmetry/core/audit/detection/corpus/attack_remote_pipe_to_shell.jsonl +1 -0
  37. agentmetry/core/audit/detection/corpus/attack_remote_staging_then_execute.jsonl +2 -0
  38. agentmetry/core/audit/detection/corpus/attack_session_tool_burst.jsonl +42 -0
  39. agentmetry/core/audit/detection/corpus/attack_single_command_exfil.jsonl +2 -0
  40. agentmetry/core/audit/detection/corpus/attack_ssh_directory_exfil.jsonl +3 -0
  41. agentmetry/core/audit/detection/corpus/attack_subagent_swarm.jsonl +6 -0
  42. agentmetry/core/audit/detection/corpus/attack_timestamp_collision.jsonl +2 -0
  43. agentmetry/core/audit/detection/corpus/attack_untrusted_input_then_action.jsonl +3 -0
  44. agentmetry/core/audit/detection/corpus/benign_authoring_merge_fixtures.jsonl +3 -0
  45. agentmetry/core/audit/detection/corpus/benign_autonomous_after_approval.jsonl +4 -0
  46. agentmetry/core/audit/detection/corpus/benign_build_artifact_cleanup.jsonl +5 -0
  47. agentmetry/core/audit/detection/corpus/benign_ci_artifact_download.jsonl +3 -0
  48. agentmetry/core/audit/detection/corpus/benign_database_migration.jsonl +5 -0
  49. agentmetry/core/audit/detection/corpus/benign_dependency_install_and_build.jsonl +5 -0
  50. agentmetry/core/audit/detection/corpus/benign_download_release_archive.jsonl +4 -0
  51. agentmetry/core/audit/detection/corpus/benign_fetch_data_then_run_repo_script.jsonl +3 -0
  52. agentmetry/core/audit/detection/corpus/benign_fetch_dataset_then_analyse.jsonl +3 -0
  53. agentmetry/core/audit/detection/corpus/benign_fetch_lockfile_then_install.jsonl +3 -0
  54. agentmetry/core/audit/detection/corpus/benign_git_review_and_push.jsonl +6 -0
  55. agentmetry/core/audit/detection/corpus/benign_human_driven_deletes.jsonl +6 -0
  56. agentmetry/core/audit/detection/corpus/benign_local_api_probing.jsonl +5 -0
  57. agentmetry/core/audit/detection/corpus/benign_long_but_calm_session.jsonl +30 -0
  58. agentmetry/core/audit/detection/corpus/benign_loopback_is_not_egress.jsonl +2 -0
  59. agentmetry/core/audit/detection/corpus/benign_loopback_pipe_to_interpreter.jsonl +3 -0
  60. agentmetry/core/audit/detection/corpus/benign_ordinary_development.jsonl +5 -0
  61. agentmetry/core/audit/detection/corpus/benign_package_manager_after_fetch.jsonl +2 -0
  62. agentmetry/core/audit/detection/corpus/benign_reading_config_that_is_not_secret.jsonl +5 -0
  63. agentmetry/core/audit/detection/corpus/benign_remote_api_call_no_credentials.jsonl +4 -0
  64. agentmetry/core/audit/detection/corpus/benign_research_then_docs.jsonl +5 -0
  65. agentmetry/core/audit/detection/corpus/benign_reversed_order_is_not_exfil.jsonl +2 -0
  66. agentmetry/core/audit/detection/corpus/benign_test_and_fix_loop.jsonl +6 -0
  67. agentmetry/core/audit/detection/corpus/benign_writing_about_credentials.jsonl +6 -0
  68. agentmetry/core/audit/detection/corpus/corpus.yaml +443 -0
  69. agentmetry/core/audit/detection/disposition.py +651 -0
  70. agentmetry/core/audit/detection/engine.py +78 -0
  71. agentmetry/core/audit/detection/live.py +127 -0
  72. agentmetry/core/audit/detection/live_store.py +355 -0
  73. agentmetry/core/audit/detection/models.py +53 -0
  74. agentmetry/core/audit/detection/rules.py +1314 -0
  75. agentmetry/core/audit/detection/traits.py +648 -0
  76. agentmetry/core/audit/detection/yaml_config.py +91 -0
  77. agentmetry/core/audit/detection/yaml_rules.py +83 -0
  78. agentmetry/core/audit/dlp/__init__.py +4 -0
  79. agentmetry/core/audit/dlp/loader.py +29 -0
  80. agentmetry/core/audit/dlp/models.py +29 -0
  81. agentmetry/core/audit/dlp/scanner.py +96 -0
  82. agentmetry/core/audit/dogfood.py +398 -0
  83. agentmetry/core/audit/evidence_pack.py +500 -0
  84. agentmetry/core/audit/external.py +213 -0
  85. agentmetry/core/audit/hashing.py +21 -0
  86. agentmetry/core/audit/hook_bootstrap.py +451 -0
  87. agentmetry/core/audit/identity.py +39 -0
  88. agentmetry/core/audit/ingest.py +242 -0
  89. agentmetry/core/audit/migrate.py +73 -0
  90. agentmetry/core/audit/mitre.py +244 -0
  91. agentmetry/core/audit/policy.py +99 -0
  92. agentmetry/core/audit/redaction.py +50 -0
  93. agentmetry/core/audit/replay.py +54 -0
  94. agentmetry/core/audit/run_context.py +129 -0
  95. agentmetry/core/audit/sinks.py +235 -0
  96. agentmetry/core/audit/spool.py +394 -0
  97. agentmetry/core/audit/tool_policy/__init__.py +4 -0
  98. agentmetry/core/audit/tool_policy/evaluator.py +198 -0
  99. agentmetry/core/audit/tool_policy/loader.py +44 -0
  100. agentmetry/core/audit/tool_policy/models.py +25 -0
  101. agentmetry/core/audit/trail_chain.py +300 -0
  102. agentmetry/core/audit/trail_db.py +491 -0
  103. agentmetry/core/audit/trail_merkle.py +332 -0
  104. agentmetry/core/auth.py +54 -0
  105. agentmetry/core/bus/__init__.py +5 -0
  106. agentmetry/core/bus/audit_exporter.py +107 -0
  107. agentmetry/core/bus/bridges.py +26 -0
  108. agentmetry/core/bus/bus.py +102 -0
  109. agentmetry/core/bus/events.py +50 -0
  110. agentmetry/core/bus/outbox.py +124 -0
  111. agentmetry/core/config.py +177 -0
  112. agentmetry/core/diagnostics/__init__.py +0 -0
  113. agentmetry/core/diagnostics/autostart.py +563 -0
  114. agentmetry/core/diagnostics/doctor.py +535 -0
  115. agentmetry/core/diagnostics/driver_paths.py +156 -0
  116. agentmetry/core/diagnostics/env_file.py +45 -0
  117. agentmetry/core/drivers/__init__.py +4 -0
  118. agentmetry/core/drivers/host.py +263 -0
  119. agentmetry/core/drivers/permissions.py +37 -0
  120. agentmetry/core/drivers/spec.py +118 -0
  121. agentmetry/core/extensions.py +107 -0
  122. agentmetry/core/health.py +26 -0
  123. agentmetry/core/version.py +13 -0
  124. agentmetry/policies/detection/manifest.yaml +43 -0
  125. agentmetry/policies/dlp/manifest.yaml +161 -0
  126. agentmetry/policies/opa/agent_rules.rego +33 -0
  127. agentmetry/policies/tool/manifest.yaml +117 -0
  128. agentmetry-0.4.0.dist-info/METADATA +86 -0
  129. agentmetry-0.4.0.dist-info/RECORD +131 -0
  130. agentmetry-0.4.0.dist-info/WHEEL +4 -0
  131. agentmetry-0.4.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,99 @@
1
+ """Local policy checks for agent tool calls.
2
+
3
+ **This is not OPA.** It is a small built-in ruleset. Real OPA/Rego evaluation is
4
+ on the roadmap; `agentmetry/policies/opa/agent_rules.rego` is a draft for that work and is
5
+ NOT evaluated by this module.
6
+
7
+ Two design rules, both learned the hard way:
8
+
9
+ 1. **Annotate, never rewrite.** A policy verdict is a separate fact from what
10
+ actually happened. The previous version set `action.outcome = "denied"` on
11
+ events that had *already executed* on the agent's machine, which put a lie in
12
+ the audit trail: an incident responder would read "denied" for a tool that
13
+ ran. It also blinded every detection rule that keys on `outcome == "success"`
14
+ (a burst of `shell.rm` stopped firing `destructive-delete-burst` precisely
15
+ because policy flagged it). The verdict now lands in its own `policy` block
16
+ and `action` is left alone.
17
+
18
+ 2. **This cannot block.** By the time an event reaches the ingest API the tool
19
+ has already run. Real prevention has to happen in the hook process,
20
+ pre-execution, the way DLP `block` mode does. The annotation records
21
+ `enforced: false` so nobody mistakes it for enforcement.
22
+
23
+ Off by default (`AGENTMETRY_POLICY_ENABLED=1` to turn on): the ruleset below is
24
+ a hardcoded starting point, not something to impose on every operator.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ from dataclasses import dataclass
30
+ from typing import Any
31
+
32
+
33
+ @dataclass(frozen=True)
34
+ class PolicyVerdict:
35
+ allowed: bool
36
+ rule_id: str = ""
37
+ reason: str = ""
38
+
39
+
40
+ def _norm(name: str) -> str:
41
+ """Fold a tool name the same way core.audit.mitre does.
42
+
43
+ Exact string matching on `tool.qualified` was trivially evadable: a rule for
44
+ "shell.rm" missed "shell.RM" and "shell_rm". Normalizing at least closes the
45
+ spelling gap. It does NOT close the real gap (see module docstring): a shell
46
+ tool running `rm -rf` under any name is still not caught here.
47
+ """
48
+ return name.lower().replace("_", "").replace("-", "")
49
+
50
+
51
+ # Tools that should not run unattended. Deliberately tiny: this is a default,
52
+ # not a policy language. The roadmap replaces it with real Rego.
53
+ _RESTRICTED: dict[str, str] = {
54
+ _norm("kubectl.exec"): "restricted:kubectl-exec",
55
+ _norm("aws.iam.delete_user"): "restricted:iam-delete-user",
56
+ _norm("shell.rm"): "restricted:shell-rm",
57
+ }
58
+
59
+
60
+ def evaluate_policy(event: dict[str, Any]) -> PolicyVerdict:
61
+ """Return a verdict for one canonical event. Never mutates the event."""
62
+ tool = event.get("tool")
63
+ qualified = str(tool.get("qualified") or "") if isinstance(tool, dict) else ""
64
+ if not qualified:
65
+ return PolicyVerdict(allowed=True)
66
+
67
+ rule_id = _RESTRICTED.get(_norm(qualified))
68
+ if rule_id is None:
69
+ return PolicyVerdict(allowed=True)
70
+
71
+ # A restricted tool is allowed when a human explicitly approved this action.
72
+ action = event.get("action") or {}
73
+ if action.get("type") == "approval_response" and action.get("outcome") == "success":
74
+ return PolicyVerdict(allowed=True)
75
+
76
+ return PolicyVerdict(
77
+ allowed=False,
78
+ rule_id=rule_id,
79
+ reason=f"{qualified} is restricted and had no recorded human approval",
80
+ )
81
+
82
+
83
+ def annotate(event: dict[str, Any]) -> None:
84
+ """Attach a policy verdict to `event['policy']`, in place.
85
+
86
+ Only writes when the verdict is a denial, and only ever touches the `policy`
87
+ key. `action.outcome` stays whatever actually happened.
88
+ """
89
+ verdict = evaluate_policy(event)
90
+ if verdict.allowed:
91
+ return
92
+ event["policy"] = {
93
+ "decision": "deny",
94
+ "rule_id": verdict.rule_id,
95
+ "engine": "builtin",
96
+ # This event already executed. We observed it; we did not stop it.
97
+ "enforced": False,
98
+ "reason": verdict.reason,
99
+ }
@@ -0,0 +1,50 @@
1
+ """Secret scrubbing for captured command text (Tier A/B).
2
+
3
+ When command logging is enabled, the raw command string can contain inline
4
+ secrets (bearer tokens, basic-auth URLs, --password values, cloud keys) that
5
+ key-based argument redaction never sees. Scrub those before storage.
6
+
7
+ Keep the pattern list in sync with the inline mirror in
8
+ `scripts/agentmetry_ingest.py` (the standalone hook cannot import this module).
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import re
14
+
15
+ _SECRET_PATTERNS: list[tuple[re.Pattern[str], str]] = [
16
+ (re.compile(r"(?i)(bearer\s+)[A-Za-z0-9._\-]+"), r"\1<redacted>"),
17
+ (re.compile(r"(?i)(authorization:\s*)\S+"), r"\1<redacted>"),
18
+ # user:pass@host in URLs
19
+ (re.compile(r"(https?://)[^/\s:@]+:[^/\s@]+@"), r"\1<redacted>@"),
20
+ # --password X / --token=X / -p X
21
+ (re.compile(r"(?i)(-{1,2}(?:password|token|secret|api[-_]?key|pwd)[=\s]+)\S+"), r"\1<redacted>"),
22
+ # key=value / key: value assignments
23
+ (
24
+ re.compile(
25
+ r"(?i)\b(password|passwd|pwd|token|secret|api[-_]?key|apikey|access[-_]?key)\s*[=:]\s*[^\s;&|\"']+"
26
+ ),
27
+ r"\1=<redacted>",
28
+ ),
29
+ (re.compile(r"\bAKIA[0-9A-Z]{16}\b"), "<redacted-aws-key>"),
30
+ (re.compile(r"\bsk-[A-Za-z0-9]{20,}\b"), "<redacted-key>"),
31
+ (re.compile(r"\bgh[pousr]_[A-Za-z0-9]{20,}\b"), "<redacted-gh-token>"),
32
+ (re.compile(r"\bxox[baprs]-[A-Za-z0-9-]{10,}\b"), "<redacted-slack-token>"),
33
+ ]
34
+
35
+
36
+ def scrub_secrets(text: str) -> str:
37
+ """Mask obvious inline secrets in a command/string. Best-effort, conservative."""
38
+ if not isinstance(text, str) or not text:
39
+ return text
40
+ out = text
41
+ for pattern, repl in _SECRET_PATTERNS:
42
+ out = pattern.sub(repl, out)
43
+ return out
44
+
45
+
46
+ def scrub_arg_values(args: object) -> object:
47
+ """Scrub secret patterns inside string values of an arguments dict."""
48
+ if not isinstance(args, dict):
49
+ return args
50
+ return {k: (scrub_secrets(v) if isinstance(v, str) else v) for k, v in args.items()}
@@ -0,0 +1,54 @@
1
+ """ASCII timeline rendering for `agentmetry replay`."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from agentmetry.core.audit.canonical import normalize_outbox_row
8
+
9
+ _ICONS = {
10
+ "session_start": "▶",
11
+ "session_end": "■",
12
+ "approval_request": "⏸",
13
+ "approval_response": "✓",
14
+ "tool_called": "🔧",
15
+ "config_change": "⚙",
16
+ }
17
+
18
+
19
+ def format_timeline(rows: list[dict[str, Any]], *, thread_id: str) -> str:
20
+ if not rows:
21
+ return f"No audit events for thread_id={thread_id}"
22
+
23
+ lines = [f"Agentmetry replay — correlation_id={thread_id}", f"{'─' * 60}"]
24
+
25
+ for row in rows:
26
+ canonical = normalize_outbox_row(row)
27
+ if canonical is None:
28
+ ts = row.get("ts", "?")
29
+ topic = row.get("topic", "?")
30
+ lines.append(f" {ts} [{topic}]")
31
+ continue
32
+
33
+ ts = canonical["timestamp_utc"][:19].replace("T", " ")
34
+ action = canonical["action"]
35
+ icon = _ICONS.get(action["type"], "·")
36
+ outcome = action["outcome"]
37
+ label = action["type"]
38
+
39
+ detail_parts: list[str] = []
40
+ if skill := canonical.get("agent", {}).get("skill_id"):
41
+ detail_parts.append(f"skill={skill}")
42
+ if tool := canonical.get("tool"):
43
+ detail_parts.append(f"tool={tool.get('qualified') or tool.get('name')}")
44
+ if outcome == "denied" and action.get("reason"):
45
+ detail_parts.append(f"reason={action['reason']}")
46
+ if action.get("reason") and action["type"] == "approval_response":
47
+ detail_parts.append(action["reason"])
48
+
49
+ detail = f" ({', '.join(detail_parts)})" if detail_parts else ""
50
+ lines.append(f" {ts} {icon} {label}/{outcome}{detail} seq={canonical.get('seq')}")
51
+
52
+ lines.append(f"{'─' * 60}")
53
+ lines.append(f"{len(rows)} event(s)")
54
+ return "\n".join(lines)
@@ -0,0 +1,129 @@
1
+ """Per-run audit context — initiator provenance and last gated tool (schema v1.1)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from agentmetry.core.config import settings
8
+
9
+ # thread_id → initiator block (set at run start, server-derived only)
10
+ _thread_initiators: dict[str, dict[str, str]] = {}
11
+ # thread_id → last successful tool call on this run (for approval binding)
12
+ _thread_last_tool: dict[str, dict[str, str]] = {}
13
+
14
+
15
+ def _operator_id() -> str:
16
+ return settings.operator_id.strip() or "local"
17
+
18
+
19
+ def build_initiator(triggered_by: str) -> dict[str, str]:
20
+ """Derive initiator from run origin. Must never trust client-supplied headers."""
21
+ operator_id = _operator_id()
22
+ if triggered_by == "manual":
23
+ return {"actor_type": "human", "trigger": "manual", "operator_id": operator_id}
24
+ if triggered_by.startswith("channel:"):
25
+ return {"actor_type": "human", "trigger": "channel", "operator_id": operator_id}
26
+ if triggered_by == "cron":
27
+ return {"actor_type": "autonomous", "trigger": "cron", "operator_id": operator_id}
28
+ if triggered_by == "vault_watch":
29
+ return {"actor_type": "autonomous", "trigger": "vault_watch", "operator_id": operator_id}
30
+ if triggered_by == "ingress":
31
+ return {"actor_type": "autonomous", "trigger": "ingress", "operator_id": operator_id}
32
+ if triggered_by == "recovery":
33
+ return {"actor_type": "autonomous", "trigger": "recovery", "operator_id": operator_id}
34
+ return {
35
+ "actor_type": "autonomous",
36
+ "trigger": triggered_by,
37
+ "operator_id": operator_id,
38
+ }
39
+
40
+
41
+ def actor_from_initiator(initiator: dict[str, str]) -> dict[str, str]:
42
+ human = initiator.get("actor_type") == "human"
43
+ return {
44
+ "type": "user" if human else "agent",
45
+ "id": initiator.get("operator_id") or _operator_id(),
46
+ "role": "operator",
47
+ }
48
+
49
+
50
+ def default_initiator() -> dict[str, str]:
51
+ return build_initiator("manual")
52
+
53
+
54
+ def set_thread_initiator(thread_id: str, triggered_by: str) -> dict[str, str]:
55
+ initiator = build_initiator(triggered_by)
56
+ _thread_initiators[thread_id] = initiator
57
+ return initiator
58
+
59
+
60
+ def get_thread_initiator(thread_id: str) -> dict[str, str] | None:
61
+ return _thread_initiators.get(thread_id)
62
+
63
+
64
+ def resolve_initiator(
65
+ payload: dict[str, Any], thread_id: str = ""
66
+ ) -> dict[str, str]:
67
+ """Read initiator from payload or thread cache; fallback manual human."""
68
+ raw = payload.get("initiator")
69
+ if isinstance(raw, dict) and raw.get("actor_type"):
70
+ return {
71
+ "actor_type": str(raw.get("actor_type") or "human"),
72
+ "trigger": str(raw.get("trigger") or "manual"),
73
+ "operator_id": str(raw.get("operator_id") or _operator_id()),
74
+ }
75
+ triggered_by = str(payload.get("triggered_by") or "")
76
+ if triggered_by:
77
+ return build_initiator(triggered_by)
78
+ if thread_id:
79
+ cached = _thread_initiators.get(thread_id)
80
+ if cached:
81
+ return cached
82
+ return default_initiator()
83
+
84
+
85
+ def record_tool_call(thread_id: str, qualified: str, arguments_sha256: str) -> None:
86
+ if not thread_id:
87
+ return
88
+ server = qualified.split(".", 1)[0] if "." in qualified else ""
89
+ _thread_last_tool[thread_id] = {
90
+ "tool": qualified,
91
+ "server": server,
92
+ "input_hash": arguments_sha256,
93
+ }
94
+
95
+
96
+ def last_gated_action(thread_id: str) -> dict[str, str] | None:
97
+ if not thread_id:
98
+ return None
99
+ action = _thread_last_tool.get(thread_id)
100
+ if not action or not action.get("tool"):
101
+ return None
102
+ return dict(action)
103
+
104
+
105
+ def clear_run_context(thread_id: str) -> None:
106
+ _thread_initiators.pop(thread_id, None)
107
+ _thread_last_tool.pop(thread_id, None)
108
+
109
+
110
+ def audit_payload(
111
+ thread_id: str,
112
+ triggered_by: str | None = None,
113
+ *,
114
+ initiator: dict[str, str] | None = None,
115
+ ) -> dict[str, Any]:
116
+ """Extra bus payload fields for canonical v1.1."""
117
+ init = initiator or (
118
+ get_thread_initiator(thread_id)
119
+ if thread_id
120
+ else None
121
+ )
122
+ if init is None and triggered_by:
123
+ init = build_initiator(triggered_by)
124
+ if init is None:
125
+ init = default_initiator()
126
+ out: dict[str, Any] = {"initiator": init}
127
+ if triggered_by:
128
+ out["triggered_by"] = triggered_by
129
+ return out
@@ -0,0 +1,235 @@
1
+ """Agentmetry forward sinks — file, webhook, Elastic ECS, Splunk HEC."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from abc import ABC, abstractmethod
7
+ from pathlib import Path
8
+ from threading import Lock
9
+ from typing import Any
10
+
11
+ import httpx
12
+
13
+ from agentmetry.core.audit.adapters.ecs import canonical_to_ecs
14
+ from agentmetry.core.audit.adapters.splunk import canonical_to_hec_event
15
+
16
+ logger = logging.getLogger(__name__)
17
+
18
+ _file_lock = Lock()
19
+
20
+
21
+ class AuditSink(ABC):
22
+ @abstractmethod
23
+ async def emit(self, canonical: dict[str, Any]) -> None:
24
+ ...
25
+
26
+
27
+ class FileAuditSink(AuditSink):
28
+ def __init__(self, path: Path) -> None:
29
+ self._path = path
30
+
31
+ async def emit(self, canonical: dict[str, Any]) -> None:
32
+ from agentmetry.core.audit.trail_chain import append_chained_line
33
+
34
+ with _file_lock:
35
+ append_chained_line(self._path, canonical)
36
+
37
+
38
+ class WebhookAuditSink(AuditSink):
39
+ """POST each event to a URL, as the canonical record or a CloudEvent.
40
+
41
+ `format="cloudevents"` wraps the same record in a CloudEvents v1.0 structured
42
+ envelope, which is what brokers speak: Knative, EventBridge, Event Grid,
43
+ Dapr and the Kafka bindings all consume it. The canonical event still travels
44
+ whole inside `data`, so nothing is lost by choosing it.
45
+
46
+ Default stays `canonical`. Changing the shape of what an existing webhook
47
+ receives because a new option appeared would break every consumer already
48
+ wired up, and silently.
49
+ """
50
+
51
+ def __init__(
52
+ self, url: str, *, timeout_seconds: float = 5.0, format: str = "canonical"
53
+ ) -> None:
54
+ self._url = url
55
+ self._timeout = timeout_seconds
56
+ self._cloudevents = (format or "").strip().lower() in ("cloudevents", "cloudevent", "ce")
57
+
58
+ async def emit(self, canonical: dict[str, Any]) -> None:
59
+ payload = canonical
60
+ # `application/cloudevents+json` is what marks structured mode; a
61
+ # consumer distinguishes it from a bare JSON body by content type alone.
62
+ content_type = "application/json"
63
+ if self._cloudevents:
64
+ from agentmetry.core.audit.adapters.cloudevents import canonical_to_cloudevent
65
+
66
+ payload = canonical_to_cloudevent(canonical)
67
+ content_type = "application/cloudevents+json; charset=utf-8"
68
+ try:
69
+ async with httpx.AsyncClient(timeout=self._timeout) as client:
70
+ response = await client.post(
71
+ self._url,
72
+ json=payload,
73
+ headers={"Content-Type": content_type, "User-Agent": "Agentmetry/1.0"},
74
+ )
75
+ response.raise_for_status()
76
+ except Exception:
77
+ logger.exception("Audit webhook POST failed → %s", self._url)
78
+
79
+
80
+ class ElasticEcsSink(AuditSink):
81
+ """Index one ECS document per event (Elasticsearch _doc API)."""
82
+
83
+ def __init__(
84
+ self,
85
+ base_url: str,
86
+ index: str,
87
+ api_key: str,
88
+ *,
89
+ timeout_seconds: float = 5.0,
90
+ verify_tls: bool = True,
91
+ ) -> None:
92
+ self._url = base_url.rstrip("/") + f"/{index}/_doc"
93
+ self._api_key = api_key
94
+ self._timeout = timeout_seconds
95
+ self._verify = verify_tls
96
+
97
+ async def emit(self, canonical: dict[str, Any]) -> None:
98
+ doc = canonical_to_ecs(canonical)
99
+ headers = {
100
+ "Content-Type": "application/json",
101
+ "Authorization": f"ApiKey {self._api_key}",
102
+ "User-Agent": "Agentmetry/1.0",
103
+ }
104
+ try:
105
+ async with httpx.AsyncClient(timeout=self._timeout, verify=self._verify) as client:
106
+ response = await client.post(self._url, json=doc, headers=headers)
107
+ response.raise_for_status()
108
+ except Exception:
109
+ logger.exception("Elastic ECS index failed → %s", self._url)
110
+
111
+
112
+ class SplunkHecSink(AuditSink):
113
+ """POST one event to Splunk HTTP Event Collector."""
114
+
115
+ def __init__(
116
+ self,
117
+ hec_url: str,
118
+ token: str,
119
+ *,
120
+ index: str = "main",
121
+ sourcetype: str = "agentmetry:json",
122
+ timeout_seconds: float = 5.0,
123
+ verify_tls: bool = True,
124
+ ) -> None:
125
+ base = hec_url.rstrip("/")
126
+ if base.endswith("/services/collector"):
127
+ self._url = base
128
+ elif base.endswith("/services/collector/event"):
129
+ self._url = base
130
+ else:
131
+ self._url = base + "/services/collector/event"
132
+ self._token = token
133
+ self._index = index
134
+ self._sourcetype = sourcetype
135
+ self._timeout = timeout_seconds
136
+ self._verify = verify_tls
137
+
138
+ async def emit(self, canonical: dict[str, Any]) -> None:
139
+ payload = canonical_to_hec_event(
140
+ canonical,
141
+ index=self._index,
142
+ sourcetype=self._sourcetype,
143
+ )
144
+ headers = {
145
+ "Authorization": f"Splunk {self._token}",
146
+ "Content-Type": "application/json",
147
+ "User-Agent": "Agentmetry/1.0",
148
+ }
149
+ try:
150
+ async with httpx.AsyncClient(timeout=self._timeout, verify=self._verify) as client:
151
+ response = await client.post(self._url, json=payload, headers=headers)
152
+ response.raise_for_status()
153
+ except Exception:
154
+ logger.exception("Splunk HEC POST failed → %s", self._url)
155
+
156
+
157
+ class MultiAuditSink(AuditSink):
158
+ def __init__(self, sinks: list[AuditSink]) -> None:
159
+ self._sinks = sinks
160
+
161
+ async def emit(self, canonical: dict[str, Any]) -> None:
162
+ for sink in self._sinks:
163
+ await sink.emit(canonical)
164
+
165
+
166
+ def parse_sink_modes(raw: str) -> set[str]:
167
+ text = raw.strip().lower()
168
+ if not text or text == "file":
169
+ return {"file"}
170
+ if text == "both":
171
+ return {"file", "webhook"}
172
+ if text == "all":
173
+ return {"file", "webhook", "elastic", "splunk"}
174
+ return {part.strip() for part in text.split(",") if part.strip()}
175
+
176
+
177
+ def build_audit_sinks(
178
+ *,
179
+ modes: set[str],
180
+ file_path: Path,
181
+ webhook_url: str,
182
+ webhook_timeout_seconds: float,
183
+ webhook_format: str = "canonical",
184
+ elastic_url: str,
185
+ elastic_index: str,
186
+ elastic_api_key: str,
187
+ elastic_verify_tls: bool,
188
+ splunk_hec_url: str,
189
+ splunk_hec_token: str,
190
+ splunk_index: str,
191
+ splunk_sourcetype: str,
192
+ splunk_verify_tls: bool,
193
+ ) -> AuditSink | None:
194
+ sinks: list[AuditSink] = []
195
+
196
+ if "file" in modes:
197
+ sinks.append(FileAuditSink(file_path))
198
+
199
+ if "webhook" in modes and webhook_url.strip():
200
+ sinks.append(
201
+ WebhookAuditSink(
202
+ webhook_url.strip(),
203
+ timeout_seconds=webhook_timeout_seconds,
204
+ format=webhook_format,
205
+ )
206
+ )
207
+
208
+ if "elastic" in modes and elastic_url.strip() and elastic_api_key.strip():
209
+ sinks.append(
210
+ ElasticEcsSink(
211
+ elastic_url.strip(),
212
+ elastic_index,
213
+ elastic_api_key.strip(),
214
+ timeout_seconds=webhook_timeout_seconds,
215
+ verify_tls=elastic_verify_tls,
216
+ )
217
+ )
218
+
219
+ if "splunk" in modes and splunk_hec_url.strip() and splunk_hec_token.strip():
220
+ sinks.append(
221
+ SplunkHecSink(
222
+ splunk_hec_url.strip(),
223
+ splunk_hec_token.strip(),
224
+ index=splunk_index,
225
+ sourcetype=splunk_sourcetype,
226
+ timeout_seconds=webhook_timeout_seconds,
227
+ verify_tls=splunk_verify_tls,
228
+ )
229
+ )
230
+
231
+ if not sinks:
232
+ return None
233
+ if len(sinks) == 1:
234
+ return sinks[0]
235
+ return MultiAuditSink(sinks)