agentmetry 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentmetry/__init__.py +12 -0
- agentmetry/api/__init__.py +0 -0
- agentmetry/api/main.py +232 -0
- agentmetry/api/routes/__init__.py +0 -0
- agentmetry/api/routes/audit.py +396 -0
- agentmetry/api/websocket.py +53 -0
- agentmetry/api/ws_bridge.py +34 -0
- agentmetry/cli/__init__.py +941 -0
- agentmetry/cli/__main__.py +5 -0
- agentmetry/core/__init__.py +0 -0
- agentmetry/core/audit/__init__.py +1 -0
- agentmetry/core/audit/adapters/__init__.py +0 -0
- agentmetry/core/audit/adapters/agt.py +304 -0
- agentmetry/core/audit/adapters/cloudevents.py +159 -0
- agentmetry/core/audit/adapters/ecs.py +103 -0
- agentmetry/core/audit/adapters/splunk.py +40 -0
- agentmetry/core/audit/alerts.py +56 -0
- agentmetry/core/audit/canonical.py +150 -0
- agentmetry/core/audit/compliance_digest.py +299 -0
- agentmetry/core/audit/detection/__init__.py +9 -0
- agentmetry/core/audit/detection/benchmark.py +194 -0
- agentmetry/core/audit/detection/corpus/attack_approval_denied_then_executed.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/attack_arbitrary_host_stage_execute.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/attack_autonomous_unapproved_write.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/attack_credential_exfil.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_credential_then_cloud_api.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_destructive_delete_burst.jsonl +6 -0
- agentmetry/core/audit/detection/corpus/attack_discovery_then_collect.jsonl +5 -0
- agentmetry/core/audit/detection/corpus/attack_dotfile_then_git_push.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_encoded_command_download.jsonl +1 -0
- agentmetry/core/audit/detection/corpus/attack_env_credential_exfil.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/attack_hashed_only_no_command.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_interpreter_egress.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_pr_merged_without_review.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_proc_substitution_cradle.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_remote_pipe_to_shell.jsonl +1 -0
- agentmetry/core/audit/detection/corpus/attack_remote_staging_then_execute.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_session_tool_burst.jsonl +42 -0
- agentmetry/core/audit/detection/corpus/attack_single_command_exfil.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_ssh_directory_exfil.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/attack_subagent_swarm.jsonl +6 -0
- agentmetry/core/audit/detection/corpus/attack_timestamp_collision.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_untrusted_input_then_action.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/benign_authoring_merge_fixtures.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/benign_autonomous_after_approval.jsonl +4 -0
- agentmetry/core/audit/detection/corpus/benign_build_artifact_cleanup.jsonl +5 -0
- agentmetry/core/audit/detection/corpus/benign_ci_artifact_download.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/benign_database_migration.jsonl +5 -0
- agentmetry/core/audit/detection/corpus/benign_dependency_install_and_build.jsonl +5 -0
- agentmetry/core/audit/detection/corpus/benign_download_release_archive.jsonl +4 -0
- agentmetry/core/audit/detection/corpus/benign_fetch_data_then_run_repo_script.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/benign_fetch_dataset_then_analyse.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/benign_fetch_lockfile_then_install.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/benign_git_review_and_push.jsonl +6 -0
- agentmetry/core/audit/detection/corpus/benign_human_driven_deletes.jsonl +6 -0
- agentmetry/core/audit/detection/corpus/benign_local_api_probing.jsonl +5 -0
- agentmetry/core/audit/detection/corpus/benign_long_but_calm_session.jsonl +30 -0
- agentmetry/core/audit/detection/corpus/benign_loopback_is_not_egress.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/benign_loopback_pipe_to_interpreter.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/benign_ordinary_development.jsonl +5 -0
- agentmetry/core/audit/detection/corpus/benign_package_manager_after_fetch.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/benign_reading_config_that_is_not_secret.jsonl +5 -0
- agentmetry/core/audit/detection/corpus/benign_remote_api_call_no_credentials.jsonl +4 -0
- agentmetry/core/audit/detection/corpus/benign_research_then_docs.jsonl +5 -0
- agentmetry/core/audit/detection/corpus/benign_reversed_order_is_not_exfil.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/benign_test_and_fix_loop.jsonl +6 -0
- agentmetry/core/audit/detection/corpus/benign_writing_about_credentials.jsonl +6 -0
- agentmetry/core/audit/detection/corpus/corpus.yaml +443 -0
- agentmetry/core/audit/detection/disposition.py +651 -0
- agentmetry/core/audit/detection/engine.py +78 -0
- agentmetry/core/audit/detection/live.py +127 -0
- agentmetry/core/audit/detection/live_store.py +355 -0
- agentmetry/core/audit/detection/models.py +53 -0
- agentmetry/core/audit/detection/rules.py +1314 -0
- agentmetry/core/audit/detection/traits.py +648 -0
- agentmetry/core/audit/detection/yaml_config.py +91 -0
- agentmetry/core/audit/detection/yaml_rules.py +83 -0
- agentmetry/core/audit/dlp/__init__.py +4 -0
- agentmetry/core/audit/dlp/loader.py +29 -0
- agentmetry/core/audit/dlp/models.py +29 -0
- agentmetry/core/audit/dlp/scanner.py +96 -0
- agentmetry/core/audit/dogfood.py +398 -0
- agentmetry/core/audit/evidence_pack.py +500 -0
- agentmetry/core/audit/external.py +213 -0
- agentmetry/core/audit/hashing.py +21 -0
- agentmetry/core/audit/hook_bootstrap.py +451 -0
- agentmetry/core/audit/identity.py +39 -0
- agentmetry/core/audit/ingest.py +242 -0
- agentmetry/core/audit/migrate.py +73 -0
- agentmetry/core/audit/mitre.py +244 -0
- agentmetry/core/audit/policy.py +99 -0
- agentmetry/core/audit/redaction.py +50 -0
- agentmetry/core/audit/replay.py +54 -0
- agentmetry/core/audit/run_context.py +129 -0
- agentmetry/core/audit/sinks.py +235 -0
- agentmetry/core/audit/spool.py +394 -0
- agentmetry/core/audit/tool_policy/__init__.py +4 -0
- agentmetry/core/audit/tool_policy/evaluator.py +198 -0
- agentmetry/core/audit/tool_policy/loader.py +44 -0
- agentmetry/core/audit/tool_policy/models.py +25 -0
- agentmetry/core/audit/trail_chain.py +300 -0
- agentmetry/core/audit/trail_db.py +491 -0
- agentmetry/core/audit/trail_merkle.py +332 -0
- agentmetry/core/auth.py +54 -0
- agentmetry/core/bus/__init__.py +5 -0
- agentmetry/core/bus/audit_exporter.py +107 -0
- agentmetry/core/bus/bridges.py +26 -0
- agentmetry/core/bus/bus.py +102 -0
- agentmetry/core/bus/events.py +50 -0
- agentmetry/core/bus/outbox.py +124 -0
- agentmetry/core/config.py +177 -0
- agentmetry/core/diagnostics/__init__.py +0 -0
- agentmetry/core/diagnostics/autostart.py +563 -0
- agentmetry/core/diagnostics/doctor.py +535 -0
- agentmetry/core/diagnostics/driver_paths.py +156 -0
- agentmetry/core/diagnostics/env_file.py +45 -0
- agentmetry/core/drivers/__init__.py +4 -0
- agentmetry/core/drivers/host.py +263 -0
- agentmetry/core/drivers/permissions.py +37 -0
- agentmetry/core/drivers/spec.py +118 -0
- agentmetry/core/extensions.py +107 -0
- agentmetry/core/health.py +26 -0
- agentmetry/core/version.py +13 -0
- agentmetry/policies/detection/manifest.yaml +43 -0
- agentmetry/policies/dlp/manifest.yaml +161 -0
- agentmetry/policies/opa/agent_rules.rego +33 -0
- agentmetry/policies/tool/manifest.yaml +117 -0
- agentmetry-0.4.0.dist-info/METADATA +86 -0
- agentmetry-0.4.0.dist-info/RECORD +131 -0
- agentmetry-0.4.0.dist-info/WHEEL +4 -0
- agentmetry-0.4.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
"""Local policy checks for agent tool calls.
|
|
2
|
+
|
|
3
|
+
**This is not OPA.** It is a small built-in ruleset. Real OPA/Rego evaluation is
|
|
4
|
+
on the roadmap; `agentmetry/policies/opa/agent_rules.rego` is a draft for that work and is
|
|
5
|
+
NOT evaluated by this module.
|
|
6
|
+
|
|
7
|
+
Two design rules, both learned the hard way:
|
|
8
|
+
|
|
9
|
+
1. **Annotate, never rewrite.** A policy verdict is a separate fact from what
|
|
10
|
+
actually happened. The previous version set `action.outcome = "denied"` on
|
|
11
|
+
events that had *already executed* on the agent's machine, which put a lie in
|
|
12
|
+
the audit trail: an incident responder would read "denied" for a tool that
|
|
13
|
+
ran. It also blinded every detection rule that keys on `outcome == "success"`
|
|
14
|
+
(a burst of `shell.rm` stopped firing `destructive-delete-burst` precisely
|
|
15
|
+
because policy flagged it). The verdict now lands in its own `policy` block
|
|
16
|
+
and `action` is left alone.
|
|
17
|
+
|
|
18
|
+
2. **This cannot block.** By the time an event reaches the ingest API the tool
|
|
19
|
+
has already run. Real prevention has to happen in the hook process,
|
|
20
|
+
pre-execution, the way DLP `block` mode does. The annotation records
|
|
21
|
+
`enforced: false` so nobody mistakes it for enforcement.
|
|
22
|
+
|
|
23
|
+
Off by default (`AGENTMETRY_POLICY_ENABLED=1` to turn on): the ruleset below is
|
|
24
|
+
a hardcoded starting point, not something to impose on every operator.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
from dataclasses import dataclass
|
|
30
|
+
from typing import Any
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass(frozen=True)
|
|
34
|
+
class PolicyVerdict:
|
|
35
|
+
allowed: bool
|
|
36
|
+
rule_id: str = ""
|
|
37
|
+
reason: str = ""
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _norm(name: str) -> str:
|
|
41
|
+
"""Fold a tool name the same way core.audit.mitre does.
|
|
42
|
+
|
|
43
|
+
Exact string matching on `tool.qualified` was trivially evadable: a rule for
|
|
44
|
+
"shell.rm" missed "shell.RM" and "shell_rm". Normalizing at least closes the
|
|
45
|
+
spelling gap. It does NOT close the real gap (see module docstring): a shell
|
|
46
|
+
tool running `rm -rf` under any name is still not caught here.
|
|
47
|
+
"""
|
|
48
|
+
return name.lower().replace("_", "").replace("-", "")
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
# Tools that should not run unattended. Deliberately tiny: this is a default,
|
|
52
|
+
# not a policy language. The roadmap replaces it with real Rego.
|
|
53
|
+
_RESTRICTED: dict[str, str] = {
|
|
54
|
+
_norm("kubectl.exec"): "restricted:kubectl-exec",
|
|
55
|
+
_norm("aws.iam.delete_user"): "restricted:iam-delete-user",
|
|
56
|
+
_norm("shell.rm"): "restricted:shell-rm",
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def evaluate_policy(event: dict[str, Any]) -> PolicyVerdict:
|
|
61
|
+
"""Return a verdict for one canonical event. Never mutates the event."""
|
|
62
|
+
tool = event.get("tool")
|
|
63
|
+
qualified = str(tool.get("qualified") or "") if isinstance(tool, dict) else ""
|
|
64
|
+
if not qualified:
|
|
65
|
+
return PolicyVerdict(allowed=True)
|
|
66
|
+
|
|
67
|
+
rule_id = _RESTRICTED.get(_norm(qualified))
|
|
68
|
+
if rule_id is None:
|
|
69
|
+
return PolicyVerdict(allowed=True)
|
|
70
|
+
|
|
71
|
+
# A restricted tool is allowed when a human explicitly approved this action.
|
|
72
|
+
action = event.get("action") or {}
|
|
73
|
+
if action.get("type") == "approval_response" and action.get("outcome") == "success":
|
|
74
|
+
return PolicyVerdict(allowed=True)
|
|
75
|
+
|
|
76
|
+
return PolicyVerdict(
|
|
77
|
+
allowed=False,
|
|
78
|
+
rule_id=rule_id,
|
|
79
|
+
reason=f"{qualified} is restricted and had no recorded human approval",
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def annotate(event: dict[str, Any]) -> None:
|
|
84
|
+
"""Attach a policy verdict to `event['policy']`, in place.
|
|
85
|
+
|
|
86
|
+
Only writes when the verdict is a denial, and only ever touches the `policy`
|
|
87
|
+
key. `action.outcome` stays whatever actually happened.
|
|
88
|
+
"""
|
|
89
|
+
verdict = evaluate_policy(event)
|
|
90
|
+
if verdict.allowed:
|
|
91
|
+
return
|
|
92
|
+
event["policy"] = {
|
|
93
|
+
"decision": "deny",
|
|
94
|
+
"rule_id": verdict.rule_id,
|
|
95
|
+
"engine": "builtin",
|
|
96
|
+
# This event already executed. We observed it; we did not stop it.
|
|
97
|
+
"enforced": False,
|
|
98
|
+
"reason": verdict.reason,
|
|
99
|
+
}
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""Secret scrubbing for captured command text (Tier A/B).
|
|
2
|
+
|
|
3
|
+
When command logging is enabled, the raw command string can contain inline
|
|
4
|
+
secrets (bearer tokens, basic-auth URLs, --password values, cloud keys) that
|
|
5
|
+
key-based argument redaction never sees. Scrub those before storage.
|
|
6
|
+
|
|
7
|
+
Keep the pattern list in sync with the inline mirror in
|
|
8
|
+
`scripts/agentmetry_ingest.py` (the standalone hook cannot import this module).
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
|
|
15
|
+
_SECRET_PATTERNS: list[tuple[re.Pattern[str], str]] = [
|
|
16
|
+
(re.compile(r"(?i)(bearer\s+)[A-Za-z0-9._\-]+"), r"\1<redacted>"),
|
|
17
|
+
(re.compile(r"(?i)(authorization:\s*)\S+"), r"\1<redacted>"),
|
|
18
|
+
# user:pass@host in URLs
|
|
19
|
+
(re.compile(r"(https?://)[^/\s:@]+:[^/\s@]+@"), r"\1<redacted>@"),
|
|
20
|
+
# --password X / --token=X / -p X
|
|
21
|
+
(re.compile(r"(?i)(-{1,2}(?:password|token|secret|api[-_]?key|pwd)[=\s]+)\S+"), r"\1<redacted>"),
|
|
22
|
+
# key=value / key: value assignments
|
|
23
|
+
(
|
|
24
|
+
re.compile(
|
|
25
|
+
r"(?i)\b(password|passwd|pwd|token|secret|api[-_]?key|apikey|access[-_]?key)\s*[=:]\s*[^\s;&|\"']+"
|
|
26
|
+
),
|
|
27
|
+
r"\1=<redacted>",
|
|
28
|
+
),
|
|
29
|
+
(re.compile(r"\bAKIA[0-9A-Z]{16}\b"), "<redacted-aws-key>"),
|
|
30
|
+
(re.compile(r"\bsk-[A-Za-z0-9]{20,}\b"), "<redacted-key>"),
|
|
31
|
+
(re.compile(r"\bgh[pousr]_[A-Za-z0-9]{20,}\b"), "<redacted-gh-token>"),
|
|
32
|
+
(re.compile(r"\bxox[baprs]-[A-Za-z0-9-]{10,}\b"), "<redacted-slack-token>"),
|
|
33
|
+
]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def scrub_secrets(text: str) -> str:
|
|
37
|
+
"""Mask obvious inline secrets in a command/string. Best-effort, conservative."""
|
|
38
|
+
if not isinstance(text, str) or not text:
|
|
39
|
+
return text
|
|
40
|
+
out = text
|
|
41
|
+
for pattern, repl in _SECRET_PATTERNS:
|
|
42
|
+
out = pattern.sub(repl, out)
|
|
43
|
+
return out
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def scrub_arg_values(args: object) -> object:
|
|
47
|
+
"""Scrub secret patterns inside string values of an arguments dict."""
|
|
48
|
+
if not isinstance(args, dict):
|
|
49
|
+
return args
|
|
50
|
+
return {k: (scrub_secrets(v) if isinstance(v, str) else v) for k, v in args.items()}
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""ASCII timeline rendering for `agentmetry replay`."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from agentmetry.core.audit.canonical import normalize_outbox_row
|
|
8
|
+
|
|
9
|
+
_ICONS = {
|
|
10
|
+
"session_start": "▶",
|
|
11
|
+
"session_end": "■",
|
|
12
|
+
"approval_request": "⏸",
|
|
13
|
+
"approval_response": "✓",
|
|
14
|
+
"tool_called": "🔧",
|
|
15
|
+
"config_change": "⚙",
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def format_timeline(rows: list[dict[str, Any]], *, thread_id: str) -> str:
|
|
20
|
+
if not rows:
|
|
21
|
+
return f"No audit events for thread_id={thread_id}"
|
|
22
|
+
|
|
23
|
+
lines = [f"Agentmetry replay — correlation_id={thread_id}", f"{'─' * 60}"]
|
|
24
|
+
|
|
25
|
+
for row in rows:
|
|
26
|
+
canonical = normalize_outbox_row(row)
|
|
27
|
+
if canonical is None:
|
|
28
|
+
ts = row.get("ts", "?")
|
|
29
|
+
topic = row.get("topic", "?")
|
|
30
|
+
lines.append(f" {ts} [{topic}]")
|
|
31
|
+
continue
|
|
32
|
+
|
|
33
|
+
ts = canonical["timestamp_utc"][:19].replace("T", " ")
|
|
34
|
+
action = canonical["action"]
|
|
35
|
+
icon = _ICONS.get(action["type"], "·")
|
|
36
|
+
outcome = action["outcome"]
|
|
37
|
+
label = action["type"]
|
|
38
|
+
|
|
39
|
+
detail_parts: list[str] = []
|
|
40
|
+
if skill := canonical.get("agent", {}).get("skill_id"):
|
|
41
|
+
detail_parts.append(f"skill={skill}")
|
|
42
|
+
if tool := canonical.get("tool"):
|
|
43
|
+
detail_parts.append(f"tool={tool.get('qualified') or tool.get('name')}")
|
|
44
|
+
if outcome == "denied" and action.get("reason"):
|
|
45
|
+
detail_parts.append(f"reason={action['reason']}")
|
|
46
|
+
if action.get("reason") and action["type"] == "approval_response":
|
|
47
|
+
detail_parts.append(action["reason"])
|
|
48
|
+
|
|
49
|
+
detail = f" ({', '.join(detail_parts)})" if detail_parts else ""
|
|
50
|
+
lines.append(f" {ts} {icon} {label}/{outcome}{detail} seq={canonical.get('seq')}")
|
|
51
|
+
|
|
52
|
+
lines.append(f"{'─' * 60}")
|
|
53
|
+
lines.append(f"{len(rows)} event(s)")
|
|
54
|
+
return "\n".join(lines)
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
"""Per-run audit context — initiator provenance and last gated tool (schema v1.1)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from agentmetry.core.config import settings
|
|
8
|
+
|
|
9
|
+
# thread_id → initiator block (set at run start, server-derived only)
|
|
10
|
+
_thread_initiators: dict[str, dict[str, str]] = {}
|
|
11
|
+
# thread_id → last successful tool call on this run (for approval binding)
|
|
12
|
+
_thread_last_tool: dict[str, dict[str, str]] = {}
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _operator_id() -> str:
|
|
16
|
+
return settings.operator_id.strip() or "local"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def build_initiator(triggered_by: str) -> dict[str, str]:
|
|
20
|
+
"""Derive initiator from run origin. Must never trust client-supplied headers."""
|
|
21
|
+
operator_id = _operator_id()
|
|
22
|
+
if triggered_by == "manual":
|
|
23
|
+
return {"actor_type": "human", "trigger": "manual", "operator_id": operator_id}
|
|
24
|
+
if triggered_by.startswith("channel:"):
|
|
25
|
+
return {"actor_type": "human", "trigger": "channel", "operator_id": operator_id}
|
|
26
|
+
if triggered_by == "cron":
|
|
27
|
+
return {"actor_type": "autonomous", "trigger": "cron", "operator_id": operator_id}
|
|
28
|
+
if triggered_by == "vault_watch":
|
|
29
|
+
return {"actor_type": "autonomous", "trigger": "vault_watch", "operator_id": operator_id}
|
|
30
|
+
if triggered_by == "ingress":
|
|
31
|
+
return {"actor_type": "autonomous", "trigger": "ingress", "operator_id": operator_id}
|
|
32
|
+
if triggered_by == "recovery":
|
|
33
|
+
return {"actor_type": "autonomous", "trigger": "recovery", "operator_id": operator_id}
|
|
34
|
+
return {
|
|
35
|
+
"actor_type": "autonomous",
|
|
36
|
+
"trigger": triggered_by,
|
|
37
|
+
"operator_id": operator_id,
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def actor_from_initiator(initiator: dict[str, str]) -> dict[str, str]:
|
|
42
|
+
human = initiator.get("actor_type") == "human"
|
|
43
|
+
return {
|
|
44
|
+
"type": "user" if human else "agent",
|
|
45
|
+
"id": initiator.get("operator_id") or _operator_id(),
|
|
46
|
+
"role": "operator",
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def default_initiator() -> dict[str, str]:
|
|
51
|
+
return build_initiator("manual")
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def set_thread_initiator(thread_id: str, triggered_by: str) -> dict[str, str]:
|
|
55
|
+
initiator = build_initiator(triggered_by)
|
|
56
|
+
_thread_initiators[thread_id] = initiator
|
|
57
|
+
return initiator
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def get_thread_initiator(thread_id: str) -> dict[str, str] | None:
|
|
61
|
+
return _thread_initiators.get(thread_id)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def resolve_initiator(
|
|
65
|
+
payload: dict[str, Any], thread_id: str = ""
|
|
66
|
+
) -> dict[str, str]:
|
|
67
|
+
"""Read initiator from payload or thread cache; fallback manual human."""
|
|
68
|
+
raw = payload.get("initiator")
|
|
69
|
+
if isinstance(raw, dict) and raw.get("actor_type"):
|
|
70
|
+
return {
|
|
71
|
+
"actor_type": str(raw.get("actor_type") or "human"),
|
|
72
|
+
"trigger": str(raw.get("trigger") or "manual"),
|
|
73
|
+
"operator_id": str(raw.get("operator_id") or _operator_id()),
|
|
74
|
+
}
|
|
75
|
+
triggered_by = str(payload.get("triggered_by") or "")
|
|
76
|
+
if triggered_by:
|
|
77
|
+
return build_initiator(triggered_by)
|
|
78
|
+
if thread_id:
|
|
79
|
+
cached = _thread_initiators.get(thread_id)
|
|
80
|
+
if cached:
|
|
81
|
+
return cached
|
|
82
|
+
return default_initiator()
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def record_tool_call(thread_id: str, qualified: str, arguments_sha256: str) -> None:
|
|
86
|
+
if not thread_id:
|
|
87
|
+
return
|
|
88
|
+
server = qualified.split(".", 1)[0] if "." in qualified else ""
|
|
89
|
+
_thread_last_tool[thread_id] = {
|
|
90
|
+
"tool": qualified,
|
|
91
|
+
"server": server,
|
|
92
|
+
"input_hash": arguments_sha256,
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def last_gated_action(thread_id: str) -> dict[str, str] | None:
|
|
97
|
+
if not thread_id:
|
|
98
|
+
return None
|
|
99
|
+
action = _thread_last_tool.get(thread_id)
|
|
100
|
+
if not action or not action.get("tool"):
|
|
101
|
+
return None
|
|
102
|
+
return dict(action)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def clear_run_context(thread_id: str) -> None:
|
|
106
|
+
_thread_initiators.pop(thread_id, None)
|
|
107
|
+
_thread_last_tool.pop(thread_id, None)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def audit_payload(
|
|
111
|
+
thread_id: str,
|
|
112
|
+
triggered_by: str | None = None,
|
|
113
|
+
*,
|
|
114
|
+
initiator: dict[str, str] | None = None,
|
|
115
|
+
) -> dict[str, Any]:
|
|
116
|
+
"""Extra bus payload fields for canonical v1.1."""
|
|
117
|
+
init = initiator or (
|
|
118
|
+
get_thread_initiator(thread_id)
|
|
119
|
+
if thread_id
|
|
120
|
+
else None
|
|
121
|
+
)
|
|
122
|
+
if init is None and triggered_by:
|
|
123
|
+
init = build_initiator(triggered_by)
|
|
124
|
+
if init is None:
|
|
125
|
+
init = default_initiator()
|
|
126
|
+
out: dict[str, Any] = {"initiator": init}
|
|
127
|
+
if triggered_by:
|
|
128
|
+
out["triggered_by"] = triggered_by
|
|
129
|
+
return out
|
|
@@ -0,0 +1,235 @@
|
|
|
1
|
+
"""Agentmetry forward sinks — file, webhook, Elastic ECS, Splunk HEC."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from abc import ABC, abstractmethod
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from threading import Lock
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
import httpx
|
|
12
|
+
|
|
13
|
+
from agentmetry.core.audit.adapters.ecs import canonical_to_ecs
|
|
14
|
+
from agentmetry.core.audit.adapters.splunk import canonical_to_hec_event
|
|
15
|
+
|
|
16
|
+
logger = logging.getLogger(__name__)
|
|
17
|
+
|
|
18
|
+
_file_lock = Lock()
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class AuditSink(ABC):
|
|
22
|
+
@abstractmethod
|
|
23
|
+
async def emit(self, canonical: dict[str, Any]) -> None:
|
|
24
|
+
...
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class FileAuditSink(AuditSink):
|
|
28
|
+
def __init__(self, path: Path) -> None:
|
|
29
|
+
self._path = path
|
|
30
|
+
|
|
31
|
+
async def emit(self, canonical: dict[str, Any]) -> None:
|
|
32
|
+
from agentmetry.core.audit.trail_chain import append_chained_line
|
|
33
|
+
|
|
34
|
+
with _file_lock:
|
|
35
|
+
append_chained_line(self._path, canonical)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class WebhookAuditSink(AuditSink):
|
|
39
|
+
"""POST each event to a URL, as the canonical record or a CloudEvent.
|
|
40
|
+
|
|
41
|
+
`format="cloudevents"` wraps the same record in a CloudEvents v1.0 structured
|
|
42
|
+
envelope, which is what brokers speak: Knative, EventBridge, Event Grid,
|
|
43
|
+
Dapr and the Kafka bindings all consume it. The canonical event still travels
|
|
44
|
+
whole inside `data`, so nothing is lost by choosing it.
|
|
45
|
+
|
|
46
|
+
Default stays `canonical`. Changing the shape of what an existing webhook
|
|
47
|
+
receives because a new option appeared would break every consumer already
|
|
48
|
+
wired up, and silently.
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
def __init__(
|
|
52
|
+
self, url: str, *, timeout_seconds: float = 5.0, format: str = "canonical"
|
|
53
|
+
) -> None:
|
|
54
|
+
self._url = url
|
|
55
|
+
self._timeout = timeout_seconds
|
|
56
|
+
self._cloudevents = (format or "").strip().lower() in ("cloudevents", "cloudevent", "ce")
|
|
57
|
+
|
|
58
|
+
async def emit(self, canonical: dict[str, Any]) -> None:
|
|
59
|
+
payload = canonical
|
|
60
|
+
# `application/cloudevents+json` is what marks structured mode; a
|
|
61
|
+
# consumer distinguishes it from a bare JSON body by content type alone.
|
|
62
|
+
content_type = "application/json"
|
|
63
|
+
if self._cloudevents:
|
|
64
|
+
from agentmetry.core.audit.adapters.cloudevents import canonical_to_cloudevent
|
|
65
|
+
|
|
66
|
+
payload = canonical_to_cloudevent(canonical)
|
|
67
|
+
content_type = "application/cloudevents+json; charset=utf-8"
|
|
68
|
+
try:
|
|
69
|
+
async with httpx.AsyncClient(timeout=self._timeout) as client:
|
|
70
|
+
response = await client.post(
|
|
71
|
+
self._url,
|
|
72
|
+
json=payload,
|
|
73
|
+
headers={"Content-Type": content_type, "User-Agent": "Agentmetry/1.0"},
|
|
74
|
+
)
|
|
75
|
+
response.raise_for_status()
|
|
76
|
+
except Exception:
|
|
77
|
+
logger.exception("Audit webhook POST failed → %s", self._url)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class ElasticEcsSink(AuditSink):
|
|
81
|
+
"""Index one ECS document per event (Elasticsearch _doc API)."""
|
|
82
|
+
|
|
83
|
+
def __init__(
|
|
84
|
+
self,
|
|
85
|
+
base_url: str,
|
|
86
|
+
index: str,
|
|
87
|
+
api_key: str,
|
|
88
|
+
*,
|
|
89
|
+
timeout_seconds: float = 5.0,
|
|
90
|
+
verify_tls: bool = True,
|
|
91
|
+
) -> None:
|
|
92
|
+
self._url = base_url.rstrip("/") + f"/{index}/_doc"
|
|
93
|
+
self._api_key = api_key
|
|
94
|
+
self._timeout = timeout_seconds
|
|
95
|
+
self._verify = verify_tls
|
|
96
|
+
|
|
97
|
+
async def emit(self, canonical: dict[str, Any]) -> None:
|
|
98
|
+
doc = canonical_to_ecs(canonical)
|
|
99
|
+
headers = {
|
|
100
|
+
"Content-Type": "application/json",
|
|
101
|
+
"Authorization": f"ApiKey {self._api_key}",
|
|
102
|
+
"User-Agent": "Agentmetry/1.0",
|
|
103
|
+
}
|
|
104
|
+
try:
|
|
105
|
+
async with httpx.AsyncClient(timeout=self._timeout, verify=self._verify) as client:
|
|
106
|
+
response = await client.post(self._url, json=doc, headers=headers)
|
|
107
|
+
response.raise_for_status()
|
|
108
|
+
except Exception:
|
|
109
|
+
logger.exception("Elastic ECS index failed → %s", self._url)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
class SplunkHecSink(AuditSink):
|
|
113
|
+
"""POST one event to Splunk HTTP Event Collector."""
|
|
114
|
+
|
|
115
|
+
def __init__(
|
|
116
|
+
self,
|
|
117
|
+
hec_url: str,
|
|
118
|
+
token: str,
|
|
119
|
+
*,
|
|
120
|
+
index: str = "main",
|
|
121
|
+
sourcetype: str = "agentmetry:json",
|
|
122
|
+
timeout_seconds: float = 5.0,
|
|
123
|
+
verify_tls: bool = True,
|
|
124
|
+
) -> None:
|
|
125
|
+
base = hec_url.rstrip("/")
|
|
126
|
+
if base.endswith("/services/collector"):
|
|
127
|
+
self._url = base
|
|
128
|
+
elif base.endswith("/services/collector/event"):
|
|
129
|
+
self._url = base
|
|
130
|
+
else:
|
|
131
|
+
self._url = base + "/services/collector/event"
|
|
132
|
+
self._token = token
|
|
133
|
+
self._index = index
|
|
134
|
+
self._sourcetype = sourcetype
|
|
135
|
+
self._timeout = timeout_seconds
|
|
136
|
+
self._verify = verify_tls
|
|
137
|
+
|
|
138
|
+
async def emit(self, canonical: dict[str, Any]) -> None:
|
|
139
|
+
payload = canonical_to_hec_event(
|
|
140
|
+
canonical,
|
|
141
|
+
index=self._index,
|
|
142
|
+
sourcetype=self._sourcetype,
|
|
143
|
+
)
|
|
144
|
+
headers = {
|
|
145
|
+
"Authorization": f"Splunk {self._token}",
|
|
146
|
+
"Content-Type": "application/json",
|
|
147
|
+
"User-Agent": "Agentmetry/1.0",
|
|
148
|
+
}
|
|
149
|
+
try:
|
|
150
|
+
async with httpx.AsyncClient(timeout=self._timeout, verify=self._verify) as client:
|
|
151
|
+
response = await client.post(self._url, json=payload, headers=headers)
|
|
152
|
+
response.raise_for_status()
|
|
153
|
+
except Exception:
|
|
154
|
+
logger.exception("Splunk HEC POST failed → %s", self._url)
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
class MultiAuditSink(AuditSink):
|
|
158
|
+
def __init__(self, sinks: list[AuditSink]) -> None:
|
|
159
|
+
self._sinks = sinks
|
|
160
|
+
|
|
161
|
+
async def emit(self, canonical: dict[str, Any]) -> None:
|
|
162
|
+
for sink in self._sinks:
|
|
163
|
+
await sink.emit(canonical)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def parse_sink_modes(raw: str) -> set[str]:
|
|
167
|
+
text = raw.strip().lower()
|
|
168
|
+
if not text or text == "file":
|
|
169
|
+
return {"file"}
|
|
170
|
+
if text == "both":
|
|
171
|
+
return {"file", "webhook"}
|
|
172
|
+
if text == "all":
|
|
173
|
+
return {"file", "webhook", "elastic", "splunk"}
|
|
174
|
+
return {part.strip() for part in text.split(",") if part.strip()}
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def build_audit_sinks(
|
|
178
|
+
*,
|
|
179
|
+
modes: set[str],
|
|
180
|
+
file_path: Path,
|
|
181
|
+
webhook_url: str,
|
|
182
|
+
webhook_timeout_seconds: float,
|
|
183
|
+
webhook_format: str = "canonical",
|
|
184
|
+
elastic_url: str,
|
|
185
|
+
elastic_index: str,
|
|
186
|
+
elastic_api_key: str,
|
|
187
|
+
elastic_verify_tls: bool,
|
|
188
|
+
splunk_hec_url: str,
|
|
189
|
+
splunk_hec_token: str,
|
|
190
|
+
splunk_index: str,
|
|
191
|
+
splunk_sourcetype: str,
|
|
192
|
+
splunk_verify_tls: bool,
|
|
193
|
+
) -> AuditSink | None:
|
|
194
|
+
sinks: list[AuditSink] = []
|
|
195
|
+
|
|
196
|
+
if "file" in modes:
|
|
197
|
+
sinks.append(FileAuditSink(file_path))
|
|
198
|
+
|
|
199
|
+
if "webhook" in modes and webhook_url.strip():
|
|
200
|
+
sinks.append(
|
|
201
|
+
WebhookAuditSink(
|
|
202
|
+
webhook_url.strip(),
|
|
203
|
+
timeout_seconds=webhook_timeout_seconds,
|
|
204
|
+
format=webhook_format,
|
|
205
|
+
)
|
|
206
|
+
)
|
|
207
|
+
|
|
208
|
+
if "elastic" in modes and elastic_url.strip() and elastic_api_key.strip():
|
|
209
|
+
sinks.append(
|
|
210
|
+
ElasticEcsSink(
|
|
211
|
+
elastic_url.strip(),
|
|
212
|
+
elastic_index,
|
|
213
|
+
elastic_api_key.strip(),
|
|
214
|
+
timeout_seconds=webhook_timeout_seconds,
|
|
215
|
+
verify_tls=elastic_verify_tls,
|
|
216
|
+
)
|
|
217
|
+
)
|
|
218
|
+
|
|
219
|
+
if "splunk" in modes and splunk_hec_url.strip() and splunk_hec_token.strip():
|
|
220
|
+
sinks.append(
|
|
221
|
+
SplunkHecSink(
|
|
222
|
+
splunk_hec_url.strip(),
|
|
223
|
+
splunk_hec_token.strip(),
|
|
224
|
+
index=splunk_index,
|
|
225
|
+
sourcetype=splunk_sourcetype,
|
|
226
|
+
timeout_seconds=webhook_timeout_seconds,
|
|
227
|
+
verify_tls=splunk_verify_tls,
|
|
228
|
+
)
|
|
229
|
+
)
|
|
230
|
+
|
|
231
|
+
if not sinks:
|
|
232
|
+
return None
|
|
233
|
+
if len(sinks) == 1:
|
|
234
|
+
return sinks[0]
|
|
235
|
+
return MultiAuditSink(sinks)
|