agentmetry 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentmetry/__init__.py +12 -0
- agentmetry/api/__init__.py +0 -0
- agentmetry/api/main.py +232 -0
- agentmetry/api/routes/__init__.py +0 -0
- agentmetry/api/routes/audit.py +396 -0
- agentmetry/api/websocket.py +53 -0
- agentmetry/api/ws_bridge.py +34 -0
- agentmetry/cli/__init__.py +941 -0
- agentmetry/cli/__main__.py +5 -0
- agentmetry/core/__init__.py +0 -0
- agentmetry/core/audit/__init__.py +1 -0
- agentmetry/core/audit/adapters/__init__.py +0 -0
- agentmetry/core/audit/adapters/agt.py +304 -0
- agentmetry/core/audit/adapters/cloudevents.py +159 -0
- agentmetry/core/audit/adapters/ecs.py +103 -0
- agentmetry/core/audit/adapters/splunk.py +40 -0
- agentmetry/core/audit/alerts.py +56 -0
- agentmetry/core/audit/canonical.py +150 -0
- agentmetry/core/audit/compliance_digest.py +299 -0
- agentmetry/core/audit/detection/__init__.py +9 -0
- agentmetry/core/audit/detection/benchmark.py +194 -0
- agentmetry/core/audit/detection/corpus/attack_approval_denied_then_executed.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/attack_arbitrary_host_stage_execute.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/attack_autonomous_unapproved_write.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/attack_credential_exfil.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_credential_then_cloud_api.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_destructive_delete_burst.jsonl +6 -0
- agentmetry/core/audit/detection/corpus/attack_discovery_then_collect.jsonl +5 -0
- agentmetry/core/audit/detection/corpus/attack_dotfile_then_git_push.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_encoded_command_download.jsonl +1 -0
- agentmetry/core/audit/detection/corpus/attack_env_credential_exfil.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/attack_hashed_only_no_command.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_interpreter_egress.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_pr_merged_without_review.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_proc_substitution_cradle.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_remote_pipe_to_shell.jsonl +1 -0
- agentmetry/core/audit/detection/corpus/attack_remote_staging_then_execute.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_session_tool_burst.jsonl +42 -0
- agentmetry/core/audit/detection/corpus/attack_single_command_exfil.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_ssh_directory_exfil.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/attack_subagent_swarm.jsonl +6 -0
- agentmetry/core/audit/detection/corpus/attack_timestamp_collision.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/attack_untrusted_input_then_action.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/benign_authoring_merge_fixtures.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/benign_autonomous_after_approval.jsonl +4 -0
- agentmetry/core/audit/detection/corpus/benign_build_artifact_cleanup.jsonl +5 -0
- agentmetry/core/audit/detection/corpus/benign_ci_artifact_download.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/benign_database_migration.jsonl +5 -0
- agentmetry/core/audit/detection/corpus/benign_dependency_install_and_build.jsonl +5 -0
- agentmetry/core/audit/detection/corpus/benign_download_release_archive.jsonl +4 -0
- agentmetry/core/audit/detection/corpus/benign_fetch_data_then_run_repo_script.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/benign_fetch_dataset_then_analyse.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/benign_fetch_lockfile_then_install.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/benign_git_review_and_push.jsonl +6 -0
- agentmetry/core/audit/detection/corpus/benign_human_driven_deletes.jsonl +6 -0
- agentmetry/core/audit/detection/corpus/benign_local_api_probing.jsonl +5 -0
- agentmetry/core/audit/detection/corpus/benign_long_but_calm_session.jsonl +30 -0
- agentmetry/core/audit/detection/corpus/benign_loopback_is_not_egress.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/benign_loopback_pipe_to_interpreter.jsonl +3 -0
- agentmetry/core/audit/detection/corpus/benign_ordinary_development.jsonl +5 -0
- agentmetry/core/audit/detection/corpus/benign_package_manager_after_fetch.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/benign_reading_config_that_is_not_secret.jsonl +5 -0
- agentmetry/core/audit/detection/corpus/benign_remote_api_call_no_credentials.jsonl +4 -0
- agentmetry/core/audit/detection/corpus/benign_research_then_docs.jsonl +5 -0
- agentmetry/core/audit/detection/corpus/benign_reversed_order_is_not_exfil.jsonl +2 -0
- agentmetry/core/audit/detection/corpus/benign_test_and_fix_loop.jsonl +6 -0
- agentmetry/core/audit/detection/corpus/benign_writing_about_credentials.jsonl +6 -0
- agentmetry/core/audit/detection/corpus/corpus.yaml +443 -0
- agentmetry/core/audit/detection/disposition.py +651 -0
- agentmetry/core/audit/detection/engine.py +78 -0
- agentmetry/core/audit/detection/live.py +127 -0
- agentmetry/core/audit/detection/live_store.py +355 -0
- agentmetry/core/audit/detection/models.py +53 -0
- agentmetry/core/audit/detection/rules.py +1314 -0
- agentmetry/core/audit/detection/traits.py +648 -0
- agentmetry/core/audit/detection/yaml_config.py +91 -0
- agentmetry/core/audit/detection/yaml_rules.py +83 -0
- agentmetry/core/audit/dlp/__init__.py +4 -0
- agentmetry/core/audit/dlp/loader.py +29 -0
- agentmetry/core/audit/dlp/models.py +29 -0
- agentmetry/core/audit/dlp/scanner.py +96 -0
- agentmetry/core/audit/dogfood.py +398 -0
- agentmetry/core/audit/evidence_pack.py +500 -0
- agentmetry/core/audit/external.py +213 -0
- agentmetry/core/audit/hashing.py +21 -0
- agentmetry/core/audit/hook_bootstrap.py +451 -0
- agentmetry/core/audit/identity.py +39 -0
- agentmetry/core/audit/ingest.py +242 -0
- agentmetry/core/audit/migrate.py +73 -0
- agentmetry/core/audit/mitre.py +244 -0
- agentmetry/core/audit/policy.py +99 -0
- agentmetry/core/audit/redaction.py +50 -0
- agentmetry/core/audit/replay.py +54 -0
- agentmetry/core/audit/run_context.py +129 -0
- agentmetry/core/audit/sinks.py +235 -0
- agentmetry/core/audit/spool.py +394 -0
- agentmetry/core/audit/tool_policy/__init__.py +4 -0
- agentmetry/core/audit/tool_policy/evaluator.py +198 -0
- agentmetry/core/audit/tool_policy/loader.py +44 -0
- agentmetry/core/audit/tool_policy/models.py +25 -0
- agentmetry/core/audit/trail_chain.py +300 -0
- agentmetry/core/audit/trail_db.py +491 -0
- agentmetry/core/audit/trail_merkle.py +332 -0
- agentmetry/core/auth.py +54 -0
- agentmetry/core/bus/__init__.py +5 -0
- agentmetry/core/bus/audit_exporter.py +107 -0
- agentmetry/core/bus/bridges.py +26 -0
- agentmetry/core/bus/bus.py +102 -0
- agentmetry/core/bus/events.py +50 -0
- agentmetry/core/bus/outbox.py +124 -0
- agentmetry/core/config.py +177 -0
- agentmetry/core/diagnostics/__init__.py +0 -0
- agentmetry/core/diagnostics/autostart.py +563 -0
- agentmetry/core/diagnostics/doctor.py +535 -0
- agentmetry/core/diagnostics/driver_paths.py +156 -0
- agentmetry/core/diagnostics/env_file.py +45 -0
- agentmetry/core/drivers/__init__.py +4 -0
- agentmetry/core/drivers/host.py +263 -0
- agentmetry/core/drivers/permissions.py +37 -0
- agentmetry/core/drivers/spec.py +118 -0
- agentmetry/core/extensions.py +107 -0
- agentmetry/core/health.py +26 -0
- agentmetry/core/version.py +13 -0
- agentmetry/policies/detection/manifest.yaml +43 -0
- agentmetry/policies/dlp/manifest.yaml +161 -0
- agentmetry/policies/opa/agent_rules.rego +33 -0
- agentmetry/policies/tool/manifest.yaml +117 -0
- agentmetry-0.4.0.dist-info/METADATA +86 -0
- agentmetry-0.4.0.dist-info/RECORD +131 -0
- agentmetry-0.4.0.dist-info/WHEEL +4 -0
- agentmetry-0.4.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,535 @@
|
|
|
1
|
+
"""Agentmetry doctor - SIEM preflight (manifests, trail chain, hooks, health).
|
|
2
|
+
|
|
3
|
+
Vault/drivers checks are optional-runtime extras and can only warn.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import json
|
|
9
|
+
import os
|
|
10
|
+
import shutil
|
|
11
|
+
import sys
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Literal
|
|
15
|
+
|
|
16
|
+
from agentmetry.core.config import settings
|
|
17
|
+
from agentmetry.core.diagnostics.driver_paths import (
|
|
18
|
+
default_python,
|
|
19
|
+
entry_has_absolute_paths,
|
|
20
|
+
normalize_drivers_file,
|
|
21
|
+
orchestrator_root,
|
|
22
|
+
resolve_driver_entry,
|
|
23
|
+
)
|
|
24
|
+
from agentmetry.core.drivers.spec import DriverSpec
|
|
25
|
+
|
|
26
|
+
Severity = Literal["ok", "warn", "fail"]
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class Finding:
|
|
31
|
+
severity: Severity
|
|
32
|
+
code: str
|
|
33
|
+
message: str
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass
|
|
37
|
+
class DoctorReport:
|
|
38
|
+
findings: list[Finding] = field(default_factory=list)
|
|
39
|
+
|
|
40
|
+
def ok(self, code: str, message: str) -> None:
|
|
41
|
+
self.findings.append(Finding("ok", code, message))
|
|
42
|
+
|
|
43
|
+
def warn(self, code: str, message: str) -> None:
|
|
44
|
+
self.findings.append(Finding("warn", code, message))
|
|
45
|
+
|
|
46
|
+
def fail(self, code: str, message: str) -> None:
|
|
47
|
+
self.findings.append(Finding("fail", code, message))
|
|
48
|
+
|
|
49
|
+
@property
|
|
50
|
+
def exit_code(self) -> int:
|
|
51
|
+
if any(f.severity == "fail" for f in self.findings):
|
|
52
|
+
return 1
|
|
53
|
+
return 0
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _check_health_endpoint(report: DoctorReport) -> None:
|
|
57
|
+
"""Is the orchestrator up? A recorder that is not running records nothing."""
|
|
58
|
+
import urllib.request
|
|
59
|
+
|
|
60
|
+
url = settings.audit_ingest_url.rstrip("/") + "/api/v1/health"
|
|
61
|
+
try:
|
|
62
|
+
with urllib.request.urlopen(url, timeout=2) as resp:
|
|
63
|
+
if resp.status == 200:
|
|
64
|
+
report.ok("orchestrator_up", f"Orchestrator responding at {url}")
|
|
65
|
+
return
|
|
66
|
+
report.warn("orchestrator_up", f"Orchestrator returned HTTP {resp.status} at {url}")
|
|
67
|
+
except Exception:
|
|
68
|
+
report.warn(
|
|
69
|
+
"orchestrator_up",
|
|
70
|
+
f"Orchestrator not reachable at {url} - start it with `agentmetry start`",
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
#: Bind addresses that keep the API on this machine.
|
|
75
|
+
_LOOPBACK = frozenset({"127.0.0.1", "localhost", "::1", ""})
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _check_exposure(report: DoctorReport) -> None:
|
|
79
|
+
"""Is the API reachable by anyone who cannot already read the trail?
|
|
80
|
+
|
|
81
|
+
`require_api_key` is deliberately a no-op when no key is set, which is the
|
|
82
|
+
right default for a recorder bound to loopback. Combined with a non-loopback
|
|
83
|
+
bind it is not a weak default, it is an open door: the ingest route accepts
|
|
84
|
+
forged events into the tamper-evident trail, the export route hands over the
|
|
85
|
+
whole evidence pack, and the disposition route lets a stranger close a
|
|
86
|
+
finding as accepted risk - a decision that is then written into the trail as
|
|
87
|
+
a legitimate human action.
|
|
88
|
+
|
|
89
|
+
The enterprise MSI reached exactly that combination by setting
|
|
90
|
+
AGENTMETRY_HOST=0.0.0.0 without setting a key, so this check fails rather
|
|
91
|
+
than warns. Any future packaging that repeats the mistake trips it here.
|
|
92
|
+
"""
|
|
93
|
+
if not settings.fleet_id.strip():
|
|
94
|
+
report.warn(
|
|
95
|
+
"fleet_id",
|
|
96
|
+
"AGENTMETRY_FLEET_ID not set - fleet SIEM queries cannot scope to "
|
|
97
|
+
"this org or business unit",
|
|
98
|
+
)
|
|
99
|
+
else:
|
|
100
|
+
report.ok("fleet_id", f"Fleet id: {settings.fleet_id.strip()}")
|
|
101
|
+
|
|
102
|
+
host = os.environ.get("AGENTMETRY_HOST", "127.0.0.1").strip()
|
|
103
|
+
has_key = bool(settings.api_key.strip())
|
|
104
|
+
|
|
105
|
+
if host in _LOOPBACK:
|
|
106
|
+
detail = "loopback only" if has_key else "loopback only (no API key needed)"
|
|
107
|
+
report.ok("exposure", f"API bound to {host or '127.0.0.1'} - {detail}")
|
|
108
|
+
return
|
|
109
|
+
|
|
110
|
+
if has_key:
|
|
111
|
+
report.ok("exposure", f"API bound to {host} with an API key set")
|
|
112
|
+
return
|
|
113
|
+
|
|
114
|
+
report.fail(
|
|
115
|
+
"exposure",
|
|
116
|
+
f"API bound to {host} with NO API key. Anyone who can reach this host "
|
|
117
|
+
"can read the trail, export evidence, inject events, and close "
|
|
118
|
+
"detections. Set AGENTMETRY_API_KEY, or bind 127.0.0.1.",
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _check_hooks_installed(report: DoctorReport) -> None:
|
|
123
|
+
"""Detect global hook installs. Absence is a warn: capture is opt-in per IDE."""
|
|
124
|
+
targets = {
|
|
125
|
+
"cursor": Path.home() / ".cursor" / "hooks.json",
|
|
126
|
+
"claude": Path.home() / ".claude" / "settings.json",
|
|
127
|
+
}
|
|
128
|
+
installed: list[str] = []
|
|
129
|
+
missing: list[str] = []
|
|
130
|
+
for name, path in targets.items():
|
|
131
|
+
try:
|
|
132
|
+
if path.is_file() and "agentmetry_ingest" in path.read_text(encoding="utf-8"):
|
|
133
|
+
installed.append(name)
|
|
134
|
+
else:
|
|
135
|
+
missing.append(name)
|
|
136
|
+
except OSError:
|
|
137
|
+
missing.append(name)
|
|
138
|
+
if installed:
|
|
139
|
+
report.ok("hooks", f"Hooks installed: {', '.join(installed)}")
|
|
140
|
+
if missing:
|
|
141
|
+
report.warn(
|
|
142
|
+
"hooks",
|
|
143
|
+
f"No hooks detected for: {', '.join(missing)} "
|
|
144
|
+
"(installed at orchestrator boot, or run scripts/install_*_hooks.ps1)",
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _check_trail(report: DoctorReport) -> None:
|
|
149
|
+
trail = Path(settings.audit_export_path)
|
|
150
|
+
if not trail.is_file():
|
|
151
|
+
report.warn(
|
|
152
|
+
"trail",
|
|
153
|
+
f"No trail yet at {trail.name} - run `python scripts/demo.py` or capture a session",
|
|
154
|
+
)
|
|
155
|
+
return
|
|
156
|
+
from agentmetry.core.audit.trail_chain import verify_trail_file
|
|
157
|
+
|
|
158
|
+
result = verify_trail_file(trail)
|
|
159
|
+
if result.ok:
|
|
160
|
+
report.ok("trail", f"Trail chain verified: {result.message}")
|
|
161
|
+
else:
|
|
162
|
+
report.fail("trail", f"Trail chain BROKEN: {result.message}")
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _check_triage(report: DoctorReport) -> None:
|
|
166
|
+
"""Surface the triage backlog without making the operator open the UI.
|
|
167
|
+
|
|
168
|
+
A growing pile of undispositioned findings is the failure mode this product
|
|
169
|
+
is most exposed to: detection keeps working, nobody answers it, and the
|
|
170
|
+
evidence pack quietly says so. Warn, never fail - an untriaged detection is
|
|
171
|
+
a task, not a broken install.
|
|
172
|
+
"""
|
|
173
|
+
from agentmetry.core.audit.detection.disposition import CLOSED_STATUSES, get_disposition_store
|
|
174
|
+
|
|
175
|
+
try:
|
|
176
|
+
counts = get_disposition_store().counts()
|
|
177
|
+
except Exception as exc: # a missing store must not sink the whole report
|
|
178
|
+
report.warn("triage", f"Could not read triage state: {exc}")
|
|
179
|
+
return
|
|
180
|
+
|
|
181
|
+
decided = sum(counts.values())
|
|
182
|
+
if not decided:
|
|
183
|
+
report.warn(
|
|
184
|
+
"triage",
|
|
185
|
+
"No detections have been dispositioned. Detections evidence that "
|
|
186
|
+
"the system noticed, not that anyone acted.",
|
|
187
|
+
)
|
|
188
|
+
return
|
|
189
|
+
|
|
190
|
+
open_findings = sum(n for s, n in counts.items() if s not in CLOSED_STATUSES)
|
|
191
|
+
report.ok(
|
|
192
|
+
"triage",
|
|
193
|
+
f"{decided} detection(s) dispositioned; {open_findings} still open",
|
|
194
|
+
)
|
|
195
|
+
|
|
196
|
+
# A decision about a rule that no longer exists is still evidence somebody
|
|
197
|
+
# reviewed something, so it is never dropped. It does need saying out loud,
|
|
198
|
+
# or an auditor reads a retired rule as an unreviewed finding.
|
|
199
|
+
try:
|
|
200
|
+
orphans = get_disposition_store().orphaned()
|
|
201
|
+
except Exception:
|
|
202
|
+
return
|
|
203
|
+
if orphans:
|
|
204
|
+
rules = sorted({str(o["rule_id"]) for o in orphans})
|
|
205
|
+
report.warn(
|
|
206
|
+
"triage_orphans",
|
|
207
|
+
f"{len(orphans)} disposition(s) reference rules that no longer exist "
|
|
208
|
+
f"({', '.join(rules[:3])}). Kept as evidence; add a RULE_ALIASES "
|
|
209
|
+
"entry if the rule was renamed rather than retired.",
|
|
210
|
+
)
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _check_autostart(report: DoctorReport) -> None:
|
|
214
|
+
"""Say whether anything will restart the recorder without a human.
|
|
215
|
+
|
|
216
|
+
`agentmetry install` has existed for a while and nothing ever mentioned it.
|
|
217
|
+
On the machine where this check was written it had never been run, and the
|
|
218
|
+
result was five days of agent activity sitting in the hook spool while the
|
|
219
|
+
trail looked healthy. A capability nobody is told about is worth about as
|
|
220
|
+
much as one that does not exist.
|
|
221
|
+
|
|
222
|
+
A warning rather than a failure: running the recorder by hand is a
|
|
223
|
+
legitimate choice, and doctor should not fail an operator for making it.
|
|
224
|
+
|
|
225
|
+
A registration that exists but does not work is a different matter, and it
|
|
226
|
+
does fail. Nobody chose that, it looks identical to working from the
|
|
227
|
+
outside, and it is the state this check was in for a day: the task launched
|
|
228
|
+
a module path a package rename had removed, exited 1 every minute, and
|
|
229
|
+
doctor called it OK because something was registered.
|
|
230
|
+
"""
|
|
231
|
+
from agentmetry.core.diagnostics import autostart
|
|
232
|
+
|
|
233
|
+
state = autostart.status()
|
|
234
|
+
if state.configured and state.healthy is False:
|
|
235
|
+
report.fail("autostart", f"Autostart is broken ({state.backend}): {state.detail}")
|
|
236
|
+
return
|
|
237
|
+
if state.configured:
|
|
238
|
+
report.ok("autostart", f"Starts automatically ({state.backend}): {state.detail}")
|
|
239
|
+
return
|
|
240
|
+
report.warn(
|
|
241
|
+
"autostart",
|
|
242
|
+
f"Nothing restarts the recorder ({state.backend}: {state.detail}). "
|
|
243
|
+
"Hooks keep capturing to the spool while it is down, and spooled events "
|
|
244
|
+
"expire after 7 days. Run `agentmetry install` to fix.",
|
|
245
|
+
)
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
# A backlog this deep, or this old, is no longer "the orchestrator restarted a
|
|
249
|
+
# moment ago". It means capture is not reaching the trail, and for a flight
|
|
250
|
+
# recorder that is a failure, not a note.
|
|
251
|
+
_SPOOL_FAIL_DEPTH = 100
|
|
252
|
+
_SPOOL_FAIL_AGE_SECONDS = 24 * 3600
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _check_spool(report: DoctorReport) -> None:
|
|
256
|
+
"""Surface events the hooks captured but the trail has not accepted.
|
|
257
|
+
|
|
258
|
+
A small spool is normal for a moment after a restart and drains on a timer.
|
|
259
|
+
A large or old one means the orchestrator is not reachable from the hooks,
|
|
260
|
+
and the operator would otherwise see a healthy-looking trail that is quietly
|
|
261
|
+
missing sessions. Events past MAX_AGE_SECONDS stop being replayable, so the
|
|
262
|
+
age is a countdown, not a statistic.
|
|
263
|
+
"""
|
|
264
|
+
from agentmetry.core.audit.spool import (
|
|
265
|
+
MAX_AGE_SECONDS,
|
|
266
|
+
expired_path,
|
|
267
|
+
spool_depth,
|
|
268
|
+
spool_oldest_age_seconds,
|
|
269
|
+
)
|
|
270
|
+
|
|
271
|
+
depth = spool_depth()
|
|
272
|
+
quarantined = expired_path()
|
|
273
|
+
|
|
274
|
+
if depth == 0:
|
|
275
|
+
if quarantined.is_file():
|
|
276
|
+
report.warn(
|
|
277
|
+
"spool",
|
|
278
|
+
f"Spool empty, but past events were quarantined unreplayed: {quarantined}",
|
|
279
|
+
)
|
|
280
|
+
else:
|
|
281
|
+
report.ok("spool", "No pending hook spool (capture is reaching the trail)")
|
|
282
|
+
return
|
|
283
|
+
|
|
284
|
+
age = spool_oldest_age_seconds() or 0.0
|
|
285
|
+
hours = age / 3600
|
|
286
|
+
message = f"{depth} event(s) pending replay; oldest {hours:.1f}h old"
|
|
287
|
+
|
|
288
|
+
if depth >= _SPOOL_FAIL_DEPTH or age >= _SPOOL_FAIL_AGE_SECONDS:
|
|
289
|
+
remaining = (MAX_AGE_SECONDS - age) / 3600
|
|
290
|
+
if remaining <= 0:
|
|
291
|
+
message += ". The oldest are past the replay window already"
|
|
292
|
+
else:
|
|
293
|
+
message += f". The oldest become unreplayable in {remaining:.0f}h"
|
|
294
|
+
message += ". Is the orchestrator running and reachable at the hook's base URL?"
|
|
295
|
+
report.fail("spool", message)
|
|
296
|
+
return
|
|
297
|
+
|
|
298
|
+
report.warn("spool", message)
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def _check_manifests(report: DoctorReport) -> None:
|
|
302
|
+
dlp_path = Path(settings.dlp_rules_path)
|
|
303
|
+
if not dlp_path.is_file():
|
|
304
|
+
report.fail("dlp", f"DLP manifest missing at {dlp_path}")
|
|
305
|
+
else:
|
|
306
|
+
try:
|
|
307
|
+
from agentmetry.core.audit.dlp.loader import load_dlp_rules
|
|
308
|
+
|
|
309
|
+
rules = load_dlp_rules(dlp_path)
|
|
310
|
+
report.ok("dlp", f"{len(rules)} DLP rules load from {dlp_path.name}")
|
|
311
|
+
except Exception as exc:
|
|
312
|
+
report.fail("dlp", f"DLP manifest failed to load: {exc}")
|
|
313
|
+
|
|
314
|
+
tp_path = Path(settings.tool_policy_path)
|
|
315
|
+
if not tp_path.is_file():
|
|
316
|
+
report.fail("tool_policy", f"Tool policy manifest missing at {tp_path}")
|
|
317
|
+
else:
|
|
318
|
+
try:
|
|
319
|
+
from agentmetry.core.audit.tool_policy.loader import load_tool_policy
|
|
320
|
+
|
|
321
|
+
rules, default = load_tool_policy(tp_path)
|
|
322
|
+
report.ok(
|
|
323
|
+
"tool_policy",
|
|
324
|
+
f"{len(rules)} tool policy rules load (default: {default})",
|
|
325
|
+
)
|
|
326
|
+
except Exception as exc:
|
|
327
|
+
report.fail("tool_policy", f"Tool policy manifest failed to load: {exc}")
|
|
328
|
+
|
|
329
|
+
det_path = Path(settings.detection_rules_path)
|
|
330
|
+
if not det_path.is_file():
|
|
331
|
+
report.fail("detection", f"Detection manifest missing at {det_path}")
|
|
332
|
+
else:
|
|
333
|
+
try:
|
|
334
|
+
from agentmetry.core.audit.detection.yaml_config import load_manifest
|
|
335
|
+
|
|
336
|
+
manifest = load_manifest(reload=True)
|
|
337
|
+
thresholds = manifest.get("thresholds") or {}
|
|
338
|
+
count_rules = manifest.get("count_rules") or []
|
|
339
|
+
report.ok(
|
|
340
|
+
"detection",
|
|
341
|
+
f"{len(count_rules)} YAML count rules + {len(thresholds)} thresholds from {det_path.name}",
|
|
342
|
+
)
|
|
343
|
+
except Exception as exc:
|
|
344
|
+
report.fail("detection", f"Detection manifest failed to load: {exc}")
|
|
345
|
+
|
|
346
|
+
report.ok(
|
|
347
|
+
"hook_enforcement",
|
|
348
|
+
f"Tool policy={settings.tool_policy_mode}, DLP={settings.dlp_mode} "
|
|
349
|
+
"(set block in .env or install.ps1 -ToolPolicyBlock)",
|
|
350
|
+
)
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
def _check_optional_vault(
|
|
354
|
+
report: DoctorReport, vault: Path, *, fix_drivers: bool
|
|
355
|
+
) -> None:
|
|
356
|
+
"""Demo MCP vault checks - optional runtime, never a doctor failure.
|
|
357
|
+
|
|
358
|
+
The SIEM records IDE hook traffic with no vault at all. These checks only
|
|
359
|
+
run when a vault directory exists, and the worst they produce is a warn.
|
|
360
|
+
"""
|
|
361
|
+
if not vault.is_dir():
|
|
362
|
+
report.ok("vault", "Demo MCP vault not present (optional) - skipped")
|
|
363
|
+
return
|
|
364
|
+
|
|
365
|
+
report.ok("vault", f"Demo vault found at {vault} (optional runtime)")
|
|
366
|
+
drivers_path = vault / ".system" / "drivers.json"
|
|
367
|
+
example_path = vault / ".system" / "drivers.json.example"
|
|
368
|
+
|
|
369
|
+
if not drivers_path.is_file():
|
|
370
|
+
if fix_drivers and example_path.is_file():
|
|
371
|
+
shutil.copy(example_path, drivers_path)
|
|
372
|
+
report.ok("drivers", f"Created {drivers_path.name} from drivers.json.example")
|
|
373
|
+
else:
|
|
374
|
+
report.warn(
|
|
375
|
+
"drivers",
|
|
376
|
+
f"No {drivers_path.name} - demo MCP drivers disabled "
|
|
377
|
+
"(copy drivers.json.example or run `agentmetry doctor --fix`)",
|
|
378
|
+
)
|
|
379
|
+
return
|
|
380
|
+
|
|
381
|
+
try:
|
|
382
|
+
raw = json.loads(drivers_path.read_text(encoding="utf-8"))
|
|
383
|
+
except json.JSONDecodeError as exc:
|
|
384
|
+
report.warn("drivers", f"drivers.json invalid JSON: {exc}")
|
|
385
|
+
return
|
|
386
|
+
|
|
387
|
+
drivers = raw.get("drivers") or []
|
|
388
|
+
report.ok("drivers", f"{len(drivers)} driver entries in drivers.json")
|
|
389
|
+
|
|
390
|
+
absolute_entries = [d.get("name", "?") for d in drivers if entry_has_absolute_paths(d)]
|
|
391
|
+
if absolute_entries:
|
|
392
|
+
if fix_drivers:
|
|
393
|
+
if normalize_drivers_file(drivers_path, vault_path=vault):
|
|
394
|
+
report.ok(
|
|
395
|
+
"drivers_portable",
|
|
396
|
+
"Rewrote drivers.json with {PYTHON}/{ORCHESTRATOR_ROOT}/{VAULT_PATH} tokens",
|
|
397
|
+
)
|
|
398
|
+
raw = json.loads(drivers_path.read_text(encoding="utf-8"))
|
|
399
|
+
drivers = raw.get("drivers") or []
|
|
400
|
+
absolute_entries = [
|
|
401
|
+
d.get("name", "?") for d in drivers if entry_has_absolute_paths(d)
|
|
402
|
+
]
|
|
403
|
+
else:
|
|
404
|
+
report.warn("drivers_portable", "Nothing to rewrite in drivers.json")
|
|
405
|
+
if absolute_entries:
|
|
406
|
+
report.warn(
|
|
407
|
+
"drivers_absolute",
|
|
408
|
+
f"Machine-specific paths in: {', '.join(absolute_entries)} "
|
|
409
|
+
"(run `agentmetry doctor --fix`)",
|
|
410
|
+
)
|
|
411
|
+
else:
|
|
412
|
+
report.ok("drivers_portable", "drivers.json uses portable path tokens")
|
|
413
|
+
|
|
414
|
+
invalid: list[str] = []
|
|
415
|
+
for entry in drivers:
|
|
416
|
+
try:
|
|
417
|
+
DriverSpec.model_validate(resolve_driver_entry(entry, vault_path=vault))
|
|
418
|
+
except Exception:
|
|
419
|
+
invalid.append(str(entry.get("name", "?")))
|
|
420
|
+
if invalid:
|
|
421
|
+
report.warn("drivers_schema", f"Invalid driver entries: {', '.join(invalid)}")
|
|
422
|
+
else:
|
|
423
|
+
report.ok("drivers_schema", "All driver entries validate")
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
def _check_extensions(report: DoctorReport) -> None:
|
|
427
|
+
"""Report enterprise extension packages (entry points) if installed."""
|
|
428
|
+
from agentmetry.core.extensions import _iter_extension_entry_points, get_extension_registry
|
|
429
|
+
|
|
430
|
+
registry = get_extension_registry()
|
|
431
|
+
if registry.loaded:
|
|
432
|
+
names = ", ".join(item.name for item in registry.loaded)
|
|
433
|
+
report.ok("extensions", f"Enterprise extensions loaded: {names}")
|
|
434
|
+
return
|
|
435
|
+
|
|
436
|
+
eps = list(_iter_extension_entry_points())
|
|
437
|
+
if not eps:
|
|
438
|
+
report.ok("extensions", "Open-source core (no enterprise extensions installed)")
|
|
439
|
+
return
|
|
440
|
+
|
|
441
|
+
names = ", ".join(sorted(ep.name for ep in eps))
|
|
442
|
+
report.ok(
|
|
443
|
+
"extensions",
|
|
444
|
+
f"Enterprise extension packages installed ({names}) - loaded on orchestrator start",
|
|
445
|
+
)
|
|
446
|
+
|
|
447
|
+
|
|
448
|
+
def run_doctor(
|
|
449
|
+
*,
|
|
450
|
+
vault_path: Path | None = None,
|
|
451
|
+
fix_drivers: bool = False,
|
|
452
|
+
) -> DoctorReport:
|
|
453
|
+
"""SIEM preflight. The recorder is the product; the demo vault is optional.
|
|
454
|
+
|
|
455
|
+
Order and severity reflect that: a missing DLP manifest or a broken trail
|
|
456
|
+
chain is a failure, a missing vault is not - the previous doctor hard-failed
|
|
457
|
+
on vault/drivers.json and returned early, so a recorder-only install (the
|
|
458
|
+
documented quick start) showed FAIL while capturing perfectly. Vault checks
|
|
459
|
+
now run last and can only warn.
|
|
460
|
+
"""
|
|
461
|
+
report = DoctorReport()
|
|
462
|
+
orch = orchestrator_root()
|
|
463
|
+
|
|
464
|
+
# --- SIEM flight recorder ------------------------------------------------
|
|
465
|
+
# A source checkout has a pyproject next to the package; an installed one
|
|
466
|
+
# does not, and never will. Failing on its absence told every pip user their
|
|
467
|
+
# working install was broken as the first line of the first command they
|
|
468
|
+
# run, which is a poor way to meet someone.
|
|
469
|
+
if (orch / "pyproject.toml").is_file():
|
|
470
|
+
report.ok("orchestrator", f"Orchestrator root {orch}")
|
|
471
|
+
elif (Path(__file__).resolve().parents[2] / "__init__.py").is_file():
|
|
472
|
+
report.ok("orchestrator", f"Installed package at {Path(__file__).resolve().parents[2]}")
|
|
473
|
+
else:
|
|
474
|
+
report.fail("orchestrator", f"Expected orchestrator at {orch}")
|
|
475
|
+
|
|
476
|
+
py = Path(default_python())
|
|
477
|
+
if py.is_file():
|
|
478
|
+
report.ok("python", f"Python interpreter {py}")
|
|
479
|
+
else:
|
|
480
|
+
report.warn("python", f"Python not found at {py} - run pip install -e '.[dev]'")
|
|
481
|
+
|
|
482
|
+
env_file = orch / ".env"
|
|
483
|
+
if env_file.is_file():
|
|
484
|
+
report.ok("env", f"Found {env_file.name} (secrets stay gitignored)")
|
|
485
|
+
else:
|
|
486
|
+
report.warn("env", f"No {env_file} - copy from .env.example if needed")
|
|
487
|
+
|
|
488
|
+
data_dir = orch / "data"
|
|
489
|
+
try:
|
|
490
|
+
data_dir.mkdir(parents=True, exist_ok=True)
|
|
491
|
+
probe = data_dir / ".doctor-probe"
|
|
492
|
+
probe.write_text("ok", encoding="utf-8")
|
|
493
|
+
probe.unlink()
|
|
494
|
+
report.ok("data", f"Data directory writable: {data_dir}")
|
|
495
|
+
except OSError as exc:
|
|
496
|
+
report.fail("data", f"Data directory not writable ({data_dir}): {exc}")
|
|
497
|
+
|
|
498
|
+
_check_manifests(report)
|
|
499
|
+
_check_exposure(report)
|
|
500
|
+
_check_trail(report)
|
|
501
|
+
_check_triage(report)
|
|
502
|
+
_check_spool(report)
|
|
503
|
+
_check_autostart(report)
|
|
504
|
+
_check_health_endpoint(report)
|
|
505
|
+
_check_hooks_installed(report)
|
|
506
|
+
_check_extensions(report)
|
|
507
|
+
|
|
508
|
+
# --- Optional governed runtime (demo vault) ------------------------------
|
|
509
|
+
vault = Path(vault_path or settings.vault_path).resolve()
|
|
510
|
+
_check_optional_vault(report, vault, fix_drivers=fix_drivers)
|
|
511
|
+
|
|
512
|
+
return report
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
def format_report(report: DoctorReport) -> str:
|
|
516
|
+
lines: list[str] = []
|
|
517
|
+
for finding in report.findings:
|
|
518
|
+
prefix = {"ok": "OK", "warn": "WARN", "fail": "FAIL"}[finding.severity]
|
|
519
|
+
lines.append(f" [{prefix}] {finding.message}")
|
|
520
|
+
return "\n".join(lines)
|
|
521
|
+
|
|
522
|
+
|
|
523
|
+
def main(argv: list[str] | None = None) -> int:
|
|
524
|
+
import argparse
|
|
525
|
+
|
|
526
|
+
parser = argparse.ArgumentParser(prog="agentmetry-doctor")
|
|
527
|
+
parser.add_argument("--fix", action="store_true", help="rewrite drivers.json to portable tokens")
|
|
528
|
+
args = parser.parse_args(argv)
|
|
529
|
+
report = run_doctor(fix_drivers=args.fix)
|
|
530
|
+
print("Agentmetry doctor\n" + format_report(report))
|
|
531
|
+
return report.exit_code
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
if __name__ == "__main__":
|
|
535
|
+
sys.exit(main())
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
"""Portable driver path tokens and drivers.json normalization."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import logging
|
|
7
|
+
import re
|
|
8
|
+
import sys
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from pydantic import ValidationError
|
|
13
|
+
|
|
14
|
+
from agentmetry.core.config import settings
|
|
15
|
+
from agentmetry.core.drivers.spec import DriverSpec
|
|
16
|
+
|
|
17
|
+
logger = logging.getLogger(__name__)
|
|
18
|
+
|
|
19
|
+
_ORCH_ROOT = Path(__file__).resolve().parents[3]
|
|
20
|
+
_REPO_ROOT = _ORCH_ROOT.parents[1]
|
|
21
|
+
|
|
22
|
+
TOKEN_PYTHON = "{PYTHON}"
|
|
23
|
+
TOKEN_ORCH = "{ORCHESTRATOR_ROOT}"
|
|
24
|
+
TOKEN_REPO = "{REPO_ROOT}"
|
|
25
|
+
TOKEN_VAULT = "{VAULT_PATH}"
|
|
26
|
+
|
|
27
|
+
_PORTABLE_TOKENS = (TOKEN_PYTHON, TOKEN_ORCH, TOKEN_REPO, TOKEN_VAULT)
|
|
28
|
+
|
|
29
|
+
_ABSOLUTE_PATH_RE = re.compile(
|
|
30
|
+
r"(?:[A-Za-z]:[\\/]|/Users/|/home/|C:/Users/|C:\\Users\\)"
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def orchestrator_root() -> Path:
|
|
35
|
+
return _ORCH_ROOT
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def repo_root() -> Path:
|
|
39
|
+
return _REPO_ROOT
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def default_python() -> str:
|
|
43
|
+
venv_py = _ORCH_ROOT / ".venv" / "Scripts" / "python.exe"
|
|
44
|
+
if venv_py.is_file():
|
|
45
|
+
return str(venv_py)
|
|
46
|
+
venv_bin = _ORCH_ROOT / ".venv" / "bin" / "python"
|
|
47
|
+
if venv_bin.is_file():
|
|
48
|
+
return str(venv_bin)
|
|
49
|
+
return sys.executable
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def placeholder_map(vault_path: Path | None = None) -> dict[str, str]:
|
|
53
|
+
vault = Path(vault_path or settings.vault_path).resolve()
|
|
54
|
+
return {
|
|
55
|
+
TOKEN_PYTHON: default_python().replace("\\", "/"),
|
|
56
|
+
TOKEN_ORCH: str(_ORCH_ROOT.resolve()).replace("\\", "/"),
|
|
57
|
+
TOKEN_REPO: str(_REPO_ROOT.resolve()).replace("\\", "/"),
|
|
58
|
+
TOKEN_VAULT: str(vault).replace("\\", "/"),
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def expand_placeholders(value: str, *, vault_path: Path | None = None) -> str:
|
|
63
|
+
text = value.replace("\\", "/")
|
|
64
|
+
for token, resolved in placeholder_map(vault_path).items():
|
|
65
|
+
text = text.replace(token, resolved)
|
|
66
|
+
return text
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def collapse_to_portable(value: str, *, vault_path: Path | None = None) -> str:
|
|
70
|
+
text = value.replace("\\", "/")
|
|
71
|
+
mapping = placeholder_map(vault_path)
|
|
72
|
+
py = mapping[TOKEN_PYTHON]
|
|
73
|
+
if py and py in text:
|
|
74
|
+
return text.replace(py, TOKEN_PYTHON)
|
|
75
|
+
for token, resolved in sorted(mapping.items(), key=lambda kv: len(kv[1]), reverse=True):
|
|
76
|
+
if token == TOKEN_PYTHON:
|
|
77
|
+
continue
|
|
78
|
+
if resolved and resolved in text:
|
|
79
|
+
text = text.replace(resolved, token)
|
|
80
|
+
return text
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def resolve_driver_entry(entry: dict[str, Any], *, vault_path: Path | None = None) -> dict[str, Any]:
|
|
84
|
+
resolved = dict(entry)
|
|
85
|
+
command = resolved.get("command")
|
|
86
|
+
if isinstance(command, str):
|
|
87
|
+
resolved["command"] = expand_placeholders(command, vault_path=vault_path)
|
|
88
|
+
args = resolved.get("args") or []
|
|
89
|
+
resolved["args"] = [
|
|
90
|
+
expand_placeholders(arg, vault_path=vault_path) if isinstance(arg, str) else arg
|
|
91
|
+
for arg in args
|
|
92
|
+
]
|
|
93
|
+
return resolved
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def normalize_driver_entry(entry: dict[str, Any], *, vault_path: Path | None = None) -> dict[str, Any]:
|
|
97
|
+
normalized = dict(entry)
|
|
98
|
+
command = normalized.get("command")
|
|
99
|
+
if isinstance(command, str):
|
|
100
|
+
normalized["command"] = collapse_to_portable(command, vault_path=vault_path)
|
|
101
|
+
args = normalized.get("args") or []
|
|
102
|
+
normalized["args"] = [
|
|
103
|
+
collapse_to_portable(arg, vault_path=vault_path) if isinstance(arg, str) else arg
|
|
104
|
+
for arg in args
|
|
105
|
+
]
|
|
106
|
+
return normalized
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def entry_has_absolute_paths(entry: dict[str, Any]) -> bool:
|
|
110
|
+
parts: list[str] = []
|
|
111
|
+
if isinstance(entry.get("command"), str):
|
|
112
|
+
parts.append(entry["command"])
|
|
113
|
+
for arg in entry.get("args") or []:
|
|
114
|
+
if isinstance(arg, str):
|
|
115
|
+
parts.append(arg)
|
|
116
|
+
combined = " ".join(parts)
|
|
117
|
+
if any(token in combined for token in _PORTABLE_TOKENS):
|
|
118
|
+
return False
|
|
119
|
+
return bool(_ABSOLUTE_PATH_RE.search(combined))
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def load_resolved_driver_specs(config_path: Path, *, vault_path: Path | None = None) -> list[DriverSpec]:
|
|
123
|
+
if not config_path.exists():
|
|
124
|
+
return []
|
|
125
|
+
try:
|
|
126
|
+
data = json.loads(config_path.read_text(encoding="utf-8"))
|
|
127
|
+
except (OSError, json.JSONDecodeError) as exc:
|
|
128
|
+
logger.warning("drivers.json unreadable (%s) — no drivers mounted", exc)
|
|
129
|
+
return []
|
|
130
|
+
|
|
131
|
+
specs: list[DriverSpec] = []
|
|
132
|
+
for entry in data.get("drivers", []):
|
|
133
|
+
try:
|
|
134
|
+
resolved = resolve_driver_entry(entry, vault_path=vault_path)
|
|
135
|
+
spec = DriverSpec.model_validate(resolved)
|
|
136
|
+
except ValidationError as exc:
|
|
137
|
+
logger.warning("Skipping invalid driver entry %r: %s", entry.get("name"), exc)
|
|
138
|
+
continue
|
|
139
|
+
if not spec.enabled:
|
|
140
|
+
logger.info("Driver %s is disabled — skipped", spec.name)
|
|
141
|
+
continue
|
|
142
|
+
specs.append(spec)
|
|
143
|
+
return specs
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def normalize_drivers_file(config_path: Path, *, vault_path: Path | None = None) -> bool:
|
|
147
|
+
if not config_path.exists():
|
|
148
|
+
return False
|
|
149
|
+
data = json.loads(config_path.read_text(encoding="utf-8"))
|
|
150
|
+
drivers = data.get("drivers") or []
|
|
151
|
+
normalized = [normalize_driver_entry(entry, vault_path=vault_path) for entry in drivers]
|
|
152
|
+
if normalized == drivers:
|
|
153
|
+
return False
|
|
154
|
+
data["drivers"] = normalized
|
|
155
|
+
config_path.write_text(json.dumps(data, indent=2) + "\n", encoding="utf-8")
|
|
156
|
+
return True
|