devopsiq 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent/__init__.py +5 -0
- agent/agent.py +232 -0
- agent/investigation.py +339 -0
- agent/prompts.py +125 -0
- agent/store.py +147 -0
- devopsiq-0.1.0.dist-info/METADATA +662 -0
- devopsiq-0.1.0.dist-info/RECORD +30 -0
- devopsiq-0.1.0.dist-info/WHEEL +5 -0
- devopsiq-0.1.0.dist-info/entry_points.txt +2 -0
- devopsiq-0.1.0.dist-info/licenses/LICENSE +21 -0
- devopsiq-0.1.0.dist-info/top_level.txt +3 -0
- main.py +310 -0
- tools/__init__.py +6 -0
- tools/ansible.py +110 -0
- tools/argocd.py +91 -0
- tools/base.py +112 -0
- tools/cloud.py +101 -0
- tools/docker.py +280 -0
- tools/git_ci.py +257 -0
- tools/helm.py +168 -0
- tools/investigation.py +357 -0
- tools/istio.py +43 -0
- tools/kubernetes.py +464 -0
- tools/monitoring.py +162 -0
- tools/newrelic.py +167 -0
- tools/preflight.py +59 -0
- tools/registry.py +49 -0
- tools/system.py +162 -0
- tools/terraform.py +90 -0
- tools/trivy.py +83 -0
tools/investigation.py
ADDED
|
@@ -0,0 +1,357 @@
|
|
|
1
|
+
"""Investigation meta-tools (Phase 4).
|
|
2
|
+
|
|
3
|
+
The model uses these three tools to maintain a first-class investigation
|
|
4
|
+
record (agent/investigation.py) while it works a reported problem: open the
|
|
5
|
+
investigation, record hypotheses / evidence / verdicts, and conclude with
|
|
6
|
+
root cause + remediation + verification.
|
|
7
|
+
|
|
8
|
+
These are the only "stateful" tools — and they mutate NOTHING outside the
|
|
9
|
+
agent's own memory: no cluster, no files, no infrastructure. The read-only
|
|
10
|
+
guarantee of the project is unaffected.
|
|
11
|
+
|
|
12
|
+
Phase 7 adds persistence: every mutation is auto-saved (best-effort) to the
|
|
13
|
+
InvestigationStore (agent/store.py), so the record survives CLI exits. The
|
|
14
|
+
save hook lives here because all record mutations funnel through the shared
|
|
15
|
+
functions below — tool executors and CLI commands are literally the same
|
|
16
|
+
code path.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from agent.investigation import (
|
|
20
|
+
CONFIDENCE_LEVELS,
|
|
21
|
+
HYPOTHESIS_STATUSES,
|
|
22
|
+
Investigation,
|
|
23
|
+
InvestigationError,
|
|
24
|
+
)
|
|
25
|
+
from agent.store import InvestigationStore, default_store
|
|
26
|
+
from tools.base import Tool, ToolError
|
|
27
|
+
from tools.registry import register
|
|
28
|
+
|
|
29
|
+
# The single active investigation for this agent process (single-threaded CLI).
|
|
30
|
+
_active: Investigation | None = None
|
|
31
|
+
|
|
32
|
+
# Persistence (Phase 7): resolved lazily (explicit > AGENT_STORE_DIR > the
|
|
33
|
+
# default home directory), disableable for tests. set_store(None) turns
|
|
34
|
+
# persistence off; _get_store() otherwise resolves and caches a default.
|
|
35
|
+
_store: InvestigationStore | None = None
|
|
36
|
+
_store_disabled: bool = False
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def set_store(store: InvestigationStore | None) -> None:
|
|
40
|
+
"""Replace the persistence store (tests pass None to disable saving)."""
|
|
41
|
+
global _store, _store_disabled
|
|
42
|
+
_store = store
|
|
43
|
+
_store_disabled = store is None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _get_store() -> InvestigationStore | None:
|
|
47
|
+
"""The process store, or None when persistence is off/unavailable."""
|
|
48
|
+
global _store
|
|
49
|
+
if _store_disabled:
|
|
50
|
+
return None
|
|
51
|
+
if _store is None:
|
|
52
|
+
try:
|
|
53
|
+
_store = default_store()
|
|
54
|
+
except Exception: # noqa: BLE001 — persistence must never break the agent
|
|
55
|
+
return None
|
|
56
|
+
return _store
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _persist() -> str:
|
|
60
|
+
"""Auto-save the active record; returns a short note for the result text.
|
|
61
|
+
|
|
62
|
+
Best-effort: an unavailable store returns "" (no note, no error) and a
|
|
63
|
+
failing save adds a one-line note instead of raising.
|
|
64
|
+
"""
|
|
65
|
+
store = _get_store()
|
|
66
|
+
if store is None:
|
|
67
|
+
return ""
|
|
68
|
+
path = store.save(_active)
|
|
69
|
+
if path is None:
|
|
70
|
+
return "\n(note: persistence unavailable — the record is not being saved)"
|
|
71
|
+
return f"\n(saved to {path})"
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
# --- thin functions shared by the CLI and the tool executors ---
|
|
75
|
+
|
|
76
|
+
def start_investigation(problem: str, hypotheses: list[str] | None = None) -> str:
|
|
77
|
+
"""Open an investigation (tool executor and /investigate both route here)."""
|
|
78
|
+
global _active
|
|
79
|
+
if _active is not None:
|
|
80
|
+
return "an investigation is already active:\n" + _active.render_status()
|
|
81
|
+
try:
|
|
82
|
+
_active = Investigation(problem, hypotheses)
|
|
83
|
+
except InvestigationError as exc:
|
|
84
|
+
raise ToolError(str(exc)) from exc
|
|
85
|
+
return (
|
|
86
|
+
"Investigation started.\n"
|
|
87
|
+
+ _active.render_status()
|
|
88
|
+
+ _persist()
|
|
89
|
+
+ "\n\nTrack hypotheses and evidence with investigation_record, and "
|
|
90
|
+
"finish with investigation_conclude."
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def record(
|
|
95
|
+
kind: str,
|
|
96
|
+
content: str,
|
|
97
|
+
hypothesis_id: str | None = None,
|
|
98
|
+
status: str | None = None,
|
|
99
|
+
) -> str:
|
|
100
|
+
inv = _require_active()
|
|
101
|
+
try:
|
|
102
|
+
kind = (kind or "").strip()
|
|
103
|
+
if kind == "hypothesis":
|
|
104
|
+
result = inv.add_hypothesis(content)
|
|
105
|
+
elif kind == "evidence":
|
|
106
|
+
result = inv.record_evidence(content, hypothesis_id)
|
|
107
|
+
elif kind == "verdict":
|
|
108
|
+
if not hypothesis_id or not status:
|
|
109
|
+
raise InvestigationError(
|
|
110
|
+
"a verdict requires both hypothesis_id and status"
|
|
111
|
+
)
|
|
112
|
+
result = inv.verify_hypothesis(hypothesis_id, status, content)
|
|
113
|
+
else:
|
|
114
|
+
raise InvestigationError(
|
|
115
|
+
f"unknown kind {kind!r}; expected hypothesis | evidence | verdict"
|
|
116
|
+
)
|
|
117
|
+
except InvestigationError as exc:
|
|
118
|
+
raise ToolError(str(exc)) from exc
|
|
119
|
+
return result + "\n" + inv.render_status() + _persist()
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def conclude_investigation(
|
|
123
|
+
*,
|
|
124
|
+
summary: str,
|
|
125
|
+
root_cause: str,
|
|
126
|
+
remediation: str | list[str],
|
|
127
|
+
verification: str | list[str],
|
|
128
|
+
confidence: str = "medium",
|
|
129
|
+
) -> str:
|
|
130
|
+
inv = _require_active()
|
|
131
|
+
try:
|
|
132
|
+
result = inv.conclude(
|
|
133
|
+
summary=summary,
|
|
134
|
+
root_cause=root_cause,
|
|
135
|
+
remediation=remediation,
|
|
136
|
+
verification=verification,
|
|
137
|
+
confidence=confidence,
|
|
138
|
+
)
|
|
139
|
+
except InvestigationError as exc:
|
|
140
|
+
raise ToolError(str(exc)) from exc
|
|
141
|
+
return result + _persist()
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def status_text() -> str | None:
|
|
145
|
+
return _active.render_status() if _active is not None else None
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def report_text() -> str | None:
|
|
149
|
+
"""Current tracker, or the canonical report once concluded."""
|
|
150
|
+
return _active.render_report() if _active is not None else None
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def report_json() -> dict | None:
|
|
154
|
+
"""Structured JSON export of the active investigation (CI/automation)."""
|
|
155
|
+
return _active.render_report_json() if _active is not None else None
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def finish_investigation() -> str:
|
|
159
|
+
global _active
|
|
160
|
+
if _active is None:
|
|
161
|
+
return "no active investigation"
|
|
162
|
+
_active = None
|
|
163
|
+
store = _get_store()
|
|
164
|
+
if store is not None:
|
|
165
|
+
store.forget() # clears memory only — the saved copy stays as history
|
|
166
|
+
return "investigation cleared"
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def resume_investigation() -> str | None:
|
|
170
|
+
"""Load the newest in-progress record from the store into memory.
|
|
171
|
+
|
|
172
|
+
Returns a short notice, or None when nothing was resumed (persistence
|
|
173
|
+
off, no saved in-progress record, or an investigation is already
|
|
174
|
+
active).
|
|
175
|
+
"""
|
|
176
|
+
global _active
|
|
177
|
+
if _active is not None:
|
|
178
|
+
return None
|
|
179
|
+
store = _get_store()
|
|
180
|
+
if store is None:
|
|
181
|
+
return None
|
|
182
|
+
inv = store.resume_latest()
|
|
183
|
+
if inv is None:
|
|
184
|
+
return None
|
|
185
|
+
_active = inv
|
|
186
|
+
return f"Resumed: {inv.problem} — /investigation to view, continue as before."
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def list_saved_text() -> str | None:
|
|
190
|
+
"""Rendered list of saved records, or None when there is nothing to show."""
|
|
191
|
+
store = _get_store()
|
|
192
|
+
if store is None:
|
|
193
|
+
return None
|
|
194
|
+
rows = store.list_saved()
|
|
195
|
+
if not rows:
|
|
196
|
+
return None
|
|
197
|
+
current = store.current_file
|
|
198
|
+
lines = ["**Saved investigations** (newest first)"]
|
|
199
|
+
for row in rows:
|
|
200
|
+
marker = " <- active" if row["file"] == current else ""
|
|
201
|
+
lines.append(f"- [{row['status']}] {row['file']}{marker} — {row['problem']}")
|
|
202
|
+
return "\n".join(lines)
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def store_directory_text() -> str | None:
|
|
206
|
+
"""Where records are being saved, or None when persistence is off."""
|
|
207
|
+
store = _get_store()
|
|
208
|
+
return str(store.directory) if store is not None else None
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _require_active() -> Investigation:
|
|
212
|
+
if _active is None:
|
|
213
|
+
raise ToolError(
|
|
214
|
+
"no active investigation. Call investigation_begin first (the "
|
|
215
|
+
"user must be reporting a concrete problem)."
|
|
216
|
+
)
|
|
217
|
+
return _active
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
# --- tool executors (adapter from parsed args dict to the functions above) ---
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def _begin_executor(args: dict) -> str:
|
|
224
|
+
return start_investigation(
|
|
225
|
+
args.get("problem"),
|
|
226
|
+
args.get("initial_hypotheses"),
|
|
227
|
+
)
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def _record_executor(args: dict) -> str:
|
|
231
|
+
return record(
|
|
232
|
+
kind=args.get("kind"),
|
|
233
|
+
content=args.get("content"),
|
|
234
|
+
hypothesis_id=args.get("hypothesis_id"),
|
|
235
|
+
status=args.get("status"),
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _conclude_executor(args: dict) -> str:
|
|
240
|
+
return conclude_investigation(
|
|
241
|
+
summary=args.get("summary"),
|
|
242
|
+
root_cause=args.get("root_cause"),
|
|
243
|
+
remediation=args.get("remediation"),
|
|
244
|
+
verification=args.get("verification"),
|
|
245
|
+
confidence=args.get("confidence", "medium"),
|
|
246
|
+
)
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
INVESTIGATION_BEGIN = Tool(
|
|
250
|
+
name="investigation_begin",
|
|
251
|
+
description=(
|
|
252
|
+
"Start a formal incident investigation for a problem the user just "
|
|
253
|
+
"reported (a failing pod, broken rollout, alert, degradation — NOT a "
|
|
254
|
+
"general question). Call this BEFORE gathering evidence, listing your "
|
|
255
|
+
"initial hypotheses. Returns the live investigation tracker."
|
|
256
|
+
),
|
|
257
|
+
parameters={
|
|
258
|
+
"type": "object",
|
|
259
|
+
"properties": {
|
|
260
|
+
"problem": {
|
|
261
|
+
"type": "string",
|
|
262
|
+
"description": "One-line statement of the problem to investigate.",
|
|
263
|
+
},
|
|
264
|
+
"initial_hypotheses": {
|
|
265
|
+
"type": "array",
|
|
266
|
+
"items": {"type": "string"},
|
|
267
|
+
"description": "2-4 plausible causes to test (will become H1, H2, ...).",
|
|
268
|
+
},
|
|
269
|
+
},
|
|
270
|
+
"required": ["problem"],
|
|
271
|
+
"additionalProperties": False,
|
|
272
|
+
},
|
|
273
|
+
executor=_begin_executor,
|
|
274
|
+
)
|
|
275
|
+
|
|
276
|
+
INVESTIGATION_RECORD = Tool(
|
|
277
|
+
name="investigation_record",
|
|
278
|
+
description=(
|
|
279
|
+
"Record something into the ACTIVE investigation: a new hypothesis "
|
|
280
|
+
"(kind=hypothesis), an evidence note linked to a hypothesis "
|
|
281
|
+
"(kind=evidence, optional hypothesis_id), or a verdict updating a "
|
|
282
|
+
"hypothesis (kind=verdict with hypothesis_id and status). Returns the "
|
|
283
|
+
"updated tracker. Requires an active investigation."
|
|
284
|
+
),
|
|
285
|
+
parameters={
|
|
286
|
+
"type": "object",
|
|
287
|
+
"properties": {
|
|
288
|
+
"kind": {
|
|
289
|
+
"type": "string",
|
|
290
|
+
"enum": ["hypothesis", "evidence", "verdict"],
|
|
291
|
+
"description": "What to record.",
|
|
292
|
+
},
|
|
293
|
+
"content": {
|
|
294
|
+
"type": "string",
|
|
295
|
+
"description": "The hypothesis statement, evidence note, or "
|
|
296
|
+
"verdict justification.",
|
|
297
|
+
},
|
|
298
|
+
"hypothesis_id": {
|
|
299
|
+
"type": "string",
|
|
300
|
+
"description": "Hypothesis id (H1, H2, ...) to link evidence to "
|
|
301
|
+
"or to give a verdict.",
|
|
302
|
+
},
|
|
303
|
+
"status": {
|
|
304
|
+
"type": "string",
|
|
305
|
+
"enum": list(HYPOTHESIS_STATUSES),
|
|
306
|
+
"description": "For kind=verdict: new status of the hypothesis.",
|
|
307
|
+
},
|
|
308
|
+
},
|
|
309
|
+
"required": ["kind", "content"],
|
|
310
|
+
"additionalProperties": False,
|
|
311
|
+
},
|
|
312
|
+
executor=_record_executor,
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
INVESTIGATION_CONCLUDE = Tool(
|
|
316
|
+
name="investigation_conclude",
|
|
317
|
+
description=(
|
|
318
|
+
"Finish the ACTIVE investigation once evidence is sufficient: name the "
|
|
319
|
+
"root cause, remediation RECOMMENDATIONS (you never execute anything), "
|
|
320
|
+
"verification steps, and your confidence. Requires an active "
|
|
321
|
+
"investigation run by investigation_begin."
|
|
322
|
+
),
|
|
323
|
+
parameters={
|
|
324
|
+
"type": "object",
|
|
325
|
+
"properties": {
|
|
326
|
+
"summary": {
|
|
327
|
+
"type": "string",
|
|
328
|
+
"description": "One-paragraph summary of the investigation.",
|
|
329
|
+
},
|
|
330
|
+
"root_cause": {
|
|
331
|
+
"type": "string",
|
|
332
|
+
"description": "The confirmed/likeliest root cause.",
|
|
333
|
+
},
|
|
334
|
+
"remediation": {
|
|
335
|
+
"oneOf": [{"type": "string"}, {"type": "array", "items": {"type": "string"}}],
|
|
336
|
+
"description": "Recommended fixes (read-only agent: these are "
|
|
337
|
+
"proposals, never executed).",
|
|
338
|
+
},
|
|
339
|
+
"verification": {
|
|
340
|
+
"oneOf": [{"type": "string"}, {"type": "array", "items": {"type": "string"}}],
|
|
341
|
+
"description": "Steps to confirm the fix worked.",
|
|
342
|
+
},
|
|
343
|
+
"confidence": {
|
|
344
|
+
"type": "string",
|
|
345
|
+
"enum": list(CONFIDENCE_LEVELS),
|
|
346
|
+
"description": "Confidence in the root cause.",
|
|
347
|
+
},
|
|
348
|
+
},
|
|
349
|
+
"required": ["summary", "root_cause", "remediation", "verification"],
|
|
350
|
+
"additionalProperties": False,
|
|
351
|
+
},
|
|
352
|
+
executor=_conclude_executor,
|
|
353
|
+
)
|
|
354
|
+
|
|
355
|
+
register(INVESTIGATION_BEGIN)
|
|
356
|
+
register(INVESTIGATION_RECORD)
|
|
357
|
+
register(INVESTIGATION_CONCLUDE)
|
tools/istio.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""Read-only Istio tools (Phase 9).
|
|
2
|
+
|
|
3
|
+
One tool: `istioctl proxy-status` — the service mesh's control-plane view of
|
|
4
|
+
every connected Envoy proxy: which proxies exist, their cluster, their sync
|
|
5
|
+
version with istiod, and which are STALE (not receiving config — the mesh
|
|
6
|
+
version of a node NotReady).
|
|
7
|
+
|
|
8
|
+
Safety model:
|
|
9
|
+
- A single fixed argv template, no arguments at all — the model cannot
|
|
10
|
+
inject anything. istioctl's mutating verbs (proxy-config with write
|
|
11
|
+
paths, install, upgrade, webhook apply, x) are unreachable.
|
|
12
|
+
- istioctl must be installed and pointed at a mesh; missing/unreachable
|
|
13
|
+
surfaces as the exact CLI error.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from tools.base import Tool, read_command_output
|
|
17
|
+
from tools.registry import register
|
|
18
|
+
|
|
19
|
+
_TIMEOUT_S = 15
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _istioctl_proxy_status(args: dict) -> str:
|
|
23
|
+
return read_command_output(
|
|
24
|
+
("istioctl", "proxy-status"),
|
|
25
|
+
timeout=_TIMEOUT_S,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
ISTIOCTL_PROXY_STATUS = Tool(
|
|
30
|
+
name="istioctl_proxy_status",
|
|
31
|
+
description=(
|
|
32
|
+
"Istio mesh proxy status (istioctl proxy-status): every Envoy "
|
|
33
|
+
"proxy (CLUSTER/CDS/LDS/RDS/ECDS out-of-sync columns) and which "
|
|
34
|
+
"ones are STALE — out of sync with istiod. Use for mesh problems: "
|
|
35
|
+
"'traffic isn't following the new config', 'a sidecar stopped "
|
|
36
|
+
"receiving updates'. Requires istioctl + a reachable mesh. "
|
|
37
|
+
"Read-only."
|
|
38
|
+
),
|
|
39
|
+
parameters={"type": "object", "properties": {}, "additionalProperties": False},
|
|
40
|
+
executor=_istioctl_proxy_status,
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
register(ISTIOCTL_PROXY_STATUS)
|