devopsiq 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
tools/investigation.py ADDED
@@ -0,0 +1,357 @@
1
+ """Investigation meta-tools (Phase 4).
2
+
3
+ The model uses these three tools to maintain a first-class investigation
4
+ record (agent/investigation.py) while it works a reported problem: open the
5
+ investigation, record hypotheses / evidence / verdicts, and conclude with
6
+ root cause + remediation + verification.
7
+
8
+ These are the only "stateful" tools — and they mutate NOTHING outside the
9
+ agent's own memory: no cluster, no files, no infrastructure. The read-only
10
+ guarantee of the project is unaffected.
11
+
12
+ Phase 7 adds persistence: every mutation is auto-saved (best-effort) to the
13
+ InvestigationStore (agent/store.py), so the record survives CLI exits. The
14
+ save hook lives here because all record mutations funnel through the shared
15
+ functions below — tool executors and CLI commands are literally the same
16
+ code path.
17
+ """
18
+
19
+ from agent.investigation import (
20
+ CONFIDENCE_LEVELS,
21
+ HYPOTHESIS_STATUSES,
22
+ Investigation,
23
+ InvestigationError,
24
+ )
25
+ from agent.store import InvestigationStore, default_store
26
+ from tools.base import Tool, ToolError
27
+ from tools.registry import register
28
+
29
+ # The single active investigation for this agent process (single-threaded CLI).
30
+ _active: Investigation | None = None
31
+
32
+ # Persistence (Phase 7): resolved lazily (explicit > AGENT_STORE_DIR > the
33
+ # default home directory), disableable for tests. set_store(None) turns
34
+ # persistence off; _get_store() otherwise resolves and caches a default.
35
+ _store: InvestigationStore | None = None
36
+ _store_disabled: bool = False
37
+
38
+
39
+ def set_store(store: InvestigationStore | None) -> None:
40
+ """Replace the persistence store (tests pass None to disable saving)."""
41
+ global _store, _store_disabled
42
+ _store = store
43
+ _store_disabled = store is None
44
+
45
+
46
+ def _get_store() -> InvestigationStore | None:
47
+ """The process store, or None when persistence is off/unavailable."""
48
+ global _store
49
+ if _store_disabled:
50
+ return None
51
+ if _store is None:
52
+ try:
53
+ _store = default_store()
54
+ except Exception: # noqa: BLE001 — persistence must never break the agent
55
+ return None
56
+ return _store
57
+
58
+
59
+ def _persist() -> str:
60
+ """Auto-save the active record; returns a short note for the result text.
61
+
62
+ Best-effort: an unavailable store returns "" (no note, no error) and a
63
+ failing save adds a one-line note instead of raising.
64
+ """
65
+ store = _get_store()
66
+ if store is None:
67
+ return ""
68
+ path = store.save(_active)
69
+ if path is None:
70
+ return "\n(note: persistence unavailable — the record is not being saved)"
71
+ return f"\n(saved to {path})"
72
+
73
+
74
+ # --- thin functions shared by the CLI and the tool executors ---
75
+
76
+ def start_investigation(problem: str, hypotheses: list[str] | None = None) -> str:
77
+ """Open an investigation (tool executor and /investigate both route here)."""
78
+ global _active
79
+ if _active is not None:
80
+ return "an investigation is already active:\n" + _active.render_status()
81
+ try:
82
+ _active = Investigation(problem, hypotheses)
83
+ except InvestigationError as exc:
84
+ raise ToolError(str(exc)) from exc
85
+ return (
86
+ "Investigation started.\n"
87
+ + _active.render_status()
88
+ + _persist()
89
+ + "\n\nTrack hypotheses and evidence with investigation_record, and "
90
+ "finish with investigation_conclude."
91
+ )
92
+
93
+
94
+ def record(
95
+ kind: str,
96
+ content: str,
97
+ hypothesis_id: str | None = None,
98
+ status: str | None = None,
99
+ ) -> str:
100
+ inv = _require_active()
101
+ try:
102
+ kind = (kind or "").strip()
103
+ if kind == "hypothesis":
104
+ result = inv.add_hypothesis(content)
105
+ elif kind == "evidence":
106
+ result = inv.record_evidence(content, hypothesis_id)
107
+ elif kind == "verdict":
108
+ if not hypothesis_id or not status:
109
+ raise InvestigationError(
110
+ "a verdict requires both hypothesis_id and status"
111
+ )
112
+ result = inv.verify_hypothesis(hypothesis_id, status, content)
113
+ else:
114
+ raise InvestigationError(
115
+ f"unknown kind {kind!r}; expected hypothesis | evidence | verdict"
116
+ )
117
+ except InvestigationError as exc:
118
+ raise ToolError(str(exc)) from exc
119
+ return result + "\n" + inv.render_status() + _persist()
120
+
121
+
122
+ def conclude_investigation(
123
+ *,
124
+ summary: str,
125
+ root_cause: str,
126
+ remediation: str | list[str],
127
+ verification: str | list[str],
128
+ confidence: str = "medium",
129
+ ) -> str:
130
+ inv = _require_active()
131
+ try:
132
+ result = inv.conclude(
133
+ summary=summary,
134
+ root_cause=root_cause,
135
+ remediation=remediation,
136
+ verification=verification,
137
+ confidence=confidence,
138
+ )
139
+ except InvestigationError as exc:
140
+ raise ToolError(str(exc)) from exc
141
+ return result + _persist()
142
+
143
+
144
+ def status_text() -> str | None:
145
+ return _active.render_status() if _active is not None else None
146
+
147
+
148
+ def report_text() -> str | None:
149
+ """Current tracker, or the canonical report once concluded."""
150
+ return _active.render_report() if _active is not None else None
151
+
152
+
153
+ def report_json() -> dict | None:
154
+ """Structured JSON export of the active investigation (CI/automation)."""
155
+ return _active.render_report_json() if _active is not None else None
156
+
157
+
158
+ def finish_investigation() -> str:
159
+ global _active
160
+ if _active is None:
161
+ return "no active investigation"
162
+ _active = None
163
+ store = _get_store()
164
+ if store is not None:
165
+ store.forget() # clears memory only — the saved copy stays as history
166
+ return "investigation cleared"
167
+
168
+
169
+ def resume_investigation() -> str | None:
170
+ """Load the newest in-progress record from the store into memory.
171
+
172
+ Returns a short notice, or None when nothing was resumed (persistence
173
+ off, no saved in-progress record, or an investigation is already
174
+ active).
175
+ """
176
+ global _active
177
+ if _active is not None:
178
+ return None
179
+ store = _get_store()
180
+ if store is None:
181
+ return None
182
+ inv = store.resume_latest()
183
+ if inv is None:
184
+ return None
185
+ _active = inv
186
+ return f"Resumed: {inv.problem} — /investigation to view, continue as before."
187
+
188
+
189
+ def list_saved_text() -> str | None:
190
+ """Rendered list of saved records, or None when there is nothing to show."""
191
+ store = _get_store()
192
+ if store is None:
193
+ return None
194
+ rows = store.list_saved()
195
+ if not rows:
196
+ return None
197
+ current = store.current_file
198
+ lines = ["**Saved investigations** (newest first)"]
199
+ for row in rows:
200
+ marker = " <- active" if row["file"] == current else ""
201
+ lines.append(f"- [{row['status']}] {row['file']}{marker} — {row['problem']}")
202
+ return "\n".join(lines)
203
+
204
+
205
+ def store_directory_text() -> str | None:
206
+ """Where records are being saved, or None when persistence is off."""
207
+ store = _get_store()
208
+ return str(store.directory) if store is not None else None
209
+
210
+
211
+ def _require_active() -> Investigation:
212
+ if _active is None:
213
+ raise ToolError(
214
+ "no active investigation. Call investigation_begin first (the "
215
+ "user must be reporting a concrete problem)."
216
+ )
217
+ return _active
218
+
219
+
220
+ # --- tool executors (adapter from parsed args dict to the functions above) ---
221
+
222
+
223
+ def _begin_executor(args: dict) -> str:
224
+ return start_investigation(
225
+ args.get("problem"),
226
+ args.get("initial_hypotheses"),
227
+ )
228
+
229
+
230
+ def _record_executor(args: dict) -> str:
231
+ return record(
232
+ kind=args.get("kind"),
233
+ content=args.get("content"),
234
+ hypothesis_id=args.get("hypothesis_id"),
235
+ status=args.get("status"),
236
+ )
237
+
238
+
239
+ def _conclude_executor(args: dict) -> str:
240
+ return conclude_investigation(
241
+ summary=args.get("summary"),
242
+ root_cause=args.get("root_cause"),
243
+ remediation=args.get("remediation"),
244
+ verification=args.get("verification"),
245
+ confidence=args.get("confidence", "medium"),
246
+ )
247
+
248
+
249
+ INVESTIGATION_BEGIN = Tool(
250
+ name="investigation_begin",
251
+ description=(
252
+ "Start a formal incident investigation for a problem the user just "
253
+ "reported (a failing pod, broken rollout, alert, degradation — NOT a "
254
+ "general question). Call this BEFORE gathering evidence, listing your "
255
+ "initial hypotheses. Returns the live investigation tracker."
256
+ ),
257
+ parameters={
258
+ "type": "object",
259
+ "properties": {
260
+ "problem": {
261
+ "type": "string",
262
+ "description": "One-line statement of the problem to investigate.",
263
+ },
264
+ "initial_hypotheses": {
265
+ "type": "array",
266
+ "items": {"type": "string"},
267
+ "description": "2-4 plausible causes to test (will become H1, H2, ...).",
268
+ },
269
+ },
270
+ "required": ["problem"],
271
+ "additionalProperties": False,
272
+ },
273
+ executor=_begin_executor,
274
+ )
275
+
276
+ INVESTIGATION_RECORD = Tool(
277
+ name="investigation_record",
278
+ description=(
279
+ "Record something into the ACTIVE investigation: a new hypothesis "
280
+ "(kind=hypothesis), an evidence note linked to a hypothesis "
281
+ "(kind=evidence, optional hypothesis_id), or a verdict updating a "
282
+ "hypothesis (kind=verdict with hypothesis_id and status). Returns the "
283
+ "updated tracker. Requires an active investigation."
284
+ ),
285
+ parameters={
286
+ "type": "object",
287
+ "properties": {
288
+ "kind": {
289
+ "type": "string",
290
+ "enum": ["hypothesis", "evidence", "verdict"],
291
+ "description": "What to record.",
292
+ },
293
+ "content": {
294
+ "type": "string",
295
+ "description": "The hypothesis statement, evidence note, or "
296
+ "verdict justification.",
297
+ },
298
+ "hypothesis_id": {
299
+ "type": "string",
300
+ "description": "Hypothesis id (H1, H2, ...) to link evidence to "
301
+ "or to give a verdict.",
302
+ },
303
+ "status": {
304
+ "type": "string",
305
+ "enum": list(HYPOTHESIS_STATUSES),
306
+ "description": "For kind=verdict: new status of the hypothesis.",
307
+ },
308
+ },
309
+ "required": ["kind", "content"],
310
+ "additionalProperties": False,
311
+ },
312
+ executor=_record_executor,
313
+ )
314
+
315
+ INVESTIGATION_CONCLUDE = Tool(
316
+ name="investigation_conclude",
317
+ description=(
318
+ "Finish the ACTIVE investigation once evidence is sufficient: name the "
319
+ "root cause, remediation RECOMMENDATIONS (you never execute anything), "
320
+ "verification steps, and your confidence. Requires an active "
321
+ "investigation run by investigation_begin."
322
+ ),
323
+ parameters={
324
+ "type": "object",
325
+ "properties": {
326
+ "summary": {
327
+ "type": "string",
328
+ "description": "One-paragraph summary of the investigation.",
329
+ },
330
+ "root_cause": {
331
+ "type": "string",
332
+ "description": "The confirmed/likeliest root cause.",
333
+ },
334
+ "remediation": {
335
+ "oneOf": [{"type": "string"}, {"type": "array", "items": {"type": "string"}}],
336
+ "description": "Recommended fixes (read-only agent: these are "
337
+ "proposals, never executed).",
338
+ },
339
+ "verification": {
340
+ "oneOf": [{"type": "string"}, {"type": "array", "items": {"type": "string"}}],
341
+ "description": "Steps to confirm the fix worked.",
342
+ },
343
+ "confidence": {
344
+ "type": "string",
345
+ "enum": list(CONFIDENCE_LEVELS),
346
+ "description": "Confidence in the root cause.",
347
+ },
348
+ },
349
+ "required": ["summary", "root_cause", "remediation", "verification"],
350
+ "additionalProperties": False,
351
+ },
352
+ executor=_conclude_executor,
353
+ )
354
+
355
+ register(INVESTIGATION_BEGIN)
356
+ register(INVESTIGATION_RECORD)
357
+ register(INVESTIGATION_CONCLUDE)
tools/istio.py ADDED
@@ -0,0 +1,43 @@
1
+ """Read-only Istio tools (Phase 9).
2
+
3
+ One tool: `istioctl proxy-status` — the service mesh's control-plane view of
4
+ every connected Envoy proxy: which proxies exist, their cluster, their sync
5
+ version with istiod, and which are STALE (not receiving config — the mesh
6
+ version of a node NotReady).
7
+
8
+ Safety model:
9
+ - A single fixed argv template, no arguments at all — the model cannot
10
+ inject anything. istioctl's mutating verbs (proxy-config with write
11
+ paths, install, upgrade, webhook apply, x) are unreachable.
12
+ - istioctl must be installed and pointed at a mesh; missing/unreachable
13
+ surfaces as the exact CLI error.
14
+ """
15
+
16
+ from tools.base import Tool, read_command_output
17
+ from tools.registry import register
18
+
19
+ _TIMEOUT_S = 15
20
+
21
+
22
+ def _istioctl_proxy_status(args: dict) -> str:
23
+ return read_command_output(
24
+ ("istioctl", "proxy-status"),
25
+ timeout=_TIMEOUT_S,
26
+ )
27
+
28
+
29
+ ISTIOCTL_PROXY_STATUS = Tool(
30
+ name="istioctl_proxy_status",
31
+ description=(
32
+ "Istio mesh proxy status (istioctl proxy-status): every Envoy "
33
+ "proxy (CLUSTER/CDS/LDS/RDS/ECDS out-of-sync columns) and which "
34
+ "ones are STALE — out of sync with istiod. Use for mesh problems: "
35
+ "'traffic isn't following the new config', 'a sidecar stopped "
36
+ "receiving updates'. Requires istioctl + a reachable mesh. "
37
+ "Read-only."
38
+ ),
39
+ parameters={"type": "object", "properties": {}, "additionalProperties": False},
40
+ executor=_istioctl_proxy_status,
41
+ )
42
+
43
+ register(ISTIOCTL_PROXY_STATUS)