devopsiq 0.1.1__tar.gz → 0.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {devopsiq-0.1.1/devopsiq.egg-info → devopsiq-0.1.2}/PKG-INFO +57 -15
- {devopsiq-0.1.1 → devopsiq-0.1.2}/README.md +55 -14
- {devopsiq-0.1.1 → devopsiq-0.1.2}/agent/agent.py +79 -4
- devopsiq-0.1.2/agent/config.py +45 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2/devopsiq.egg-info}/PKG-INFO +57 -15
- {devopsiq-0.1.1 → devopsiq-0.1.2}/devopsiq.egg-info/SOURCES.txt +1 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/main.py +46 -14
- {devopsiq-0.1.1 → devopsiq-0.1.2}/pyproject.toml +2 -1
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tests/test_automation.py +228 -5
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tests/test_phase7.py +6 -5
- {devopsiq-0.1.1 → devopsiq-0.1.2}/LICENSE +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/agent/__init__.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/agent/investigation.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/agent/prompts.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/agent/store.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/devopsiq.egg-info/dependency_links.txt +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/devopsiq.egg-info/entry_points.txt +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/devopsiq.egg-info/requires.txt +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/devopsiq.egg-info/top_level.txt +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/setup.cfg +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tests/test_phase2.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tests/test_phase3.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tests/test_phase4.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tests/test_phase5.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tests/test_phase8.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tests/test_phase9.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/__init__.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/ansible.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/argocd.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/base.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/cloud.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/docker.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/git_ci.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/helm.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/investigation.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/istio.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/kubernetes.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/monitoring.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/newrelic.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/preflight.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/registry.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/system.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/terraform.py +0 -0
- {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/trivy.py +0 -0
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: devopsiq
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.2
|
|
4
4
|
Summary: A from-scratch, read-only AI agent that investigates real DevOps problems — Kubernetes, Docker, Helm, Argo CD, Istio, Terraform, GitHub Actions, cloud, New Relic and more.
|
|
5
5
|
Author: DevOpsAbhii
|
|
6
6
|
License-Expression: MIT
|
|
7
7
|
Project-URL: Homepage, https://github.com/DevOpsAbhii/devops-ai-agent
|
|
8
8
|
Project-URL: Repository, https://github.com/DevOpsAbhii/devops-ai-agent
|
|
9
9
|
Project-URL: Issues, https://github.com/DevOpsAbhii/devops-ai-agent/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/DevOpsAbhii/devops-ai-agent/blob/main/CHANGELOG.md
|
|
10
11
|
Keywords: devops,kubernetes,docker,helm,argocd,istio,terraform,github-actions,observability,newrelic,incident-response,root-cause-analysis,ai-agent,cli
|
|
11
12
|
Classifier: Development Status :: 4 - Beta
|
|
12
13
|
Classifier: Environment :: Console
|
|
@@ -29,6 +30,13 @@ Dynamic: license-file
|
|
|
29
30
|
|
|
30
31
|
# DevOps AI Agent
|
|
31
32
|
|
|
33
|
+
[](https://pypi.org/project/devopsiq/)
|
|
34
|
+
[](https://pypi.org/project/devopsiq/)
|
|
35
|
+
[](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
|
|
36
|
+
[](https://github.com/DevOpsAbhii/devops-ai-agent/pkgs/container/devops-ai-agent)
|
|
37
|
+
[](LICENSE)
|
|
38
|
+
[](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
|
|
39
|
+
|
|
32
40
|
An AI agent that investigates real DevOps problems. The end goal: ask it
|
|
33
41
|
something like *"Why is my Kubernetes pod in CrashLoopBackOff?"* and have it
|
|
34
42
|
gather evidence, reason about the evidence, identify the likely root cause,
|
|
@@ -55,7 +63,14 @@ from environment config), and Ansible listing (inventory, playbook tasks).
|
|
|
55
63
|
open alerts over NerdGraph — credentials from env), Trivy image
|
|
56
64
|
vulnerability scanning, Helm releases (list/status/history), Argo CD
|
|
57
65
|
(GitOps app sync/health), Istio mesh proxy status, and Docker Compose
|
|
58
|
-
(project and service listing).
|
|
66
|
+
(project and service listing). **Phase 10 ships it as a product:** the
|
|
67
|
+
[`devopsiq` package on PyPI](https://pypi.org/project/devopsiq/) and a
|
|
68
|
+
prebuilt multi-arch Docker image (`ghcr.io/devopsabhii/devops-ai-agent`) —
|
|
69
|
+
both cut automatically by pushing a `v*` tag (test gate → PyPI via Trusted
|
|
70
|
+
Publishing + GHCR in parallel). **No API key? The agent still runs:** it
|
|
71
|
+
starts in model-less mode — record commands (`/report`, `/investigations`)
|
|
72
|
+
and the whole 58-tool layer work without a key; only questions to the model
|
|
73
|
+
need one. The repository is git-tracked.
|
|
59
74
|
Every phase still built from scratch — no LangChain, LangGraph,
|
|
60
75
|
AutoGen, CrewAI, or MCP.
|
|
61
76
|
|
|
@@ -73,7 +88,7 @@ model is called, how conversation history flows, how tool selection +
|
|
|
73
88
|
execution + evidence feedback work — instead of depending on a framework
|
|
74
89
|
for it.
|
|
75
90
|
|
|
76
|
-
|
|
91
|
+
Today the agent:
|
|
77
92
|
- holds a conversation with **GLM 5.3** through **OpenRouter**;
|
|
78
93
|
- has **58 real, read-only tools** across fifteen domains: host facts,
|
|
79
94
|
Kubernetes (12 tools), Linux system (4), Docker + Compose (10),
|
|
@@ -133,8 +148,8 @@ Module map:
|
|
|
133
148
|
|
|
134
149
|
| Path | Responsibility |
|
|
135
150
|
| --------------------------- | ------------------------------------------------------------ |
|
|
136
|
-
| `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages |
|
|
137
|
-
| `agent/agent.py` | `DevOpsAgent` — client, history, `ask()`, `_complete()` loop |
|
|
151
|
+
| `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages, model-less start |
|
|
152
|
+
| `agent/agent.py` | `DevOpsAgent` — client (`None` = model-less mode), history, `ask()`, `_complete()` loop |
|
|
138
153
|
| `agent/prompts.py` | The system prompt (versioned/tested separately) |
|
|
139
154
|
| `agent/investigation.py` | First-class investigation record: hypotheses, verdicts, evidence, report renderer (pure data) |
|
|
140
155
|
| `agent/store.py` | `InvestigationStore` — one JSON file per record, atomic writes, resume/list (Phase 7) |
|
|
@@ -163,6 +178,11 @@ Module map:
|
|
|
163
178
|
| `tests/test_phase7.py` | Offline suite: round-trip serialization, store files, auto-save, resume, CLI flags (Phase 7) |
|
|
164
179
|
| `tests/test_phase8.py` | Offline suite: argv templates + validation for the 21 Phase 8 tools (fake CLIs, env-based monitoring) |
|
|
165
180
|
| `tests/test_phase9.py` | Offline suite: New Relic env-credential + payload tests, trivy/helm/argocd/istio/compose argv templates (Phase 9) |
|
|
181
|
+
| `pyproject.toml` | Package `devopsiq`: metadata, MIT, console script `devopsiq = main:main` (Phase 10) |
|
|
182
|
+
| `Dockerfile` | Prebuilt image: slim base + kubectl/helm/trivy/gh + app, non-root, `/data` record store (Phase 10) |
|
|
183
|
+
| `.github/workflows/release.yml` | Tag-driven release: test gate → PyPI (Trusted Publishing) + GHCR multi-arch (Phase 10) |
|
|
184
|
+
| `docs/generate_pdf.py` | Builds the project documentation PDF from live source |
|
|
185
|
+
| `docs/generate_release_guide.py` | Builds the release & update playbook PDF |
|
|
166
186
|
|
|
167
187
|
### The tool-use loop
|
|
168
188
|
|
|
@@ -209,6 +229,7 @@ the CLI exposes it directly:
|
|
|
209
229
|
| `/investigations` | List saved records on disk (newest first; `← active` marks the live one) |
|
|
210
230
|
| `/report` | Show the canonical report (once concluded) |
|
|
211
231
|
| `/endinvestigation` | Clear the record (memory only — the saved copy stays as history) |
|
|
232
|
+
| `/model [<name>]` | Show the active model, or save a default (config.json) and switch to it |
|
|
212
233
|
|
|
213
234
|
The report the agent ends with and `/report` render are kept consistent by
|
|
214
235
|
construction: `tool` results confirm each record call and the tracker, and
|
|
@@ -396,16 +417,22 @@ official `openai` Python SDK works as our client with two config lines:
|
|
|
396
417
|
OpenAI(api_key=..., base_url="https://openrouter.ai/api/v1")
|
|
397
418
|
```
|
|
398
419
|
|
|
399
|
-
Swapping to another OpenRouter model later is a one-
|
|
400
|
-
|
|
401
|
-
`agent/agent.py` configuration.
|
|
420
|
+
Swapping to another OpenRouter model later is a one-liner (env var, `/model`
|
|
421
|
+
command, or `--model` flag); moving to any other OpenAI-compatible provider
|
|
422
|
+
changes only `agent/agent.py` configuration.
|
|
402
423
|
|
|
403
424
|
**Your model, your choice.** The default is baked in as a fallback, never a
|
|
404
|
-
restriction
|
|
405
|
-
|
|
425
|
+
restriction. Four ways to set the model — the first one set wins:
|
|
426
|
+
|
|
427
|
+
1. `--model NAME` flag (one run: `devopsiq --model openai/gpt-5.2 "why?"`)
|
|
428
|
+
2. `/model <name>` in the REPL — saves it to `~/.devops-ai-agent/config.json`
|
|
429
|
+
so every future run uses it
|
|
430
|
+
3. `OPENROUTER_MODEL` (shell or `.env`)
|
|
431
|
+
4. built-in default (`z-ai/glm-5.3`)
|
|
406
432
|
|
|
407
433
|
```bash
|
|
408
434
|
export OPENROUTER_MODEL=anthropic/claude-sonnet-5 # or openai/gpt-5.2, google/gemini-2.5-pro, ...
|
|
435
|
+
devopsiq /model openai/gpt-5.2 # or save a default from the REPL
|
|
409
436
|
```
|
|
410
437
|
|
|
411
438
|
Two things to weigh when picking: the agent is a tool-use loop, so choose a
|
|
@@ -558,6 +585,7 @@ setup/API errors — so it drops straight into a pipeline:
|
|
|
558
585
|
.venv/bin/python main.py --json "why is api-5d6f crash-looping?" # structured report
|
|
559
586
|
.venv/bin/python main.py --resume --json "any update?" # continue a prior run
|
|
560
587
|
.venv/bin/python main.py --store-dir /tmp/runs --out report.json "..." # pipeline paths
|
|
588
|
+
.venv/bin/python main.py --model openai/gpt-5.2 "why is it down?" # one-run model override
|
|
561
589
|
```
|
|
562
590
|
|
|
563
591
|
With `--json` the stdout is one JSON document (see `render_report_json` in
|
|
@@ -599,8 +627,8 @@ Tools: ansible_inventory, ansible_playbook_tasks, aws_identity,
|
|
|
599
627
|
k8s_top_nodes, k8s_top_pods, loki_query, prom_query,
|
|
600
628
|
sys_open_ports, sys_service_logs, sys_service_status,
|
|
601
629
|
sys_top_processes, system_info, tf_plan, tf_show, tf_state_list
|
|
602
|
-
Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation
|
|
603
|
-
One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] "<problem>"
|
|
630
|
+
Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation, /model [<name>]
|
|
631
|
+
One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] [--model NAME] "<problem>"
|
|
604
632
|
Type 'exit' to quit.
|
|
605
633
|
|
|
606
634
|
You: The checkout service container keeps exiting in Docker. Investigate.
|
|
@@ -617,7 +645,7 @@ You: exit
|
|
|
617
645
|
Type `exit` / `quit`, or press Ctrl-D / Ctrl-C to leave. Slash commands are
|
|
618
646
|
handled locally and never reach the model.
|
|
619
647
|
|
|
620
|
-
## 8. Current limitations
|
|
648
|
+
## 8. Current limitations
|
|
621
649
|
|
|
622
650
|
- **Each domain needs its CLI installed and reachable.** Missing CLIs,
|
|
623
651
|
unauthenticated `gh`, a dead docker daemon, an uninitialized terraform
|
|
@@ -702,8 +730,22 @@ handled locally and never reach the model.
|
|
|
702
730
|
`helm_history` — reads only), Argo CD (`argocd_apps`,
|
|
703
731
|
`argocd_app_status`), Istio (`istioctl_proxy_status`), Docker Compose
|
|
704
732
|
(`docker_compose_ls`, `docker_compose_ps` — live-verified on this host).
|
|
733
|
+
- **Phase 10 — distribution. ✅ Done.** The `devopsiq` package on PyPI
|
|
734
|
+
(console command `devopsiq`, MIT) and the prebuilt multi-arch GHCR image,
|
|
735
|
+
cut by a tag-driven release workflow: an offline test gate, then PyPI
|
|
736
|
+
(Trusted Publishing — no token in the repo) and GHCR (amd64 + arm64) in
|
|
737
|
+
parallel. **`v0.1.1` added model-less mode:** with no API key the agent
|
|
738
|
+
still constructs — record commands (`/report`, `/investigations`) and the
|
|
739
|
+
58-tool layer work; only model questions exit 1 with the setup message.
|
|
740
|
+
Verified end-to-end against the published package.
|
|
741
|
+
- **Model choice + resilient calls (v0.1.2). ✅ Done.** The model is
|
|
742
|
+
set by a 4-rung ladder — `--model` flag > `/model`-saved config file
|
|
743
|
+
(`~/.devops-ai-agent/config.json`) > `OPENROUTER_MODEL` > baked-in
|
|
744
|
+
default — and transient model-call failures (timeouts, connection
|
|
745
|
+
errors, 429s, 5xx) retry with exponential backoff (2s→4s→8s + jitter,
|
|
746
|
+
honoring a 429's Retry-After); a rejected key fails immediately.
|
|
705
747
|
- **Later — region-scoped cloud resources** (ec2 describe-*, compute
|
|
706
748
|
instances list, ...) behind the same template pattern; more observability
|
|
707
749
|
depth (New Relic dashboards/entities, Prometheus range queries); streaming;
|
|
708
|
-
conversation-history persistence; and a human-approval gate
|
|
709
|
-
mutating action is ever allowed.
|
|
750
|
+
conversation-history persistence; and a human-approval gate
|
|
751
|
+
before any mutating action is ever allowed.
|
|
@@ -1,5 +1,12 @@
|
|
|
1
1
|
# DevOps AI Agent
|
|
2
2
|
|
|
3
|
+
[](https://pypi.org/project/devopsiq/)
|
|
4
|
+
[](https://pypi.org/project/devopsiq/)
|
|
5
|
+
[](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
|
|
6
|
+
[](https://github.com/DevOpsAbhii/devops-ai-agent/pkgs/container/devops-ai-agent)
|
|
7
|
+
[](LICENSE)
|
|
8
|
+
[](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
|
|
9
|
+
|
|
3
10
|
An AI agent that investigates real DevOps problems. The end goal: ask it
|
|
4
11
|
something like *"Why is my Kubernetes pod in CrashLoopBackOff?"* and have it
|
|
5
12
|
gather evidence, reason about the evidence, identify the likely root cause,
|
|
@@ -26,7 +33,14 @@ from environment config), and Ansible listing (inventory, playbook tasks).
|
|
|
26
33
|
open alerts over NerdGraph — credentials from env), Trivy image
|
|
27
34
|
vulnerability scanning, Helm releases (list/status/history), Argo CD
|
|
28
35
|
(GitOps app sync/health), Istio mesh proxy status, and Docker Compose
|
|
29
|
-
(project and service listing).
|
|
36
|
+
(project and service listing). **Phase 10 ships it as a product:** the
|
|
37
|
+
[`devopsiq` package on PyPI](https://pypi.org/project/devopsiq/) and a
|
|
38
|
+
prebuilt multi-arch Docker image (`ghcr.io/devopsabhii/devops-ai-agent`) —
|
|
39
|
+
both cut automatically by pushing a `v*` tag (test gate → PyPI via Trusted
|
|
40
|
+
Publishing + GHCR in parallel). **No API key? The agent still runs:** it
|
|
41
|
+
starts in model-less mode — record commands (`/report`, `/investigations`)
|
|
42
|
+
and the whole 58-tool layer work without a key; only questions to the model
|
|
43
|
+
need one. The repository is git-tracked.
|
|
30
44
|
Every phase still built from scratch — no LangChain, LangGraph,
|
|
31
45
|
AutoGen, CrewAI, or MCP.
|
|
32
46
|
|
|
@@ -44,7 +58,7 @@ model is called, how conversation history flows, how tool selection +
|
|
|
44
58
|
execution + evidence feedback work — instead of depending on a framework
|
|
45
59
|
for it.
|
|
46
60
|
|
|
47
|
-
|
|
61
|
+
Today the agent:
|
|
48
62
|
- holds a conversation with **GLM 5.3** through **OpenRouter**;
|
|
49
63
|
- has **58 real, read-only tools** across fifteen domains: host facts,
|
|
50
64
|
Kubernetes (12 tools), Linux system (4), Docker + Compose (10),
|
|
@@ -104,8 +118,8 @@ Module map:
|
|
|
104
118
|
|
|
105
119
|
| Path | Responsibility |
|
|
106
120
|
| --------------------------- | ------------------------------------------------------------ |
|
|
107
|
-
| `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages |
|
|
108
|
-
| `agent/agent.py` | `DevOpsAgent` — client, history, `ask()`, `_complete()` loop |
|
|
121
|
+
| `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages, model-less start |
|
|
122
|
+
| `agent/agent.py` | `DevOpsAgent` — client (`None` = model-less mode), history, `ask()`, `_complete()` loop |
|
|
109
123
|
| `agent/prompts.py` | The system prompt (versioned/tested separately) |
|
|
110
124
|
| `agent/investigation.py` | First-class investigation record: hypotheses, verdicts, evidence, report renderer (pure data) |
|
|
111
125
|
| `agent/store.py` | `InvestigationStore` — one JSON file per record, atomic writes, resume/list (Phase 7) |
|
|
@@ -134,6 +148,11 @@ Module map:
|
|
|
134
148
|
| `tests/test_phase7.py` | Offline suite: round-trip serialization, store files, auto-save, resume, CLI flags (Phase 7) |
|
|
135
149
|
| `tests/test_phase8.py` | Offline suite: argv templates + validation for the 21 Phase 8 tools (fake CLIs, env-based monitoring) |
|
|
136
150
|
| `tests/test_phase9.py` | Offline suite: New Relic env-credential + payload tests, trivy/helm/argocd/istio/compose argv templates (Phase 9) |
|
|
151
|
+
| `pyproject.toml` | Package `devopsiq`: metadata, MIT, console script `devopsiq = main:main` (Phase 10) |
|
|
152
|
+
| `Dockerfile` | Prebuilt image: slim base + kubectl/helm/trivy/gh + app, non-root, `/data` record store (Phase 10) |
|
|
153
|
+
| `.github/workflows/release.yml` | Tag-driven release: test gate → PyPI (Trusted Publishing) + GHCR multi-arch (Phase 10) |
|
|
154
|
+
| `docs/generate_pdf.py` | Builds the project documentation PDF from live source |
|
|
155
|
+
| `docs/generate_release_guide.py` | Builds the release & update playbook PDF |
|
|
137
156
|
|
|
138
157
|
### The tool-use loop
|
|
139
158
|
|
|
@@ -180,6 +199,7 @@ the CLI exposes it directly:
|
|
|
180
199
|
| `/investigations` | List saved records on disk (newest first; `← active` marks the live one) |
|
|
181
200
|
| `/report` | Show the canonical report (once concluded) |
|
|
182
201
|
| `/endinvestigation` | Clear the record (memory only — the saved copy stays as history) |
|
|
202
|
+
| `/model [<name>]` | Show the active model, or save a default (config.json) and switch to it |
|
|
183
203
|
|
|
184
204
|
The report the agent ends with and `/report` render are kept consistent by
|
|
185
205
|
construction: `tool` results confirm each record call and the tracker, and
|
|
@@ -367,16 +387,22 @@ official `openai` Python SDK works as our client with two config lines:
|
|
|
367
387
|
OpenAI(api_key=..., base_url="https://openrouter.ai/api/v1")
|
|
368
388
|
```
|
|
369
389
|
|
|
370
|
-
Swapping to another OpenRouter model later is a one-
|
|
371
|
-
|
|
372
|
-
`agent/agent.py` configuration.
|
|
390
|
+
Swapping to another OpenRouter model later is a one-liner (env var, `/model`
|
|
391
|
+
command, or `--model` flag); moving to any other OpenAI-compatible provider
|
|
392
|
+
changes only `agent/agent.py` configuration.
|
|
373
393
|
|
|
374
394
|
**Your model, your choice.** The default is baked in as a fallback, never a
|
|
375
|
-
restriction
|
|
376
|
-
|
|
395
|
+
restriction. Four ways to set the model — the first one set wins:
|
|
396
|
+
|
|
397
|
+
1. `--model NAME` flag (one run: `devopsiq --model openai/gpt-5.2 "why?"`)
|
|
398
|
+
2. `/model <name>` in the REPL — saves it to `~/.devops-ai-agent/config.json`
|
|
399
|
+
so every future run uses it
|
|
400
|
+
3. `OPENROUTER_MODEL` (shell or `.env`)
|
|
401
|
+
4. built-in default (`z-ai/glm-5.3`)
|
|
377
402
|
|
|
378
403
|
```bash
|
|
379
404
|
export OPENROUTER_MODEL=anthropic/claude-sonnet-5 # or openai/gpt-5.2, google/gemini-2.5-pro, ...
|
|
405
|
+
devopsiq /model openai/gpt-5.2 # or save a default from the REPL
|
|
380
406
|
```
|
|
381
407
|
|
|
382
408
|
Two things to weigh when picking: the agent is a tool-use loop, so choose a
|
|
@@ -529,6 +555,7 @@ setup/API errors — so it drops straight into a pipeline:
|
|
|
529
555
|
.venv/bin/python main.py --json "why is api-5d6f crash-looping?" # structured report
|
|
530
556
|
.venv/bin/python main.py --resume --json "any update?" # continue a prior run
|
|
531
557
|
.venv/bin/python main.py --store-dir /tmp/runs --out report.json "..." # pipeline paths
|
|
558
|
+
.venv/bin/python main.py --model openai/gpt-5.2 "why is it down?" # one-run model override
|
|
532
559
|
```
|
|
533
560
|
|
|
534
561
|
With `--json` the stdout is one JSON document (see `render_report_json` in
|
|
@@ -570,8 +597,8 @@ Tools: ansible_inventory, ansible_playbook_tasks, aws_identity,
|
|
|
570
597
|
k8s_top_nodes, k8s_top_pods, loki_query, prom_query,
|
|
571
598
|
sys_open_ports, sys_service_logs, sys_service_status,
|
|
572
599
|
sys_top_processes, system_info, tf_plan, tf_show, tf_state_list
|
|
573
|
-
Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation
|
|
574
|
-
One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] "<problem>"
|
|
600
|
+
Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation, /model [<name>]
|
|
601
|
+
One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] [--model NAME] "<problem>"
|
|
575
602
|
Type 'exit' to quit.
|
|
576
603
|
|
|
577
604
|
You: The checkout service container keeps exiting in Docker. Investigate.
|
|
@@ -588,7 +615,7 @@ You: exit
|
|
|
588
615
|
Type `exit` / `quit`, or press Ctrl-D / Ctrl-C to leave. Slash commands are
|
|
589
616
|
handled locally and never reach the model.
|
|
590
617
|
|
|
591
|
-
## 8. Current limitations
|
|
618
|
+
## 8. Current limitations
|
|
592
619
|
|
|
593
620
|
- **Each domain needs its CLI installed and reachable.** Missing CLIs,
|
|
594
621
|
unauthenticated `gh`, a dead docker daemon, an uninitialized terraform
|
|
@@ -673,8 +700,22 @@ handled locally and never reach the model.
|
|
|
673
700
|
`helm_history` — reads only), Argo CD (`argocd_apps`,
|
|
674
701
|
`argocd_app_status`), Istio (`istioctl_proxy_status`), Docker Compose
|
|
675
702
|
(`docker_compose_ls`, `docker_compose_ps` — live-verified on this host).
|
|
703
|
+
- **Phase 10 — distribution. ✅ Done.** The `devopsiq` package on PyPI
|
|
704
|
+
(console command `devopsiq`, MIT) and the prebuilt multi-arch GHCR image,
|
|
705
|
+
cut by a tag-driven release workflow: an offline test gate, then PyPI
|
|
706
|
+
(Trusted Publishing — no token in the repo) and GHCR (amd64 + arm64) in
|
|
707
|
+
parallel. **`v0.1.1` added model-less mode:** with no API key the agent
|
|
708
|
+
still constructs — record commands (`/report`, `/investigations`) and the
|
|
709
|
+
58-tool layer work; only model questions exit 1 with the setup message.
|
|
710
|
+
Verified end-to-end against the published package.
|
|
711
|
+
- **Model choice + resilient calls (v0.1.2). ✅ Done.** The model is
|
|
712
|
+
set by a 4-rung ladder — `--model` flag > `/model`-saved config file
|
|
713
|
+
(`~/.devops-ai-agent/config.json`) > `OPENROUTER_MODEL` > baked-in
|
|
714
|
+
default — and transient model-call failures (timeouts, connection
|
|
715
|
+
errors, 429s, 5xx) retry with exponential backoff (2s→4s→8s + jitter,
|
|
716
|
+
honoring a 429's Retry-After); a rejected key fails immediately.
|
|
676
717
|
- **Later — region-scoped cloud resources** (ec2 describe-*, compute
|
|
677
718
|
instances list, ...) behind the same template pattern; more observability
|
|
678
719
|
depth (New Relic dashboards/entities, Prometheus range queries); streaming;
|
|
679
|
-
conversation-history persistence; and a human-approval gate
|
|
680
|
-
mutating action is ever allowed.
|
|
720
|
+
conversation-history persistence; and a human-approval gate
|
|
721
|
+
before any mutating action is ever allowed.
|
|
@@ -26,9 +26,18 @@ exits; the delegates below expose resume/list/store-location to the CLI.
|
|
|
26
26
|
"""
|
|
27
27
|
|
|
28
28
|
import os
|
|
29
|
+
import random
|
|
30
|
+
import time
|
|
31
|
+
|
|
32
|
+
from openai import (
|
|
33
|
+
APIConnectionError,
|
|
34
|
+
APIStatusError,
|
|
35
|
+
APITimeoutError,
|
|
36
|
+
OpenAI,
|
|
37
|
+
RateLimitError,
|
|
38
|
+
)
|
|
29
39
|
|
|
30
|
-
from
|
|
31
|
-
|
|
40
|
+
from agent.config import load_user_config
|
|
32
41
|
from agent.prompts import SYSTEM_PROMPT
|
|
33
42
|
from agent.store import InvestigationStore
|
|
34
43
|
from tools import ( # noqa: F401 — side effect: each module registers its tools
|
|
@@ -68,6 +77,50 @@ PLACEHOLDER_KEY = "your_key_here"
|
|
|
68
77
|
# Safety valve: the model gets at most this many tool-use turns before we stop.
|
|
69
78
|
MAX_TOOL_ITERATIONS = 10
|
|
70
79
|
|
|
80
|
+
# Transient-failure retries for the model call itself (network blips, 429,
|
|
81
|
+
# 5xx). MAX_MODEL_RETRIES attempts beyond the first, exponential backoff
|
|
82
|
+
# starting at RETRY_BASE_DELAY (2s -> 4s -> 8s) plus jitter, capped at
|
|
83
|
+
# RETRY_MAX_DELAY. A 429's Retry-After header wins when present.
|
|
84
|
+
MAX_MODEL_RETRIES = 3
|
|
85
|
+
RETRY_BASE_DELAY = 2.0
|
|
86
|
+
RETRY_MAX_DELAY = 60.0
|
|
87
|
+
|
|
88
|
+
_sleep = time.sleep # test seam: offline tests patch agent.agent._sleep
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _is_transient(exc: Exception) -> bool:
|
|
92
|
+
"""True when retrying `exc` can plausibly succeed.
|
|
93
|
+
|
|
94
|
+
Timeouts, connection failures, rate limits, and server-side 5xx are
|
|
95
|
+
transient. A rejected key (401) or any other client 4xx is not — the
|
|
96
|
+
same request would fail identically forever, so it fails immediately.
|
|
97
|
+
"""
|
|
98
|
+
if isinstance(exc, (APITimeoutError, APIConnectionError, RateLimitError)):
|
|
99
|
+
return True
|
|
100
|
+
if isinstance(exc, APIStatusError):
|
|
101
|
+
return exc.status_code >= 500
|
|
102
|
+
return False
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _retry_delay(exc: Exception, attempt: int) -> float:
|
|
106
|
+
"""Seconds to wait before retry number `attempt` (0-based).
|
|
107
|
+
|
|
108
|
+
Exponential backoff with jitter, except on a rate limit carrying a
|
|
109
|
+
Retry-After header — the server knows its own budget, so it wins
|
|
110
|
+
(clamped to [1, RETRY_MAX_DELAY] so a bad header cannot hurt us).
|
|
111
|
+
"""
|
|
112
|
+
response = getattr(exc, "response", None)
|
|
113
|
+
retry_after = getattr(response, "headers", {}).get("retry-after")
|
|
114
|
+
if retry_after:
|
|
115
|
+
try:
|
|
116
|
+
return min(max(float(retry_after), 1.0), RETRY_MAX_DELAY)
|
|
117
|
+
except (TypeError, ValueError):
|
|
118
|
+
pass # non-numeric header: fall through to backoff
|
|
119
|
+
return min(
|
|
120
|
+
RETRY_BASE_DELAY * (2 ** attempt) + random.uniform(0, 1),
|
|
121
|
+
RETRY_MAX_DELAY,
|
|
122
|
+
)
|
|
123
|
+
|
|
71
124
|
|
|
72
125
|
class DevOpsAgent:
|
|
73
126
|
"""Minimal DevOps investigation assistant (read-only by design)."""
|
|
@@ -81,7 +134,12 @@ class DevOpsAgent:
|
|
|
81
134
|
# Configuration resolution order: explicit argument > environment > default.
|
|
82
135
|
self.api_key = api_key or os.getenv("OPENROUTER_API_KEY")
|
|
83
136
|
|
|
84
|
-
self.model =
|
|
137
|
+
self.model = (
|
|
138
|
+
model # 1. explicit --model flag
|
|
139
|
+
or load_user_config().get("model") # 2. ~/.devops-ai-agent/config.json
|
|
140
|
+
or os.getenv("OPENROUTER_MODEL") # 3. environment / .env
|
|
141
|
+
or DEFAULT_MODEL # 4. built-in default
|
|
142
|
+
)
|
|
85
143
|
self.base_url = base_url or os.getenv("OPENROUTER_BASE_URL", DEFAULT_BASE_URL)
|
|
86
144
|
|
|
87
145
|
# Model-less mode: a missing (or placeholder) key no longer blocks
|
|
@@ -197,7 +255,7 @@ class DevOpsAgent:
|
|
|
197
255
|
if self.tools:
|
|
198
256
|
request["tools"] = [tool.schema() for tool in self.tools]
|
|
199
257
|
|
|
200
|
-
response = self.
|
|
258
|
+
response = self._call_model(request)
|
|
201
259
|
message = response.choices[0].message
|
|
202
260
|
|
|
203
261
|
if not message.tool_calls:
|
|
@@ -219,6 +277,23 @@ class DevOpsAgent:
|
|
|
219
277
|
f"The model did not finish after {MAX_TOOL_ITERATIONS} tool-use turns."
|
|
220
278
|
)
|
|
221
279
|
|
|
280
|
+
def _call_model(self, request: dict):
|
|
281
|
+
"""One model call with bounded retries on transient failures.
|
|
282
|
+
|
|
283
|
+
Timeouts, connection errors, rate limits (429) and server-side 5xx
|
|
284
|
+
are retried up to MAX_MODEL_RETRIES times with exponential backoff
|
|
285
|
+
(2s, 4s, 8s + jitter); a 429's Retry-After header wins when present.
|
|
286
|
+
Client errors — a rejected key (401) most notably — fail immediately,
|
|
287
|
+
because retrying the identical request cannot fix them.
|
|
288
|
+
"""
|
|
289
|
+
for attempt in range(MAX_MODEL_RETRIES + 1):
|
|
290
|
+
try:
|
|
291
|
+
return self.client.chat.completions.create(**request)
|
|
292
|
+
except Exception as exc: # noqa: BLE001 — classified right below
|
|
293
|
+
if attempt >= MAX_MODEL_RETRIES or not _is_transient(exc):
|
|
294
|
+
raise
|
|
295
|
+
_sleep(_retry_delay(exc, attempt))
|
|
296
|
+
|
|
222
297
|
|
|
223
298
|
def _echo_tool_request(message) -> dict:
|
|
224
299
|
"""Rebuild the assistant turn that requested tools, verbatim.
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""Per-user preferences: ~/.devops-ai-agent/config.json.
|
|
2
|
+
|
|
3
|
+
The `devopsiq` equivalent of "my settings" — currently the preferred model,
|
|
4
|
+
written by the /model command and read by the agent at startup. Precedence
|
|
5
|
+
for the model, highest first:
|
|
6
|
+
|
|
7
|
+
--model flag > config.json > OPENROUTER_MODEL env > built-in default
|
|
8
|
+
|
|
9
|
+
The file is tiny JSON, one flat object. Every read is tolerant: a missing
|
|
10
|
+
file, an unreadable home directory, or corrupt content yields {} — a broken
|
|
11
|
+
config never blocks the agent (the same degrade-don't-crash rule the
|
|
12
|
+
investigation store follows). Tests point AGENT_CONFIG_FILE at a tmp path.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import os
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
|
|
19
|
+
DEFAULT_CONFIG_PATH = Path.home() / ".devops-ai-agent" / "config.json"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def config_path() -> Path:
|
|
23
|
+
"""The active config file (AGENT_CONFIG_FILE overrides; tests use tmp)."""
|
|
24
|
+
return Path(os.getenv("AGENT_CONFIG_FILE") or DEFAULT_CONFIG_PATH)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def load_user_config() -> dict:
|
|
28
|
+
"""Read the config file; {} when missing or unreadable — never raise."""
|
|
29
|
+
try:
|
|
30
|
+
raw = config_path().read_text(encoding="utf-8")
|
|
31
|
+
data = json.loads(raw)
|
|
32
|
+
return data if isinstance(data, dict) else {}
|
|
33
|
+
except (OSError, ValueError):
|
|
34
|
+
return {}
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def save_user_config(update: dict) -> Path:
|
|
38
|
+
"""Merge `update` into the config file (create the directory if needed)."""
|
|
39
|
+
path = config_path()
|
|
40
|
+
data = load_user_config()
|
|
41
|
+
data.update(update)
|
|
42
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
43
|
+
path.write_text(json.dumps(data, indent=2, sort_keys=True) + "\n",
|
|
44
|
+
encoding="utf-8")
|
|
45
|
+
return path
|
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: devopsiq
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.2
|
|
4
4
|
Summary: A from-scratch, read-only AI agent that investigates real DevOps problems — Kubernetes, Docker, Helm, Argo CD, Istio, Terraform, GitHub Actions, cloud, New Relic and more.
|
|
5
5
|
Author: DevOpsAbhii
|
|
6
6
|
License-Expression: MIT
|
|
7
7
|
Project-URL: Homepage, https://github.com/DevOpsAbhii/devops-ai-agent
|
|
8
8
|
Project-URL: Repository, https://github.com/DevOpsAbhii/devops-ai-agent
|
|
9
9
|
Project-URL: Issues, https://github.com/DevOpsAbhii/devops-ai-agent/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/DevOpsAbhii/devops-ai-agent/blob/main/CHANGELOG.md
|
|
10
11
|
Keywords: devops,kubernetes,docker,helm,argocd,istio,terraform,github-actions,observability,newrelic,incident-response,root-cause-analysis,ai-agent,cli
|
|
11
12
|
Classifier: Development Status :: 4 - Beta
|
|
12
13
|
Classifier: Environment :: Console
|
|
@@ -29,6 +30,13 @@ Dynamic: license-file
|
|
|
29
30
|
|
|
30
31
|
# DevOps AI Agent
|
|
31
32
|
|
|
33
|
+
[](https://pypi.org/project/devopsiq/)
|
|
34
|
+
[](https://pypi.org/project/devopsiq/)
|
|
35
|
+
[](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
|
|
36
|
+
[](https://github.com/DevOpsAbhii/devops-ai-agent/pkgs/container/devops-ai-agent)
|
|
37
|
+
[](LICENSE)
|
|
38
|
+
[](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
|
|
39
|
+
|
|
32
40
|
An AI agent that investigates real DevOps problems. The end goal: ask it
|
|
33
41
|
something like *"Why is my Kubernetes pod in CrashLoopBackOff?"* and have it
|
|
34
42
|
gather evidence, reason about the evidence, identify the likely root cause,
|
|
@@ -55,7 +63,14 @@ from environment config), and Ansible listing (inventory, playbook tasks).
|
|
|
55
63
|
open alerts over NerdGraph — credentials from env), Trivy image
|
|
56
64
|
vulnerability scanning, Helm releases (list/status/history), Argo CD
|
|
57
65
|
(GitOps app sync/health), Istio mesh proxy status, and Docker Compose
|
|
58
|
-
(project and service listing).
|
|
66
|
+
(project and service listing). **Phase 10 ships it as a product:** the
|
|
67
|
+
[`devopsiq` package on PyPI](https://pypi.org/project/devopsiq/) and a
|
|
68
|
+
prebuilt multi-arch Docker image (`ghcr.io/devopsabhii/devops-ai-agent`) —
|
|
69
|
+
both cut automatically by pushing a `v*` tag (test gate → PyPI via Trusted
|
|
70
|
+
Publishing + GHCR in parallel). **No API key? The agent still runs:** it
|
|
71
|
+
starts in model-less mode — record commands (`/report`, `/investigations`)
|
|
72
|
+
and the whole 58-tool layer work without a key; only questions to the model
|
|
73
|
+
need one. The repository is git-tracked.
|
|
59
74
|
Every phase still built from scratch — no LangChain, LangGraph,
|
|
60
75
|
AutoGen, CrewAI, or MCP.
|
|
61
76
|
|
|
@@ -73,7 +88,7 @@ model is called, how conversation history flows, how tool selection +
|
|
|
73
88
|
execution + evidence feedback work — instead of depending on a framework
|
|
74
89
|
for it.
|
|
75
90
|
|
|
76
|
-
|
|
91
|
+
Today the agent:
|
|
77
92
|
- holds a conversation with **GLM 5.3** through **OpenRouter**;
|
|
78
93
|
- has **58 real, read-only tools** across fifteen domains: host facts,
|
|
79
94
|
Kubernetes (12 tools), Linux system (4), Docker + Compose (10),
|
|
@@ -133,8 +148,8 @@ Module map:
|
|
|
133
148
|
|
|
134
149
|
| Path | Responsibility |
|
|
135
150
|
| --------------------------- | ------------------------------------------------------------ |
|
|
136
|
-
| `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages |
|
|
137
|
-
| `agent/agent.py` | `DevOpsAgent` — client, history, `ask()`, `_complete()` loop |
|
|
151
|
+
| `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages, model-less start |
|
|
152
|
+
| `agent/agent.py` | `DevOpsAgent` — client (`None` = model-less mode), history, `ask()`, `_complete()` loop |
|
|
138
153
|
| `agent/prompts.py` | The system prompt (versioned/tested separately) |
|
|
139
154
|
| `agent/investigation.py` | First-class investigation record: hypotheses, verdicts, evidence, report renderer (pure data) |
|
|
140
155
|
| `agent/store.py` | `InvestigationStore` — one JSON file per record, atomic writes, resume/list (Phase 7) |
|
|
@@ -163,6 +178,11 @@ Module map:
|
|
|
163
178
|
| `tests/test_phase7.py` | Offline suite: round-trip serialization, store files, auto-save, resume, CLI flags (Phase 7) |
|
|
164
179
|
| `tests/test_phase8.py` | Offline suite: argv templates + validation for the 21 Phase 8 tools (fake CLIs, env-based monitoring) |
|
|
165
180
|
| `tests/test_phase9.py` | Offline suite: New Relic env-credential + payload tests, trivy/helm/argocd/istio/compose argv templates (Phase 9) |
|
|
181
|
+
| `pyproject.toml` | Package `devopsiq`: metadata, MIT, console script `devopsiq = main:main` (Phase 10) |
|
|
182
|
+
| `Dockerfile` | Prebuilt image: slim base + kubectl/helm/trivy/gh + app, non-root, `/data` record store (Phase 10) |
|
|
183
|
+
| `.github/workflows/release.yml` | Tag-driven release: test gate → PyPI (Trusted Publishing) + GHCR multi-arch (Phase 10) |
|
|
184
|
+
| `docs/generate_pdf.py` | Builds the project documentation PDF from live source |
|
|
185
|
+
| `docs/generate_release_guide.py` | Builds the release & update playbook PDF |
|
|
166
186
|
|
|
167
187
|
### The tool-use loop
|
|
168
188
|
|
|
@@ -209,6 +229,7 @@ the CLI exposes it directly:
|
|
|
209
229
|
| `/investigations` | List saved records on disk (newest first; `← active` marks the live one) |
|
|
210
230
|
| `/report` | Show the canonical report (once concluded) |
|
|
211
231
|
| `/endinvestigation` | Clear the record (memory only — the saved copy stays as history) |
|
|
232
|
+
| `/model [<name>]` | Show the active model, or save a default (config.json) and switch to it |
|
|
212
233
|
|
|
213
234
|
The report the agent ends with and `/report` render are kept consistent by
|
|
214
235
|
construction: `tool` results confirm each record call and the tracker, and
|
|
@@ -396,16 +417,22 @@ official `openai` Python SDK works as our client with two config lines:
|
|
|
396
417
|
OpenAI(api_key=..., base_url="https://openrouter.ai/api/v1")
|
|
397
418
|
```
|
|
398
419
|
|
|
399
|
-
Swapping to another OpenRouter model later is a one-
|
|
400
|
-
|
|
401
|
-
`agent/agent.py` configuration.
|
|
420
|
+
Swapping to another OpenRouter model later is a one-liner (env var, `/model`
|
|
421
|
+
command, or `--model` flag); moving to any other OpenAI-compatible provider
|
|
422
|
+
changes only `agent/agent.py` configuration.
|
|
402
423
|
|
|
403
424
|
**Your model, your choice.** The default is baked in as a fallback, never a
|
|
404
|
-
restriction
|
|
405
|
-
|
|
425
|
+
restriction. Four ways to set the model — the first one set wins:
|
|
426
|
+
|
|
427
|
+
1. `--model NAME` flag (one run: `devopsiq --model openai/gpt-5.2 "why?"`)
|
|
428
|
+
2. `/model <name>` in the REPL — saves it to `~/.devops-ai-agent/config.json`
|
|
429
|
+
so every future run uses it
|
|
430
|
+
3. `OPENROUTER_MODEL` (shell or `.env`)
|
|
431
|
+
4. built-in default (`z-ai/glm-5.3`)
|
|
406
432
|
|
|
407
433
|
```bash
|
|
408
434
|
export OPENROUTER_MODEL=anthropic/claude-sonnet-5 # or openai/gpt-5.2, google/gemini-2.5-pro, ...
|
|
435
|
+
devopsiq /model openai/gpt-5.2 # or save a default from the REPL
|
|
409
436
|
```
|
|
410
437
|
|
|
411
438
|
Two things to weigh when picking: the agent is a tool-use loop, so choose a
|
|
@@ -558,6 +585,7 @@ setup/API errors — so it drops straight into a pipeline:
|
|
|
558
585
|
.venv/bin/python main.py --json "why is api-5d6f crash-looping?" # structured report
|
|
559
586
|
.venv/bin/python main.py --resume --json "any update?" # continue a prior run
|
|
560
587
|
.venv/bin/python main.py --store-dir /tmp/runs --out report.json "..." # pipeline paths
|
|
588
|
+
.venv/bin/python main.py --model openai/gpt-5.2 "why is it down?" # one-run model override
|
|
561
589
|
```
|
|
562
590
|
|
|
563
591
|
With `--json` the stdout is one JSON document (see `render_report_json` in
|
|
@@ -599,8 +627,8 @@ Tools: ansible_inventory, ansible_playbook_tasks, aws_identity,
|
|
|
599
627
|
k8s_top_nodes, k8s_top_pods, loki_query, prom_query,
|
|
600
628
|
sys_open_ports, sys_service_logs, sys_service_status,
|
|
601
629
|
sys_top_processes, system_info, tf_plan, tf_show, tf_state_list
|
|
602
|
-
Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation
|
|
603
|
-
One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] "<problem>"
|
|
630
|
+
Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation, /model [<name>]
|
|
631
|
+
One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] [--model NAME] "<problem>"
|
|
604
632
|
Type 'exit' to quit.
|
|
605
633
|
|
|
606
634
|
You: The checkout service container keeps exiting in Docker. Investigate.
|
|
@@ -617,7 +645,7 @@ You: exit
|
|
|
617
645
|
Type `exit` / `quit`, or press Ctrl-D / Ctrl-C to leave. Slash commands are
|
|
618
646
|
handled locally and never reach the model.
|
|
619
647
|
|
|
620
|
-
## 8. Current limitations
|
|
648
|
+
## 8. Current limitations
|
|
621
649
|
|
|
622
650
|
- **Each domain needs its CLI installed and reachable.** Missing CLIs,
|
|
623
651
|
unauthenticated `gh`, a dead docker daemon, an uninitialized terraform
|
|
@@ -702,8 +730,22 @@ handled locally and never reach the model.
|
|
|
702
730
|
`helm_history` — reads only), Argo CD (`argocd_apps`,
|
|
703
731
|
`argocd_app_status`), Istio (`istioctl_proxy_status`), Docker Compose
|
|
704
732
|
(`docker_compose_ls`, `docker_compose_ps` — live-verified on this host).
|
|
733
|
+
- **Phase 10 — distribution. ✅ Done.** The `devopsiq` package on PyPI
|
|
734
|
+
(console command `devopsiq`, MIT) and the prebuilt multi-arch GHCR image,
|
|
735
|
+
cut by a tag-driven release workflow: an offline test gate, then PyPI
|
|
736
|
+
(Trusted Publishing — no token in the repo) and GHCR (amd64 + arm64) in
|
|
737
|
+
parallel. **`v0.1.1` added model-less mode:** with no API key the agent
|
|
738
|
+
still constructs — record commands (`/report`, `/investigations`) and the
|
|
739
|
+
58-tool layer work; only model questions exit 1 with the setup message.
|
|
740
|
+
Verified end-to-end against the published package.
|
|
741
|
+
- **Model choice + resilient calls (v0.1.2). ✅ Done.** The model is
|
|
742
|
+
set by a 4-rung ladder — `--model` flag > `/model`-saved config file
|
|
743
|
+
(`~/.devops-ai-agent/config.json`) > `OPENROUTER_MODEL` > baked-in
|
|
744
|
+
default — and transient model-call failures (timeouts, connection
|
|
745
|
+
errors, 429s, 5xx) retry with exponential backoff (2s→4s→8s + jitter,
|
|
746
|
+
honoring a 429's Retry-After); a rejected key fails immediately.
|
|
705
747
|
- **Later — region-scoped cloud resources** (ec2 describe-*, compute
|
|
706
748
|
instances list, ...) behind the same template pattern; more observability
|
|
707
749
|
depth (New Relic dashboards/entities, Prometheus range queries); streaming;
|
|
708
|
-
conversation-history persistence; and a human-approval gate
|
|
709
|
-
mutating action is ever allowed.
|
|
750
|
+
conversation-history persistence; and a human-approval gate
|
|
751
|
+
before any mutating action is ever allowed.
|
|
@@ -10,6 +10,7 @@ One-shot (for cron / CI / scripts):
|
|
|
10
10
|
python main.py --json "why is api-5d6f crash-looping?" # structured report
|
|
11
11
|
python main.py --resume --json "any update?" # continue prior run
|
|
12
12
|
python main.py --store-dir DIR --out report.json "..." # pipeline paths
|
|
13
|
+
python main.py --model openai/gpt-4o-mini "..." # one-run model override
|
|
13
14
|
|
|
14
15
|
One-shot mode sends the problem once, prints the agent's report, and exits
|
|
15
16
|
with a status code (0 = completed, 1 = setup/API error). With --json the
|
|
@@ -41,6 +42,7 @@ from openai import (
|
|
|
41
42
|
)
|
|
42
43
|
|
|
43
44
|
from agent.agent import DevOpsAgent
|
|
45
|
+
from agent.config import config_path, save_user_config
|
|
44
46
|
|
|
45
47
|
EXIT_WORDS = {"exit", "quit"}
|
|
46
48
|
|
|
@@ -55,6 +57,7 @@ COMMAND_ALIASES = {
|
|
|
55
57
|
"/report": "/report",
|
|
56
58
|
"/endinvestigation": "/endinvestigation",
|
|
57
59
|
"/end": "/endinvestigation",
|
|
60
|
+
"/model": "/model",
|
|
58
61
|
}
|
|
59
62
|
|
|
60
63
|
|
|
@@ -82,8 +85,32 @@ def handle_command(text: str, agent: DevOpsAgent) -> str | None:
|
|
|
82
85
|
return agent.investigation_report_text() or "(no investigation recorded)"
|
|
83
86
|
if canonical == "/endinvestigation":
|
|
84
87
|
return agent.end_investigation()
|
|
88
|
+
if canonical == "/model":
|
|
89
|
+
return handle_model_command(rest, agent)
|
|
85
90
|
return (f"unknown command: {cmd}. Try /investigate <problem>, "
|
|
86
|
-
"/investigation, /investigations, /report, /endinvestigation"
|
|
91
|
+
"/investigation, /investigations, /report, /endinvestigation, "
|
|
92
|
+
"/model")
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def handle_model_command(rest: str, agent: DevOpsAgent) -> str:
|
|
96
|
+
"""/model (show the active model and where it came from) or /model <name>.
|
|
97
|
+
|
|
98
|
+
With a name: save it as the user's preference in the config file
|
|
99
|
+
(~/.devops-ai-agent/config.json) and switch this session to it. The
|
|
100
|
+
next run picks it up automatically — the flag and env still outrank it.
|
|
101
|
+
"""
|
|
102
|
+
name = rest.strip()
|
|
103
|
+
if not name:
|
|
104
|
+
return (f"Model: {agent.model}\n"
|
|
105
|
+
f"Config file: {config_path()} "
|
|
106
|
+
"(save a default with /model <name>)")
|
|
107
|
+
if len(name.split()) != 1:
|
|
108
|
+
return "Model name must be a single token, e.g. /model openai/gpt-4o-mini"
|
|
109
|
+
path = save_user_config({"model": name})
|
|
110
|
+
agent.model = name
|
|
111
|
+
return (f"Model set to {name} (saved in {path}).\n"
|
|
112
|
+
"This session now uses it; the --model flag and OPENROUTER_MODEL "
|
|
113
|
+
"still override it per run.")
|
|
87
114
|
|
|
88
115
|
|
|
89
116
|
class OneShotArgs(NamedTuple):
|
|
@@ -94,9 +121,10 @@ class OneShotArgs(NamedTuple):
|
|
|
94
121
|
store_dir: str | None # --store-dir PATH (persistence override)
|
|
95
122
|
resume: bool # --resume (continue the newest in-progress record)
|
|
96
123
|
out: str | None # --out PATH (also write the JSON report there)
|
|
124
|
+
model: str | None # --model NAME (one-run model override)
|
|
97
125
|
|
|
98
126
|
|
|
99
|
-
_VALUE_FLAGS = ("--store-dir", "--out")
|
|
127
|
+
_VALUE_FLAGS = ("--store-dir", "--out", "--model")
|
|
100
128
|
|
|
101
129
|
|
|
102
130
|
def _flag_value(argv: list[str], flag: str) -> str | None:
|
|
@@ -114,14 +142,16 @@ def parse_args(argv: list[str]) -> OneShotArgs:
|
|
|
114
142
|
--json switches the one-shot output from markdown to the structured JSON
|
|
115
143
|
report; --resume continues the newest in-progress record from the store;
|
|
116
144
|
--store-dir PATH overrides the persistence directory for this run; --out
|
|
117
|
-
PATH additionally writes the JSON report to an exact path
|
|
145
|
+
PATH additionally writes the JSON report to an exact path; --model NAME
|
|
146
|
+
overrides the model for this run (highest model precedence). Example:
|
|
118
147
|
`python main.py --json --out r.json "why is it down?"` ->
|
|
119
|
-
("why is it down?", True, None, False, "r.json").
|
|
148
|
+
("why is it down?", True, None, False, "r.json", None).
|
|
120
149
|
"""
|
|
121
150
|
as_json = "--json" in argv
|
|
122
151
|
do_resume = "--resume" in argv
|
|
123
152
|
store_dir = _flag_value(argv, "--store-dir")
|
|
124
153
|
out = _flag_value(argv, "--out")
|
|
154
|
+
model = _flag_value(argv, "--model")
|
|
125
155
|
positionals: list[str] = []
|
|
126
156
|
skip_next = False
|
|
127
157
|
for arg in argv:
|
|
@@ -136,8 +166,8 @@ def parse_args(argv: list[str]) -> OneShotArgs:
|
|
|
136
166
|
positionals.append(arg)
|
|
137
167
|
text = " ".join(positionals).strip()
|
|
138
168
|
if not positionals or not text:
|
|
139
|
-
return OneShotArgs(None, as_json, store_dir, do_resume, out)
|
|
140
|
-
return OneShotArgs(text, as_json, store_dir, do_resume, out)
|
|
169
|
+
return OneShotArgs(None, as_json, store_dir, do_resume, out, model)
|
|
170
|
+
return OneShotArgs(text, as_json, store_dir, do_resume, out, model)
|
|
141
171
|
|
|
142
172
|
|
|
143
173
|
def run_one_shot(
|
|
@@ -147,6 +177,7 @@ def run_one_shot(
|
|
|
147
177
|
store_dir: str | None = None,
|
|
148
178
|
resume: bool = False,
|
|
149
179
|
out: str | None = None,
|
|
180
|
+
model: str | None = None,
|
|
150
181
|
) -> int:
|
|
151
182
|
"""Non-interactive single run: send `task`, print the result, exit cleanly.
|
|
152
183
|
|
|
@@ -155,14 +186,14 @@ def run_one_shot(
|
|
|
155
186
|
opened an investigation this is a hard failure (exit 1) — a caller asked
|
|
156
187
|
for a report and there is none to give. The record is auto-saved on every
|
|
157
188
|
mutation regardless; store_dir points persistence somewhere else for this
|
|
158
|
-
run, resume continues the newest in-progress record before asking,
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
fresh one from the environment.
|
|
189
|
+
run, resume continues the newest in-progress record before asking, out
|
|
190
|
+
additionally writes the JSON report to an exact path, and model overrides
|
|
191
|
+
the model for this run. `agent` lets embedders reuse a configured agent
|
|
192
|
+
(also the test seam); default builds a fresh one from the environment.
|
|
162
193
|
"""
|
|
163
194
|
if agent is None:
|
|
164
195
|
try:
|
|
165
|
-
agent = DevOpsAgent()
|
|
196
|
+
agent = DevOpsAgent(model=model)
|
|
166
197
|
except ValueError as exc:
|
|
167
198
|
print(f"[setup] {exc}", file=sys.stderr)
|
|
168
199
|
return 1
|
|
@@ -245,10 +276,11 @@ def main() -> int:
|
|
|
245
276
|
store_dir=args.store_dir,
|
|
246
277
|
resume=args.resume,
|
|
247
278
|
out=args.out,
|
|
279
|
+
model=args.model,
|
|
248
280
|
)
|
|
249
281
|
|
|
250
282
|
try:
|
|
251
|
-
agent = DevOpsAgent()
|
|
283
|
+
agent = DevOpsAgent(model=args.model)
|
|
252
284
|
except ValueError as exc:
|
|
253
285
|
print(f"[setup] {exc}", file=sys.stderr)
|
|
254
286
|
return 1
|
|
@@ -268,9 +300,9 @@ def main() -> int:
|
|
|
268
300
|
print(f"Store: {agent.store_dir or '(persistence off)'}")
|
|
269
301
|
print(f"Tools: {tool_names}")
|
|
270
302
|
print("Commands: /investigate <problem>, /investigation, /investigations, "
|
|
271
|
-
"/report, /endinvestigation")
|
|
303
|
+
"/report, /endinvestigation, /model [<name>]")
|
|
272
304
|
print("One-shot: python main.py [--json] [--resume] [--out report.json] "
|
|
273
|
-
"[--store-dir DIR] \"<problem>\"")
|
|
305
|
+
"[--store-dir DIR] [--model NAME] \"<problem>\"")
|
|
274
306
|
print("Type 'exit' to quit.")
|
|
275
307
|
print()
|
|
276
308
|
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "devopsiq"
|
|
7
|
-
version = "0.1.
|
|
7
|
+
version = "0.1.2"
|
|
8
8
|
description = "A from-scratch, read-only AI agent that investigates real DevOps problems — Kubernetes, Docker, Helm, Argo CD, Istio, Terraform, GitHub Actions, cloud, New Relic and more."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -38,6 +38,7 @@ dependencies = [
|
|
|
38
38
|
Homepage = "https://github.com/DevOpsAbhii/devops-ai-agent"
|
|
39
39
|
Repository = "https://github.com/DevOpsAbhii/devops-ai-agent"
|
|
40
40
|
Issues = "https://github.com/DevOpsAbhii/devops-ai-agent/issues"
|
|
41
|
+
Changelog = "https://github.com/DevOpsAbhii/devops-ai-agent/blob/main/CHANGELOG.md"
|
|
41
42
|
|
|
42
43
|
[project.scripts]
|
|
43
44
|
devopsiq = "main:main"
|
|
@@ -11,12 +11,25 @@ import contextlib
|
|
|
11
11
|
import io
|
|
12
12
|
import json
|
|
13
13
|
import os
|
|
14
|
+
import tempfile
|
|
15
|
+
import time
|
|
14
16
|
import types
|
|
15
17
|
import unittest
|
|
18
|
+
from pathlib import Path
|
|
16
19
|
|
|
20
|
+
import httpx2 as httpx # openai 3.19 depends on the httpx2 fork
|
|
17
21
|
import main as main_module
|
|
18
|
-
|
|
22
|
+
import agent.agent as agent_module
|
|
23
|
+
from agent import config as user_config
|
|
24
|
+
from agent.agent import DEFAULT_MODEL, DevOpsAgent
|
|
19
25
|
from agent.investigation import Investigation
|
|
26
|
+
from openai import (
|
|
27
|
+
APIConnectionError,
|
|
28
|
+
APIStatusError,
|
|
29
|
+
APITimeoutError,
|
|
30
|
+
AuthenticationError,
|
|
31
|
+
RateLimitError,
|
|
32
|
+
)
|
|
20
33
|
from tools import investigation as inv_tools
|
|
21
34
|
|
|
22
35
|
|
|
@@ -187,19 +200,27 @@ class OneShotCliTests(unittest.TestCase):
|
|
|
187
200
|
|
|
188
201
|
def test_parse_args(self):
|
|
189
202
|
self.assertEqual(
|
|
190
|
-
main_module.parse_args([]), (None, False, None, False, None)
|
|
203
|
+
main_module.parse_args([]), (None, False, None, False, None, None)
|
|
191
204
|
)
|
|
192
205
|
self.assertEqual(
|
|
193
206
|
main_module.parse_args(["why is it down?"]),
|
|
194
|
-
("why is it down?", False, None, False, None),
|
|
207
|
+
("why is it down?", False, None, False, None, None),
|
|
195
208
|
)
|
|
196
209
|
self.assertEqual(
|
|
197
210
|
main_module.parse_args(["--json", "why", "is it down?"]),
|
|
198
|
-
("why is it down?", True, None, False, None),
|
|
211
|
+
("why is it down?", True, None, False, None, None),
|
|
199
212
|
)
|
|
200
213
|
self.assertEqual(
|
|
201
214
|
main_module.parse_args(["--json"]),
|
|
202
|
-
(None, True, None, False, None),
|
|
215
|
+
(None, True, None, False, None, None),
|
|
216
|
+
)
|
|
217
|
+
self.assertEqual(
|
|
218
|
+
main_module.parse_args(["--model", "openai/gpt-4o-mini", "why?"]),
|
|
219
|
+
("why?", False, None, False, None, "openai/gpt-4o-mini"),
|
|
220
|
+
)
|
|
221
|
+
self.assertEqual(
|
|
222
|
+
main_module.parse_args(["--model", "anthropic/claude-sonnet-5"]),
|
|
223
|
+
(None, False, None, False, None, "anthropic/claude-sonnet-5"),
|
|
203
224
|
)
|
|
204
225
|
|
|
205
226
|
def test_slash_command_routes_without_model(self):
|
|
@@ -300,5 +321,207 @@ class KeylessTests(unittest.TestCase):
|
|
|
300
321
|
self.assertIn("OPENROUTER_API_KEY", err.getvalue())
|
|
301
322
|
|
|
302
323
|
|
|
324
|
+
class ModelChoiceTests(unittest.TestCase):
|
|
325
|
+
"""Model preference ladder + the /model command + agent/config.py.
|
|
326
|
+
|
|
327
|
+
Precedence, highest first: explicit model argument > config.json >
|
|
328
|
+
OPENROUTER_MODEL env > built-in default. Tests point AGENT_CONFIG_FILE
|
|
329
|
+
at a tmp file so the developer's real ~/.devops-ai-agent/config.json
|
|
330
|
+
is never read or written.
|
|
331
|
+
"""
|
|
332
|
+
|
|
333
|
+
def setUp(self):
|
|
334
|
+
tmp = tempfile.NamedTemporaryFile(suffix=".json", delete=False)
|
|
335
|
+
tmp.close()
|
|
336
|
+
self.config_file = tmp.name
|
|
337
|
+
os.environ["AGENT_CONFIG_FILE"] = self.config_file
|
|
338
|
+
self.addCleanup(self._restore)
|
|
339
|
+
|
|
340
|
+
def _restore(self):
|
|
341
|
+
os.environ.pop("AGENT_CONFIG_FILE", None)
|
|
342
|
+
Path(self.config_file).unlink(missing_ok=True)
|
|
343
|
+
|
|
344
|
+
def _set_env(self, name, value):
|
|
345
|
+
old = os.environ.get(name)
|
|
346
|
+
os.environ[name] = value
|
|
347
|
+
self.addCleanup(lambda: (
|
|
348
|
+
os.environ.pop(name, None) if old is None
|
|
349
|
+
else os.environ.__setitem__(name, old)
|
|
350
|
+
))
|
|
351
|
+
|
|
352
|
+
def test_config_round_trip_and_corrupt_tolerance(self):
|
|
353
|
+
self.assertEqual(user_config.load_user_config(), {}) # missing file
|
|
354
|
+
path = user_config.save_user_config({"model": "openai/gpt-4o-mini"})
|
|
355
|
+
self.assertEqual(path, Path(self.config_file))
|
|
356
|
+
self.assertEqual(user_config.load_user_config(),
|
|
357
|
+
{"model": "openai/gpt-4o-mini"})
|
|
358
|
+
# Merging keeps existing keys and overwrites the given one.
|
|
359
|
+
user_config.save_user_config({"model": "anthropic/claude-sonnet-5"})
|
|
360
|
+
self.assertEqual(user_config.load_user_config(),
|
|
361
|
+
{"model": "anthropic/claude-sonnet-5"})
|
|
362
|
+
# Corrupt content degrades to {} instead of raising.
|
|
363
|
+
Path(self.config_file).write_text("{not json", encoding="utf-8")
|
|
364
|
+
self.assertEqual(user_config.load_user_config(), {})
|
|
365
|
+
|
|
366
|
+
def test_model_precedence_ladder(self):
|
|
367
|
+
# 4. built-in default
|
|
368
|
+
self.assertEqual(DevOpsAgent().model, DEFAULT_MODEL)
|
|
369
|
+
# 3. environment
|
|
370
|
+
self._set_env("OPENROUTER_MODEL", "env/model")
|
|
371
|
+
self.assertEqual(DevOpsAgent().model, "env/model")
|
|
372
|
+
# 2. config file (outranks env)
|
|
373
|
+
user_config.save_user_config({"model": "config/model"})
|
|
374
|
+
self.assertEqual(DevOpsAgent().model, "config/model")
|
|
375
|
+
# 1. explicit argument (outranks everything)
|
|
376
|
+
self.assertEqual(DevOpsAgent(model="flag/model").model, "flag/model")
|
|
377
|
+
|
|
378
|
+
def test_model_command_shows_current_model(self):
|
|
379
|
+
agent = DevOpsAgent()
|
|
380
|
+
before = agent.model
|
|
381
|
+
text = main_module.handle_command("/model", agent)
|
|
382
|
+
self.assertIn(before, text)
|
|
383
|
+
self.assertIn(str(user_config.config_path()), text)
|
|
384
|
+
self.assertEqual(agent.model, before) # show-only changes nothing
|
|
385
|
+
|
|
386
|
+
def test_model_command_saves_and_switches(self):
|
|
387
|
+
agent = DevOpsAgent()
|
|
388
|
+
out = io.StringIO()
|
|
389
|
+
with contextlib.redirect_stdout(out):
|
|
390
|
+
text = main_module.handle_command(
|
|
391
|
+
"/model openai/gpt-4o-mini", agent)
|
|
392
|
+
self.assertIn("openai/gpt-4o-mini", text)
|
|
393
|
+
self.assertEqual(agent.model, "openai/gpt-4o-mini")
|
|
394
|
+
self.assertEqual(
|
|
395
|
+
user_config.load_user_config(), {"model": "openai/gpt-4o-mini"})
|
|
396
|
+
# A multi-token name is rejected; the config stays untouched.
|
|
397
|
+
text = main_module.handle_command("/model two words", agent)
|
|
398
|
+
self.assertIn("single token", text)
|
|
399
|
+
self.assertEqual(agent.model, "openai/gpt-4o-mini")
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
class FlakyChat:
|
|
403
|
+
"""Chat stub that raises the queued errors, then answers with content."""
|
|
404
|
+
|
|
405
|
+
def __init__(self, errors: list[Exception], content: str = "ok"):
|
|
406
|
+
self._errors = list(errors)
|
|
407
|
+
self._content = content
|
|
408
|
+
self.calls = 0
|
|
409
|
+
|
|
410
|
+
def create(self, **kwargs): # noqa: D102
|
|
411
|
+
self.calls += 1
|
|
412
|
+
if self._errors:
|
|
413
|
+
raise self._errors.pop(0)
|
|
414
|
+
message = types.SimpleNamespace(content=self._content, tool_calls=None)
|
|
415
|
+
return types.SimpleNamespace(
|
|
416
|
+
choices=[types.SimpleNamespace(message=message)]
|
|
417
|
+
)
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def _api_status_error(status: int, headers: dict | None = None):
|
|
421
|
+
"""Build an openai status error offline (no HTTP round-trip)."""
|
|
422
|
+
request = httpx.Request(
|
|
423
|
+
"POST", "https://openrouter.ai/api/v1/chat/completions")
|
|
424
|
+
response = httpx.Response(
|
|
425
|
+
status, request=request, headers=headers or {})
|
|
426
|
+
return APIStatusError("boom", response=response, body=None)
|
|
427
|
+
|
|
428
|
+
|
|
429
|
+
def _api_connection_error():
|
|
430
|
+
request = httpx.Request(
|
|
431
|
+
"POST", "https://openrouter.ai/api/v1/chat/completions")
|
|
432
|
+
return APIConnectionError(request=request)
|
|
433
|
+
|
|
434
|
+
|
|
435
|
+
class RetryTests(unittest.TestCase):
|
|
436
|
+
"""_call_model: transient failures retry with backoff, others fail fast.
|
|
437
|
+
|
|
438
|
+
Sleeps are intercepted via the agent.agent._sleep seam, so the tests
|
|
439
|
+
run instantly; the recorded delays prove the backoff schedule.
|
|
440
|
+
"""
|
|
441
|
+
|
|
442
|
+
def setUp(self):
|
|
443
|
+
self._old_key = os.environ.get("OPENROUTER_API_KEY")
|
|
444
|
+
os.environ["OPENROUTER_API_KEY"] = "test-key-for-offline-tests"
|
|
445
|
+
self.agent = DevOpsAgent()
|
|
446
|
+
self.sleeps: list[float] = []
|
|
447
|
+
agent_module._sleep = self.sleeps.append
|
|
448
|
+
self.addCleanup(self._restore)
|
|
449
|
+
|
|
450
|
+
def _restore(self):
|
|
451
|
+
agent_module._sleep = time.sleep
|
|
452
|
+
os.environ.pop("OPENROUTER_API_KEY", None)
|
|
453
|
+
if self._old_key is not None:
|
|
454
|
+
os.environ["OPENROUTER_API_KEY"] = self._old_key
|
|
455
|
+
|
|
456
|
+
def test_connection_errors_are_retried_then_succeed(self):
|
|
457
|
+
chat = FlakyChat([_api_connection_error(), _api_connection_error()])
|
|
458
|
+
self.agent.client.chat.completions = chat
|
|
459
|
+
self.assertEqual(self.agent.ask("why is it down?"), "ok")
|
|
460
|
+
self.assertEqual(chat.calls, 3) # original + 2 retries
|
|
461
|
+
# Backoff: attempt 0 -> ~2s (+jitter), attempt 1 -> ~4s (+jitter).
|
|
462
|
+
self.assertGreaterEqual(self.sleeps[0], 2.0)
|
|
463
|
+
self.assertLessEqual(self.sleeps[0], 3.0)
|
|
464
|
+
self.assertGreaterEqual(self.sleeps[1], 4.0)
|
|
465
|
+
self.assertLessEqual(self.sleeps[1], 5.0)
|
|
466
|
+
|
|
467
|
+
def test_timeout_is_transient(self):
|
|
468
|
+
request = httpx.Request("POST", "https://openrouter.ai/api/v1")
|
|
469
|
+
chat = FlakyChat([APITimeoutError(request)])
|
|
470
|
+
self.agent.client.chat.completions = chat
|
|
471
|
+
self.assertEqual(self.agent.ask("hello"), "ok")
|
|
472
|
+
self.assertEqual(chat.calls, 2)
|
|
473
|
+
|
|
474
|
+
def test_rate_limit_honors_retry_after_header(self):
|
|
475
|
+
request = httpx.Request("POST", "https://openrouter.ai/api/v1")
|
|
476
|
+
response = httpx.Response(
|
|
477
|
+
429, request=request, headers={"retry-after": "7"})
|
|
478
|
+
chat = FlakyChat([RateLimitError("slow down", response=response,
|
|
479
|
+
body=None)])
|
|
480
|
+
self.agent.client.chat.completions = chat
|
|
481
|
+
self.assertEqual(self.agent.ask("hello"), "ok")
|
|
482
|
+
self.assertEqual(self.sleeps, [7.0])
|
|
483
|
+
|
|
484
|
+
def test_rate_limit_without_header_uses_backoff(self):
|
|
485
|
+
request = httpx.Request("POST", "https://openrouter.ai/api/v1")
|
|
486
|
+
response = httpx.Response(429, request=request)
|
|
487
|
+
chat = FlakyChat([RateLimitError("slow down", response=response,
|
|
488
|
+
body=None)])
|
|
489
|
+
self.agent.client.chat.completions = chat
|
|
490
|
+
self.assertEqual(self.agent.ask("hello"), "ok")
|
|
491
|
+
self.assertGreaterEqual(self.sleeps[0], 2.0)
|
|
492
|
+
|
|
493
|
+
def test_auth_error_fails_immediately(self):
|
|
494
|
+
request = httpx.Request("POST", "https://openrouter.ai/api/v1")
|
|
495
|
+
response = httpx.Response(401, request=request)
|
|
496
|
+
chat = FlakyChat([AuthenticationError("bad key", response=response,
|
|
497
|
+
body=None)])
|
|
498
|
+
self.agent.client.chat.completions = chat
|
|
499
|
+
with self.assertRaises(AuthenticationError):
|
|
500
|
+
self.agent.ask("hello")
|
|
501
|
+
self.assertEqual(chat.calls, 1) # no retry on a rejected key
|
|
502
|
+
self.assertEqual(self.sleeps, [])
|
|
503
|
+
|
|
504
|
+
def test_client_400_fails_immediately(self):
|
|
505
|
+
chat = FlakyChat([_api_status_error(400)])
|
|
506
|
+
self.agent.client.chat.completions = chat
|
|
507
|
+
with self.assertRaises(APIStatusError):
|
|
508
|
+
self.agent.ask("hello")
|
|
509
|
+
self.assertEqual(chat.calls, 1)
|
|
510
|
+
|
|
511
|
+
def test_server_500_is_retried(self):
|
|
512
|
+
chat = FlakyChat([_api_status_error(500), _api_status_error(502)])
|
|
513
|
+
self.agent.client.chat.completions = chat
|
|
514
|
+
self.assertEqual(self.agent.ask("hello"), "ok")
|
|
515
|
+
self.assertEqual(chat.calls, 3)
|
|
516
|
+
|
|
517
|
+
def test_retries_are_exhausted_loudly(self):
|
|
518
|
+
chat = FlakyChat([_api_connection_error() for _ in range(4)])
|
|
519
|
+
self.agent.client.chat.completions = chat
|
|
520
|
+
with self.assertRaises(APIConnectionError):
|
|
521
|
+
self.agent.ask("hello")
|
|
522
|
+
self.assertEqual(chat.calls, 4) # original + 3 retries, then raise
|
|
523
|
+
self.assertEqual(len(self.sleeps), 3) # 2s, 4s, 8s schedule
|
|
524
|
+
|
|
525
|
+
|
|
303
526
|
if __name__ == "__main__":
|
|
304
527
|
unittest.main()
|
|
@@ -329,11 +329,11 @@ class CliTests(unittest.TestCase):
|
|
|
329
329
|
|
|
330
330
|
def test_parse_args_defaults(self):
|
|
331
331
|
self.assertEqual(
|
|
332
|
-
main_module.parse_args([]), (None, False, None, False, None)
|
|
332
|
+
main_module.parse_args([]), (None, False, None, False, None, None)
|
|
333
333
|
)
|
|
334
334
|
self.assertEqual(
|
|
335
335
|
main_module.parse_args(["why is it down?"]),
|
|
336
|
-
("why is it down?", False, None, False, None),
|
|
336
|
+
("why is it down?", False, None, False, None, None),
|
|
337
337
|
)
|
|
338
338
|
|
|
339
339
|
def test_parse_args_all_flags(self):
|
|
@@ -341,18 +341,19 @@ class CliTests(unittest.TestCase):
|
|
|
341
341
|
["--json", "--resume", "--store-dir", "/tmp/s", "--out", "r.json",
|
|
342
342
|
"why", "is it down?"]
|
|
343
343
|
)
|
|
344
|
-
self.assertEqual(
|
|
344
|
+
self.assertEqual(
|
|
345
|
+
args, ("why is it down?", True, "/tmp/s", True, "r.json", None))
|
|
345
346
|
|
|
346
347
|
def test_parse_args_flags_only_runs_repl(self):
|
|
347
348
|
self.assertEqual(
|
|
348
349
|
main_module.parse_args(["--json", "--resume"]),
|
|
349
|
-
(None, True, None, True, None),
|
|
350
|
+
(None, True, None, True, None, None),
|
|
350
351
|
)
|
|
351
352
|
|
|
352
353
|
def test_parse_args_flag_without_value_is_ignored(self):
|
|
353
354
|
self.assertEqual(
|
|
354
355
|
main_module.parse_args(["--store-dir"]),
|
|
355
|
-
(None, False, None, False, None),
|
|
356
|
+
(None, False, None, False, None, None),
|
|
356
357
|
)
|
|
357
358
|
|
|
358
359
|
# --- --store-dir ----------------------------------------------------------
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|