devopsiq 0.1.1__tar.gz → 0.1.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. {devopsiq-0.1.1/devopsiq.egg-info → devopsiq-0.1.3}/PKG-INFO +61 -15
  2. {devopsiq-0.1.1 → devopsiq-0.1.3}/README.md +59 -14
  3. {devopsiq-0.1.1 → devopsiq-0.1.3}/agent/agent.py +79 -4
  4. devopsiq-0.1.3/agent/config.py +45 -0
  5. {devopsiq-0.1.1 → devopsiq-0.1.3/devopsiq.egg-info}/PKG-INFO +61 -15
  6. {devopsiq-0.1.1 → devopsiq-0.1.3}/devopsiq.egg-info/SOURCES.txt +1 -0
  7. {devopsiq-0.1.1 → devopsiq-0.1.3}/main.py +46 -14
  8. {devopsiq-0.1.1 → devopsiq-0.1.3}/pyproject.toml +2 -1
  9. {devopsiq-0.1.1 → devopsiq-0.1.3}/tests/test_automation.py +228 -5
  10. {devopsiq-0.1.1 → devopsiq-0.1.3}/tests/test_phase7.py +6 -5
  11. {devopsiq-0.1.1 → devopsiq-0.1.3}/LICENSE +0 -0
  12. {devopsiq-0.1.1 → devopsiq-0.1.3}/agent/__init__.py +0 -0
  13. {devopsiq-0.1.1 → devopsiq-0.1.3}/agent/investigation.py +0 -0
  14. {devopsiq-0.1.1 → devopsiq-0.1.3}/agent/prompts.py +0 -0
  15. {devopsiq-0.1.1 → devopsiq-0.1.3}/agent/store.py +0 -0
  16. {devopsiq-0.1.1 → devopsiq-0.1.3}/devopsiq.egg-info/dependency_links.txt +0 -0
  17. {devopsiq-0.1.1 → devopsiq-0.1.3}/devopsiq.egg-info/entry_points.txt +0 -0
  18. {devopsiq-0.1.1 → devopsiq-0.1.3}/devopsiq.egg-info/requires.txt +0 -0
  19. {devopsiq-0.1.1 → devopsiq-0.1.3}/devopsiq.egg-info/top_level.txt +0 -0
  20. {devopsiq-0.1.1 → devopsiq-0.1.3}/setup.cfg +0 -0
  21. {devopsiq-0.1.1 → devopsiq-0.1.3}/tests/test_phase2.py +0 -0
  22. {devopsiq-0.1.1 → devopsiq-0.1.3}/tests/test_phase3.py +0 -0
  23. {devopsiq-0.1.1 → devopsiq-0.1.3}/tests/test_phase4.py +0 -0
  24. {devopsiq-0.1.1 → devopsiq-0.1.3}/tests/test_phase5.py +0 -0
  25. {devopsiq-0.1.1 → devopsiq-0.1.3}/tests/test_phase8.py +0 -0
  26. {devopsiq-0.1.1 → devopsiq-0.1.3}/tests/test_phase9.py +0 -0
  27. {devopsiq-0.1.1 → devopsiq-0.1.3}/tools/__init__.py +0 -0
  28. {devopsiq-0.1.1 → devopsiq-0.1.3}/tools/ansible.py +0 -0
  29. {devopsiq-0.1.1 → devopsiq-0.1.3}/tools/argocd.py +0 -0
  30. {devopsiq-0.1.1 → devopsiq-0.1.3}/tools/base.py +0 -0
  31. {devopsiq-0.1.1 → devopsiq-0.1.3}/tools/cloud.py +0 -0
  32. {devopsiq-0.1.1 → devopsiq-0.1.3}/tools/docker.py +0 -0
  33. {devopsiq-0.1.1 → devopsiq-0.1.3}/tools/git_ci.py +0 -0
  34. {devopsiq-0.1.1 → devopsiq-0.1.3}/tools/helm.py +0 -0
  35. {devopsiq-0.1.1 → devopsiq-0.1.3}/tools/investigation.py +0 -0
  36. {devopsiq-0.1.1 → devopsiq-0.1.3}/tools/istio.py +0 -0
  37. {devopsiq-0.1.1 → devopsiq-0.1.3}/tools/kubernetes.py +0 -0
  38. {devopsiq-0.1.1 → devopsiq-0.1.3}/tools/monitoring.py +0 -0
  39. {devopsiq-0.1.1 → devopsiq-0.1.3}/tools/newrelic.py +0 -0
  40. {devopsiq-0.1.1 → devopsiq-0.1.3}/tools/preflight.py +0 -0
  41. {devopsiq-0.1.1 → devopsiq-0.1.3}/tools/registry.py +0 -0
  42. {devopsiq-0.1.1 → devopsiq-0.1.3}/tools/system.py +0 -0
  43. {devopsiq-0.1.1 → devopsiq-0.1.3}/tools/terraform.py +0 -0
  44. {devopsiq-0.1.1 → devopsiq-0.1.3}/tools/trivy.py +0 -0
@@ -1,12 +1,13 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: devopsiq
3
- Version: 0.1.1
3
+ Version: 0.1.3
4
4
  Summary: A from-scratch, read-only AI agent that investigates real DevOps problems — Kubernetes, Docker, Helm, Argo CD, Istio, Terraform, GitHub Actions, cloud, New Relic and more.
5
5
  Author: DevOpsAbhii
6
6
  License-Expression: MIT
7
7
  Project-URL: Homepage, https://github.com/DevOpsAbhii/devops-ai-agent
8
8
  Project-URL: Repository, https://github.com/DevOpsAbhii/devops-ai-agent
9
9
  Project-URL: Issues, https://github.com/DevOpsAbhii/devops-ai-agent/issues
10
+ Project-URL: Changelog, https://github.com/DevOpsAbhii/devops-ai-agent/blob/main/CHANGELOG.md
10
11
  Keywords: devops,kubernetes,docker,helm,argocd,istio,terraform,github-actions,observability,newrelic,incident-response,root-cause-analysis,ai-agent,cli
11
12
  Classifier: Development Status :: 4 - Beta
12
13
  Classifier: Environment :: Console
@@ -29,6 +30,13 @@ Dynamic: license-file
29
30
 
30
31
  # DevOps AI Agent
31
32
 
33
+ [![PyPI](https://img.shields.io/pypi/v/devopsiq)](https://pypi.org/project/devopsiq/)
34
+ [![Python](https://img.shields.io/pypi/pyversions/devopsiq)](https://pypi.org/project/devopsiq/)
35
+ [![Release](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml/badge.svg)](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
36
+ [![Docker](https://img.shields.io/badge/ghcr-devops--ai--agent-2496ED?logo=docker&logoColor=white)](https://github.com/DevOpsAbhii/devops-ai-agent/pkgs/container/devops-ai-agent)
37
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
38
+ [![Tests](https://img.shields.io/badge/tests-181%20offline-brightgreen)](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
39
+
32
40
  An AI agent that investigates real DevOps problems. The end goal: ask it
33
41
  something like *"Why is my Kubernetes pod in CrashLoopBackOff?"* and have it
34
42
  gather evidence, reason about the evidence, identify the likely root cause,
@@ -55,10 +63,21 @@ from environment config), and Ansible listing (inventory, playbook tasks).
55
63
  open alerts over NerdGraph — credentials from env), Trivy image
56
64
  vulnerability scanning, Helm releases (list/status/history), Argo CD
57
65
  (GitOps app sync/health), Istio mesh proxy status, and Docker Compose
58
- (project and service listing). The repository is git-tracked.
66
+ (project and service listing). **Phase 10 ships it as a product:** the
67
+ [`devopsiq` package on PyPI](https://pypi.org/project/devopsiq/) and a
68
+ prebuilt multi-arch Docker image (`ghcr.io/devopsabhii/devops-ai-agent`) —
69
+ both cut automatically by pushing a `v*` tag (test gate → PyPI via Trusted
70
+ Publishing + GHCR in parallel). **No API key? The agent still runs:** it
71
+ starts in model-less mode — record commands (`/report`, `/investigations`)
72
+ and the whole 58-tool layer work without a key; only questions to the model
73
+ need one. The repository is git-tracked.
59
74
  Every phase still built from scratch — no LangChain, LangGraph,
60
75
  AutoGen, CrewAI, or MCP.
61
76
 
77
+ > **New here?** The [Install & run guide (PDF)](DevOps_User_Guide.pdf) walks
78
+ > you from zero to investigating — three install routes, credentials
79
+ > (or none at all), model choice, and every command.
80
+
62
81
  > **Want to use this in your own project?** See [INTEGRATION.md](INTEGRATION.md)
63
82
  > for the three integration levels: drive it as a CLI from cron/CI (exit
64
83
  > codes + `--json`), embed it as a Python library, or extend it with your
@@ -73,7 +92,7 @@ model is called, how conversation history flows, how tool selection +
73
92
  execution + evidence feedback work — instead of depending on a framework
74
93
  for it.
75
94
 
76
- In Phase 5 the agent:
95
+ Today the agent:
77
96
  - holds a conversation with **GLM 5.3** through **OpenRouter**;
78
97
  - has **58 real, read-only tools** across fifteen domains: host facts,
79
98
  Kubernetes (12 tools), Linux system (4), Docker + Compose (10),
@@ -133,8 +152,8 @@ Module map:
133
152
 
134
153
  | Path | Responsibility |
135
154
  | --------------------------- | ------------------------------------------------------------ |
136
- | `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages |
137
- | `agent/agent.py` | `DevOpsAgent` — client, history, `ask()`, `_complete()` loop |
155
+ | `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages, model-less start |
156
+ | `agent/agent.py` | `DevOpsAgent` — client (`None` = model-less mode), history, `ask()`, `_complete()` loop |
138
157
  | `agent/prompts.py` | The system prompt (versioned/tested separately) |
139
158
  | `agent/investigation.py` | First-class investigation record: hypotheses, verdicts, evidence, report renderer (pure data) |
140
159
  | `agent/store.py` | `InvestigationStore` — one JSON file per record, atomic writes, resume/list (Phase 7) |
@@ -163,6 +182,11 @@ Module map:
163
182
  | `tests/test_phase7.py` | Offline suite: round-trip serialization, store files, auto-save, resume, CLI flags (Phase 7) |
164
183
  | `tests/test_phase8.py` | Offline suite: argv templates + validation for the 21 Phase 8 tools (fake CLIs, env-based monitoring) |
165
184
  | `tests/test_phase9.py` | Offline suite: New Relic env-credential + payload tests, trivy/helm/argocd/istio/compose argv templates (Phase 9) |
185
+ | `pyproject.toml` | Package `devopsiq`: metadata, MIT, console script `devopsiq = main:main` (Phase 10) |
186
+ | `Dockerfile` | Prebuilt image: slim base + kubectl/helm/trivy/gh + app, non-root, `/data` record store (Phase 10) |
187
+ | `.github/workflows/release.yml` | Tag-driven release: test gate → PyPI (Trusted Publishing) + GHCR multi-arch (Phase 10) |
188
+ | `docs/generate_pdf.py` | Builds the project documentation PDF from live source |
189
+ | `docs/generate_release_guide.py` | Builds the release & update playbook PDF |
166
190
 
167
191
  ### The tool-use loop
168
192
 
@@ -209,6 +233,7 @@ the CLI exposes it directly:
209
233
  | `/investigations` | List saved records on disk (newest first; `← active` marks the live one) |
210
234
  | `/report` | Show the canonical report (once concluded) |
211
235
  | `/endinvestigation` | Clear the record (memory only — the saved copy stays as history) |
236
+ | `/model [<name>]` | Show the active model, or save a default (config.json) and switch to it |
212
237
 
213
238
  The report the agent ends with and `/report` render are kept consistent by
214
239
  construction: `tool` results confirm each record call and the tracker, and
@@ -396,16 +421,22 @@ official `openai` Python SDK works as our client with two config lines:
396
421
  OpenAI(api_key=..., base_url="https://openrouter.ai/api/v1")
397
422
  ```
398
423
 
399
- Swapping to another OpenRouter model later is a one-line change (an env var
400
- today); moving to any other OpenAI-compatible provider changes only
401
- `agent/agent.py` configuration.
424
+ Swapping to another OpenRouter model later is a one-liner (env var, `/model`
425
+ command, or `--model` flag); moving to any other OpenAI-compatible provider
426
+ changes only `agent/agent.py` configuration.
402
427
 
403
428
  **Your model, your choice.** The default is baked in as a fallback, never a
404
- restriction — set `OPENROUTER_MODEL` (shell or `.env`) to any model on
405
- OpenRouter:
429
+ restriction. Four ways to set the model — the first one set wins:
430
+
431
+ 1. `--model NAME` flag (one run: `devopsiq --model openai/gpt-5.2 "why?"`)
432
+ 2. `/model <name>` in the REPL — saves it to `~/.devops-ai-agent/config.json`
433
+ so every future run uses it
434
+ 3. `OPENROUTER_MODEL` (shell or `.env`)
435
+ 4. built-in default (`z-ai/glm-5.3`)
406
436
 
407
437
  ```bash
408
438
  export OPENROUTER_MODEL=anthropic/claude-sonnet-5 # or openai/gpt-5.2, google/gemini-2.5-pro, ...
439
+ devopsiq /model openai/gpt-5.2 # or save a default from the REPL
409
440
  ```
410
441
 
411
442
  Two things to weigh when picking: the agent is a tool-use loop, so choose a
@@ -558,6 +589,7 @@ setup/API errors — so it drops straight into a pipeline:
558
589
  .venv/bin/python main.py --json "why is api-5d6f crash-looping?" # structured report
559
590
  .venv/bin/python main.py --resume --json "any update?" # continue a prior run
560
591
  .venv/bin/python main.py --store-dir /tmp/runs --out report.json "..." # pipeline paths
592
+ .venv/bin/python main.py --model openai/gpt-5.2 "why is it down?" # one-run model override
561
593
  ```
562
594
 
563
595
  With `--json` the stdout is one JSON document (see `render_report_json` in
@@ -599,8 +631,8 @@ Tools: ansible_inventory, ansible_playbook_tasks, aws_identity,
599
631
  k8s_top_nodes, k8s_top_pods, loki_query, prom_query,
600
632
  sys_open_ports, sys_service_logs, sys_service_status,
601
633
  sys_top_processes, system_info, tf_plan, tf_show, tf_state_list
602
- Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation
603
- One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] "<problem>"
634
+ Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation, /model [<name>]
635
+ One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] [--model NAME] "<problem>"
604
636
  Type 'exit' to quit.
605
637
 
606
638
  You: The checkout service container keeps exiting in Docker. Investigate.
@@ -617,7 +649,7 @@ You: exit
617
649
  Type `exit` / `quit`, or press Ctrl-D / Ctrl-C to leave. Slash commands are
618
650
  handled locally and never reach the model.
619
651
 
620
- ## 8. Current limitations (Phase 9)
652
+ ## 8. Current limitations
621
653
 
622
654
  - **Each domain needs its CLI installed and reachable.** Missing CLIs,
623
655
  unauthenticated `gh`, a dead docker daemon, an uninitialized terraform
@@ -702,8 +734,22 @@ handled locally and never reach the model.
702
734
  `helm_history` — reads only), Argo CD (`argocd_apps`,
703
735
  `argocd_app_status`), Istio (`istioctl_proxy_status`), Docker Compose
704
736
  (`docker_compose_ls`, `docker_compose_ps` — live-verified on this host).
737
+ - **Phase 10 — distribution. ✅ Done.** The `devopsiq` package on PyPI
738
+ (console command `devopsiq`, MIT) and the prebuilt multi-arch GHCR image,
739
+ cut by a tag-driven release workflow: an offline test gate, then PyPI
740
+ (Trusted Publishing — no token in the repo) and GHCR (amd64 + arm64) in
741
+ parallel. **`v0.1.1` added model-less mode:** with no API key the agent
742
+ still constructs — record commands (`/report`, `/investigations`) and the
743
+ 58-tool layer work; only model questions exit 1 with the setup message.
744
+ Verified end-to-end against the published package.
745
+ - **Model choice + resilient calls (v0.1.2). ✅ Done.** The model is
746
+ set by a 4-rung ladder — `--model` flag > `/model`-saved config file
747
+ (`~/.devops-ai-agent/config.json`) > `OPENROUTER_MODEL` > baked-in
748
+ default — and transient model-call failures (timeouts, connection
749
+ errors, 429s, 5xx) retry with exponential backoff (2s→4s→8s + jitter,
750
+ honoring a 429's Retry-After); a rejected key fails immediately.
705
751
  - **Later — region-scoped cloud resources** (ec2 describe-*, compute
706
752
  instances list, ...) behind the same template pattern; more observability
707
753
  depth (New Relic dashboards/entities, Prometheus range queries); streaming;
708
- conversation-history persistence; and a human-approval gate before any
709
- mutating action is ever allowed.
754
+ conversation-history persistence; and a human-approval gate
755
+ before any mutating action is ever allowed.
@@ -1,5 +1,12 @@
1
1
  # DevOps AI Agent
2
2
 
3
+ [![PyPI](https://img.shields.io/pypi/v/devopsiq)](https://pypi.org/project/devopsiq/)
4
+ [![Python](https://img.shields.io/pypi/pyversions/devopsiq)](https://pypi.org/project/devopsiq/)
5
+ [![Release](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml/badge.svg)](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
6
+ [![Docker](https://img.shields.io/badge/ghcr-devops--ai--agent-2496ED?logo=docker&logoColor=white)](https://github.com/DevOpsAbhii/devops-ai-agent/pkgs/container/devops-ai-agent)
7
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
8
+ [![Tests](https://img.shields.io/badge/tests-181%20offline-brightgreen)](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
9
+
3
10
  An AI agent that investigates real DevOps problems. The end goal: ask it
4
11
  something like *"Why is my Kubernetes pod in CrashLoopBackOff?"* and have it
5
12
  gather evidence, reason about the evidence, identify the likely root cause,
@@ -26,10 +33,21 @@ from environment config), and Ansible listing (inventory, playbook tasks).
26
33
  open alerts over NerdGraph — credentials from env), Trivy image
27
34
  vulnerability scanning, Helm releases (list/status/history), Argo CD
28
35
  (GitOps app sync/health), Istio mesh proxy status, and Docker Compose
29
- (project and service listing). The repository is git-tracked.
36
+ (project and service listing). **Phase 10 ships it as a product:** the
37
+ [`devopsiq` package on PyPI](https://pypi.org/project/devopsiq/) and a
38
+ prebuilt multi-arch Docker image (`ghcr.io/devopsabhii/devops-ai-agent`) —
39
+ both cut automatically by pushing a `v*` tag (test gate → PyPI via Trusted
40
+ Publishing + GHCR in parallel). **No API key? The agent still runs:** it
41
+ starts in model-less mode — record commands (`/report`, `/investigations`)
42
+ and the whole 58-tool layer work without a key; only questions to the model
43
+ need one. The repository is git-tracked.
30
44
  Every phase still built from scratch — no LangChain, LangGraph,
31
45
  AutoGen, CrewAI, or MCP.
32
46
 
47
+ > **New here?** The [Install & run guide (PDF)](DevOps_User_Guide.pdf) walks
48
+ > you from zero to investigating — three install routes, credentials
49
+ > (or none at all), model choice, and every command.
50
+
33
51
  > **Want to use this in your own project?** See [INTEGRATION.md](INTEGRATION.md)
34
52
  > for the three integration levels: drive it as a CLI from cron/CI (exit
35
53
  > codes + `--json`), embed it as a Python library, or extend it with your
@@ -44,7 +62,7 @@ model is called, how conversation history flows, how tool selection +
44
62
  execution + evidence feedback work — instead of depending on a framework
45
63
  for it.
46
64
 
47
- In Phase 5 the agent:
65
+ Today the agent:
48
66
  - holds a conversation with **GLM 5.3** through **OpenRouter**;
49
67
  - has **58 real, read-only tools** across fifteen domains: host facts,
50
68
  Kubernetes (12 tools), Linux system (4), Docker + Compose (10),
@@ -104,8 +122,8 @@ Module map:
104
122
 
105
123
  | Path | Responsibility |
106
124
  | --------------------------- | ------------------------------------------------------------ |
107
- | `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages |
108
- | `agent/agent.py` | `DevOpsAgent` — client, history, `ask()`, `_complete()` loop |
125
+ | `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages, model-less start |
126
+ | `agent/agent.py` | `DevOpsAgent` — client (`None` = model-less mode), history, `ask()`, `_complete()` loop |
109
127
  | `agent/prompts.py` | The system prompt (versioned/tested separately) |
110
128
  | `agent/investigation.py` | First-class investigation record: hypotheses, verdicts, evidence, report renderer (pure data) |
111
129
  | `agent/store.py` | `InvestigationStore` — one JSON file per record, atomic writes, resume/list (Phase 7) |
@@ -134,6 +152,11 @@ Module map:
134
152
  | `tests/test_phase7.py` | Offline suite: round-trip serialization, store files, auto-save, resume, CLI flags (Phase 7) |
135
153
  | `tests/test_phase8.py` | Offline suite: argv templates + validation for the 21 Phase 8 tools (fake CLIs, env-based monitoring) |
136
154
  | `tests/test_phase9.py` | Offline suite: New Relic env-credential + payload tests, trivy/helm/argocd/istio/compose argv templates (Phase 9) |
155
+ | `pyproject.toml` | Package `devopsiq`: metadata, MIT, console script `devopsiq = main:main` (Phase 10) |
156
+ | `Dockerfile` | Prebuilt image: slim base + kubectl/helm/trivy/gh + app, non-root, `/data` record store (Phase 10) |
157
+ | `.github/workflows/release.yml` | Tag-driven release: test gate → PyPI (Trusted Publishing) + GHCR multi-arch (Phase 10) |
158
+ | `docs/generate_pdf.py` | Builds the project documentation PDF from live source |
159
+ | `docs/generate_release_guide.py` | Builds the release & update playbook PDF |
137
160
 
138
161
  ### The tool-use loop
139
162
 
@@ -180,6 +203,7 @@ the CLI exposes it directly:
180
203
  | `/investigations` | List saved records on disk (newest first; `← active` marks the live one) |
181
204
  | `/report` | Show the canonical report (once concluded) |
182
205
  | `/endinvestigation` | Clear the record (memory only — the saved copy stays as history) |
206
+ | `/model [<name>]` | Show the active model, or save a default (config.json) and switch to it |
183
207
 
184
208
  The report the agent ends with and `/report` render are kept consistent by
185
209
  construction: `tool` results confirm each record call and the tracker, and
@@ -367,16 +391,22 @@ official `openai` Python SDK works as our client with two config lines:
367
391
  OpenAI(api_key=..., base_url="https://openrouter.ai/api/v1")
368
392
  ```
369
393
 
370
- Swapping to another OpenRouter model later is a one-line change (an env var
371
- today); moving to any other OpenAI-compatible provider changes only
372
- `agent/agent.py` configuration.
394
+ Swapping to another OpenRouter model later is a one-liner (env var, `/model`
395
+ command, or `--model` flag); moving to any other OpenAI-compatible provider
396
+ changes only `agent/agent.py` configuration.
373
397
 
374
398
  **Your model, your choice.** The default is baked in as a fallback, never a
375
- restriction — set `OPENROUTER_MODEL` (shell or `.env`) to any model on
376
- OpenRouter:
399
+ restriction. Four ways to set the model — the first one set wins:
400
+
401
+ 1. `--model NAME` flag (one run: `devopsiq --model openai/gpt-5.2 "why?"`)
402
+ 2. `/model <name>` in the REPL — saves it to `~/.devops-ai-agent/config.json`
403
+ so every future run uses it
404
+ 3. `OPENROUTER_MODEL` (shell or `.env`)
405
+ 4. built-in default (`z-ai/glm-5.3`)
377
406
 
378
407
  ```bash
379
408
  export OPENROUTER_MODEL=anthropic/claude-sonnet-5 # or openai/gpt-5.2, google/gemini-2.5-pro, ...
409
+ devopsiq /model openai/gpt-5.2 # or save a default from the REPL
380
410
  ```
381
411
 
382
412
  Two things to weigh when picking: the agent is a tool-use loop, so choose a
@@ -529,6 +559,7 @@ setup/API errors — so it drops straight into a pipeline:
529
559
  .venv/bin/python main.py --json "why is api-5d6f crash-looping?" # structured report
530
560
  .venv/bin/python main.py --resume --json "any update?" # continue a prior run
531
561
  .venv/bin/python main.py --store-dir /tmp/runs --out report.json "..." # pipeline paths
562
+ .venv/bin/python main.py --model openai/gpt-5.2 "why is it down?" # one-run model override
532
563
  ```
533
564
 
534
565
  With `--json` the stdout is one JSON document (see `render_report_json` in
@@ -570,8 +601,8 @@ Tools: ansible_inventory, ansible_playbook_tasks, aws_identity,
570
601
  k8s_top_nodes, k8s_top_pods, loki_query, prom_query,
571
602
  sys_open_ports, sys_service_logs, sys_service_status,
572
603
  sys_top_processes, system_info, tf_plan, tf_show, tf_state_list
573
- Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation
574
- One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] "<problem>"
604
+ Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation, /model [<name>]
605
+ One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] [--model NAME] "<problem>"
575
606
  Type 'exit' to quit.
576
607
 
577
608
  You: The checkout service container keeps exiting in Docker. Investigate.
@@ -588,7 +619,7 @@ You: exit
588
619
  Type `exit` / `quit`, or press Ctrl-D / Ctrl-C to leave. Slash commands are
589
620
  handled locally and never reach the model.
590
621
 
591
- ## 8. Current limitations (Phase 9)
622
+ ## 8. Current limitations
592
623
 
593
624
  - **Each domain needs its CLI installed and reachable.** Missing CLIs,
594
625
  unauthenticated `gh`, a dead docker daemon, an uninitialized terraform
@@ -673,8 +704,22 @@ handled locally and never reach the model.
673
704
  `helm_history` — reads only), Argo CD (`argocd_apps`,
674
705
  `argocd_app_status`), Istio (`istioctl_proxy_status`), Docker Compose
675
706
  (`docker_compose_ls`, `docker_compose_ps` — live-verified on this host).
707
+ - **Phase 10 — distribution. ✅ Done.** The `devopsiq` package on PyPI
708
+ (console command `devopsiq`, MIT) and the prebuilt multi-arch GHCR image,
709
+ cut by a tag-driven release workflow: an offline test gate, then PyPI
710
+ (Trusted Publishing — no token in the repo) and GHCR (amd64 + arm64) in
711
+ parallel. **`v0.1.1` added model-less mode:** with no API key the agent
712
+ still constructs — record commands (`/report`, `/investigations`) and the
713
+ 58-tool layer work; only model questions exit 1 with the setup message.
714
+ Verified end-to-end against the published package.
715
+ - **Model choice + resilient calls (v0.1.2). ✅ Done.** The model is
716
+ set by a 4-rung ladder — `--model` flag > `/model`-saved config file
717
+ (`~/.devops-ai-agent/config.json`) > `OPENROUTER_MODEL` > baked-in
718
+ default — and transient model-call failures (timeouts, connection
719
+ errors, 429s, 5xx) retry with exponential backoff (2s→4s→8s + jitter,
720
+ honoring a 429's Retry-After); a rejected key fails immediately.
676
721
  - **Later — region-scoped cloud resources** (ec2 describe-*, compute
677
722
  instances list, ...) behind the same template pattern; more observability
678
723
  depth (New Relic dashboards/entities, Prometheus range queries); streaming;
679
- conversation-history persistence; and a human-approval gate before any
680
- mutating action is ever allowed.
724
+ conversation-history persistence; and a human-approval gate
725
+ before any mutating action is ever allowed.
@@ -26,9 +26,18 @@ exits; the delegates below expose resume/list/store-location to the CLI.
26
26
  """
27
27
 
28
28
  import os
29
+ import random
30
+ import time
31
+
32
+ from openai import (
33
+ APIConnectionError,
34
+ APIStatusError,
35
+ APITimeoutError,
36
+ OpenAI,
37
+ RateLimitError,
38
+ )
29
39
 
30
- from openai import OpenAI
31
-
40
+ from agent.config import load_user_config
32
41
  from agent.prompts import SYSTEM_PROMPT
33
42
  from agent.store import InvestigationStore
34
43
  from tools import ( # noqa: F401 — side effect: each module registers its tools
@@ -68,6 +77,50 @@ PLACEHOLDER_KEY = "your_key_here"
68
77
  # Safety valve: the model gets at most this many tool-use turns before we stop.
69
78
  MAX_TOOL_ITERATIONS = 10
70
79
 
80
+ # Transient-failure retries for the model call itself (network blips, 429,
81
+ # 5xx). MAX_MODEL_RETRIES attempts beyond the first, exponential backoff
82
+ # starting at RETRY_BASE_DELAY (2s -> 4s -> 8s) plus jitter, capped at
83
+ # RETRY_MAX_DELAY. A 429's Retry-After header wins when present.
84
+ MAX_MODEL_RETRIES = 3
85
+ RETRY_BASE_DELAY = 2.0
86
+ RETRY_MAX_DELAY = 60.0
87
+
88
+ _sleep = time.sleep # test seam: offline tests patch agent.agent._sleep
89
+
90
+
91
+ def _is_transient(exc: Exception) -> bool:
92
+ """True when retrying `exc` can plausibly succeed.
93
+
94
+ Timeouts, connection failures, rate limits, and server-side 5xx are
95
+ transient. A rejected key (401) or any other client 4xx is not — the
96
+ same request would fail identically forever, so it fails immediately.
97
+ """
98
+ if isinstance(exc, (APITimeoutError, APIConnectionError, RateLimitError)):
99
+ return True
100
+ if isinstance(exc, APIStatusError):
101
+ return exc.status_code >= 500
102
+ return False
103
+
104
+
105
+ def _retry_delay(exc: Exception, attempt: int) -> float:
106
+ """Seconds to wait before retry number `attempt` (0-based).
107
+
108
+ Exponential backoff with jitter, except on a rate limit carrying a
109
+ Retry-After header — the server knows its own budget, so it wins
110
+ (clamped to [1, RETRY_MAX_DELAY] so a bad header cannot hurt us).
111
+ """
112
+ response = getattr(exc, "response", None)
113
+ retry_after = getattr(response, "headers", {}).get("retry-after")
114
+ if retry_after:
115
+ try:
116
+ return min(max(float(retry_after), 1.0), RETRY_MAX_DELAY)
117
+ except (TypeError, ValueError):
118
+ pass # non-numeric header: fall through to backoff
119
+ return min(
120
+ RETRY_BASE_DELAY * (2 ** attempt) + random.uniform(0, 1),
121
+ RETRY_MAX_DELAY,
122
+ )
123
+
71
124
 
72
125
  class DevOpsAgent:
73
126
  """Minimal DevOps investigation assistant (read-only by design)."""
@@ -81,7 +134,12 @@ class DevOpsAgent:
81
134
  # Configuration resolution order: explicit argument > environment > default.
82
135
  self.api_key = api_key or os.getenv("OPENROUTER_API_KEY")
83
136
 
84
- self.model = model or os.getenv("OPENROUTER_MODEL", DEFAULT_MODEL)
137
+ self.model = (
138
+ model # 1. explicit --model flag
139
+ or load_user_config().get("model") # 2. ~/.devops-ai-agent/config.json
140
+ or os.getenv("OPENROUTER_MODEL") # 3. environment / .env
141
+ or DEFAULT_MODEL # 4. built-in default
142
+ )
85
143
  self.base_url = base_url or os.getenv("OPENROUTER_BASE_URL", DEFAULT_BASE_URL)
86
144
 
87
145
  # Model-less mode: a missing (or placeholder) key no longer blocks
@@ -197,7 +255,7 @@ class DevOpsAgent:
197
255
  if self.tools:
198
256
  request["tools"] = [tool.schema() for tool in self.tools]
199
257
 
200
- response = self.client.chat.completions.create(**request)
258
+ response = self._call_model(request)
201
259
  message = response.choices[0].message
202
260
 
203
261
  if not message.tool_calls:
@@ -219,6 +277,23 @@ class DevOpsAgent:
219
277
  f"The model did not finish after {MAX_TOOL_ITERATIONS} tool-use turns."
220
278
  )
221
279
 
280
+ def _call_model(self, request: dict):
281
+ """One model call with bounded retries on transient failures.
282
+
283
+ Timeouts, connection errors, rate limits (429) and server-side 5xx
284
+ are retried up to MAX_MODEL_RETRIES times with exponential backoff
285
+ (2s, 4s, 8s + jitter); a 429's Retry-After header wins when present.
286
+ Client errors — a rejected key (401) most notably — fail immediately,
287
+ because retrying the identical request cannot fix them.
288
+ """
289
+ for attempt in range(MAX_MODEL_RETRIES + 1):
290
+ try:
291
+ return self.client.chat.completions.create(**request)
292
+ except Exception as exc: # noqa: BLE001 — classified right below
293
+ if attempt >= MAX_MODEL_RETRIES or not _is_transient(exc):
294
+ raise
295
+ _sleep(_retry_delay(exc, attempt))
296
+
222
297
 
223
298
  def _echo_tool_request(message) -> dict:
224
299
  """Rebuild the assistant turn that requested tools, verbatim.
@@ -0,0 +1,45 @@
1
+ """Per-user preferences: ~/.devops-ai-agent/config.json.
2
+
3
+ The `devopsiq` equivalent of "my settings" — currently the preferred model,
4
+ written by the /model command and read by the agent at startup. Precedence
5
+ for the model, highest first:
6
+
7
+ --model flag > config.json > OPENROUTER_MODEL env > built-in default
8
+
9
+ The file is tiny JSON, one flat object. Every read is tolerant: a missing
10
+ file, an unreadable home directory, or corrupt content yields {} — a broken
11
+ config never blocks the agent (the same degrade-don't-crash rule the
12
+ investigation store follows). Tests point AGENT_CONFIG_FILE at a tmp path.
13
+ """
14
+
15
+ import json
16
+ import os
17
+ from pathlib import Path
18
+
19
+ DEFAULT_CONFIG_PATH = Path.home() / ".devops-ai-agent" / "config.json"
20
+
21
+
22
+ def config_path() -> Path:
23
+ """The active config file (AGENT_CONFIG_FILE overrides; tests use tmp)."""
24
+ return Path(os.getenv("AGENT_CONFIG_FILE") or DEFAULT_CONFIG_PATH)
25
+
26
+
27
+ def load_user_config() -> dict:
28
+ """Read the config file; {} when missing or unreadable — never raise."""
29
+ try:
30
+ raw = config_path().read_text(encoding="utf-8")
31
+ data = json.loads(raw)
32
+ return data if isinstance(data, dict) else {}
33
+ except (OSError, ValueError):
34
+ return {}
35
+
36
+
37
+ def save_user_config(update: dict) -> Path:
38
+ """Merge `update` into the config file (create the directory if needed)."""
39
+ path = config_path()
40
+ data = load_user_config()
41
+ data.update(update)
42
+ path.parent.mkdir(parents=True, exist_ok=True)
43
+ path.write_text(json.dumps(data, indent=2, sort_keys=True) + "\n",
44
+ encoding="utf-8")
45
+ return path
@@ -1,12 +1,13 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: devopsiq
3
- Version: 0.1.1
3
+ Version: 0.1.3
4
4
  Summary: A from-scratch, read-only AI agent that investigates real DevOps problems — Kubernetes, Docker, Helm, Argo CD, Istio, Terraform, GitHub Actions, cloud, New Relic and more.
5
5
  Author: DevOpsAbhii
6
6
  License-Expression: MIT
7
7
  Project-URL: Homepage, https://github.com/DevOpsAbhii/devops-ai-agent
8
8
  Project-URL: Repository, https://github.com/DevOpsAbhii/devops-ai-agent
9
9
  Project-URL: Issues, https://github.com/DevOpsAbhii/devops-ai-agent/issues
10
+ Project-URL: Changelog, https://github.com/DevOpsAbhii/devops-ai-agent/blob/main/CHANGELOG.md
10
11
  Keywords: devops,kubernetes,docker,helm,argocd,istio,terraform,github-actions,observability,newrelic,incident-response,root-cause-analysis,ai-agent,cli
11
12
  Classifier: Development Status :: 4 - Beta
12
13
  Classifier: Environment :: Console
@@ -29,6 +30,13 @@ Dynamic: license-file
29
30
 
30
31
  # DevOps AI Agent
31
32
 
33
+ [![PyPI](https://img.shields.io/pypi/v/devopsiq)](https://pypi.org/project/devopsiq/)
34
+ [![Python](https://img.shields.io/pypi/pyversions/devopsiq)](https://pypi.org/project/devopsiq/)
35
+ [![Release](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml/badge.svg)](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
36
+ [![Docker](https://img.shields.io/badge/ghcr-devops--ai--agent-2496ED?logo=docker&logoColor=white)](https://github.com/DevOpsAbhii/devops-ai-agent/pkgs/container/devops-ai-agent)
37
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
38
+ [![Tests](https://img.shields.io/badge/tests-181%20offline-brightgreen)](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
39
+
32
40
  An AI agent that investigates real DevOps problems. The end goal: ask it
33
41
  something like *"Why is my Kubernetes pod in CrashLoopBackOff?"* and have it
34
42
  gather evidence, reason about the evidence, identify the likely root cause,
@@ -55,10 +63,21 @@ from environment config), and Ansible listing (inventory, playbook tasks).
55
63
  open alerts over NerdGraph — credentials from env), Trivy image
56
64
  vulnerability scanning, Helm releases (list/status/history), Argo CD
57
65
  (GitOps app sync/health), Istio mesh proxy status, and Docker Compose
58
- (project and service listing). The repository is git-tracked.
66
+ (project and service listing). **Phase 10 ships it as a product:** the
67
+ [`devopsiq` package on PyPI](https://pypi.org/project/devopsiq/) and a
68
+ prebuilt multi-arch Docker image (`ghcr.io/devopsabhii/devops-ai-agent`) —
69
+ both cut automatically by pushing a `v*` tag (test gate → PyPI via Trusted
70
+ Publishing + GHCR in parallel). **No API key? The agent still runs:** it
71
+ starts in model-less mode — record commands (`/report`, `/investigations`)
72
+ and the whole 58-tool layer work without a key; only questions to the model
73
+ need one. The repository is git-tracked.
59
74
  Every phase still built from scratch — no LangChain, LangGraph,
60
75
  AutoGen, CrewAI, or MCP.
61
76
 
77
+ > **New here?** The [Install & run guide (PDF)](DevOps_User_Guide.pdf) walks
78
+ > you from zero to investigating — three install routes, credentials
79
+ > (or none at all), model choice, and every command.
80
+
62
81
  > **Want to use this in your own project?** See [INTEGRATION.md](INTEGRATION.md)
63
82
  > for the three integration levels: drive it as a CLI from cron/CI (exit
64
83
  > codes + `--json`), embed it as a Python library, or extend it with your
@@ -73,7 +92,7 @@ model is called, how conversation history flows, how tool selection +
73
92
  execution + evidence feedback work — instead of depending on a framework
74
93
  for it.
75
94
 
76
- In Phase 5 the agent:
95
+ Today the agent:
77
96
  - holds a conversation with **GLM 5.3** through **OpenRouter**;
78
97
  - has **58 real, read-only tools** across fifteen domains: host facts,
79
98
  Kubernetes (12 tools), Linux system (4), Docker + Compose (10),
@@ -133,8 +152,8 @@ Module map:
133
152
 
134
153
  | Path | Responsibility |
135
154
  | --------------------------- | ------------------------------------------------------------ |
136
- | `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages |
137
- | `agent/agent.py` | `DevOpsAgent` — client, history, `ask()`, `_complete()` loop |
155
+ | `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages, model-less start |
156
+ | `agent/agent.py` | `DevOpsAgent` — client (`None` = model-less mode), history, `ask()`, `_complete()` loop |
138
157
  | `agent/prompts.py` | The system prompt (versioned/tested separately) |
139
158
  | `agent/investigation.py` | First-class investigation record: hypotheses, verdicts, evidence, report renderer (pure data) |
140
159
  | `agent/store.py` | `InvestigationStore` — one JSON file per record, atomic writes, resume/list (Phase 7) |
@@ -163,6 +182,11 @@ Module map:
163
182
  | `tests/test_phase7.py` | Offline suite: round-trip serialization, store files, auto-save, resume, CLI flags (Phase 7) |
164
183
  | `tests/test_phase8.py` | Offline suite: argv templates + validation for the 21 Phase 8 tools (fake CLIs, env-based monitoring) |
165
184
  | `tests/test_phase9.py` | Offline suite: New Relic env-credential + payload tests, trivy/helm/argocd/istio/compose argv templates (Phase 9) |
185
+ | `pyproject.toml` | Package `devopsiq`: metadata, MIT, console script `devopsiq = main:main` (Phase 10) |
186
+ | `Dockerfile` | Prebuilt image: slim base + kubectl/helm/trivy/gh + app, non-root, `/data` record store (Phase 10) |
187
+ | `.github/workflows/release.yml` | Tag-driven release: test gate → PyPI (Trusted Publishing) + GHCR multi-arch (Phase 10) |
188
+ | `docs/generate_pdf.py` | Builds the project documentation PDF from live source |
189
+ | `docs/generate_release_guide.py` | Builds the release & update playbook PDF |
166
190
 
167
191
  ### The tool-use loop
168
192
 
@@ -209,6 +233,7 @@ the CLI exposes it directly:
209
233
  | `/investigations` | List saved records on disk (newest first; `← active` marks the live one) |
210
234
  | `/report` | Show the canonical report (once concluded) |
211
235
  | `/endinvestigation` | Clear the record (memory only — the saved copy stays as history) |
236
+ | `/model [<name>]` | Show the active model, or save a default (config.json) and switch to it |
212
237
 
213
238
  The report the agent ends with and `/report` render are kept consistent by
214
239
  construction: `tool` results confirm each record call and the tracker, and
@@ -396,16 +421,22 @@ official `openai` Python SDK works as our client with two config lines:
396
421
  OpenAI(api_key=..., base_url="https://openrouter.ai/api/v1")
397
422
  ```
398
423
 
399
- Swapping to another OpenRouter model later is a one-line change (an env var
400
- today); moving to any other OpenAI-compatible provider changes only
401
- `agent/agent.py` configuration.
424
+ Swapping to another OpenRouter model later is a one-liner (env var, `/model`
425
+ command, or `--model` flag); moving to any other OpenAI-compatible provider
426
+ changes only `agent/agent.py` configuration.
402
427
 
403
428
  **Your model, your choice.** The default is baked in as a fallback, never a
404
- restriction — set `OPENROUTER_MODEL` (shell or `.env`) to any model on
405
- OpenRouter:
429
+ restriction. Four ways to set the model — the first one set wins:
430
+
431
+ 1. `--model NAME` flag (one run: `devopsiq --model openai/gpt-5.2 "why?"`)
432
+ 2. `/model <name>` in the REPL — saves it to `~/.devops-ai-agent/config.json`
433
+ so every future run uses it
434
+ 3. `OPENROUTER_MODEL` (shell or `.env`)
435
+ 4. built-in default (`z-ai/glm-5.3`)
406
436
 
407
437
  ```bash
408
438
  export OPENROUTER_MODEL=anthropic/claude-sonnet-5 # or openai/gpt-5.2, google/gemini-2.5-pro, ...
439
+ devopsiq /model openai/gpt-5.2 # or save a default from the REPL
409
440
  ```
410
441
 
411
442
  Two things to weigh when picking: the agent is a tool-use loop, so choose a
@@ -558,6 +589,7 @@ setup/API errors — so it drops straight into a pipeline:
558
589
  .venv/bin/python main.py --json "why is api-5d6f crash-looping?" # structured report
559
590
  .venv/bin/python main.py --resume --json "any update?" # continue a prior run
560
591
  .venv/bin/python main.py --store-dir /tmp/runs --out report.json "..." # pipeline paths
592
+ .venv/bin/python main.py --model openai/gpt-5.2 "why is it down?" # one-run model override
561
593
  ```
562
594
 
563
595
  With `--json` the stdout is one JSON document (see `render_report_json` in
@@ -599,8 +631,8 @@ Tools: ansible_inventory, ansible_playbook_tasks, aws_identity,
599
631
  k8s_top_nodes, k8s_top_pods, loki_query, prom_query,
600
632
  sys_open_ports, sys_service_logs, sys_service_status,
601
633
  sys_top_processes, system_info, tf_plan, tf_show, tf_state_list
602
- Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation
603
- One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] "<problem>"
634
+ Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation, /model [<name>]
635
+ One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] [--model NAME] "<problem>"
604
636
  Type 'exit' to quit.
605
637
 
606
638
  You: The checkout service container keeps exiting in Docker. Investigate.
@@ -617,7 +649,7 @@ You: exit
617
649
  Type `exit` / `quit`, or press Ctrl-D / Ctrl-C to leave. Slash commands are
618
650
  handled locally and never reach the model.
619
651
 
620
- ## 8. Current limitations (Phase 9)
652
+ ## 8. Current limitations
621
653
 
622
654
  - **Each domain needs its CLI installed and reachable.** Missing CLIs,
623
655
  unauthenticated `gh`, a dead docker daemon, an uninitialized terraform
@@ -702,8 +734,22 @@ handled locally and never reach the model.
702
734
  `helm_history` — reads only), Argo CD (`argocd_apps`,
703
735
  `argocd_app_status`), Istio (`istioctl_proxy_status`), Docker Compose
704
736
  (`docker_compose_ls`, `docker_compose_ps` — live-verified on this host).
737
+ - **Phase 10 — distribution. ✅ Done.** The `devopsiq` package on PyPI
738
+ (console command `devopsiq`, MIT) and the prebuilt multi-arch GHCR image,
739
+ cut by a tag-driven release workflow: an offline test gate, then PyPI
740
+ (Trusted Publishing — no token in the repo) and GHCR (amd64 + arm64) in
741
+ parallel. **`v0.1.1` added model-less mode:** with no API key the agent
742
+ still constructs — record commands (`/report`, `/investigations`) and the
743
+ 58-tool layer work; only model questions exit 1 with the setup message.
744
+ Verified end-to-end against the published package.
745
+ - **Model choice + resilient calls (v0.1.2). ✅ Done.** The model is
746
+ set by a 4-rung ladder — `--model` flag > `/model`-saved config file
747
+ (`~/.devops-ai-agent/config.json`) > `OPENROUTER_MODEL` > baked-in
748
+ default — and transient model-call failures (timeouts, connection
749
+ errors, 429s, 5xx) retry with exponential backoff (2s→4s→8s + jitter,
750
+ honoring a 429's Retry-After); a rejected key fails immediately.
705
751
  - **Later — region-scoped cloud resources** (ec2 describe-*, compute
706
752
  instances list, ...) behind the same template pattern; more observability
707
753
  depth (New Relic dashboards/entities, Prometheus range queries); streaming;
708
- conversation-history persistence; and a human-approval gate before any
709
- mutating action is ever allowed.
754
+ conversation-history persistence; and a human-approval gate
755
+ before any mutating action is ever allowed.
@@ -4,6 +4,7 @@ main.py
4
4
  pyproject.toml
5
5
  agent/__init__.py
6
6
  agent/agent.py
7
+ agent/config.py
7
8
  agent/investigation.py
8
9
  agent/prompts.py
9
10
  agent/store.py
@@ -10,6 +10,7 @@ One-shot (for cron / CI / scripts):
10
10
  python main.py --json "why is api-5d6f crash-looping?" # structured report
11
11
  python main.py --resume --json "any update?" # continue prior run
12
12
  python main.py --store-dir DIR --out report.json "..." # pipeline paths
13
+ python main.py --model openai/gpt-4o-mini "..." # one-run model override
13
14
 
14
15
  One-shot mode sends the problem once, prints the agent's report, and exits
15
16
  with a status code (0 = completed, 1 = setup/API error). With --json the
@@ -41,6 +42,7 @@ from openai import (
41
42
  )
42
43
 
43
44
  from agent.agent import DevOpsAgent
45
+ from agent.config import config_path, save_user_config
44
46
 
45
47
  EXIT_WORDS = {"exit", "quit"}
46
48
 
@@ -55,6 +57,7 @@ COMMAND_ALIASES = {
55
57
  "/report": "/report",
56
58
  "/endinvestigation": "/endinvestigation",
57
59
  "/end": "/endinvestigation",
60
+ "/model": "/model",
58
61
  }
59
62
 
60
63
 
@@ -82,8 +85,32 @@ def handle_command(text: str, agent: DevOpsAgent) -> str | None:
82
85
  return agent.investigation_report_text() or "(no investigation recorded)"
83
86
  if canonical == "/endinvestigation":
84
87
  return agent.end_investigation()
88
+ if canonical == "/model":
89
+ return handle_model_command(rest, agent)
85
90
  return (f"unknown command: {cmd}. Try /investigate <problem>, "
86
- "/investigation, /investigations, /report, /endinvestigation")
91
+ "/investigation, /investigations, /report, /endinvestigation, "
92
+ "/model")
93
+
94
+
95
+ def handle_model_command(rest: str, agent: DevOpsAgent) -> str:
96
+ """/model (show the active model and where it came from) or /model <name>.
97
+
98
+ With a name: save it as the user's preference in the config file
99
+ (~/.devops-ai-agent/config.json) and switch this session to it. The
100
+ next run picks it up automatically — the flag and env still outrank it.
101
+ """
102
+ name = rest.strip()
103
+ if not name:
104
+ return (f"Model: {agent.model}\n"
105
+ f"Config file: {config_path()} "
106
+ "(save a default with /model <name>)")
107
+ if len(name.split()) != 1:
108
+ return "Model name must be a single token, e.g. /model openai/gpt-4o-mini"
109
+ path = save_user_config({"model": name})
110
+ agent.model = name
111
+ return (f"Model set to {name} (saved in {path}).\n"
112
+ "This session now uses it; the --model flag and OPENROUTER_MODEL "
113
+ "still override it per run.")
87
114
 
88
115
 
89
116
  class OneShotArgs(NamedTuple):
@@ -94,9 +121,10 @@ class OneShotArgs(NamedTuple):
94
121
  store_dir: str | None # --store-dir PATH (persistence override)
95
122
  resume: bool # --resume (continue the newest in-progress record)
96
123
  out: str | None # --out PATH (also write the JSON report there)
124
+ model: str | None # --model NAME (one-run model override)
97
125
 
98
126
 
99
- _VALUE_FLAGS = ("--store-dir", "--out")
127
+ _VALUE_FLAGS = ("--store-dir", "--out", "--model")
100
128
 
101
129
 
102
130
  def _flag_value(argv: list[str], flag: str) -> str | None:
@@ -114,14 +142,16 @@ def parse_args(argv: list[str]) -> OneShotArgs:
114
142
  --json switches the one-shot output from markdown to the structured JSON
115
143
  report; --resume continues the newest in-progress record from the store;
116
144
  --store-dir PATH overrides the persistence directory for this run; --out
117
- PATH additionally writes the JSON report to an exact path. Example:
145
+ PATH additionally writes the JSON report to an exact path; --model NAME
146
+ overrides the model for this run (highest model precedence). Example:
118
147
  `python main.py --json --out r.json "why is it down?"` ->
119
- ("why is it down?", True, None, False, "r.json").
148
+ ("why is it down?", True, None, False, "r.json", None).
120
149
  """
121
150
  as_json = "--json" in argv
122
151
  do_resume = "--resume" in argv
123
152
  store_dir = _flag_value(argv, "--store-dir")
124
153
  out = _flag_value(argv, "--out")
154
+ model = _flag_value(argv, "--model")
125
155
  positionals: list[str] = []
126
156
  skip_next = False
127
157
  for arg in argv:
@@ -136,8 +166,8 @@ def parse_args(argv: list[str]) -> OneShotArgs:
136
166
  positionals.append(arg)
137
167
  text = " ".join(positionals).strip()
138
168
  if not positionals or not text:
139
- return OneShotArgs(None, as_json, store_dir, do_resume, out)
140
- return OneShotArgs(text, as_json, store_dir, do_resume, out)
169
+ return OneShotArgs(None, as_json, store_dir, do_resume, out, model)
170
+ return OneShotArgs(text, as_json, store_dir, do_resume, out, model)
141
171
 
142
172
 
143
173
  def run_one_shot(
@@ -147,6 +177,7 @@ def run_one_shot(
147
177
  store_dir: str | None = None,
148
178
  resume: bool = False,
149
179
  out: str | None = None,
180
+ model: str | None = None,
150
181
  ) -> int:
151
182
  """Non-interactive single run: send `task`, print the result, exit cleanly.
152
183
 
@@ -155,14 +186,14 @@ def run_one_shot(
155
186
  opened an investigation this is a hard failure (exit 1) — a caller asked
156
187
  for a report and there is none to give. The record is auto-saved on every
157
188
  mutation regardless; store_dir points persistence somewhere else for this
158
- run, resume continues the newest in-progress record before asking, and
159
- out additionally writes the JSON report to an exact path. `agent` lets
160
- embedders reuse a configured agent (also the test seam); default builds a
161
- fresh one from the environment.
189
+ run, resume continues the newest in-progress record before asking, out
190
+ additionally writes the JSON report to an exact path, and model overrides
191
+ the model for this run. `agent` lets embedders reuse a configured agent
192
+ (also the test seam); default builds a fresh one from the environment.
162
193
  """
163
194
  if agent is None:
164
195
  try:
165
- agent = DevOpsAgent()
196
+ agent = DevOpsAgent(model=model)
166
197
  except ValueError as exc:
167
198
  print(f"[setup] {exc}", file=sys.stderr)
168
199
  return 1
@@ -245,10 +276,11 @@ def main() -> int:
245
276
  store_dir=args.store_dir,
246
277
  resume=args.resume,
247
278
  out=args.out,
279
+ model=args.model,
248
280
  )
249
281
 
250
282
  try:
251
- agent = DevOpsAgent()
283
+ agent = DevOpsAgent(model=args.model)
252
284
  except ValueError as exc:
253
285
  print(f"[setup] {exc}", file=sys.stderr)
254
286
  return 1
@@ -268,9 +300,9 @@ def main() -> int:
268
300
  print(f"Store: {agent.store_dir or '(persistence off)'}")
269
301
  print(f"Tools: {tool_names}")
270
302
  print("Commands: /investigate <problem>, /investigation, /investigations, "
271
- "/report, /endinvestigation")
303
+ "/report, /endinvestigation, /model [<name>]")
272
304
  print("One-shot: python main.py [--json] [--resume] [--out report.json] "
273
- "[--store-dir DIR] \"<problem>\"")
305
+ "[--store-dir DIR] [--model NAME] \"<problem>\"")
274
306
  print("Type 'exit' to quit.")
275
307
  print()
276
308
 
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "devopsiq"
7
- version = "0.1.1"
7
+ version = "0.1.3"
8
8
  description = "A from-scratch, read-only AI agent that investigates real DevOps problems — Kubernetes, Docker, Helm, Argo CD, Istio, Terraform, GitHub Actions, cloud, New Relic and more."
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -38,6 +38,7 @@ dependencies = [
38
38
  Homepage = "https://github.com/DevOpsAbhii/devops-ai-agent"
39
39
  Repository = "https://github.com/DevOpsAbhii/devops-ai-agent"
40
40
  Issues = "https://github.com/DevOpsAbhii/devops-ai-agent/issues"
41
+ Changelog = "https://github.com/DevOpsAbhii/devops-ai-agent/blob/main/CHANGELOG.md"
41
42
 
42
43
  [project.scripts]
43
44
  devopsiq = "main:main"
@@ -11,12 +11,25 @@ import contextlib
11
11
  import io
12
12
  import json
13
13
  import os
14
+ import tempfile
15
+ import time
14
16
  import types
15
17
  import unittest
18
+ from pathlib import Path
16
19
 
20
+ import httpx2 as httpx # openai 3.19 depends on the httpx2 fork
17
21
  import main as main_module
18
- from agent.agent import DevOpsAgent
22
+ import agent.agent as agent_module
23
+ from agent import config as user_config
24
+ from agent.agent import DEFAULT_MODEL, DevOpsAgent
19
25
  from agent.investigation import Investigation
26
+ from openai import (
27
+ APIConnectionError,
28
+ APIStatusError,
29
+ APITimeoutError,
30
+ AuthenticationError,
31
+ RateLimitError,
32
+ )
20
33
  from tools import investigation as inv_tools
21
34
 
22
35
 
@@ -187,19 +200,27 @@ class OneShotCliTests(unittest.TestCase):
187
200
 
188
201
  def test_parse_args(self):
189
202
  self.assertEqual(
190
- main_module.parse_args([]), (None, False, None, False, None)
203
+ main_module.parse_args([]), (None, False, None, False, None, None)
191
204
  )
192
205
  self.assertEqual(
193
206
  main_module.parse_args(["why is it down?"]),
194
- ("why is it down?", False, None, False, None),
207
+ ("why is it down?", False, None, False, None, None),
195
208
  )
196
209
  self.assertEqual(
197
210
  main_module.parse_args(["--json", "why", "is it down?"]),
198
- ("why is it down?", True, None, False, None),
211
+ ("why is it down?", True, None, False, None, None),
199
212
  )
200
213
  self.assertEqual(
201
214
  main_module.parse_args(["--json"]),
202
- (None, True, None, False, None),
215
+ (None, True, None, False, None, None),
216
+ )
217
+ self.assertEqual(
218
+ main_module.parse_args(["--model", "openai/gpt-4o-mini", "why?"]),
219
+ ("why?", False, None, False, None, "openai/gpt-4o-mini"),
220
+ )
221
+ self.assertEqual(
222
+ main_module.parse_args(["--model", "anthropic/claude-sonnet-5"]),
223
+ (None, False, None, False, None, "anthropic/claude-sonnet-5"),
203
224
  )
204
225
 
205
226
  def test_slash_command_routes_without_model(self):
@@ -300,5 +321,207 @@ class KeylessTests(unittest.TestCase):
300
321
  self.assertIn("OPENROUTER_API_KEY", err.getvalue())
301
322
 
302
323
 
324
+ class ModelChoiceTests(unittest.TestCase):
325
+ """Model preference ladder + the /model command + agent/config.py.
326
+
327
+ Precedence, highest first: explicit model argument > config.json >
328
+ OPENROUTER_MODEL env > built-in default. Tests point AGENT_CONFIG_FILE
329
+ at a tmp file so the developer's real ~/.devops-ai-agent/config.json
330
+ is never read or written.
331
+ """
332
+
333
+ def setUp(self):
334
+ tmp = tempfile.NamedTemporaryFile(suffix=".json", delete=False)
335
+ tmp.close()
336
+ self.config_file = tmp.name
337
+ os.environ["AGENT_CONFIG_FILE"] = self.config_file
338
+ self.addCleanup(self._restore)
339
+
340
+ def _restore(self):
341
+ os.environ.pop("AGENT_CONFIG_FILE", None)
342
+ Path(self.config_file).unlink(missing_ok=True)
343
+
344
+ def _set_env(self, name, value):
345
+ old = os.environ.get(name)
346
+ os.environ[name] = value
347
+ self.addCleanup(lambda: (
348
+ os.environ.pop(name, None) if old is None
349
+ else os.environ.__setitem__(name, old)
350
+ ))
351
+
352
+ def test_config_round_trip_and_corrupt_tolerance(self):
353
+ self.assertEqual(user_config.load_user_config(), {}) # missing file
354
+ path = user_config.save_user_config({"model": "openai/gpt-4o-mini"})
355
+ self.assertEqual(path, Path(self.config_file))
356
+ self.assertEqual(user_config.load_user_config(),
357
+ {"model": "openai/gpt-4o-mini"})
358
+ # Merging keeps existing keys and overwrites the given one.
359
+ user_config.save_user_config({"model": "anthropic/claude-sonnet-5"})
360
+ self.assertEqual(user_config.load_user_config(),
361
+ {"model": "anthropic/claude-sonnet-5"})
362
+ # Corrupt content degrades to {} instead of raising.
363
+ Path(self.config_file).write_text("{not json", encoding="utf-8")
364
+ self.assertEqual(user_config.load_user_config(), {})
365
+
366
+ def test_model_precedence_ladder(self):
367
+ # 4. built-in default
368
+ self.assertEqual(DevOpsAgent().model, DEFAULT_MODEL)
369
+ # 3. environment
370
+ self._set_env("OPENROUTER_MODEL", "env/model")
371
+ self.assertEqual(DevOpsAgent().model, "env/model")
372
+ # 2. config file (outranks env)
373
+ user_config.save_user_config({"model": "config/model"})
374
+ self.assertEqual(DevOpsAgent().model, "config/model")
375
+ # 1. explicit argument (outranks everything)
376
+ self.assertEqual(DevOpsAgent(model="flag/model").model, "flag/model")
377
+
378
+ def test_model_command_shows_current_model(self):
379
+ agent = DevOpsAgent()
380
+ before = agent.model
381
+ text = main_module.handle_command("/model", agent)
382
+ self.assertIn(before, text)
383
+ self.assertIn(str(user_config.config_path()), text)
384
+ self.assertEqual(agent.model, before) # show-only changes nothing
385
+
386
+ def test_model_command_saves_and_switches(self):
387
+ agent = DevOpsAgent()
388
+ out = io.StringIO()
389
+ with contextlib.redirect_stdout(out):
390
+ text = main_module.handle_command(
391
+ "/model openai/gpt-4o-mini", agent)
392
+ self.assertIn("openai/gpt-4o-mini", text)
393
+ self.assertEqual(agent.model, "openai/gpt-4o-mini")
394
+ self.assertEqual(
395
+ user_config.load_user_config(), {"model": "openai/gpt-4o-mini"})
396
+ # A multi-token name is rejected; the config stays untouched.
397
+ text = main_module.handle_command("/model two words", agent)
398
+ self.assertIn("single token", text)
399
+ self.assertEqual(agent.model, "openai/gpt-4o-mini")
400
+
401
+
402
+ class FlakyChat:
403
+ """Chat stub that raises the queued errors, then answers with content."""
404
+
405
+ def __init__(self, errors: list[Exception], content: str = "ok"):
406
+ self._errors = list(errors)
407
+ self._content = content
408
+ self.calls = 0
409
+
410
+ def create(self, **kwargs): # noqa: D102
411
+ self.calls += 1
412
+ if self._errors:
413
+ raise self._errors.pop(0)
414
+ message = types.SimpleNamespace(content=self._content, tool_calls=None)
415
+ return types.SimpleNamespace(
416
+ choices=[types.SimpleNamespace(message=message)]
417
+ )
418
+
419
+
420
+ def _api_status_error(status: int, headers: dict | None = None):
421
+ """Build an openai status error offline (no HTTP round-trip)."""
422
+ request = httpx.Request(
423
+ "POST", "https://openrouter.ai/api/v1/chat/completions")
424
+ response = httpx.Response(
425
+ status, request=request, headers=headers or {})
426
+ return APIStatusError("boom", response=response, body=None)
427
+
428
+
429
+ def _api_connection_error():
430
+ request = httpx.Request(
431
+ "POST", "https://openrouter.ai/api/v1/chat/completions")
432
+ return APIConnectionError(request=request)
433
+
434
+
435
+ class RetryTests(unittest.TestCase):
436
+ """_call_model: transient failures retry with backoff, others fail fast.
437
+
438
+ Sleeps are intercepted via the agent.agent._sleep seam, so the tests
439
+ run instantly; the recorded delays prove the backoff schedule.
440
+ """
441
+
442
+ def setUp(self):
443
+ self._old_key = os.environ.get("OPENROUTER_API_KEY")
444
+ os.environ["OPENROUTER_API_KEY"] = "test-key-for-offline-tests"
445
+ self.agent = DevOpsAgent()
446
+ self.sleeps: list[float] = []
447
+ agent_module._sleep = self.sleeps.append
448
+ self.addCleanup(self._restore)
449
+
450
+ def _restore(self):
451
+ agent_module._sleep = time.sleep
452
+ os.environ.pop("OPENROUTER_API_KEY", None)
453
+ if self._old_key is not None:
454
+ os.environ["OPENROUTER_API_KEY"] = self._old_key
455
+
456
+ def test_connection_errors_are_retried_then_succeed(self):
457
+ chat = FlakyChat([_api_connection_error(), _api_connection_error()])
458
+ self.agent.client.chat.completions = chat
459
+ self.assertEqual(self.agent.ask("why is it down?"), "ok")
460
+ self.assertEqual(chat.calls, 3) # original + 2 retries
461
+ # Backoff: attempt 0 -> ~2s (+jitter), attempt 1 -> ~4s (+jitter).
462
+ self.assertGreaterEqual(self.sleeps[0], 2.0)
463
+ self.assertLessEqual(self.sleeps[0], 3.0)
464
+ self.assertGreaterEqual(self.sleeps[1], 4.0)
465
+ self.assertLessEqual(self.sleeps[1], 5.0)
466
+
467
+ def test_timeout_is_transient(self):
468
+ request = httpx.Request("POST", "https://openrouter.ai/api/v1")
469
+ chat = FlakyChat([APITimeoutError(request)])
470
+ self.agent.client.chat.completions = chat
471
+ self.assertEqual(self.agent.ask("hello"), "ok")
472
+ self.assertEqual(chat.calls, 2)
473
+
474
+ def test_rate_limit_honors_retry_after_header(self):
475
+ request = httpx.Request("POST", "https://openrouter.ai/api/v1")
476
+ response = httpx.Response(
477
+ 429, request=request, headers={"retry-after": "7"})
478
+ chat = FlakyChat([RateLimitError("slow down", response=response,
479
+ body=None)])
480
+ self.agent.client.chat.completions = chat
481
+ self.assertEqual(self.agent.ask("hello"), "ok")
482
+ self.assertEqual(self.sleeps, [7.0])
483
+
484
+ def test_rate_limit_without_header_uses_backoff(self):
485
+ request = httpx.Request("POST", "https://openrouter.ai/api/v1")
486
+ response = httpx.Response(429, request=request)
487
+ chat = FlakyChat([RateLimitError("slow down", response=response,
488
+ body=None)])
489
+ self.agent.client.chat.completions = chat
490
+ self.assertEqual(self.agent.ask("hello"), "ok")
491
+ self.assertGreaterEqual(self.sleeps[0], 2.0)
492
+
493
+ def test_auth_error_fails_immediately(self):
494
+ request = httpx.Request("POST", "https://openrouter.ai/api/v1")
495
+ response = httpx.Response(401, request=request)
496
+ chat = FlakyChat([AuthenticationError("bad key", response=response,
497
+ body=None)])
498
+ self.agent.client.chat.completions = chat
499
+ with self.assertRaises(AuthenticationError):
500
+ self.agent.ask("hello")
501
+ self.assertEqual(chat.calls, 1) # no retry on a rejected key
502
+ self.assertEqual(self.sleeps, [])
503
+
504
+ def test_client_400_fails_immediately(self):
505
+ chat = FlakyChat([_api_status_error(400)])
506
+ self.agent.client.chat.completions = chat
507
+ with self.assertRaises(APIStatusError):
508
+ self.agent.ask("hello")
509
+ self.assertEqual(chat.calls, 1)
510
+
511
+ def test_server_500_is_retried(self):
512
+ chat = FlakyChat([_api_status_error(500), _api_status_error(502)])
513
+ self.agent.client.chat.completions = chat
514
+ self.assertEqual(self.agent.ask("hello"), "ok")
515
+ self.assertEqual(chat.calls, 3)
516
+
517
+ def test_retries_are_exhausted_loudly(self):
518
+ chat = FlakyChat([_api_connection_error() for _ in range(4)])
519
+ self.agent.client.chat.completions = chat
520
+ with self.assertRaises(APIConnectionError):
521
+ self.agent.ask("hello")
522
+ self.assertEqual(chat.calls, 4) # original + 3 retries, then raise
523
+ self.assertEqual(len(self.sleeps), 3) # 2s, 4s, 8s schedule
524
+
525
+
303
526
  if __name__ == "__main__":
304
527
  unittest.main()
@@ -329,11 +329,11 @@ class CliTests(unittest.TestCase):
329
329
 
330
330
  def test_parse_args_defaults(self):
331
331
  self.assertEqual(
332
- main_module.parse_args([]), (None, False, None, False, None)
332
+ main_module.parse_args([]), (None, False, None, False, None, None)
333
333
  )
334
334
  self.assertEqual(
335
335
  main_module.parse_args(["why is it down?"]),
336
- ("why is it down?", False, None, False, None),
336
+ ("why is it down?", False, None, False, None, None),
337
337
  )
338
338
 
339
339
  def test_parse_args_all_flags(self):
@@ -341,18 +341,19 @@ class CliTests(unittest.TestCase):
341
341
  ["--json", "--resume", "--store-dir", "/tmp/s", "--out", "r.json",
342
342
  "why", "is it down?"]
343
343
  )
344
- self.assertEqual(args, ("why is it down?", True, "/tmp/s", True, "r.json"))
344
+ self.assertEqual(
345
+ args, ("why is it down?", True, "/tmp/s", True, "r.json", None))
345
346
 
346
347
  def test_parse_args_flags_only_runs_repl(self):
347
348
  self.assertEqual(
348
349
  main_module.parse_args(["--json", "--resume"]),
349
- (None, True, None, True, None),
350
+ (None, True, None, True, None, None),
350
351
  )
351
352
 
352
353
  def test_parse_args_flag_without_value_is_ignored(self):
353
354
  self.assertEqual(
354
355
  main_module.parse_args(["--store-dir"]),
355
- (None, False, None, False, None),
356
+ (None, False, None, False, None, None),
356
357
  )
357
358
 
358
359
  # --- --store-dir ----------------------------------------------------------
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes