devopsiq 0.1.1__tar.gz → 0.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. {devopsiq-0.1.1/devopsiq.egg-info → devopsiq-0.1.2}/PKG-INFO +57 -15
  2. {devopsiq-0.1.1 → devopsiq-0.1.2}/README.md +55 -14
  3. {devopsiq-0.1.1 → devopsiq-0.1.2}/agent/agent.py +79 -4
  4. devopsiq-0.1.2/agent/config.py +45 -0
  5. {devopsiq-0.1.1 → devopsiq-0.1.2/devopsiq.egg-info}/PKG-INFO +57 -15
  6. {devopsiq-0.1.1 → devopsiq-0.1.2}/devopsiq.egg-info/SOURCES.txt +1 -0
  7. {devopsiq-0.1.1 → devopsiq-0.1.2}/main.py +46 -14
  8. {devopsiq-0.1.1 → devopsiq-0.1.2}/pyproject.toml +2 -1
  9. {devopsiq-0.1.1 → devopsiq-0.1.2}/tests/test_automation.py +228 -5
  10. {devopsiq-0.1.1 → devopsiq-0.1.2}/tests/test_phase7.py +6 -5
  11. {devopsiq-0.1.1 → devopsiq-0.1.2}/LICENSE +0 -0
  12. {devopsiq-0.1.1 → devopsiq-0.1.2}/agent/__init__.py +0 -0
  13. {devopsiq-0.1.1 → devopsiq-0.1.2}/agent/investigation.py +0 -0
  14. {devopsiq-0.1.1 → devopsiq-0.1.2}/agent/prompts.py +0 -0
  15. {devopsiq-0.1.1 → devopsiq-0.1.2}/agent/store.py +0 -0
  16. {devopsiq-0.1.1 → devopsiq-0.1.2}/devopsiq.egg-info/dependency_links.txt +0 -0
  17. {devopsiq-0.1.1 → devopsiq-0.1.2}/devopsiq.egg-info/entry_points.txt +0 -0
  18. {devopsiq-0.1.1 → devopsiq-0.1.2}/devopsiq.egg-info/requires.txt +0 -0
  19. {devopsiq-0.1.1 → devopsiq-0.1.2}/devopsiq.egg-info/top_level.txt +0 -0
  20. {devopsiq-0.1.1 → devopsiq-0.1.2}/setup.cfg +0 -0
  21. {devopsiq-0.1.1 → devopsiq-0.1.2}/tests/test_phase2.py +0 -0
  22. {devopsiq-0.1.1 → devopsiq-0.1.2}/tests/test_phase3.py +0 -0
  23. {devopsiq-0.1.1 → devopsiq-0.1.2}/tests/test_phase4.py +0 -0
  24. {devopsiq-0.1.1 → devopsiq-0.1.2}/tests/test_phase5.py +0 -0
  25. {devopsiq-0.1.1 → devopsiq-0.1.2}/tests/test_phase8.py +0 -0
  26. {devopsiq-0.1.1 → devopsiq-0.1.2}/tests/test_phase9.py +0 -0
  27. {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/__init__.py +0 -0
  28. {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/ansible.py +0 -0
  29. {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/argocd.py +0 -0
  30. {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/base.py +0 -0
  31. {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/cloud.py +0 -0
  32. {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/docker.py +0 -0
  33. {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/git_ci.py +0 -0
  34. {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/helm.py +0 -0
  35. {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/investigation.py +0 -0
  36. {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/istio.py +0 -0
  37. {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/kubernetes.py +0 -0
  38. {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/monitoring.py +0 -0
  39. {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/newrelic.py +0 -0
  40. {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/preflight.py +0 -0
  41. {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/registry.py +0 -0
  42. {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/system.py +0 -0
  43. {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/terraform.py +0 -0
  44. {devopsiq-0.1.1 → devopsiq-0.1.2}/tools/trivy.py +0 -0
@@ -1,12 +1,13 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: devopsiq
3
- Version: 0.1.1
3
+ Version: 0.1.2
4
4
  Summary: A from-scratch, read-only AI agent that investigates real DevOps problems — Kubernetes, Docker, Helm, Argo CD, Istio, Terraform, GitHub Actions, cloud, New Relic and more.
5
5
  Author: DevOpsAbhii
6
6
  License-Expression: MIT
7
7
  Project-URL: Homepage, https://github.com/DevOpsAbhii/devops-ai-agent
8
8
  Project-URL: Repository, https://github.com/DevOpsAbhii/devops-ai-agent
9
9
  Project-URL: Issues, https://github.com/DevOpsAbhii/devops-ai-agent/issues
10
+ Project-URL: Changelog, https://github.com/DevOpsAbhii/devops-ai-agent/blob/main/CHANGELOG.md
10
11
  Keywords: devops,kubernetes,docker,helm,argocd,istio,terraform,github-actions,observability,newrelic,incident-response,root-cause-analysis,ai-agent,cli
11
12
  Classifier: Development Status :: 4 - Beta
12
13
  Classifier: Environment :: Console
@@ -29,6 +30,13 @@ Dynamic: license-file
29
30
 
30
31
  # DevOps AI Agent
31
32
 
33
+ [![PyPI](https://img.shields.io/pypi/v/devopsiq)](https://pypi.org/project/devopsiq/)
34
+ [![Python](https://img.shields.io/pypi/pyversions/devopsiq)](https://pypi.org/project/devopsiq/)
35
+ [![Release](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml/badge.svg)](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
36
+ [![Docker](https://img.shields.io/badge/ghcr-devops--ai--agent-2496ED?logo=docker&logoColor=white)](https://github.com/DevOpsAbhii/devops-ai-agent/pkgs/container/devops-ai-agent)
37
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
38
+ [![Tests](https://img.shields.io/badge/tests-181%20offline-brightgreen)](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
39
+
32
40
  An AI agent that investigates real DevOps problems. The end goal: ask it
33
41
  something like *"Why is my Kubernetes pod in CrashLoopBackOff?"* and have it
34
42
  gather evidence, reason about the evidence, identify the likely root cause,
@@ -55,7 +63,14 @@ from environment config), and Ansible listing (inventory, playbook tasks).
55
63
  open alerts over NerdGraph — credentials from env), Trivy image
56
64
  vulnerability scanning, Helm releases (list/status/history), Argo CD
57
65
  (GitOps app sync/health), Istio mesh proxy status, and Docker Compose
58
- (project and service listing). The repository is git-tracked.
66
+ (project and service listing). **Phase 10 ships it as a product:** the
67
+ [`devopsiq` package on PyPI](https://pypi.org/project/devopsiq/) and a
68
+ prebuilt multi-arch Docker image (`ghcr.io/devopsabhii/devops-ai-agent`) —
69
+ both cut automatically by pushing a `v*` tag (test gate → PyPI via Trusted
70
+ Publishing + GHCR in parallel). **No API key? The agent still runs:** it
71
+ starts in model-less mode — record commands (`/report`, `/investigations`)
72
+ and the whole 58-tool layer work without a key; only questions to the model
73
+ need one. The repository is git-tracked.
59
74
  Every phase still built from scratch — no LangChain, LangGraph,
60
75
  AutoGen, CrewAI, or MCP.
61
76
 
@@ -73,7 +88,7 @@ model is called, how conversation history flows, how tool selection +
73
88
  execution + evidence feedback work — instead of depending on a framework
74
89
  for it.
75
90
 
76
- In Phase 5 the agent:
91
+ Today the agent:
77
92
  - holds a conversation with **GLM 5.3** through **OpenRouter**;
78
93
  - has **58 real, read-only tools** across fifteen domains: host facts,
79
94
  Kubernetes (12 tools), Linux system (4), Docker + Compose (10),
@@ -133,8 +148,8 @@ Module map:
133
148
 
134
149
  | Path | Responsibility |
135
150
  | --------------------------- | ------------------------------------------------------------ |
136
- | `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages |
137
- | `agent/agent.py` | `DevOpsAgent` — client, history, `ask()`, `_complete()` loop |
151
+ | `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages, model-less start |
152
+ | `agent/agent.py` | `DevOpsAgent` — client (`None` = model-less mode), history, `ask()`, `_complete()` loop |
138
153
  | `agent/prompts.py` | The system prompt (versioned/tested separately) |
139
154
  | `agent/investigation.py` | First-class investigation record: hypotheses, verdicts, evidence, report renderer (pure data) |
140
155
  | `agent/store.py` | `InvestigationStore` — one JSON file per record, atomic writes, resume/list (Phase 7) |
@@ -163,6 +178,11 @@ Module map:
163
178
  | `tests/test_phase7.py` | Offline suite: round-trip serialization, store files, auto-save, resume, CLI flags (Phase 7) |
164
179
  | `tests/test_phase8.py` | Offline suite: argv templates + validation for the 21 Phase 8 tools (fake CLIs, env-based monitoring) |
165
180
  | `tests/test_phase9.py` | Offline suite: New Relic env-credential + payload tests, trivy/helm/argocd/istio/compose argv templates (Phase 9) |
181
+ | `pyproject.toml` | Package `devopsiq`: metadata, MIT, console script `devopsiq = main:main` (Phase 10) |
182
+ | `Dockerfile` | Prebuilt image: slim base + kubectl/helm/trivy/gh + app, non-root, `/data` record store (Phase 10) |
183
+ | `.github/workflows/release.yml` | Tag-driven release: test gate → PyPI (Trusted Publishing) + GHCR multi-arch (Phase 10) |
184
+ | `docs/generate_pdf.py` | Builds the project documentation PDF from live source |
185
+ | `docs/generate_release_guide.py` | Builds the release & update playbook PDF |
166
186
 
167
187
  ### The tool-use loop
168
188
 
@@ -209,6 +229,7 @@ the CLI exposes it directly:
209
229
  | `/investigations` | List saved records on disk (newest first; `← active` marks the live one) |
210
230
  | `/report` | Show the canonical report (once concluded) |
211
231
  | `/endinvestigation` | Clear the record (memory only — the saved copy stays as history) |
232
+ | `/model [<name>]` | Show the active model, or save a default (config.json) and switch to it |
212
233
 
213
234
  The report the agent ends with and `/report` render are kept consistent by
214
235
  construction: `tool` results confirm each record call and the tracker, and
@@ -396,16 +417,22 @@ official `openai` Python SDK works as our client with two config lines:
396
417
  OpenAI(api_key=..., base_url="https://openrouter.ai/api/v1")
397
418
  ```
398
419
 
399
- Swapping to another OpenRouter model later is a one-line change (an env var
400
- today); moving to any other OpenAI-compatible provider changes only
401
- `agent/agent.py` configuration.
420
+ Swapping to another OpenRouter model later is a one-liner (env var, `/model`
421
+ command, or `--model` flag); moving to any other OpenAI-compatible provider
422
+ changes only `agent/agent.py` configuration.
402
423
 
403
424
  **Your model, your choice.** The default is baked in as a fallback, never a
404
- restriction — set `OPENROUTER_MODEL` (shell or `.env`) to any model on
405
- OpenRouter:
425
+ restriction. Four ways to set the model — the first one set wins:
426
+
427
+ 1. `--model NAME` flag (one run: `devopsiq --model openai/gpt-5.2 "why?"`)
428
+ 2. `/model <name>` in the REPL — saves it to `~/.devops-ai-agent/config.json`
429
+ so every future run uses it
430
+ 3. `OPENROUTER_MODEL` (shell or `.env`)
431
+ 4. built-in default (`z-ai/glm-5.3`)
406
432
 
407
433
  ```bash
408
434
  export OPENROUTER_MODEL=anthropic/claude-sonnet-5 # or openai/gpt-5.2, google/gemini-2.5-pro, ...
435
+ devopsiq /model openai/gpt-5.2 # or save a default from the REPL
409
436
  ```
410
437
 
411
438
  Two things to weigh when picking: the agent is a tool-use loop, so choose a
@@ -558,6 +585,7 @@ setup/API errors — so it drops straight into a pipeline:
558
585
  .venv/bin/python main.py --json "why is api-5d6f crash-looping?" # structured report
559
586
  .venv/bin/python main.py --resume --json "any update?" # continue a prior run
560
587
  .venv/bin/python main.py --store-dir /tmp/runs --out report.json "..." # pipeline paths
588
+ .venv/bin/python main.py --model openai/gpt-5.2 "why is it down?" # one-run model override
561
589
  ```
562
590
 
563
591
  With `--json` the stdout is one JSON document (see `render_report_json` in
@@ -599,8 +627,8 @@ Tools: ansible_inventory, ansible_playbook_tasks, aws_identity,
599
627
  k8s_top_nodes, k8s_top_pods, loki_query, prom_query,
600
628
  sys_open_ports, sys_service_logs, sys_service_status,
601
629
  sys_top_processes, system_info, tf_plan, tf_show, tf_state_list
602
- Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation
603
- One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] "<problem>"
630
+ Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation, /model [<name>]
631
+ One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] [--model NAME] "<problem>"
604
632
  Type 'exit' to quit.
605
633
 
606
634
  You: The checkout service container keeps exiting in Docker. Investigate.
@@ -617,7 +645,7 @@ You: exit
617
645
  Type `exit` / `quit`, or press Ctrl-D / Ctrl-C to leave. Slash commands are
618
646
  handled locally and never reach the model.
619
647
 
620
- ## 8. Current limitations (Phase 9)
648
+ ## 8. Current limitations
621
649
 
622
650
  - **Each domain needs its CLI installed and reachable.** Missing CLIs,
623
651
  unauthenticated `gh`, a dead docker daemon, an uninitialized terraform
@@ -702,8 +730,22 @@ handled locally and never reach the model.
702
730
  `helm_history` — reads only), Argo CD (`argocd_apps`,
703
731
  `argocd_app_status`), Istio (`istioctl_proxy_status`), Docker Compose
704
732
  (`docker_compose_ls`, `docker_compose_ps` — live-verified on this host).
733
+ - **Phase 10 — distribution. ✅ Done.** The `devopsiq` package on PyPI
734
+ (console command `devopsiq`, MIT) and the prebuilt multi-arch GHCR image,
735
+ cut by a tag-driven release workflow: an offline test gate, then PyPI
736
+ (Trusted Publishing — no token in the repo) and GHCR (amd64 + arm64) in
737
+ parallel. **`v0.1.1` added model-less mode:** with no API key the agent
738
+ still constructs — record commands (`/report`, `/investigations`) and the
739
+ 58-tool layer work; only model questions exit 1 with the setup message.
740
+ Verified end-to-end against the published package.
741
+ - **Model choice + resilient calls (v0.1.2). ✅ Done.** The model is
742
+ set by a 4-rung ladder — `--model` flag > `/model`-saved config file
743
+ (`~/.devops-ai-agent/config.json`) > `OPENROUTER_MODEL` > baked-in
744
+ default — and transient model-call failures (timeouts, connection
745
+ errors, 429s, 5xx) retry with exponential backoff (2s→4s→8s + jitter,
746
+ honoring a 429's Retry-After); a rejected key fails immediately.
705
747
  - **Later — region-scoped cloud resources** (ec2 describe-*, compute
706
748
  instances list, ...) behind the same template pattern; more observability
707
749
  depth (New Relic dashboards/entities, Prometheus range queries); streaming;
708
- conversation-history persistence; and a human-approval gate before any
709
- mutating action is ever allowed.
750
+ conversation-history persistence; and a human-approval gate
751
+ before any mutating action is ever allowed.
@@ -1,5 +1,12 @@
1
1
  # DevOps AI Agent
2
2
 
3
+ [![PyPI](https://img.shields.io/pypi/v/devopsiq)](https://pypi.org/project/devopsiq/)
4
+ [![Python](https://img.shields.io/pypi/pyversions/devopsiq)](https://pypi.org/project/devopsiq/)
5
+ [![Release](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml/badge.svg)](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
6
+ [![Docker](https://img.shields.io/badge/ghcr-devops--ai--agent-2496ED?logo=docker&logoColor=white)](https://github.com/DevOpsAbhii/devops-ai-agent/pkgs/container/devops-ai-agent)
7
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
8
+ [![Tests](https://img.shields.io/badge/tests-181%20offline-brightgreen)](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
9
+
3
10
  An AI agent that investigates real DevOps problems. The end goal: ask it
4
11
  something like *"Why is my Kubernetes pod in CrashLoopBackOff?"* and have it
5
12
  gather evidence, reason about the evidence, identify the likely root cause,
@@ -26,7 +33,14 @@ from environment config), and Ansible listing (inventory, playbook tasks).
26
33
  open alerts over NerdGraph — credentials from env), Trivy image
27
34
  vulnerability scanning, Helm releases (list/status/history), Argo CD
28
35
  (GitOps app sync/health), Istio mesh proxy status, and Docker Compose
29
- (project and service listing). The repository is git-tracked.
36
+ (project and service listing). **Phase 10 ships it as a product:** the
37
+ [`devopsiq` package on PyPI](https://pypi.org/project/devopsiq/) and a
38
+ prebuilt multi-arch Docker image (`ghcr.io/devopsabhii/devops-ai-agent`) —
39
+ both cut automatically by pushing a `v*` tag (test gate → PyPI via Trusted
40
+ Publishing + GHCR in parallel). **No API key? The agent still runs:** it
41
+ starts in model-less mode — record commands (`/report`, `/investigations`)
42
+ and the whole 58-tool layer work without a key; only questions to the model
43
+ need one. The repository is git-tracked.
30
44
  Every phase still built from scratch — no LangChain, LangGraph,
31
45
  AutoGen, CrewAI, or MCP.
32
46
 
@@ -44,7 +58,7 @@ model is called, how conversation history flows, how tool selection +
44
58
  execution + evidence feedback work — instead of depending on a framework
45
59
  for it.
46
60
 
47
- In Phase 5 the agent:
61
+ Today the agent:
48
62
  - holds a conversation with **GLM 5.3** through **OpenRouter**;
49
63
  - has **58 real, read-only tools** across fifteen domains: host facts,
50
64
  Kubernetes (12 tools), Linux system (4), Docker + Compose (10),
@@ -104,8 +118,8 @@ Module map:
104
118
 
105
119
  | Path | Responsibility |
106
120
  | --------------------------- | ------------------------------------------------------------ |
107
- | `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages |
108
- | `agent/agent.py` | `DevOpsAgent` — client, history, `ask()`, `_complete()` loop |
121
+ | `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages, model-less start |
122
+ | `agent/agent.py` | `DevOpsAgent` — client (`None` = model-less mode), history, `ask()`, `_complete()` loop |
109
123
  | `agent/prompts.py` | The system prompt (versioned/tested separately) |
110
124
  | `agent/investigation.py` | First-class investigation record: hypotheses, verdicts, evidence, report renderer (pure data) |
111
125
  | `agent/store.py` | `InvestigationStore` — one JSON file per record, atomic writes, resume/list (Phase 7) |
@@ -134,6 +148,11 @@ Module map:
134
148
  | `tests/test_phase7.py` | Offline suite: round-trip serialization, store files, auto-save, resume, CLI flags (Phase 7) |
135
149
  | `tests/test_phase8.py` | Offline suite: argv templates + validation for the 21 Phase 8 tools (fake CLIs, env-based monitoring) |
136
150
  | `tests/test_phase9.py` | Offline suite: New Relic env-credential + payload tests, trivy/helm/argocd/istio/compose argv templates (Phase 9) |
151
+ | `pyproject.toml` | Package `devopsiq`: metadata, MIT, console script `devopsiq = main:main` (Phase 10) |
152
+ | `Dockerfile` | Prebuilt image: slim base + kubectl/helm/trivy/gh + app, non-root, `/data` record store (Phase 10) |
153
+ | `.github/workflows/release.yml` | Tag-driven release: test gate → PyPI (Trusted Publishing) + GHCR multi-arch (Phase 10) |
154
+ | `docs/generate_pdf.py` | Builds the project documentation PDF from live source |
155
+ | `docs/generate_release_guide.py` | Builds the release & update playbook PDF |
137
156
 
138
157
  ### The tool-use loop
139
158
 
@@ -180,6 +199,7 @@ the CLI exposes it directly:
180
199
  | `/investigations` | List saved records on disk (newest first; `← active` marks the live one) |
181
200
  | `/report` | Show the canonical report (once concluded) |
182
201
  | `/endinvestigation` | Clear the record (memory only — the saved copy stays as history) |
202
+ | `/model [<name>]` | Show the active model, or save a default (config.json) and switch to it |
183
203
 
184
204
  The report the agent ends with and `/report` render are kept consistent by
185
205
  construction: `tool` results confirm each record call and the tracker, and
@@ -367,16 +387,22 @@ official `openai` Python SDK works as our client with two config lines:
367
387
  OpenAI(api_key=..., base_url="https://openrouter.ai/api/v1")
368
388
  ```
369
389
 
370
- Swapping to another OpenRouter model later is a one-line change (an env var
371
- today); moving to any other OpenAI-compatible provider changes only
372
- `agent/agent.py` configuration.
390
+ Swapping to another OpenRouter model later is a one-liner (env var, `/model`
391
+ command, or `--model` flag); moving to any other OpenAI-compatible provider
392
+ changes only `agent/agent.py` configuration.
373
393
 
374
394
  **Your model, your choice.** The default is baked in as a fallback, never a
375
- restriction — set `OPENROUTER_MODEL` (shell or `.env`) to any model on
376
- OpenRouter:
395
+ restriction. Four ways to set the model — the first one set wins:
396
+
397
+ 1. `--model NAME` flag (one run: `devopsiq --model openai/gpt-5.2 "why?"`)
398
+ 2. `/model <name>` in the REPL — saves it to `~/.devops-ai-agent/config.json`
399
+ so every future run uses it
400
+ 3. `OPENROUTER_MODEL` (shell or `.env`)
401
+ 4. built-in default (`z-ai/glm-5.3`)
377
402
 
378
403
  ```bash
379
404
  export OPENROUTER_MODEL=anthropic/claude-sonnet-5 # or openai/gpt-5.2, google/gemini-2.5-pro, ...
405
+ devopsiq /model openai/gpt-5.2 # or save a default from the REPL
380
406
  ```
381
407
 
382
408
  Two things to weigh when picking: the agent is a tool-use loop, so choose a
@@ -529,6 +555,7 @@ setup/API errors — so it drops straight into a pipeline:
529
555
  .venv/bin/python main.py --json "why is api-5d6f crash-looping?" # structured report
530
556
  .venv/bin/python main.py --resume --json "any update?" # continue a prior run
531
557
  .venv/bin/python main.py --store-dir /tmp/runs --out report.json "..." # pipeline paths
558
+ .venv/bin/python main.py --model openai/gpt-5.2 "why is it down?" # one-run model override
532
559
  ```
533
560
 
534
561
  With `--json` the stdout is one JSON document (see `render_report_json` in
@@ -570,8 +597,8 @@ Tools: ansible_inventory, ansible_playbook_tasks, aws_identity,
570
597
  k8s_top_nodes, k8s_top_pods, loki_query, prom_query,
571
598
  sys_open_ports, sys_service_logs, sys_service_status,
572
599
  sys_top_processes, system_info, tf_plan, tf_show, tf_state_list
573
- Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation
574
- One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] "<problem>"
600
+ Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation, /model [<name>]
601
+ One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] [--model NAME] "<problem>"
575
602
  Type 'exit' to quit.
576
603
 
577
604
  You: The checkout service container keeps exiting in Docker. Investigate.
@@ -588,7 +615,7 @@ You: exit
588
615
  Type `exit` / `quit`, or press Ctrl-D / Ctrl-C to leave. Slash commands are
589
616
  handled locally and never reach the model.
590
617
 
591
- ## 8. Current limitations (Phase 9)
618
+ ## 8. Current limitations
592
619
 
593
620
  - **Each domain needs its CLI installed and reachable.** Missing CLIs,
594
621
  unauthenticated `gh`, a dead docker daemon, an uninitialized terraform
@@ -673,8 +700,22 @@ handled locally and never reach the model.
673
700
  `helm_history` — reads only), Argo CD (`argocd_apps`,
674
701
  `argocd_app_status`), Istio (`istioctl_proxy_status`), Docker Compose
675
702
  (`docker_compose_ls`, `docker_compose_ps` — live-verified on this host).
703
+ - **Phase 10 — distribution. ✅ Done.** The `devopsiq` package on PyPI
704
+ (console command `devopsiq`, MIT) and the prebuilt multi-arch GHCR image,
705
+ cut by a tag-driven release workflow: an offline test gate, then PyPI
706
+ (Trusted Publishing — no token in the repo) and GHCR (amd64 + arm64) in
707
+ parallel. **`v0.1.1` added model-less mode:** with no API key the agent
708
+ still constructs — record commands (`/report`, `/investigations`) and the
709
+ 58-tool layer work; only model questions exit 1 with the setup message.
710
+ Verified end-to-end against the published package.
711
+ - **Model choice + resilient calls (v0.1.2). ✅ Done.** The model is
712
+ set by a 4-rung ladder — `--model` flag > `/model`-saved config file
713
+ (`~/.devops-ai-agent/config.json`) > `OPENROUTER_MODEL` > baked-in
714
+ default — and transient model-call failures (timeouts, connection
715
+ errors, 429s, 5xx) retry with exponential backoff (2s→4s→8s + jitter,
716
+ honoring a 429's Retry-After); a rejected key fails immediately.
676
717
  - **Later — region-scoped cloud resources** (ec2 describe-*, compute
677
718
  instances list, ...) behind the same template pattern; more observability
678
719
  depth (New Relic dashboards/entities, Prometheus range queries); streaming;
679
- conversation-history persistence; and a human-approval gate before any
680
- mutating action is ever allowed.
720
+ conversation-history persistence; and a human-approval gate
721
+ before any mutating action is ever allowed.
@@ -26,9 +26,18 @@ exits; the delegates below expose resume/list/store-location to the CLI.
26
26
  """
27
27
 
28
28
  import os
29
+ import random
30
+ import time
31
+
32
+ from openai import (
33
+ APIConnectionError,
34
+ APIStatusError,
35
+ APITimeoutError,
36
+ OpenAI,
37
+ RateLimitError,
38
+ )
29
39
 
30
- from openai import OpenAI
31
-
40
+ from agent.config import load_user_config
32
41
  from agent.prompts import SYSTEM_PROMPT
33
42
  from agent.store import InvestigationStore
34
43
  from tools import ( # noqa: F401 — side effect: each module registers its tools
@@ -68,6 +77,50 @@ PLACEHOLDER_KEY = "your_key_here"
68
77
  # Safety valve: the model gets at most this many tool-use turns before we stop.
69
78
  MAX_TOOL_ITERATIONS = 10
70
79
 
80
+ # Transient-failure retries for the model call itself (network blips, 429,
81
+ # 5xx). MAX_MODEL_RETRIES attempts beyond the first, exponential backoff
82
+ # starting at RETRY_BASE_DELAY (2s -> 4s -> 8s) plus jitter, capped at
83
+ # RETRY_MAX_DELAY. A 429's Retry-After header wins when present.
84
+ MAX_MODEL_RETRIES = 3
85
+ RETRY_BASE_DELAY = 2.0
86
+ RETRY_MAX_DELAY = 60.0
87
+
88
+ _sleep = time.sleep # test seam: offline tests patch agent.agent._sleep
89
+
90
+
91
+ def _is_transient(exc: Exception) -> bool:
92
+ """True when retrying `exc` can plausibly succeed.
93
+
94
+ Timeouts, connection failures, rate limits, and server-side 5xx are
95
+ transient. A rejected key (401) or any other client 4xx is not — the
96
+ same request would fail identically forever, so it fails immediately.
97
+ """
98
+ if isinstance(exc, (APITimeoutError, APIConnectionError, RateLimitError)):
99
+ return True
100
+ if isinstance(exc, APIStatusError):
101
+ return exc.status_code >= 500
102
+ return False
103
+
104
+
105
+ def _retry_delay(exc: Exception, attempt: int) -> float:
106
+ """Seconds to wait before retry number `attempt` (0-based).
107
+
108
+ Exponential backoff with jitter, except on a rate limit carrying a
109
+ Retry-After header — the server knows its own budget, so it wins
110
+ (clamped to [1, RETRY_MAX_DELAY] so a bad header cannot hurt us).
111
+ """
112
+ response = getattr(exc, "response", None)
113
+ retry_after = getattr(response, "headers", {}).get("retry-after")
114
+ if retry_after:
115
+ try:
116
+ return min(max(float(retry_after), 1.0), RETRY_MAX_DELAY)
117
+ except (TypeError, ValueError):
118
+ pass # non-numeric header: fall through to backoff
119
+ return min(
120
+ RETRY_BASE_DELAY * (2 ** attempt) + random.uniform(0, 1),
121
+ RETRY_MAX_DELAY,
122
+ )
123
+
71
124
 
72
125
  class DevOpsAgent:
73
126
  """Minimal DevOps investigation assistant (read-only by design)."""
@@ -81,7 +134,12 @@ class DevOpsAgent:
81
134
  # Configuration resolution order: explicit argument > environment > default.
82
135
  self.api_key = api_key or os.getenv("OPENROUTER_API_KEY")
83
136
 
84
- self.model = model or os.getenv("OPENROUTER_MODEL", DEFAULT_MODEL)
137
+ self.model = (
138
+ model # 1. explicit --model flag
139
+ or load_user_config().get("model") # 2. ~/.devops-ai-agent/config.json
140
+ or os.getenv("OPENROUTER_MODEL") # 3. environment / .env
141
+ or DEFAULT_MODEL # 4. built-in default
142
+ )
85
143
  self.base_url = base_url or os.getenv("OPENROUTER_BASE_URL", DEFAULT_BASE_URL)
86
144
 
87
145
  # Model-less mode: a missing (or placeholder) key no longer blocks
@@ -197,7 +255,7 @@ class DevOpsAgent:
197
255
  if self.tools:
198
256
  request["tools"] = [tool.schema() for tool in self.tools]
199
257
 
200
- response = self.client.chat.completions.create(**request)
258
+ response = self._call_model(request)
201
259
  message = response.choices[0].message
202
260
 
203
261
  if not message.tool_calls:
@@ -219,6 +277,23 @@ class DevOpsAgent:
219
277
  f"The model did not finish after {MAX_TOOL_ITERATIONS} tool-use turns."
220
278
  )
221
279
 
280
+ def _call_model(self, request: dict):
281
+ """One model call with bounded retries on transient failures.
282
+
283
+ Timeouts, connection errors, rate limits (429) and server-side 5xx
284
+ are retried up to MAX_MODEL_RETRIES times with exponential backoff
285
+ (2s, 4s, 8s + jitter); a 429's Retry-After header wins when present.
286
+ Client errors — a rejected key (401) most notably — fail immediately,
287
+ because retrying the identical request cannot fix them.
288
+ """
289
+ for attempt in range(MAX_MODEL_RETRIES + 1):
290
+ try:
291
+ return self.client.chat.completions.create(**request)
292
+ except Exception as exc: # noqa: BLE001 — classified right below
293
+ if attempt >= MAX_MODEL_RETRIES or not _is_transient(exc):
294
+ raise
295
+ _sleep(_retry_delay(exc, attempt))
296
+
222
297
 
223
298
  def _echo_tool_request(message) -> dict:
224
299
  """Rebuild the assistant turn that requested tools, verbatim.
@@ -0,0 +1,45 @@
1
+ """Per-user preferences: ~/.devops-ai-agent/config.json.
2
+
3
+ The `devopsiq` equivalent of "my settings" — currently the preferred model,
4
+ written by the /model command and read by the agent at startup. Precedence
5
+ for the model, highest first:
6
+
7
+ --model flag > config.json > OPENROUTER_MODEL env > built-in default
8
+
9
+ The file is tiny JSON, one flat object. Every read is tolerant: a missing
10
+ file, an unreadable home directory, or corrupt content yields {} — a broken
11
+ config never blocks the agent (the same degrade-don't-crash rule the
12
+ investigation store follows). Tests point AGENT_CONFIG_FILE at a tmp path.
13
+ """
14
+
15
+ import json
16
+ import os
17
+ from pathlib import Path
18
+
19
+ DEFAULT_CONFIG_PATH = Path.home() / ".devops-ai-agent" / "config.json"
20
+
21
+
22
+ def config_path() -> Path:
23
+ """The active config file (AGENT_CONFIG_FILE overrides; tests use tmp)."""
24
+ return Path(os.getenv("AGENT_CONFIG_FILE") or DEFAULT_CONFIG_PATH)
25
+
26
+
27
+ def load_user_config() -> dict:
28
+ """Read the config file; {} when missing or unreadable — never raise."""
29
+ try:
30
+ raw = config_path().read_text(encoding="utf-8")
31
+ data = json.loads(raw)
32
+ return data if isinstance(data, dict) else {}
33
+ except (OSError, ValueError):
34
+ return {}
35
+
36
+
37
+ def save_user_config(update: dict) -> Path:
38
+ """Merge `update` into the config file (create the directory if needed)."""
39
+ path = config_path()
40
+ data = load_user_config()
41
+ data.update(update)
42
+ path.parent.mkdir(parents=True, exist_ok=True)
43
+ path.write_text(json.dumps(data, indent=2, sort_keys=True) + "\n",
44
+ encoding="utf-8")
45
+ return path
@@ -1,12 +1,13 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: devopsiq
3
- Version: 0.1.1
3
+ Version: 0.1.2
4
4
  Summary: A from-scratch, read-only AI agent that investigates real DevOps problems — Kubernetes, Docker, Helm, Argo CD, Istio, Terraform, GitHub Actions, cloud, New Relic and more.
5
5
  Author: DevOpsAbhii
6
6
  License-Expression: MIT
7
7
  Project-URL: Homepage, https://github.com/DevOpsAbhii/devops-ai-agent
8
8
  Project-URL: Repository, https://github.com/DevOpsAbhii/devops-ai-agent
9
9
  Project-URL: Issues, https://github.com/DevOpsAbhii/devops-ai-agent/issues
10
+ Project-URL: Changelog, https://github.com/DevOpsAbhii/devops-ai-agent/blob/main/CHANGELOG.md
10
11
  Keywords: devops,kubernetes,docker,helm,argocd,istio,terraform,github-actions,observability,newrelic,incident-response,root-cause-analysis,ai-agent,cli
11
12
  Classifier: Development Status :: 4 - Beta
12
13
  Classifier: Environment :: Console
@@ -29,6 +30,13 @@ Dynamic: license-file
29
30
 
30
31
  # DevOps AI Agent
31
32
 
33
+ [![PyPI](https://img.shields.io/pypi/v/devopsiq)](https://pypi.org/project/devopsiq/)
34
+ [![Python](https://img.shields.io/pypi/pyversions/devopsiq)](https://pypi.org/project/devopsiq/)
35
+ [![Release](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml/badge.svg)](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
36
+ [![Docker](https://img.shields.io/badge/ghcr-devops--ai--agent-2496ED?logo=docker&logoColor=white)](https://github.com/DevOpsAbhii/devops-ai-agent/pkgs/container/devops-ai-agent)
37
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
38
+ [![Tests](https://img.shields.io/badge/tests-181%20offline-brightgreen)](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
39
+
32
40
  An AI agent that investigates real DevOps problems. The end goal: ask it
33
41
  something like *"Why is my Kubernetes pod in CrashLoopBackOff?"* and have it
34
42
  gather evidence, reason about the evidence, identify the likely root cause,
@@ -55,7 +63,14 @@ from environment config), and Ansible listing (inventory, playbook tasks).
55
63
  open alerts over NerdGraph — credentials from env), Trivy image
56
64
  vulnerability scanning, Helm releases (list/status/history), Argo CD
57
65
  (GitOps app sync/health), Istio mesh proxy status, and Docker Compose
58
- (project and service listing). The repository is git-tracked.
66
+ (project and service listing). **Phase 10 ships it as a product:** the
67
+ [`devopsiq` package on PyPI](https://pypi.org/project/devopsiq/) and a
68
+ prebuilt multi-arch Docker image (`ghcr.io/devopsabhii/devops-ai-agent`) —
69
+ both cut automatically by pushing a `v*` tag (test gate → PyPI via Trusted
70
+ Publishing + GHCR in parallel). **No API key? The agent still runs:** it
71
+ starts in model-less mode — record commands (`/report`, `/investigations`)
72
+ and the whole 58-tool layer work without a key; only questions to the model
73
+ need one. The repository is git-tracked.
59
74
  Every phase still built from scratch — no LangChain, LangGraph,
60
75
  AutoGen, CrewAI, or MCP.
61
76
 
@@ -73,7 +88,7 @@ model is called, how conversation history flows, how tool selection +
73
88
  execution + evidence feedback work — instead of depending on a framework
74
89
  for it.
75
90
 
76
- In Phase 5 the agent:
91
+ Today the agent:
77
92
  - holds a conversation with **GLM 5.3** through **OpenRouter**;
78
93
  - has **58 real, read-only tools** across fifteen domains: host facts,
79
94
  Kubernetes (12 tools), Linux system (4), Docker + Compose (10),
@@ -133,8 +148,8 @@ Module map:
133
148
 
134
149
  | Path | Responsibility |
135
150
  | --------------------------- | ------------------------------------------------------------ |
136
- | `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages |
137
- | `agent/agent.py` | `DevOpsAgent` — client, history, `ask()`, `_complete()` loop |
151
+ | `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages, model-less start |
152
+ | `agent/agent.py` | `DevOpsAgent` — client (`None` = model-less mode), history, `ask()`, `_complete()` loop |
138
153
  | `agent/prompts.py` | The system prompt (versioned/tested separately) |
139
154
  | `agent/investigation.py` | First-class investigation record: hypotheses, verdicts, evidence, report renderer (pure data) |
140
155
  | `agent/store.py` | `InvestigationStore` — one JSON file per record, atomic writes, resume/list (Phase 7) |
@@ -163,6 +178,11 @@ Module map:
163
178
  | `tests/test_phase7.py` | Offline suite: round-trip serialization, store files, auto-save, resume, CLI flags (Phase 7) |
164
179
  | `tests/test_phase8.py` | Offline suite: argv templates + validation for the 21 Phase 8 tools (fake CLIs, env-based monitoring) |
165
180
  | `tests/test_phase9.py` | Offline suite: New Relic env-credential + payload tests, trivy/helm/argocd/istio/compose argv templates (Phase 9) |
181
+ | `pyproject.toml` | Package `devopsiq`: metadata, MIT, console script `devopsiq = main:main` (Phase 10) |
182
+ | `Dockerfile` | Prebuilt image: slim base + kubectl/helm/trivy/gh + app, non-root, `/data` record store (Phase 10) |
183
+ | `.github/workflows/release.yml` | Tag-driven release: test gate → PyPI (Trusted Publishing) + GHCR multi-arch (Phase 10) |
184
+ | `docs/generate_pdf.py` | Builds the project documentation PDF from live source |
185
+ | `docs/generate_release_guide.py` | Builds the release & update playbook PDF |
166
186
 
167
187
  ### The tool-use loop
168
188
 
@@ -209,6 +229,7 @@ the CLI exposes it directly:
209
229
  | `/investigations` | List saved records on disk (newest first; `← active` marks the live one) |
210
230
  | `/report` | Show the canonical report (once concluded) |
211
231
  | `/endinvestigation` | Clear the record (memory only — the saved copy stays as history) |
232
+ | `/model [<name>]` | Show the active model, or save a default (config.json) and switch to it |
212
233
 
213
234
  The report the agent ends with and `/report` render are kept consistent by
214
235
  construction: `tool` results confirm each record call and the tracker, and
@@ -396,16 +417,22 @@ official `openai` Python SDK works as our client with two config lines:
396
417
  OpenAI(api_key=..., base_url="https://openrouter.ai/api/v1")
397
418
  ```
398
419
 
399
- Swapping to another OpenRouter model later is a one-line change (an env var
400
- today); moving to any other OpenAI-compatible provider changes only
401
- `agent/agent.py` configuration.
420
+ Swapping to another OpenRouter model later is a one-liner (env var, `/model`
421
+ command, or `--model` flag); moving to any other OpenAI-compatible provider
422
+ changes only `agent/agent.py` configuration.
402
423
 
403
424
  **Your model, your choice.** The default is baked in as a fallback, never a
404
- restriction — set `OPENROUTER_MODEL` (shell or `.env`) to any model on
405
- OpenRouter:
425
+ restriction. Four ways to set the model — the first one set wins:
426
+
427
+ 1. `--model NAME` flag (one run: `devopsiq --model openai/gpt-5.2 "why?"`)
428
+ 2. `/model <name>` in the REPL — saves it to `~/.devops-ai-agent/config.json`
429
+ so every future run uses it
430
+ 3. `OPENROUTER_MODEL` (shell or `.env`)
431
+ 4. built-in default (`z-ai/glm-5.3`)
406
432
 
407
433
  ```bash
408
434
  export OPENROUTER_MODEL=anthropic/claude-sonnet-5 # or openai/gpt-5.2, google/gemini-2.5-pro, ...
435
+ devopsiq /model openai/gpt-5.2 # or save a default from the REPL
409
436
  ```
410
437
 
411
438
  Two things to weigh when picking: the agent is a tool-use loop, so choose a
@@ -558,6 +585,7 @@ setup/API errors — so it drops straight into a pipeline:
558
585
  .venv/bin/python main.py --json "why is api-5d6f crash-looping?" # structured report
559
586
  .venv/bin/python main.py --resume --json "any update?" # continue a prior run
560
587
  .venv/bin/python main.py --store-dir /tmp/runs --out report.json "..." # pipeline paths
588
+ .venv/bin/python main.py --model openai/gpt-5.2 "why is it down?" # one-run model override
561
589
  ```
562
590
 
563
591
  With `--json` the stdout is one JSON document (see `render_report_json` in
@@ -599,8 +627,8 @@ Tools: ansible_inventory, ansible_playbook_tasks, aws_identity,
599
627
  k8s_top_nodes, k8s_top_pods, loki_query, prom_query,
600
628
  sys_open_ports, sys_service_logs, sys_service_status,
601
629
  sys_top_processes, system_info, tf_plan, tf_show, tf_state_list
602
- Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation
603
- One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] "<problem>"
630
+ Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation, /model [<name>]
631
+ One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] [--model NAME] "<problem>"
604
632
  Type 'exit' to quit.
605
633
 
606
634
  You: The checkout service container keeps exiting in Docker. Investigate.
@@ -617,7 +645,7 @@ You: exit
617
645
  Type `exit` / `quit`, or press Ctrl-D / Ctrl-C to leave. Slash commands are
618
646
  handled locally and never reach the model.
619
647
 
620
- ## 8. Current limitations (Phase 9)
648
+ ## 8. Current limitations
621
649
 
622
650
  - **Each domain needs its CLI installed and reachable.** Missing CLIs,
623
651
  unauthenticated `gh`, a dead docker daemon, an uninitialized terraform
@@ -702,8 +730,22 @@ handled locally and never reach the model.
702
730
  `helm_history` — reads only), Argo CD (`argocd_apps`,
703
731
  `argocd_app_status`), Istio (`istioctl_proxy_status`), Docker Compose
704
732
  (`docker_compose_ls`, `docker_compose_ps` — live-verified on this host).
733
+ - **Phase 10 — distribution. ✅ Done.** The `devopsiq` package on PyPI
734
+ (console command `devopsiq`, MIT) and the prebuilt multi-arch GHCR image,
735
+ cut by a tag-driven release workflow: an offline test gate, then PyPI
736
+ (Trusted Publishing — no token in the repo) and GHCR (amd64 + arm64) in
737
+ parallel. **`v0.1.1` added model-less mode:** with no API key the agent
738
+ still constructs — record commands (`/report`, `/investigations`) and the
739
+ 58-tool layer work; only model questions exit 1 with the setup message.
740
+ Verified end-to-end against the published package.
741
+ - **Model choice + resilient calls (v0.1.2). ✅ Done.** The model is
742
+ set by a 4-rung ladder — `--model` flag > `/model`-saved config file
743
+ (`~/.devops-ai-agent/config.json`) > `OPENROUTER_MODEL` > baked-in
744
+ default — and transient model-call failures (timeouts, connection
745
+ errors, 429s, 5xx) retry with exponential backoff (2s→4s→8s + jitter,
746
+ honoring a 429's Retry-After); a rejected key fails immediately.
705
747
  - **Later — region-scoped cloud resources** (ec2 describe-*, compute
706
748
  instances list, ...) behind the same template pattern; more observability
707
749
  depth (New Relic dashboards/entities, Prometheus range queries); streaming;
708
- conversation-history persistence; and a human-approval gate before any
709
- mutating action is ever allowed.
750
+ conversation-history persistence; and a human-approval gate
751
+ before any mutating action is ever allowed.
@@ -4,6 +4,7 @@ main.py
4
4
  pyproject.toml
5
5
  agent/__init__.py
6
6
  agent/agent.py
7
+ agent/config.py
7
8
  agent/investigation.py
8
9
  agent/prompts.py
9
10
  agent/store.py
@@ -10,6 +10,7 @@ One-shot (for cron / CI / scripts):
10
10
  python main.py --json "why is api-5d6f crash-looping?" # structured report
11
11
  python main.py --resume --json "any update?" # continue prior run
12
12
  python main.py --store-dir DIR --out report.json "..." # pipeline paths
13
+ python main.py --model openai/gpt-4o-mini "..." # one-run model override
13
14
 
14
15
  One-shot mode sends the problem once, prints the agent's report, and exits
15
16
  with a status code (0 = completed, 1 = setup/API error). With --json the
@@ -41,6 +42,7 @@ from openai import (
41
42
  )
42
43
 
43
44
  from agent.agent import DevOpsAgent
45
+ from agent.config import config_path, save_user_config
44
46
 
45
47
  EXIT_WORDS = {"exit", "quit"}
46
48
 
@@ -55,6 +57,7 @@ COMMAND_ALIASES = {
55
57
  "/report": "/report",
56
58
  "/endinvestigation": "/endinvestigation",
57
59
  "/end": "/endinvestigation",
60
+ "/model": "/model",
58
61
  }
59
62
 
60
63
 
@@ -82,8 +85,32 @@ def handle_command(text: str, agent: DevOpsAgent) -> str | None:
82
85
  return agent.investigation_report_text() or "(no investigation recorded)"
83
86
  if canonical == "/endinvestigation":
84
87
  return agent.end_investigation()
88
+ if canonical == "/model":
89
+ return handle_model_command(rest, agent)
85
90
  return (f"unknown command: {cmd}. Try /investigate <problem>, "
86
- "/investigation, /investigations, /report, /endinvestigation")
91
+ "/investigation, /investigations, /report, /endinvestigation, "
92
+ "/model")
93
+
94
+
95
+ def handle_model_command(rest: str, agent: DevOpsAgent) -> str:
96
+ """/model (show the active model and where it came from) or /model <name>.
97
+
98
+ With a name: save it as the user's preference in the config file
99
+ (~/.devops-ai-agent/config.json) and switch this session to it. The
100
+ next run picks it up automatically — the flag and env still outrank it.
101
+ """
102
+ name = rest.strip()
103
+ if not name:
104
+ return (f"Model: {agent.model}\n"
105
+ f"Config file: {config_path()} "
106
+ "(save a default with /model <name>)")
107
+ if len(name.split()) != 1:
108
+ return "Model name must be a single token, e.g. /model openai/gpt-4o-mini"
109
+ path = save_user_config({"model": name})
110
+ agent.model = name
111
+ return (f"Model set to {name} (saved in {path}).\n"
112
+ "This session now uses it; the --model flag and OPENROUTER_MODEL "
113
+ "still override it per run.")
87
114
 
88
115
 
89
116
  class OneShotArgs(NamedTuple):
@@ -94,9 +121,10 @@ class OneShotArgs(NamedTuple):
94
121
  store_dir: str | None # --store-dir PATH (persistence override)
95
122
  resume: bool # --resume (continue the newest in-progress record)
96
123
  out: str | None # --out PATH (also write the JSON report there)
124
+ model: str | None # --model NAME (one-run model override)
97
125
 
98
126
 
99
- _VALUE_FLAGS = ("--store-dir", "--out")
127
+ _VALUE_FLAGS = ("--store-dir", "--out", "--model")
100
128
 
101
129
 
102
130
  def _flag_value(argv: list[str], flag: str) -> str | None:
@@ -114,14 +142,16 @@ def parse_args(argv: list[str]) -> OneShotArgs:
114
142
  --json switches the one-shot output from markdown to the structured JSON
115
143
  report; --resume continues the newest in-progress record from the store;
116
144
  --store-dir PATH overrides the persistence directory for this run; --out
117
- PATH additionally writes the JSON report to an exact path. Example:
145
+ PATH additionally writes the JSON report to an exact path; --model NAME
146
+ overrides the model for this run (highest model precedence). Example:
118
147
  `python main.py --json --out r.json "why is it down?"` ->
119
- ("why is it down?", True, None, False, "r.json").
148
+ ("why is it down?", True, None, False, "r.json", None).
120
149
  """
121
150
  as_json = "--json" in argv
122
151
  do_resume = "--resume" in argv
123
152
  store_dir = _flag_value(argv, "--store-dir")
124
153
  out = _flag_value(argv, "--out")
154
+ model = _flag_value(argv, "--model")
125
155
  positionals: list[str] = []
126
156
  skip_next = False
127
157
  for arg in argv:
@@ -136,8 +166,8 @@ def parse_args(argv: list[str]) -> OneShotArgs:
136
166
  positionals.append(arg)
137
167
  text = " ".join(positionals).strip()
138
168
  if not positionals or not text:
139
- return OneShotArgs(None, as_json, store_dir, do_resume, out)
140
- return OneShotArgs(text, as_json, store_dir, do_resume, out)
169
+ return OneShotArgs(None, as_json, store_dir, do_resume, out, model)
170
+ return OneShotArgs(text, as_json, store_dir, do_resume, out, model)
141
171
 
142
172
 
143
173
  def run_one_shot(
@@ -147,6 +177,7 @@ def run_one_shot(
147
177
  store_dir: str | None = None,
148
178
  resume: bool = False,
149
179
  out: str | None = None,
180
+ model: str | None = None,
150
181
  ) -> int:
151
182
  """Non-interactive single run: send `task`, print the result, exit cleanly.
152
183
 
@@ -155,14 +186,14 @@ def run_one_shot(
155
186
  opened an investigation this is a hard failure (exit 1) — a caller asked
156
187
  for a report and there is none to give. The record is auto-saved on every
157
188
  mutation regardless; store_dir points persistence somewhere else for this
158
- run, resume continues the newest in-progress record before asking, and
159
- out additionally writes the JSON report to an exact path. `agent` lets
160
- embedders reuse a configured agent (also the test seam); default builds a
161
- fresh one from the environment.
189
+ run, resume continues the newest in-progress record before asking, out
190
+ additionally writes the JSON report to an exact path, and model overrides
191
+ the model for this run. `agent` lets embedders reuse a configured agent
192
+ (also the test seam); default builds a fresh one from the environment.
162
193
  """
163
194
  if agent is None:
164
195
  try:
165
- agent = DevOpsAgent()
196
+ agent = DevOpsAgent(model=model)
166
197
  except ValueError as exc:
167
198
  print(f"[setup] {exc}", file=sys.stderr)
168
199
  return 1
@@ -245,10 +276,11 @@ def main() -> int:
245
276
  store_dir=args.store_dir,
246
277
  resume=args.resume,
247
278
  out=args.out,
279
+ model=args.model,
248
280
  )
249
281
 
250
282
  try:
251
- agent = DevOpsAgent()
283
+ agent = DevOpsAgent(model=args.model)
252
284
  except ValueError as exc:
253
285
  print(f"[setup] {exc}", file=sys.stderr)
254
286
  return 1
@@ -268,9 +300,9 @@ def main() -> int:
268
300
  print(f"Store: {agent.store_dir or '(persistence off)'}")
269
301
  print(f"Tools: {tool_names}")
270
302
  print("Commands: /investigate <problem>, /investigation, /investigations, "
271
- "/report, /endinvestigation")
303
+ "/report, /endinvestigation, /model [<name>]")
272
304
  print("One-shot: python main.py [--json] [--resume] [--out report.json] "
273
- "[--store-dir DIR] \"<problem>\"")
305
+ "[--store-dir DIR] [--model NAME] \"<problem>\"")
274
306
  print("Type 'exit' to quit.")
275
307
  print()
276
308
 
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "devopsiq"
7
- version = "0.1.1"
7
+ version = "0.1.2"
8
8
  description = "A from-scratch, read-only AI agent that investigates real DevOps problems — Kubernetes, Docker, Helm, Argo CD, Istio, Terraform, GitHub Actions, cloud, New Relic and more."
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -38,6 +38,7 @@ dependencies = [
38
38
  Homepage = "https://github.com/DevOpsAbhii/devops-ai-agent"
39
39
  Repository = "https://github.com/DevOpsAbhii/devops-ai-agent"
40
40
  Issues = "https://github.com/DevOpsAbhii/devops-ai-agent/issues"
41
+ Changelog = "https://github.com/DevOpsAbhii/devops-ai-agent/blob/main/CHANGELOG.md"
41
42
 
42
43
  [project.scripts]
43
44
  devopsiq = "main:main"
@@ -11,12 +11,25 @@ import contextlib
11
11
  import io
12
12
  import json
13
13
  import os
14
+ import tempfile
15
+ import time
14
16
  import types
15
17
  import unittest
18
+ from pathlib import Path
16
19
 
20
+ import httpx2 as httpx # openai 3.19 depends on the httpx2 fork
17
21
  import main as main_module
18
- from agent.agent import DevOpsAgent
22
+ import agent.agent as agent_module
23
+ from agent import config as user_config
24
+ from agent.agent import DEFAULT_MODEL, DevOpsAgent
19
25
  from agent.investigation import Investigation
26
+ from openai import (
27
+ APIConnectionError,
28
+ APIStatusError,
29
+ APITimeoutError,
30
+ AuthenticationError,
31
+ RateLimitError,
32
+ )
20
33
  from tools import investigation as inv_tools
21
34
 
22
35
 
@@ -187,19 +200,27 @@ class OneShotCliTests(unittest.TestCase):
187
200
 
188
201
  def test_parse_args(self):
189
202
  self.assertEqual(
190
- main_module.parse_args([]), (None, False, None, False, None)
203
+ main_module.parse_args([]), (None, False, None, False, None, None)
191
204
  )
192
205
  self.assertEqual(
193
206
  main_module.parse_args(["why is it down?"]),
194
- ("why is it down?", False, None, False, None),
207
+ ("why is it down?", False, None, False, None, None),
195
208
  )
196
209
  self.assertEqual(
197
210
  main_module.parse_args(["--json", "why", "is it down?"]),
198
- ("why is it down?", True, None, False, None),
211
+ ("why is it down?", True, None, False, None, None),
199
212
  )
200
213
  self.assertEqual(
201
214
  main_module.parse_args(["--json"]),
202
- (None, True, None, False, None),
215
+ (None, True, None, False, None, None),
216
+ )
217
+ self.assertEqual(
218
+ main_module.parse_args(["--model", "openai/gpt-4o-mini", "why?"]),
219
+ ("why?", False, None, False, None, "openai/gpt-4o-mini"),
220
+ )
221
+ self.assertEqual(
222
+ main_module.parse_args(["--model", "anthropic/claude-sonnet-5"]),
223
+ (None, False, None, False, None, "anthropic/claude-sonnet-5"),
203
224
  )
204
225
 
205
226
  def test_slash_command_routes_without_model(self):
@@ -300,5 +321,207 @@ class KeylessTests(unittest.TestCase):
300
321
  self.assertIn("OPENROUTER_API_KEY", err.getvalue())
301
322
 
302
323
 
324
+ class ModelChoiceTests(unittest.TestCase):
325
+ """Model preference ladder + the /model command + agent/config.py.
326
+
327
+ Precedence, highest first: explicit model argument > config.json >
328
+ OPENROUTER_MODEL env > built-in default. Tests point AGENT_CONFIG_FILE
329
+ at a tmp file so the developer's real ~/.devops-ai-agent/config.json
330
+ is never read or written.
331
+ """
332
+
333
+ def setUp(self):
334
+ tmp = tempfile.NamedTemporaryFile(suffix=".json", delete=False)
335
+ tmp.close()
336
+ self.config_file = tmp.name
337
+ os.environ["AGENT_CONFIG_FILE"] = self.config_file
338
+ self.addCleanup(self._restore)
339
+
340
+ def _restore(self):
341
+ os.environ.pop("AGENT_CONFIG_FILE", None)
342
+ Path(self.config_file).unlink(missing_ok=True)
343
+
344
+ def _set_env(self, name, value):
345
+ old = os.environ.get(name)
346
+ os.environ[name] = value
347
+ self.addCleanup(lambda: (
348
+ os.environ.pop(name, None) if old is None
349
+ else os.environ.__setitem__(name, old)
350
+ ))
351
+
352
+ def test_config_round_trip_and_corrupt_tolerance(self):
353
+ self.assertEqual(user_config.load_user_config(), {}) # missing file
354
+ path = user_config.save_user_config({"model": "openai/gpt-4o-mini"})
355
+ self.assertEqual(path, Path(self.config_file))
356
+ self.assertEqual(user_config.load_user_config(),
357
+ {"model": "openai/gpt-4o-mini"})
358
+ # Merging keeps existing keys and overwrites the given one.
359
+ user_config.save_user_config({"model": "anthropic/claude-sonnet-5"})
360
+ self.assertEqual(user_config.load_user_config(),
361
+ {"model": "anthropic/claude-sonnet-5"})
362
+ # Corrupt content degrades to {} instead of raising.
363
+ Path(self.config_file).write_text("{not json", encoding="utf-8")
364
+ self.assertEqual(user_config.load_user_config(), {})
365
+
366
+ def test_model_precedence_ladder(self):
367
+ # 4. built-in default
368
+ self.assertEqual(DevOpsAgent().model, DEFAULT_MODEL)
369
+ # 3. environment
370
+ self._set_env("OPENROUTER_MODEL", "env/model")
371
+ self.assertEqual(DevOpsAgent().model, "env/model")
372
+ # 2. config file (outranks env)
373
+ user_config.save_user_config({"model": "config/model"})
374
+ self.assertEqual(DevOpsAgent().model, "config/model")
375
+ # 1. explicit argument (outranks everything)
376
+ self.assertEqual(DevOpsAgent(model="flag/model").model, "flag/model")
377
+
378
+ def test_model_command_shows_current_model(self):
379
+ agent = DevOpsAgent()
380
+ before = agent.model
381
+ text = main_module.handle_command("/model", agent)
382
+ self.assertIn(before, text)
383
+ self.assertIn(str(user_config.config_path()), text)
384
+ self.assertEqual(agent.model, before) # show-only changes nothing
385
+
386
+ def test_model_command_saves_and_switches(self):
387
+ agent = DevOpsAgent()
388
+ out = io.StringIO()
389
+ with contextlib.redirect_stdout(out):
390
+ text = main_module.handle_command(
391
+ "/model openai/gpt-4o-mini", agent)
392
+ self.assertIn("openai/gpt-4o-mini", text)
393
+ self.assertEqual(agent.model, "openai/gpt-4o-mini")
394
+ self.assertEqual(
395
+ user_config.load_user_config(), {"model": "openai/gpt-4o-mini"})
396
+ # A multi-token name is rejected; the config stays untouched.
397
+ text = main_module.handle_command("/model two words", agent)
398
+ self.assertIn("single token", text)
399
+ self.assertEqual(agent.model, "openai/gpt-4o-mini")
400
+
401
+
402
+ class FlakyChat:
403
+ """Chat stub that raises the queued errors, then answers with content."""
404
+
405
+ def __init__(self, errors: list[Exception], content: str = "ok"):
406
+ self._errors = list(errors)
407
+ self._content = content
408
+ self.calls = 0
409
+
410
+ def create(self, **kwargs): # noqa: D102
411
+ self.calls += 1
412
+ if self._errors:
413
+ raise self._errors.pop(0)
414
+ message = types.SimpleNamespace(content=self._content, tool_calls=None)
415
+ return types.SimpleNamespace(
416
+ choices=[types.SimpleNamespace(message=message)]
417
+ )
418
+
419
+
420
+ def _api_status_error(status: int, headers: dict | None = None):
421
+ """Build an openai status error offline (no HTTP round-trip)."""
422
+ request = httpx.Request(
423
+ "POST", "https://openrouter.ai/api/v1/chat/completions")
424
+ response = httpx.Response(
425
+ status, request=request, headers=headers or {})
426
+ return APIStatusError("boom", response=response, body=None)
427
+
428
+
429
+ def _api_connection_error():
430
+ request = httpx.Request(
431
+ "POST", "https://openrouter.ai/api/v1/chat/completions")
432
+ return APIConnectionError(request=request)
433
+
434
+
435
+ class RetryTests(unittest.TestCase):
436
+ """_call_model: transient failures retry with backoff, others fail fast.
437
+
438
+ Sleeps are intercepted via the agent.agent._sleep seam, so the tests
439
+ run instantly; the recorded delays prove the backoff schedule.
440
+ """
441
+
442
+ def setUp(self):
443
+ self._old_key = os.environ.get("OPENROUTER_API_KEY")
444
+ os.environ["OPENROUTER_API_KEY"] = "test-key-for-offline-tests"
445
+ self.agent = DevOpsAgent()
446
+ self.sleeps: list[float] = []
447
+ agent_module._sleep = self.sleeps.append
448
+ self.addCleanup(self._restore)
449
+
450
+ def _restore(self):
451
+ agent_module._sleep = time.sleep
452
+ os.environ.pop("OPENROUTER_API_KEY", None)
453
+ if self._old_key is not None:
454
+ os.environ["OPENROUTER_API_KEY"] = self._old_key
455
+
456
+ def test_connection_errors_are_retried_then_succeed(self):
457
+ chat = FlakyChat([_api_connection_error(), _api_connection_error()])
458
+ self.agent.client.chat.completions = chat
459
+ self.assertEqual(self.agent.ask("why is it down?"), "ok")
460
+ self.assertEqual(chat.calls, 3) # original + 2 retries
461
+ # Backoff: attempt 0 -> ~2s (+jitter), attempt 1 -> ~4s (+jitter).
462
+ self.assertGreaterEqual(self.sleeps[0], 2.0)
463
+ self.assertLessEqual(self.sleeps[0], 3.0)
464
+ self.assertGreaterEqual(self.sleeps[1], 4.0)
465
+ self.assertLessEqual(self.sleeps[1], 5.0)
466
+
467
+ def test_timeout_is_transient(self):
468
+ request = httpx.Request("POST", "https://openrouter.ai/api/v1")
469
+ chat = FlakyChat([APITimeoutError(request)])
470
+ self.agent.client.chat.completions = chat
471
+ self.assertEqual(self.agent.ask("hello"), "ok")
472
+ self.assertEqual(chat.calls, 2)
473
+
474
+ def test_rate_limit_honors_retry_after_header(self):
475
+ request = httpx.Request("POST", "https://openrouter.ai/api/v1")
476
+ response = httpx.Response(
477
+ 429, request=request, headers={"retry-after": "7"})
478
+ chat = FlakyChat([RateLimitError("slow down", response=response,
479
+ body=None)])
480
+ self.agent.client.chat.completions = chat
481
+ self.assertEqual(self.agent.ask("hello"), "ok")
482
+ self.assertEqual(self.sleeps, [7.0])
483
+
484
+ def test_rate_limit_without_header_uses_backoff(self):
485
+ request = httpx.Request("POST", "https://openrouter.ai/api/v1")
486
+ response = httpx.Response(429, request=request)
487
+ chat = FlakyChat([RateLimitError("slow down", response=response,
488
+ body=None)])
489
+ self.agent.client.chat.completions = chat
490
+ self.assertEqual(self.agent.ask("hello"), "ok")
491
+ self.assertGreaterEqual(self.sleeps[0], 2.0)
492
+
493
+ def test_auth_error_fails_immediately(self):
494
+ request = httpx.Request("POST", "https://openrouter.ai/api/v1")
495
+ response = httpx.Response(401, request=request)
496
+ chat = FlakyChat([AuthenticationError("bad key", response=response,
497
+ body=None)])
498
+ self.agent.client.chat.completions = chat
499
+ with self.assertRaises(AuthenticationError):
500
+ self.agent.ask("hello")
501
+ self.assertEqual(chat.calls, 1) # no retry on a rejected key
502
+ self.assertEqual(self.sleeps, [])
503
+
504
+ def test_client_400_fails_immediately(self):
505
+ chat = FlakyChat([_api_status_error(400)])
506
+ self.agent.client.chat.completions = chat
507
+ with self.assertRaises(APIStatusError):
508
+ self.agent.ask("hello")
509
+ self.assertEqual(chat.calls, 1)
510
+
511
+ def test_server_500_is_retried(self):
512
+ chat = FlakyChat([_api_status_error(500), _api_status_error(502)])
513
+ self.agent.client.chat.completions = chat
514
+ self.assertEqual(self.agent.ask("hello"), "ok")
515
+ self.assertEqual(chat.calls, 3)
516
+
517
+ def test_retries_are_exhausted_loudly(self):
518
+ chat = FlakyChat([_api_connection_error() for _ in range(4)])
519
+ self.agent.client.chat.completions = chat
520
+ with self.assertRaises(APIConnectionError):
521
+ self.agent.ask("hello")
522
+ self.assertEqual(chat.calls, 4) # original + 3 retries, then raise
523
+ self.assertEqual(len(self.sleeps), 3) # 2s, 4s, 8s schedule
524
+
525
+
303
526
  if __name__ == "__main__":
304
527
  unittest.main()
@@ -329,11 +329,11 @@ class CliTests(unittest.TestCase):
329
329
 
330
330
  def test_parse_args_defaults(self):
331
331
  self.assertEqual(
332
- main_module.parse_args([]), (None, False, None, False, None)
332
+ main_module.parse_args([]), (None, False, None, False, None, None)
333
333
  )
334
334
  self.assertEqual(
335
335
  main_module.parse_args(["why is it down?"]),
336
- ("why is it down?", False, None, False, None),
336
+ ("why is it down?", False, None, False, None, None),
337
337
  )
338
338
 
339
339
  def test_parse_args_all_flags(self):
@@ -341,18 +341,19 @@ class CliTests(unittest.TestCase):
341
341
  ["--json", "--resume", "--store-dir", "/tmp/s", "--out", "r.json",
342
342
  "why", "is it down?"]
343
343
  )
344
- self.assertEqual(args, ("why is it down?", True, "/tmp/s", True, "r.json"))
344
+ self.assertEqual(
345
+ args, ("why is it down?", True, "/tmp/s", True, "r.json", None))
345
346
 
346
347
  def test_parse_args_flags_only_runs_repl(self):
347
348
  self.assertEqual(
348
349
  main_module.parse_args(["--json", "--resume"]),
349
- (None, True, None, True, None),
350
+ (None, True, None, True, None, None),
350
351
  )
351
352
 
352
353
  def test_parse_args_flag_without_value_is_ignored(self):
353
354
  self.assertEqual(
354
355
  main_module.parse_args(["--store-dir"]),
355
- (None, False, None, False, None),
356
+ (None, False, None, False, None, None),
356
357
  )
357
358
 
358
359
  # --- --store-dir ----------------------------------------------------------
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes