devopsiq 0.1.0__tar.gz → 0.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. {devopsiq-0.1.0/devopsiq.egg-info → devopsiq-0.1.2}/PKG-INFO +103 -14
  2. {devopsiq-0.1.0 → devopsiq-0.1.2}/README.md +101 -13
  3. {devopsiq-0.1.0 → devopsiq-0.1.2}/agent/agent.py +101 -14
  4. devopsiq-0.1.2/agent/config.py +45 -0
  5. {devopsiq-0.1.0 → devopsiq-0.1.2/devopsiq.egg-info}/PKG-INFO +103 -14
  6. {devopsiq-0.1.0 → devopsiq-0.1.2}/devopsiq.egg-info/SOURCES.txt +1 -0
  7. {devopsiq-0.1.0 → devopsiq-0.1.2}/main.py +53 -14
  8. {devopsiq-0.1.0 → devopsiq-0.1.2}/pyproject.toml +2 -1
  9. devopsiq-0.1.2/tests/test_automation.py +527 -0
  10. {devopsiq-0.1.0 → devopsiq-0.1.2}/tests/test_phase7.py +6 -5
  11. devopsiq-0.1.0/tests/test_automation.py +0 -251
  12. {devopsiq-0.1.0 → devopsiq-0.1.2}/LICENSE +0 -0
  13. {devopsiq-0.1.0 → devopsiq-0.1.2}/agent/__init__.py +0 -0
  14. {devopsiq-0.1.0 → devopsiq-0.1.2}/agent/investigation.py +0 -0
  15. {devopsiq-0.1.0 → devopsiq-0.1.2}/agent/prompts.py +0 -0
  16. {devopsiq-0.1.0 → devopsiq-0.1.2}/agent/store.py +0 -0
  17. {devopsiq-0.1.0 → devopsiq-0.1.2}/devopsiq.egg-info/dependency_links.txt +0 -0
  18. {devopsiq-0.1.0 → devopsiq-0.1.2}/devopsiq.egg-info/entry_points.txt +0 -0
  19. {devopsiq-0.1.0 → devopsiq-0.1.2}/devopsiq.egg-info/requires.txt +0 -0
  20. {devopsiq-0.1.0 → devopsiq-0.1.2}/devopsiq.egg-info/top_level.txt +0 -0
  21. {devopsiq-0.1.0 → devopsiq-0.1.2}/setup.cfg +0 -0
  22. {devopsiq-0.1.0 → devopsiq-0.1.2}/tests/test_phase2.py +0 -0
  23. {devopsiq-0.1.0 → devopsiq-0.1.2}/tests/test_phase3.py +0 -0
  24. {devopsiq-0.1.0 → devopsiq-0.1.2}/tests/test_phase4.py +0 -0
  25. {devopsiq-0.1.0 → devopsiq-0.1.2}/tests/test_phase5.py +0 -0
  26. {devopsiq-0.1.0 → devopsiq-0.1.2}/tests/test_phase8.py +0 -0
  27. {devopsiq-0.1.0 → devopsiq-0.1.2}/tests/test_phase9.py +0 -0
  28. {devopsiq-0.1.0 → devopsiq-0.1.2}/tools/__init__.py +0 -0
  29. {devopsiq-0.1.0 → devopsiq-0.1.2}/tools/ansible.py +0 -0
  30. {devopsiq-0.1.0 → devopsiq-0.1.2}/tools/argocd.py +0 -0
  31. {devopsiq-0.1.0 → devopsiq-0.1.2}/tools/base.py +0 -0
  32. {devopsiq-0.1.0 → devopsiq-0.1.2}/tools/cloud.py +0 -0
  33. {devopsiq-0.1.0 → devopsiq-0.1.2}/tools/docker.py +0 -0
  34. {devopsiq-0.1.0 → devopsiq-0.1.2}/tools/git_ci.py +0 -0
  35. {devopsiq-0.1.0 → devopsiq-0.1.2}/tools/helm.py +0 -0
  36. {devopsiq-0.1.0 → devopsiq-0.1.2}/tools/investigation.py +0 -0
  37. {devopsiq-0.1.0 → devopsiq-0.1.2}/tools/istio.py +0 -0
  38. {devopsiq-0.1.0 → devopsiq-0.1.2}/tools/kubernetes.py +0 -0
  39. {devopsiq-0.1.0 → devopsiq-0.1.2}/tools/monitoring.py +0 -0
  40. {devopsiq-0.1.0 → devopsiq-0.1.2}/tools/newrelic.py +0 -0
  41. {devopsiq-0.1.0 → devopsiq-0.1.2}/tools/preflight.py +0 -0
  42. {devopsiq-0.1.0 → devopsiq-0.1.2}/tools/registry.py +0 -0
  43. {devopsiq-0.1.0 → devopsiq-0.1.2}/tools/system.py +0 -0
  44. {devopsiq-0.1.0 → devopsiq-0.1.2}/tools/terraform.py +0 -0
  45. {devopsiq-0.1.0 → devopsiq-0.1.2}/tools/trivy.py +0 -0
@@ -1,12 +1,13 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: devopsiq
3
- Version: 0.1.0
3
+ Version: 0.1.2
4
4
  Summary: A from-scratch, read-only AI agent that investigates real DevOps problems — Kubernetes, Docker, Helm, Argo CD, Istio, Terraform, GitHub Actions, cloud, New Relic and more.
5
5
  Author: DevOpsAbhii
6
6
  License-Expression: MIT
7
7
  Project-URL: Homepage, https://github.com/DevOpsAbhii/devops-ai-agent
8
8
  Project-URL: Repository, https://github.com/DevOpsAbhii/devops-ai-agent
9
9
  Project-URL: Issues, https://github.com/DevOpsAbhii/devops-ai-agent/issues
10
+ Project-URL: Changelog, https://github.com/DevOpsAbhii/devops-ai-agent/blob/main/CHANGELOG.md
10
11
  Keywords: devops,kubernetes,docker,helm,argocd,istio,terraform,github-actions,observability,newrelic,incident-response,root-cause-analysis,ai-agent,cli
11
12
  Classifier: Development Status :: 4 - Beta
12
13
  Classifier: Environment :: Console
@@ -29,6 +30,13 @@ Dynamic: license-file
29
30
 
30
31
  # DevOps AI Agent
31
32
 
33
+ [![PyPI](https://img.shields.io/pypi/v/devopsiq)](https://pypi.org/project/devopsiq/)
34
+ [![Python](https://img.shields.io/pypi/pyversions/devopsiq)](https://pypi.org/project/devopsiq/)
35
+ [![Release](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml/badge.svg)](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
36
+ [![Docker](https://img.shields.io/badge/ghcr-devops--ai--agent-2496ED?logo=docker&logoColor=white)](https://github.com/DevOpsAbhii/devops-ai-agent/pkgs/container/devops-ai-agent)
37
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
38
+ [![Tests](https://img.shields.io/badge/tests-181%20offline-brightgreen)](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
39
+
32
40
  An AI agent that investigates real DevOps problems. The end goal: ask it
33
41
  something like *"Why is my Kubernetes pod in CrashLoopBackOff?"* and have it
34
42
  gather evidence, reason about the evidence, identify the likely root cause,
@@ -55,7 +63,14 @@ from environment config), and Ansible listing (inventory, playbook tasks).
55
63
  open alerts over NerdGraph — credentials from env), Trivy image
56
64
  vulnerability scanning, Helm releases (list/status/history), Argo CD
57
65
  (GitOps app sync/health), Istio mesh proxy status, and Docker Compose
58
- (project and service listing). The repository is git-tracked.
66
+ (project and service listing). **Phase 10 ships it as a product:** the
67
+ [`devopsiq` package on PyPI](https://pypi.org/project/devopsiq/) and a
68
+ prebuilt multi-arch Docker image (`ghcr.io/devopsabhii/devops-ai-agent`) —
69
+ both cut automatically by pushing a `v*` tag (test gate → PyPI via Trusted
70
+ Publishing + GHCR in parallel). **No API key? The agent still runs:** it
71
+ starts in model-less mode — record commands (`/report`, `/investigations`)
72
+ and the whole 58-tool layer work without a key; only questions to the model
73
+ need one. The repository is git-tracked.
59
74
  Every phase still built from scratch — no LangChain, LangGraph,
60
75
  AutoGen, CrewAI, or MCP.
61
76
 
@@ -73,7 +88,7 @@ model is called, how conversation history flows, how tool selection +
73
88
  execution + evidence feedback work — instead of depending on a framework
74
89
  for it.
75
90
 
76
- In Phase 5 the agent:
91
+ Today the agent:
77
92
  - holds a conversation with **GLM 5.3** through **OpenRouter**;
78
93
  - has **58 real, read-only tools** across fifteen domains: host facts,
79
94
  Kubernetes (12 tools), Linux system (4), Docker + Compose (10),
@@ -133,8 +148,8 @@ Module map:
133
148
 
134
149
  | Path | Responsibility |
135
150
  | --------------------------- | ------------------------------------------------------------ |
136
- | `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages |
137
- | `agent/agent.py` | `DevOpsAgent` — client, history, `ask()`, `_complete()` loop |
151
+ | `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages, model-less start |
152
+ | `agent/agent.py` | `DevOpsAgent` — client (`None` = model-less mode), history, `ask()`, `_complete()` loop |
138
153
  | `agent/prompts.py` | The system prompt (versioned/tested separately) |
139
154
  | `agent/investigation.py` | First-class investigation record: hypotheses, verdicts, evidence, report renderer (pure data) |
140
155
  | `agent/store.py` | `InvestigationStore` — one JSON file per record, atomic writes, resume/list (Phase 7) |
@@ -163,6 +178,11 @@ Module map:
163
178
  | `tests/test_phase7.py` | Offline suite: round-trip serialization, store files, auto-save, resume, CLI flags (Phase 7) |
164
179
  | `tests/test_phase8.py` | Offline suite: argv templates + validation for the 21 Phase 8 tools (fake CLIs, env-based monitoring) |
165
180
  | `tests/test_phase9.py` | Offline suite: New Relic env-credential + payload tests, trivy/helm/argocd/istio/compose argv templates (Phase 9) |
181
+ | `pyproject.toml` | Package `devopsiq`: metadata, MIT, console script `devopsiq = main:main` (Phase 10) |
182
+ | `Dockerfile` | Prebuilt image: slim base + kubectl/helm/trivy/gh + app, non-root, `/data` record store (Phase 10) |
183
+ | `.github/workflows/release.yml` | Tag-driven release: test gate → PyPI (Trusted Publishing) + GHCR multi-arch (Phase 10) |
184
+ | `docs/generate_pdf.py` | Builds the project documentation PDF from live source |
185
+ | `docs/generate_release_guide.py` | Builds the release & update playbook PDF |
166
186
 
167
187
  ### The tool-use loop
168
188
 
@@ -209,6 +229,7 @@ the CLI exposes it directly:
209
229
  | `/investigations` | List saved records on disk (newest first; `← active` marks the live one) |
210
230
  | `/report` | Show the canonical report (once concluded) |
211
231
  | `/endinvestigation` | Clear the record (memory only — the saved copy stays as history) |
232
+ | `/model [<name>]` | Show the active model, or save a default (config.json) and switch to it |
212
233
 
213
234
  The report the agent ends with and `/report` render are kept consistent by
214
235
  construction: `tool` results confirm each record call and the tracker, and
@@ -396,9 +417,39 @@ official `openai` Python SDK works as our client with two config lines:
396
417
  OpenAI(api_key=..., base_url="https://openrouter.ai/api/v1")
397
418
  ```
398
419
 
399
- Swapping to another OpenRouter model later is a one-line change (an env var
400
- today); moving to any other OpenAI-compatible provider changes only
401
- `agent/agent.py` configuration.
420
+ Swapping to another OpenRouter model later is a one-liner (env var, `/model`
421
+ command, or `--model` flag); moving to any other OpenAI-compatible provider
422
+ changes only `agent/agent.py` configuration.
423
+
424
+ **Your model, your choice.** The default is baked in as a fallback, never a
425
+ restriction. Four ways to set the model — the first one set wins:
426
+
427
+ 1. `--model NAME` flag (one run: `devopsiq --model openai/gpt-5.2 "why?"`)
428
+ 2. `/model <name>` in the REPL — saves it to `~/.devops-ai-agent/config.json`
429
+ so every future run uses it
430
+ 3. `OPENROUTER_MODEL` (shell or `.env`)
431
+ 4. built-in default (`z-ai/glm-5.3`)
432
+
433
+ ```bash
434
+ export OPENROUTER_MODEL=anthropic/claude-sonnet-5 # or openai/gpt-5.2, google/gemini-2.5-pro, ...
435
+ devopsiq /model openai/gpt-5.2 # or save a default from the REPL
436
+ ```
437
+
438
+ Two things to weigh when picking: the agent is a tool-use loop, so choose a
439
+ model with solid function-calling (a chat-only model will answer from
440
+ imagination instead of gathering evidence); and an investigation makes
441
+ several model calls per run, so price-per-call multiplies — that is why the
442
+ default is a cheap, reliable tool-caller rather than the biggest model.
443
+
444
+ **No OpenRouter at all?** `OPENROUTER_BASE_URL` points the same client at
445
+ any OpenAI-compatible endpoint — including a local one. Ollama, free and
446
+ offline:
447
+
448
+ ```bash
449
+ export OPENROUTER_BASE_URL=http://localhost:11434/v1
450
+ export OPENROUTER_MODEL=llama3.2 # any Ollama model that does tools
451
+ devopsiq "docker container checkout keeps exiting, investigate"
452
+ ```
402
453
 
403
454
  ## 5. Installation
404
455
 
@@ -463,6 +514,27 @@ Recommended CLIs, by domain:
463
514
  | New Relic | `curl` + `NEW_RELIC_API_KEY` / `NEW_RELIC_ACCOUNT_ID` env vars |
464
515
  | Ansible | `ansible-inventory`, `ansible-playbook` |
465
516
 
517
+ ### No API key? You still have the tools
518
+
519
+ The model is the only part that needs a key — the agent builds without one
520
+ (model-less mode), and two things keep working:
521
+
522
+ - **Record commands**: `devopsiq /report`, `devopsiq /investigations`, and
523
+ the same commands inside the REPL, all work with no key configured. Only
524
+ actual questions exit 1 with the setup message naming `OPENROUTER_API_KEY`.
525
+ - **The whole 58-tool layer** via the Python library, with no model and no
526
+ key (see INTEGRATION.md):
527
+
528
+ ```python
529
+ import agent.agent # registers every tool
530
+ from tools.registry import execute_tool
531
+
532
+ print(execute_tool("k8s_pods", '{"namespace": "prod"}'))
533
+ ```
534
+
535
+ And the fully key-free *agent* path is a local model (Ollama example in
536
+ section 4 above) — no cloud account needed at all.
537
+
466
538
  ## 6. Environment setup
467
539
 
468
540
  ```bash
@@ -490,7 +562,9 @@ NEW_RELIC_ACCOUNT_ID=1234567 # (Phase 9) numeric account
490
562
 
491
563
  `.env` is gitignored; the API key is never hard-coded in Python, printed, or
492
564
  logged. If `OPENROUTER_API_KEY` is already set in your shell, the shell value
493
- wins and `.env` is not consulted. Kubernetes tools use kubectl's own config
565
+ wins and `.env` is not consulted. Without any key the agent still starts in
566
+ model-less mode (record commands + the tool layer — see "No API key?" in
567
+ section 5). Kubernetes tools use kubectl's own config
494
568
  (`~/.kube/config` or `KUBECONFIG`); no agent-side config is needed.
495
569
 
496
570
  ## 7. How to run the agent
@@ -511,6 +585,7 @@ setup/API errors — so it drops straight into a pipeline:
511
585
  .venv/bin/python main.py --json "why is api-5d6f crash-looping?" # structured report
512
586
  .venv/bin/python main.py --resume --json "any update?" # continue a prior run
513
587
  .venv/bin/python main.py --store-dir /tmp/runs --out report.json "..." # pipeline paths
588
+ .venv/bin/python main.py --model openai/gpt-5.2 "why is it down?" # one-run model override
514
589
  ```
515
590
 
516
591
  With `--json` the stdout is one JSON document (see `render_report_json` in
@@ -552,8 +627,8 @@ Tools: ansible_inventory, ansible_playbook_tasks, aws_identity,
552
627
  k8s_top_nodes, k8s_top_pods, loki_query, prom_query,
553
628
  sys_open_ports, sys_service_logs, sys_service_status,
554
629
  sys_top_processes, system_info, tf_plan, tf_show, tf_state_list
555
- Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation
556
- One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] "<problem>"
630
+ Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation, /model [<name>]
631
+ One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] [--model NAME] "<problem>"
557
632
  Type 'exit' to quit.
558
633
 
559
634
  You: The checkout service container keeps exiting in Docker. Investigate.
@@ -570,7 +645,7 @@ You: exit
570
645
  Type `exit` / `quit`, or press Ctrl-D / Ctrl-C to leave. Slash commands are
571
646
  handled locally and never reach the model.
572
647
 
573
- ## 8. Current limitations (Phase 9)
648
+ ## 8. Current limitations
574
649
 
575
650
  - **Each domain needs its CLI installed and reachable.** Missing CLIs,
576
651
  unauthenticated `gh`, a dead docker daemon, an uninitialized terraform
@@ -655,8 +730,22 @@ handled locally and never reach the model.
655
730
  `helm_history` — reads only), Argo CD (`argocd_apps`,
656
731
  `argocd_app_status`), Istio (`istioctl_proxy_status`), Docker Compose
657
732
  (`docker_compose_ls`, `docker_compose_ps` — live-verified on this host).
733
+ - **Phase 10 — distribution. ✅ Done.** The `devopsiq` package on PyPI
734
+ (console command `devopsiq`, MIT) and the prebuilt multi-arch GHCR image,
735
+ cut by a tag-driven release workflow: an offline test gate, then PyPI
736
+ (Trusted Publishing — no token in the repo) and GHCR (amd64 + arm64) in
737
+ parallel. **`v0.1.1` added model-less mode:** with no API key the agent
738
+ still constructs — record commands (`/report`, `/investigations`) and the
739
+ 58-tool layer work; only model questions exit 1 with the setup message.
740
+ Verified end-to-end against the published package.
741
+ - **Model choice + resilient calls (v0.1.2). ✅ Done.** The model is
742
+ set by a 4-rung ladder — `--model` flag > `/model`-saved config file
743
+ (`~/.devops-ai-agent/config.json`) > `OPENROUTER_MODEL` > baked-in
744
+ default — and transient model-call failures (timeouts, connection
745
+ errors, 429s, 5xx) retry with exponential backoff (2s→4s→8s + jitter,
746
+ honoring a 429's Retry-After); a rejected key fails immediately.
658
747
  - **Later — region-scoped cloud resources** (ec2 describe-*, compute
659
748
  instances list, ...) behind the same template pattern; more observability
660
749
  depth (New Relic dashboards/entities, Prometheus range queries); streaming;
661
- conversation-history persistence; and a human-approval gate before any
662
- mutating action is ever allowed.
750
+ conversation-history persistence; and a human-approval gate
751
+ before any mutating action is ever allowed.
@@ -1,5 +1,12 @@
1
1
  # DevOps AI Agent
2
2
 
3
+ [![PyPI](https://img.shields.io/pypi/v/devopsiq)](https://pypi.org/project/devopsiq/)
4
+ [![Python](https://img.shields.io/pypi/pyversions/devopsiq)](https://pypi.org/project/devopsiq/)
5
+ [![Release](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml/badge.svg)](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
6
+ [![Docker](https://img.shields.io/badge/ghcr-devops--ai--agent-2496ED?logo=docker&logoColor=white)](https://github.com/DevOpsAbhii/devops-ai-agent/pkgs/container/devops-ai-agent)
7
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
8
+ [![Tests](https://img.shields.io/badge/tests-181%20offline-brightgreen)](https://github.com/DevOpsAbhii/devops-ai-agent/actions/workflows/release.yml)
9
+
3
10
  An AI agent that investigates real DevOps problems. The end goal: ask it
4
11
  something like *"Why is my Kubernetes pod in CrashLoopBackOff?"* and have it
5
12
  gather evidence, reason about the evidence, identify the likely root cause,
@@ -26,7 +33,14 @@ from environment config), and Ansible listing (inventory, playbook tasks).
26
33
  open alerts over NerdGraph — credentials from env), Trivy image
27
34
  vulnerability scanning, Helm releases (list/status/history), Argo CD
28
35
  (GitOps app sync/health), Istio mesh proxy status, and Docker Compose
29
- (project and service listing). The repository is git-tracked.
36
+ (project and service listing). **Phase 10 ships it as a product:** the
37
+ [`devopsiq` package on PyPI](https://pypi.org/project/devopsiq/) and a
38
+ prebuilt multi-arch Docker image (`ghcr.io/devopsabhii/devops-ai-agent`) —
39
+ both cut automatically by pushing a `v*` tag (test gate → PyPI via Trusted
40
+ Publishing + GHCR in parallel). **No API key? The agent still runs:** it
41
+ starts in model-less mode — record commands (`/report`, `/investigations`)
42
+ and the whole 58-tool layer work without a key; only questions to the model
43
+ need one. The repository is git-tracked.
30
44
  Every phase still built from scratch — no LangChain, LangGraph,
31
45
  AutoGen, CrewAI, or MCP.
32
46
 
@@ -44,7 +58,7 @@ model is called, how conversation history flows, how tool selection +
44
58
  execution + evidence feedback work — instead of depending on a framework
45
59
  for it.
46
60
 
47
- In Phase 5 the agent:
61
+ Today the agent:
48
62
  - holds a conversation with **GLM 5.3** through **OpenRouter**;
49
63
  - has **58 real, read-only tools** across fifteen domains: host facts,
50
64
  Kubernetes (12 tools), Linux system (4), Docker + Compose (10),
@@ -104,8 +118,8 @@ Module map:
104
118
 
105
119
  | Path | Responsibility |
106
120
  | --------------------------- | ------------------------------------------------------------ |
107
- | `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages |
108
- | `agent/agent.py` | `DevOpsAgent` — client, history, `ask()`, `_complete()` loop |
121
+ | `main.py` | REPL loop, one-shot CLI, environment loading, `/investigate` commands, error messages, model-less start |
122
+ | `agent/agent.py` | `DevOpsAgent` — client (`None` = model-less mode), history, `ask()`, `_complete()` loop |
109
123
  | `agent/prompts.py` | The system prompt (versioned/tested separately) |
110
124
  | `agent/investigation.py` | First-class investigation record: hypotheses, verdicts, evidence, report renderer (pure data) |
111
125
  | `agent/store.py` | `InvestigationStore` — one JSON file per record, atomic writes, resume/list (Phase 7) |
@@ -134,6 +148,11 @@ Module map:
134
148
  | `tests/test_phase7.py` | Offline suite: round-trip serialization, store files, auto-save, resume, CLI flags (Phase 7) |
135
149
  | `tests/test_phase8.py` | Offline suite: argv templates + validation for the 21 Phase 8 tools (fake CLIs, env-based monitoring) |
136
150
  | `tests/test_phase9.py` | Offline suite: New Relic env-credential + payload tests, trivy/helm/argocd/istio/compose argv templates (Phase 9) |
151
+ | `pyproject.toml` | Package `devopsiq`: metadata, MIT, console script `devopsiq = main:main` (Phase 10) |
152
+ | `Dockerfile` | Prebuilt image: slim base + kubectl/helm/trivy/gh + app, non-root, `/data` record store (Phase 10) |
153
+ | `.github/workflows/release.yml` | Tag-driven release: test gate → PyPI (Trusted Publishing) + GHCR multi-arch (Phase 10) |
154
+ | `docs/generate_pdf.py` | Builds the project documentation PDF from live source |
155
+ | `docs/generate_release_guide.py` | Builds the release & update playbook PDF |
137
156
 
138
157
  ### The tool-use loop
139
158
 
@@ -180,6 +199,7 @@ the CLI exposes it directly:
180
199
  | `/investigations` | List saved records on disk (newest first; `← active` marks the live one) |
181
200
  | `/report` | Show the canonical report (once concluded) |
182
201
  | `/endinvestigation` | Clear the record (memory only — the saved copy stays as history) |
202
+ | `/model [<name>]` | Show the active model, or save a default (config.json) and switch to it |
183
203
 
184
204
  The report the agent ends with and `/report` render are kept consistent by
185
205
  construction: `tool` results confirm each record call and the tracker, and
@@ -367,9 +387,39 @@ official `openai` Python SDK works as our client with two config lines:
367
387
  OpenAI(api_key=..., base_url="https://openrouter.ai/api/v1")
368
388
  ```
369
389
 
370
- Swapping to another OpenRouter model later is a one-line change (an env var
371
- today); moving to any other OpenAI-compatible provider changes only
372
- `agent/agent.py` configuration.
390
+ Swapping to another OpenRouter model later is a one-liner (env var, `/model`
391
+ command, or `--model` flag); moving to any other OpenAI-compatible provider
392
+ changes only `agent/agent.py` configuration.
393
+
394
+ **Your model, your choice.** The default is baked in as a fallback, never a
395
+ restriction. Four ways to set the model — the first one set wins:
396
+
397
+ 1. `--model NAME` flag (one run: `devopsiq --model openai/gpt-5.2 "why?"`)
398
+ 2. `/model <name>` in the REPL — saves it to `~/.devops-ai-agent/config.json`
399
+ so every future run uses it
400
+ 3. `OPENROUTER_MODEL` (shell or `.env`)
401
+ 4. built-in default (`z-ai/glm-5.3`)
402
+
403
+ ```bash
404
+ export OPENROUTER_MODEL=anthropic/claude-sonnet-5 # or openai/gpt-5.2, google/gemini-2.5-pro, ...
405
+ devopsiq /model openai/gpt-5.2 # or save a default from the REPL
406
+ ```
407
+
408
+ Two things to weigh when picking: the agent is a tool-use loop, so choose a
409
+ model with solid function-calling (a chat-only model will answer from
410
+ imagination instead of gathering evidence); and an investigation makes
411
+ several model calls per run, so price-per-call multiplies — that is why the
412
+ default is a cheap, reliable tool-caller rather than the biggest model.
413
+
414
+ **No OpenRouter at all?** `OPENROUTER_BASE_URL` points the same client at
415
+ any OpenAI-compatible endpoint — including a local one. Ollama, free and
416
+ offline:
417
+
418
+ ```bash
419
+ export OPENROUTER_BASE_URL=http://localhost:11434/v1
420
+ export OPENROUTER_MODEL=llama3.2 # any Ollama model that does tools
421
+ devopsiq "docker container checkout keeps exiting, investigate"
422
+ ```
373
423
 
374
424
  ## 5. Installation
375
425
 
@@ -434,6 +484,27 @@ Recommended CLIs, by domain:
434
484
  | New Relic | `curl` + `NEW_RELIC_API_KEY` / `NEW_RELIC_ACCOUNT_ID` env vars |
435
485
  | Ansible | `ansible-inventory`, `ansible-playbook` |
436
486
 
487
+ ### No API key? You still have the tools
488
+
489
+ The model is the only part that needs a key — the agent builds without one
490
+ (model-less mode), and two things keep working:
491
+
492
+ - **Record commands**: `devopsiq /report`, `devopsiq /investigations`, and
493
+ the same commands inside the REPL, all work with no key configured. Only
494
+ actual questions exit 1 with the setup message naming `OPENROUTER_API_KEY`.
495
+ - **The whole 58-tool layer** via the Python library, with no model and no
496
+ key (see INTEGRATION.md):
497
+
498
+ ```python
499
+ import agent.agent # registers every tool
500
+ from tools.registry import execute_tool
501
+
502
+ print(execute_tool("k8s_pods", '{"namespace": "prod"}'))
503
+ ```
504
+
505
+ And the fully key-free *agent* path is a local model (Ollama example in
506
+ section 4 above) — no cloud account needed at all.
507
+
437
508
  ## 6. Environment setup
438
509
 
439
510
  ```bash
@@ -461,7 +532,9 @@ NEW_RELIC_ACCOUNT_ID=1234567 # (Phase 9) numeric account
461
532
 
462
533
  `.env` is gitignored; the API key is never hard-coded in Python, printed, or
463
534
  logged. If `OPENROUTER_API_KEY` is already set in your shell, the shell value
464
- wins and `.env` is not consulted. Kubernetes tools use kubectl's own config
535
+ wins and `.env` is not consulted. Without any key the agent still starts in
536
+ model-less mode (record commands + the tool layer — see "No API key?" in
537
+ section 5). Kubernetes tools use kubectl's own config
465
538
  (`~/.kube/config` or `KUBECONFIG`); no agent-side config is needed.
466
539
 
467
540
  ## 7. How to run the agent
@@ -482,6 +555,7 @@ setup/API errors — so it drops straight into a pipeline:
482
555
  .venv/bin/python main.py --json "why is api-5d6f crash-looping?" # structured report
483
556
  .venv/bin/python main.py --resume --json "any update?" # continue a prior run
484
557
  .venv/bin/python main.py --store-dir /tmp/runs --out report.json "..." # pipeline paths
558
+ .venv/bin/python main.py --model openai/gpt-5.2 "why is it down?" # one-run model override
485
559
  ```
486
560
 
487
561
  With `--json` the stdout is one JSON document (see `render_report_json` in
@@ -523,8 +597,8 @@ Tools: ansible_inventory, ansible_playbook_tasks, aws_identity,
523
597
  k8s_top_nodes, k8s_top_pods, loki_query, prom_query,
524
598
  sys_open_ports, sys_service_logs, sys_service_status,
525
599
  sys_top_processes, system_info, tf_plan, tf_show, tf_state_list
526
- Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation
527
- One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] "<problem>"
600
+ Commands: /investigate <problem>, /investigation, /investigations, /report, /endinvestigation, /model [<name>]
601
+ One-shot: python main.py [--json] [--resume] [--out report.json] [--store-dir DIR] [--model NAME] "<problem>"
528
602
  Type 'exit' to quit.
529
603
 
530
604
  You: The checkout service container keeps exiting in Docker. Investigate.
@@ -541,7 +615,7 @@ You: exit
541
615
  Type `exit` / `quit`, or press Ctrl-D / Ctrl-C to leave. Slash commands are
542
616
  handled locally and never reach the model.
543
617
 
544
- ## 8. Current limitations (Phase 9)
618
+ ## 8. Current limitations
545
619
 
546
620
  - **Each domain needs its CLI installed and reachable.** Missing CLIs,
547
621
  unauthenticated `gh`, a dead docker daemon, an uninitialized terraform
@@ -626,8 +700,22 @@ handled locally and never reach the model.
626
700
  `helm_history` — reads only), Argo CD (`argocd_apps`,
627
701
  `argocd_app_status`), Istio (`istioctl_proxy_status`), Docker Compose
628
702
  (`docker_compose_ls`, `docker_compose_ps` — live-verified on this host).
703
+ - **Phase 10 — distribution. ✅ Done.** The `devopsiq` package on PyPI
704
+ (console command `devopsiq`, MIT) and the prebuilt multi-arch GHCR image,
705
+ cut by a tag-driven release workflow: an offline test gate, then PyPI
706
+ (Trusted Publishing — no token in the repo) and GHCR (amd64 + arm64) in
707
+ parallel. **`v0.1.1` added model-less mode:** with no API key the agent
708
+ still constructs — record commands (`/report`, `/investigations`) and the
709
+ 58-tool layer work; only model questions exit 1 with the setup message.
710
+ Verified end-to-end against the published package.
711
+ - **Model choice + resilient calls (v0.1.2). ✅ Done.** The model is
712
+ set by a 4-rung ladder — `--model` flag > `/model`-saved config file
713
+ (`~/.devops-ai-agent/config.json`) > `OPENROUTER_MODEL` > baked-in
714
+ default — and transient model-call failures (timeouts, connection
715
+ errors, 429s, 5xx) retry with exponential backoff (2s→4s→8s + jitter,
716
+ honoring a 429's Retry-After); a rejected key fails immediately.
629
717
  - **Later — region-scoped cloud resources** (ec2 describe-*, compute
630
718
  instances list, ...) behind the same template pattern; more observability
631
719
  depth (New Relic dashboards/entities, Prometheus range queries); streaming;
632
- conversation-history persistence; and a human-approval gate before any
633
- mutating action is ever allowed.
720
+ conversation-history persistence; and a human-approval gate
721
+ before any mutating action is ever allowed.
@@ -26,9 +26,18 @@ exits; the delegates below expose resume/list/store-location to the CLI.
26
26
  """
27
27
 
28
28
  import os
29
+ import random
30
+ import time
31
+
32
+ from openai import (
33
+ APIConnectionError,
34
+ APIStatusError,
35
+ APITimeoutError,
36
+ OpenAI,
37
+ RateLimitError,
38
+ )
29
39
 
30
- from openai import OpenAI
31
-
40
+ from agent.config import load_user_config
32
41
  from agent.prompts import SYSTEM_PROMPT
33
42
  from agent.store import InvestigationStore
34
43
  from tools import ( # noqa: F401 — side effect: each module registers its tools
@@ -68,6 +77,50 @@ PLACEHOLDER_KEY = "your_key_here"
68
77
  # Safety valve: the model gets at most this many tool-use turns before we stop.
69
78
  MAX_TOOL_ITERATIONS = 10
70
79
 
80
+ # Transient-failure retries for the model call itself (network blips, 429,
81
+ # 5xx). MAX_MODEL_RETRIES attempts beyond the first, exponential backoff
82
+ # starting at RETRY_BASE_DELAY (2s -> 4s -> 8s) plus jitter, capped at
83
+ # RETRY_MAX_DELAY. A 429's Retry-After header wins when present.
84
+ MAX_MODEL_RETRIES = 3
85
+ RETRY_BASE_DELAY = 2.0
86
+ RETRY_MAX_DELAY = 60.0
87
+
88
+ _sleep = time.sleep # test seam: offline tests patch agent.agent._sleep
89
+
90
+
91
+ def _is_transient(exc: Exception) -> bool:
92
+ """True when retrying `exc` can plausibly succeed.
93
+
94
+ Timeouts, connection failures, rate limits, and server-side 5xx are
95
+ transient. A rejected key (401) or any other client 4xx is not — the
96
+ same request would fail identically forever, so it fails immediately.
97
+ """
98
+ if isinstance(exc, (APITimeoutError, APIConnectionError, RateLimitError)):
99
+ return True
100
+ if isinstance(exc, APIStatusError):
101
+ return exc.status_code >= 500
102
+ return False
103
+
104
+
105
+ def _retry_delay(exc: Exception, attempt: int) -> float:
106
+ """Seconds to wait before retry number `attempt` (0-based).
107
+
108
+ Exponential backoff with jitter, except on a rate limit carrying a
109
+ Retry-After header — the server knows its own budget, so it wins
110
+ (clamped to [1, RETRY_MAX_DELAY] so a bad header cannot hurt us).
111
+ """
112
+ response = getattr(exc, "response", None)
113
+ retry_after = getattr(response, "headers", {}).get("retry-after")
114
+ if retry_after:
115
+ try:
116
+ return min(max(float(retry_after), 1.0), RETRY_MAX_DELAY)
117
+ except (TypeError, ValueError):
118
+ pass # non-numeric header: fall through to backoff
119
+ return min(
120
+ RETRY_BASE_DELAY * (2 ** attempt) + random.uniform(0, 1),
121
+ RETRY_MAX_DELAY,
122
+ )
123
+
71
124
 
72
125
  class DevOpsAgent:
73
126
  """Minimal DevOps investigation assistant (read-only by design)."""
@@ -80,23 +133,35 @@ class DevOpsAgent:
80
133
  ) -> None:
81
134
  # Configuration resolution order: explicit argument > environment > default.
82
135
  self.api_key = api_key or os.getenv("OPENROUTER_API_KEY")
136
+
137
+ self.model = (
138
+ model # 1. explicit --model flag
139
+ or load_user_config().get("model") # 2. ~/.devops-ai-agent/config.json
140
+ or os.getenv("OPENROUTER_MODEL") # 3. environment / .env
141
+ or DEFAULT_MODEL # 4. built-in default
142
+ )
143
+ self.base_url = base_url or os.getenv("OPENROUTER_BASE_URL", DEFAULT_BASE_URL)
144
+
145
+ # Model-less mode: a missing (or placeholder) key no longer blocks
146
+ # construction. Record operations (the slash commands) and the tool
147
+ # layer never call the model, so the agent builds with client=None
148
+ # and only the chat path (_complete) demands a working key.
149
+ self.client = None
150
+ self.model_error: str | None = None
83
151
  if not self.api_key:
84
- raise ValueError(
152
+ self.model_error = (
85
153
  "OPENROUTER_API_KEY is not set. Copy .env.example to .env and "
86
154
  "fill in your key, or export OPENROUTER_API_KEY in your shell."
87
155
  )
88
- if self.api_key.strip().lower() == PLACEHOLDER_KEY:
89
- raise ValueError(
156
+ elif self.api_key.strip().lower() == PLACEHOLDER_KEY:
157
+ self.model_error = (
90
158
  "OPENROUTER_API_KEY still has the placeholder value. Edit .env "
91
159
  "and replace `your_key_here` with your real key."
92
160
  )
93
-
94
- self.model = model or os.getenv("OPENROUTER_MODEL", DEFAULT_MODEL)
95
- self.base_url = base_url or os.getenv("OPENROUTER_BASE_URL", DEFAULT_BASE_URL)
96
-
97
- # OpenRouter exposes an OpenAI-compatible API, so the official OpenAI
98
- # SDK is a drop-in client — just pointed at OpenRouter's base URL.
99
- self.client = OpenAI(api_key=self.api_key, base_url=self.base_url)
161
+ else:
162
+ # OpenRouter exposes an OpenAI-compatible API, so the official
163
+ # OpenAI SDK is a drop-in client — just pointed at the base URL.
164
+ self.client = OpenAI(api_key=self.api_key, base_url=self.base_url)
100
165
 
101
166
  # Short-term conversation history. Starts with the system prompt;
102
167
  # grows as user, model, and tool results exchange turns.
@@ -110,7 +175,8 @@ class DevOpsAgent:
110
175
 
111
176
  The message, any tool exchanges, and the final assistant reply are all
112
177
  kept in this session's history so the model retains context across
113
- turns. Raises ValueError on empty input.
178
+ turns. Raises ValueError on empty input, or when no API key is
179
+ configured (model-less mode).
114
180
  """
115
181
  message = user_message.strip()
116
182
  if not message:
@@ -180,12 +246,16 @@ class DevOpsAgent:
180
246
  model again. Stops when the model answers in plain text, or when the
181
247
  iteration cap is reached.
182
248
  """
249
+ if self.client is None:
250
+ # Model-less construction: the first chat attempt is where the
251
+ # missing key finally surfaces, as a normal caught ValueError.
252
+ raise ValueError(self.model_error)
183
253
  for _ in range(MAX_TOOL_ITERATIONS):
184
254
  request: dict = {"model": self.model, "messages": self.messages}
185
255
  if self.tools:
186
256
  request["tools"] = [tool.schema() for tool in self.tools]
187
257
 
188
- response = self.client.chat.completions.create(**request)
258
+ response = self._call_model(request)
189
259
  message = response.choices[0].message
190
260
 
191
261
  if not message.tool_calls:
@@ -207,6 +277,23 @@ class DevOpsAgent:
207
277
  f"The model did not finish after {MAX_TOOL_ITERATIONS} tool-use turns."
208
278
  )
209
279
 
280
+ def _call_model(self, request: dict):
281
+ """One model call with bounded retries on transient failures.
282
+
283
+ Timeouts, connection errors, rate limits (429) and server-side 5xx
284
+ are retried up to MAX_MODEL_RETRIES times with exponential backoff
285
+ (2s, 4s, 8s + jitter); a 429's Retry-After header wins when present.
286
+ Client errors — a rejected key (401) most notably — fail immediately,
287
+ because retrying the identical request cannot fix them.
288
+ """
289
+ for attempt in range(MAX_MODEL_RETRIES + 1):
290
+ try:
291
+ return self.client.chat.completions.create(**request)
292
+ except Exception as exc: # noqa: BLE001 — classified right below
293
+ if attempt >= MAX_MODEL_RETRIES or not _is_transient(exc):
294
+ raise
295
+ _sleep(_retry_delay(exc, attempt))
296
+
210
297
 
211
298
  def _echo_tool_request(message) -> dict:
212
299
  """Rebuild the assistant turn that requested tools, verbatim.
@@ -0,0 +1,45 @@
1
+ """Per-user preferences: ~/.devops-ai-agent/config.json.
2
+
3
+ The `devopsiq` equivalent of "my settings" — currently the preferred model,
4
+ written by the /model command and read by the agent at startup. Precedence
5
+ for the model, highest first:
6
+
7
+ --model flag > config.json > OPENROUTER_MODEL env > built-in default
8
+
9
+ The file is tiny JSON, one flat object. Every read is tolerant: a missing
10
+ file, an unreadable home directory, or corrupt content yields {} — a broken
11
+ config never blocks the agent (the same degrade-don't-crash rule the
12
+ investigation store follows). Tests point AGENT_CONFIG_FILE at a tmp path.
13
+ """
14
+
15
+ import json
16
+ import os
17
+ from pathlib import Path
18
+
19
+ DEFAULT_CONFIG_PATH = Path.home() / ".devops-ai-agent" / "config.json"
20
+
21
+
22
+ def config_path() -> Path:
23
+ """The active config file (AGENT_CONFIG_FILE overrides; tests use tmp)."""
24
+ return Path(os.getenv("AGENT_CONFIG_FILE") or DEFAULT_CONFIG_PATH)
25
+
26
+
27
+ def load_user_config() -> dict:
28
+ """Read the config file; {} when missing or unreadable — never raise."""
29
+ try:
30
+ raw = config_path().read_text(encoding="utf-8")
31
+ data = json.loads(raw)
32
+ return data if isinstance(data, dict) else {}
33
+ except (OSError, ValueError):
34
+ return {}
35
+
36
+
37
+ def save_user_config(update: dict) -> Path:
38
+ """Merge `update` into the config file (create the directory if needed)."""
39
+ path = config_path()
40
+ data = load_user_config()
41
+ data.update(update)
42
+ path.parent.mkdir(parents=True, exist_ok=True)
43
+ path.write_text(json.dumps(data, indent=2, sort_keys=True) + "\n",
44
+ encoding="utf-8")
45
+ return path