graph-agents-cli 0.3.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (291) hide show
  1. graph_agents_cli/__init__.py +26 -0
  2. graph_agents_cli/_api_policy.py +2145 -0
  3. graph_agents_cli/_approvals.py +400 -0
  4. graph_agents_cli/_build.py +186 -0
  5. graph_agents_cli/_build_info.json +7 -0
  6. graph_agents_cli/_chat_client.py +462 -0
  7. graph_agents_cli/_click.py +157 -0
  8. graph_agents_cli/_defaults.py +139 -0
  9. graph_agents_cli/_experiments.py +64 -0
  10. graph_agents_cli/_http.py +192 -0
  11. graph_agents_cli/_output.py +83 -0
  12. graph_agents_cli/_project.py +462 -0
  13. graph_agents_cli/_remote.py +220 -0
  14. graph_agents_cli/_response_schema.py +264 -0
  15. graph_agents_cli/_runner.py +319 -0
  16. graph_agents_cli/_skills_check.py +274 -0
  17. graph_agents_cli/_tools.py +189 -0
  18. graph_agents_cli/_trust.py +66 -0
  19. graph_agents_cli/api/__init__.py +15 -0
  20. graph_agents_cli/api/_changes.py +506 -0
  21. graph_agents_cli/api/_files.py +658 -0
  22. graph_agents_cli/api/cmd_api.py +2480 -0
  23. graph_agents_cli/deploy/__init__.py +15 -0
  24. graph_agents_cli/deploy/_config.py +171 -0
  25. graph_agents_cli/deploy/_image.py +128 -0
  26. graph_agents_cli/deploy/_kube.py +286 -0
  27. graph_agents_cli/deploy/_modes.py +234 -0
  28. graph_agents_cli/deploy/_preflight.py +370 -0
  29. graph_agents_cli/deploy/_values.py +168 -0
  30. graph_agents_cli/deploy/cmd_deploy.py +1866 -0
  31. graph_agents_cli/deploy/gitops.py +562 -0
  32. graph_agents_cli/deploy/local_load.py +273 -0
  33. graph_agents_cli/dev/__init__.py +13 -0
  34. graph_agents_cli/dev/cmd_build.py +131 -0
  35. graph_agents_cli/dev/cmd_install.py +78 -0
  36. graph_agents_cli/dev/cmd_lint.py +119 -0
  37. graph_agents_cli/dev/cmd_playground.py +297 -0
  38. graph_agents_cli/dev/policy_check.py +1287 -0
  39. graph_agents_cli/eval/__init__.py +22 -0
  40. graph_agents_cli/eval/_client.py +670 -0
  41. graph_agents_cli/eval/_common.py +177 -0
  42. graph_agents_cli/eval/_judge.py +168 -0
  43. graph_agents_cli/eval/_judge_runner.py +238 -0
  44. graph_agents_cli/eval/_paths.py +212 -0
  45. graph_agents_cli/eval/checks.py +581 -0
  46. graph_agents_cli/eval/cmd_analyze.py +278 -0
  47. graph_agents_cli/eval/cmd_compare.py +284 -0
  48. graph_agents_cli/eval/cmd_eval_group.py +80 -0
  49. graph_agents_cli/eval/cmd_generate.py +558 -0
  50. graph_agents_cli/eval/cmd_grade.py +466 -0
  51. graph_agents_cli/eval/cmd_metric.py +156 -0
  52. graph_agents_cli/eval/cmd_run.py +370 -0
  53. graph_agents_cli/eval/cmd_submit.py +400 -0
  54. graph_agents_cli/eval/config.py +435 -0
  55. graph_agents_cli/eval/dataset.py +350 -0
  56. graph_agents_cli/eval/gate.py +420 -0
  57. graph_agents_cli/eval/transcript.py +192 -0
  58. graph_agents_cli/extension/__init__.py +13 -0
  59. graph_agents_cli/extension/_compat.py +86 -0
  60. graph_agents_cli/extension/_loader.py +293 -0
  61. graph_agents_cli/extension/_manifest.py +135 -0
  62. graph_agents_cli/extension/_overrides.py +195 -0
  63. graph_agents_cli/extension/_paths.py +91 -0
  64. graph_agents_cli/extension/_refs.py +193 -0
  65. graph_agents_cli/extension/_resolver.py +453 -0
  66. graph_agents_cli/extension/_schema.py +106 -0
  67. graph_agents_cli/extension/_spec.py +253 -0
  68. graph_agents_cli/extension/_sync.py +102 -0
  69. graph_agents_cli/extension/_trust.py +58 -0
  70. graph_agents_cli/extension/cmd_extension_add.py +259 -0
  71. graph_agents_cli/extension/cmd_extension_group.py +57 -0
  72. graph_agents_cli/extension/cmd_extension_list.py +56 -0
  73. graph_agents_cli/extension/cmd_extension_remove.py +61 -0
  74. graph_agents_cli/extension/cmd_extension_update.py +195 -0
  75. graph_agents_cli/info/__init__.py +13 -0
  76. graph_agents_cli/info/cmd_info.py +222 -0
  77. graph_agents_cli/infra/__init__.py +15 -0
  78. graph_agents_cli/infra/checks.py +1169 -0
  79. graph_agents_cli/infra/cmd_infra.py +103 -0
  80. graph_agents_cli/main.py +591 -0
  81. graph_agents_cli/peer/__init__.py +15 -0
  82. graph_agents_cli/peer/_generate.py +254 -0
  83. graph_agents_cli/peer/cmd_peer.py +1151 -0
  84. graph_agents_cli/run/__init__.py +13 -0
  85. graph_agents_cli/run/_local_server.py +1157 -0
  86. graph_agents_cli/run/_signals.py +141 -0
  87. graph_agents_cli/run/cmd_approvals.py +530 -0
  88. graph_agents_cli/run/cmd_run.py +1421 -0
  89. graph_agents_cli/scaffold/__init__.py +19 -0
  90. graph_agents_cli/scaffold/agents/README.md +24 -0
  91. graph_agents_cli/scaffold/agents/empty_py/.template/templateconfig.yaml +22 -0
  92. graph_agents_cli/scaffold/agents/langgraph/.env.example +292 -0
  93. graph_agents_cli/scaffold/agents/langgraph/.template/templateconfig.yaml +28 -0
  94. graph_agents_cli/scaffold/agents/langgraph/Dockerfile +59 -0
  95. graph_agents_cli/scaffold/agents/langgraph/Dockerfile.langgraph-server +59 -0
  96. graph_agents_cli/scaffold/agents/langgraph/README.md +571 -0
  97. graph_agents_cli/scaffold/agents/langgraph/api-policy.yaml +60 -0
  98. graph_agents_cli/scaffold/agents/langgraph/app/__init__.py +20 -0
  99. graph_agents_cli/scaffold/agents/langgraph/app/agent.py +174 -0
  100. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/__init__.py +15 -0
  101. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a.py +2162 -0
  102. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a_client.py +1167 -0
  103. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/api_client.py +4220 -0
  104. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/approvals.py +1349 -0
  105. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/auth.py +1986 -0
  106. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/chat.py +2962 -0
  107. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/checkpointer.py +432 -0
  108. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/content.py +569 -0
  109. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/db.py +580 -0
  110. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/limits.py +203 -0
  111. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/metrics.py +231 -0
  112. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/middleware.py +361 -0
  113. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/model.py +611 -0
  114. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/playground.py +230 -0
  115. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/run_locks.py +459 -0
  116. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/structured.py +755 -0
  117. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/telemetry.py +681 -0
  118. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/threads.py +493 -0
  119. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/token_exchange.py +959 -0
  120. graph_agents_cli/scaffold/agents/langgraph/app/fast_api_app.py +770 -0
  121. graph_agents_cli/scaffold/agents/langgraph/app/policies/__init__.py +55 -0
  122. graph_agents_cli/scaffold/agents/langgraph/app/policies/custom.py +97 -0
  123. graph_agents_cli/scaffold/agents/langgraph/app/tools/__init__.py +46 -0
  124. graph_agents_cli/scaffold/agents/langgraph/app/tools/example_api.py +92 -0
  125. graph_agents_cli/scaffold/agents/langgraph/app/tools/weather.py +33 -0
  126. graph_agents_cli/scaffold/agents/langgraph/langgraph.json +14 -0
  127. graph_agents_cli/scaffold/agents/langgraph/pyproject.toml +78 -0
  128. graph_agents_cli/scaffold/agents/langgraph/tests/conftest.py +376 -0
  129. graph_agents_cli/scaffold/agents/langgraph/tests/eval/datasets/basic-dataset.json +53 -0
  130. graph_agents_cli/scaffold/agents/langgraph/tests/eval/eval_config.yaml +32 -0
  131. graph_agents_cli/scaffold/agents/langgraph/tests/integration/approval_graph.py +137 -0
  132. graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_issuer.py +216 -0
  133. graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_openai.py +357 -0
  134. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_outcomes.py +569 -0
  135. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_relay.py +479 -0
  136. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_api_surface.py +812 -0
  137. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals.py +1367 -0
  138. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals_server.py +794 -0
  139. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor.py +497 -0
  140. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor_server.py +247 -0
  141. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_history_repair.py +278 -0
  142. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_model_apis.py +242 -0
  143. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_postgres.py +637 -0
  144. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_resilience_postgres.py +770 -0
  145. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_runtime_guardrails.py +854 -0
  146. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_e2e.py +340 -0
  147. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_runtime.py +989 -0
  148. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_answers.py +584 -0
  149. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_server.py +222 -0
  150. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_token_exchange_issuer.py +650 -0
  151. graph_agents_cli/scaffold/agents/langgraph/tests/load_test/.results/.placeholder +0 -0
  152. graph_agents_cli/scaffold/agents/langgraph/tests/load_test/README.md +22 -0
  153. graph_agents_cli/scaffold/agents/langgraph/tests/load_test/conftest.py +21 -0
  154. graph_agents_cli/scaffold/agents/langgraph/tests/load_test/load_test.py +81 -0
  155. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_client.py +824 -0
  156. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_scoping.py +724 -0
  157. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client.py +1214 -0
  158. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client_hardening.py +716 -0
  159. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_policy_rpc.py +767 -0
  160. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_approval_ledger.py +1536 -0
  161. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_fake_model.py +115 -0
  162. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_jwt_policy.py +991 -0
  163. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_limits.py +310 -0
  164. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging.py +148 -0
  165. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging_hardening.py +271 -0
  166. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_policy.py +378 -0
  167. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_resilience.py +610 -0
  168. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_server_auth.py +702 -0
  169. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_structured.py +673 -0
  170. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_telemetry.py +404 -0
  171. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_thread_listing.py +255 -0
  172. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_threads.py +268 -0
  173. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_token_exchange.py +1320 -0
  174. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_untrusted_content.py +393 -0
  175. graph_agents_cli/scaffold/agents/langgraph/uv-fastapi.lock +2084 -0
  176. graph_agents_cli/scaffold/agents/langgraph/uv-langgraph-server.lock +2106 -0
  177. graph_agents_cli/scaffold/agents/langgraph/{{cookiecutter.agent_guidance_filename}} +129 -0
  178. graph_agents_cli/scaffold/base_templates/_shared/graph-agents-cli-manifest.yaml +36 -0
  179. graph_agents_cli/scaffold/base_templates/python/.dockerignore +32 -0
  180. graph_agents_cli/scaffold/base_templates/python/.github/CODEOWNERS +30 -0
  181. graph_agents_cli/scaffold/base_templates/python/.github/agent.env +7 -0
  182. graph_agents_cli/scaffold/base_templates/python/.github/workflows/pr_checks.yaml +214 -0
  183. graph_agents_cli/scaffold/base_templates/python/.gitignore +209 -0
  184. graph_agents_cli/scaffold/base_templates/python/tests/unit/test_dummy.py +23 -0
  185. graph_agents_cli/scaffold/base_templates/python/{{cookiecutter.agent_guidance_filename}} +35 -0
  186. graph_agents_cli/scaffold/cmd_scaffold_group.py +49 -0
  187. graph_agents_cli/scaffold/commands/__init__.py +13 -0
  188. graph_agents_cli/scaffold/commands/create.py +1424 -0
  189. graph_agents_cli/scaffold/commands/enhance.py +1652 -0
  190. graph_agents_cli/scaffold/commands/upgrade.py +570 -0
  191. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/agent.env +12 -0
  192. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/promote-to-prod.yaml +371 -0
  193. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/staging.yaml +450 -0
  194. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-dev.yaml +43 -0
  195. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-prod.yaml +41 -0
  196. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-staging.yaml +43 -0
  197. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/.helmignore +14 -0
  198. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/Chart.yaml +21 -0
  199. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/examples/networkpolicy.yaml +103 -0
  200. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/NOTES.txt +48 -0
  201. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/_helpers.tpl +189 -0
  202. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/certificate.yaml +15 -0
  203. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/configmap.yaml +10 -0
  204. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/deployment.yaml +199 -0
  205. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/hpa.yaml +22 -0
  206. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/httproute.yaml +30 -0
  207. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/ingress.yaml +39 -0
  208. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/networkpolicy.yaml +48 -0
  209. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/pdb.yaml +13 -0
  210. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/postgresql-secret.yaml +37 -0
  211. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/service.yaml +15 -0
  212. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/serviceaccount.yaml +13 -0
  213. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/servicemonitor.yaml +42 -0
  214. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-dev.yaml +22 -0
  215. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-prod.yaml +45 -0
  216. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-staging.yaml +29 -0
  217. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values.yaml +396 -0
  218. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/tests/integration/test_chart.py +269 -0
  219. graph_agents_cli/scaffold/deployment_targets/none/README.md +5 -0
  220. graph_agents_cli/scaffold/deployment_targets/none/python/README.md +6 -0
  221. graph_agents_cli/scaffold/utils/__init__.py +13 -0
  222. graph_agents_cli/scaffold/utils/backup.py +212 -0
  223. graph_agents_cli/scaffold/utils/build_record.py +257 -0
  224. graph_agents_cli/scaffold/utils/cli_options.py +184 -0
  225. graph_agents_cli/scaffold/utils/fs.py +83 -0
  226. graph_agents_cli/scaffold/utils/generate_locks.py +214 -0
  227. graph_agents_cli/scaffold/utils/generation_metadata.py +88 -0
  228. graph_agents_cli/scaffold/utils/keyedit.py +768 -0
  229. graph_agents_cli/scaffold/utils/keymerge.py +537 -0
  230. graph_agents_cli/scaffold/utils/language.py +138 -0
  231. graph_agents_cli/scaffold/utils/lock_utils.py +94 -0
  232. graph_agents_cli/scaffold/utils/logging.py +77 -0
  233. graph_agents_cli/scaffold/utils/manifest.py +292 -0
  234. graph_agents_cli/scaffold/utils/merge.py +970 -0
  235. graph_agents_cli/scaffold/utils/merge3.py +216 -0
  236. graph_agents_cli/scaffold/utils/openapi_seed.py +199 -0
  237. graph_agents_cli/scaffold/utils/remote_template.py +376 -0
  238. graph_agents_cli/scaffold/utils/template.py +1352 -0
  239. graph_agents_cli/scaffold/utils/upgrade.py +894 -0
  240. graph_agents_cli/scaffold/utils/version.py +438 -0
  241. graph_agents_cli/secrets/__init__.py +15 -0
  242. graph_agents_cli/secrets/_apply.py +954 -0
  243. graph_agents_cli/secrets/_required.py +188 -0
  244. graph_agents_cli/secrets/cmd_secrets.py +211 -0
  245. graph_agents_cli/setup/__init__.py +13 -0
  246. graph_agents_cli/setup/_antigravity.py +221 -0
  247. graph_agents_cli/setup/cmd_auth.py +1030 -0
  248. graph_agents_cli/setup/cmd_dev_token.py +513 -0
  249. graph_agents_cli/setup/cmd_setup.py +428 -0
  250. graph_agents_cli/setup/cmd_update.py +140 -0
  251. graph_agents_cli/skills/__init__.py +13 -0
  252. graph_agents_cli/skills/_bundle.py +65 -0
  253. graph_agents_cli/skills/data/README.md +19 -0
  254. graph_agents_cli/skills/data/graph-agents-cli-deploy/SKILL.md +357 -0
  255. graph_agents_cli/skills/data/graph-agents-cli-deploy/references/github-settings.md +113 -0
  256. graph_agents_cli/skills/data/graph-agents-cli-deploy/references/gitops.md +137 -0
  257. graph_agents_cli/skills/data/graph-agents-cli-deploy/references/kubernetes.md +315 -0
  258. graph_agents_cli/skills/data/graph-agents-cli-deploy/references/secrets.md +160 -0
  259. graph_agents_cli/skills/data/graph-agents-cli-eval/SKILL.md +303 -0
  260. graph_agents_cli/skills/data/graph-agents-cli-eval/references/dataset_schema.md +282 -0
  261. graph_agents_cli/skills/data/graph-agents-cli-eval/references/metrics-guide.md +143 -0
  262. graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/SKILL.md +659 -0
  263. graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langchain-models.md +124 -0
  264. graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langgraph.md +235 -0
  265. graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/template-contract.md +477 -0
  266. graph_agents_cli/skills/data/graph-agents-cli-observability/SKILL.md +231 -0
  267. graph_agents_cli/skills/data/graph-agents-cli-observability/references/langsmith.md +46 -0
  268. graph_agents_cli/skills/data/graph-agents-cli-observability/references/otel.md +59 -0
  269. graph_agents_cli/skills/data/graph-agents-cli-scaffold/SKILL.md +414 -0
  270. graph_agents_cli/skills/data/graph-agents-cli-scaffold/references/flags.md +134 -0
  271. graph_agents_cli/skills/data/graph-agents-cli-workflow/SKILL.md +478 -0
  272. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/brainstorming.md +118 -0
  273. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/commands.md +419 -0
  274. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/extension.md +156 -0
  275. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/internals.md +67 -0
  276. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/spec-template.md +56 -0
  277. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/terminology.md +119 -0
  278. graph_agents_cli/system/__init__.py +15 -0
  279. graph_agents_cli/system/_apply.py +519 -0
  280. graph_agents_cli/system/_checks.py +1023 -0
  281. graph_agents_cli/system/_deploy.py +215 -0
  282. graph_agents_cli/system/_model.py +363 -0
  283. graph_agents_cli/system/_system.py +664 -0
  284. graph_agents_cli/system/_views.py +208 -0
  285. graph_agents_cli/system/cmd_system.py +423 -0
  286. graph_agents_cli-0.3.1.dist-info/METADATA +162 -0
  287. graph_agents_cli-0.3.1.dist-info/RECORD +291 -0
  288. graph_agents_cli-0.3.1.dist-info/WHEEL +4 -0
  289. graph_agents_cli-0.3.1.dist-info/entry_points.txt +2 -0
  290. graph_agents_cli-0.3.1.dist-info/licenses/LICENSE +201 -0
  291. graph_agents_cli-0.3.1.dist-info/licenses/NOTICE +19 -0
@@ -0,0 +1,1866 @@
1
+ # Copyright 2026 graph-agents-cli contributors
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # https://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+ """graph-agents-cli deploy command — deploy the agent to Kubernetes.
15
+
16
+ Every check that needs only the project (chart and values, image reference,
17
+ env file) runs before anything touches a cluster, so a configuration error
18
+ never leaves a half-done deploy behind. Then the kube context is printed (and
19
+ confirmed outside ``dev``), and only then are images built and loaded or
20
+ pushed, the Secret applied, its required keys verified, and helm run.
21
+
22
+ A failed rollout leaves the environment as it was: the release is rolled
23
+ back to its last good revision, and the app Secret this run applied is put
24
+ back to its previous values (only while nobody else changed it since).
25
+ ``--status`` and ``--restart`` wait for the rollout for a bounded time and
26
+ explain a rollout that does not finish.
27
+ """
28
+
29
+ from __future__ import annotations
30
+
31
+ import datetime as _dt
32
+ import json
33
+ import re
34
+ from dataclasses import dataclass
35
+ from pathlib import Path
36
+ from typing import Any
37
+
38
+ import click
39
+
40
+ from graph_agents_cli._output import Console
41
+ from graph_agents_cli.deploy import _image, _kube, _modes, _preflight, gitops, local_load
42
+ from graph_agents_cli.deploy._config import DeploySettings, load_settings
43
+ from graph_agents_cli.deploy._kube import ConfigError, Refused, Target
44
+ from graph_agents_cli.deploy._modes import ResolvedContext
45
+ from graph_agents_cli.deploy._values import load_chart_values, split_image_ref
46
+ from graph_agents_cli.secrets import _apply as secrets_apply
47
+ from graph_agents_cli.secrets import _required
48
+
49
+ PROTECTED_ENVS = _modes.PROTECTED_ENVS
50
+ DEFAULT_TIMEOUT = "5m"
51
+ # `--status` only reads: it reports within a minute unless told otherwise.
52
+ DEFAULT_STATUS_TIMEOUT = "60s"
53
+ # Warning events older than the start of this run (less this clock-skew
54
+ # allowance) belong to earlier rollouts and are not printed.
55
+ EVENT_CLOCK_SKEW_S = 30
56
+ _DURATION = re.compile(r"^(?=\d)(?:(\d+)h)?(?:(\d+)m)?(?:(\d+)s)?$")
57
+ DIAGNOSTIC_PODS = 3
58
+ DIAGNOSTIC_LOG_LINES = 40
59
+ DIAGNOSTIC_EVENTS = 15
60
+
61
+
62
+ def _timeout_option(_ctx: click.Context, _param: click.Parameter, value: str | None) -> str | None:
63
+ """A helm duration: ``300`` (seconds), ``300s``, ``10m`` or ``1h30m``; never zero."""
64
+ if value is None:
65
+ return None
66
+ raw = value.strip().lower()
67
+ if raw.isdigit():
68
+ raw += "s"
69
+ match = _DURATION.match(raw)
70
+ if not match or not any(match.groups()) or not any(int(g or 0) for g in match.groups()):
71
+ raise click.BadParameter(
72
+ f"{value!r} is not a duration; use for example 300s, 10m or 1h30m."
73
+ )
74
+ return raw
75
+
76
+
77
+ @click.command("deploy")
78
+ @click.option("--env", "env", required=True, help="Target environment (dev, staging, prod).")
79
+ @click.option(
80
+ "--image",
81
+ "image",
82
+ default=None,
83
+ help="Image reference to deploy instead of building one (CI mode).",
84
+ )
85
+ @click.option(
86
+ "--env-file",
87
+ "env_file",
88
+ default=None,
89
+ help="Env file for the Secret; defaults to .env.<env> (dev also falls back to .env).",
90
+ )
91
+ @click.option(
92
+ "--context",
93
+ "context",
94
+ default=None,
95
+ help="Kube context to use instead of environments.<env>.context.",
96
+ )
97
+ @click.option(
98
+ "--yes",
99
+ "-y",
100
+ "yes",
101
+ is_flag=True,
102
+ help="Accept the kubeconfig's current context outside dev without prompting.",
103
+ )
104
+ @click.option(
105
+ "--status",
106
+ "status",
107
+ is_flag=True,
108
+ help="Report the rollout, pods and warning events instead of deploying (exit 1 when not "
109
+ "ready within --timeout, default 60s).",
110
+ )
111
+ @click.option(
112
+ "--restart",
113
+ "restart",
114
+ is_flag=True,
115
+ help="Rollout-restart the Deployment (after a Secret rotation) and wait for the new pods.",
116
+ )
117
+ @click.option(
118
+ "--force-direct",
119
+ "force_direct",
120
+ is_flag=True,
121
+ help="helm-push mode: allow a deploy to staging/prod from outside CI.",
122
+ )
123
+ @click.option(
124
+ "--dry-run",
125
+ "dry_run",
126
+ is_flag=True,
127
+ help="Print every command; run `helm template` instead of upgrade.",
128
+ )
129
+ @click.option(
130
+ "--tag",
131
+ "tag",
132
+ default=None,
133
+ help="Image tag for a local build (default: short git sha, plus -dirty-<time> for "
134
+ "uncommitted changes; a timestamp outside git).",
135
+ )
136
+ @click.option(
137
+ "--timeout",
138
+ "timeout",
139
+ default=None,
140
+ callback=_timeout_option,
141
+ help=f"How long to wait for the rollout (e.g. 300s, 10m; default {DEFAULT_TIMEOUT}, "
142
+ f"{DEFAULT_STATUS_TIMEOUT} for --status).",
143
+ )
144
+ @click.option(
145
+ "--atomic/--no-atomic",
146
+ "atomic",
147
+ default=True,
148
+ show_default=True,
149
+ help="Roll back a failed rollout (after printing pod diagnostics).",
150
+ )
151
+ @click.option(
152
+ "--rotate-api-key",
153
+ "rotate_api_key",
154
+ is_flag=True,
155
+ help="Replace the live API_KEY with the one in the env file (otherwise the live key wins).",
156
+ )
157
+ def cmd_deploy(
158
+ env: str,
159
+ image: str | None,
160
+ env_file: str | None,
161
+ context: str | None,
162
+ yes: bool,
163
+ status: bool,
164
+ restart: bool,
165
+ force_direct: bool,
166
+ dry_run: bool,
167
+ tag: str | None,
168
+ timeout: str | None,
169
+ atomic: bool,
170
+ rotate_api_key: bool,
171
+ ) -> None:
172
+ """Deploy the agent to Kubernetes (mode depends on the project's CD setting).
173
+
174
+ \b
175
+ Modes (from create_params.cd in the manifest):
176
+ skip direct: build, local-load or push, apply the Secret, helm upgrade
177
+ helm-push CI builds and pushes; deploy --image runs helm only
178
+ argocd never runs helm: writes image.tag into values-<env>.yaml and opens a PR
179
+ \b
180
+ Outside dev the kube context must be recorded in the manifest
181
+ (environments.<env>.context) or passed with --context; the kubeconfig's
182
+ current context is used only after a confirmation (or --yes).
183
+ """
184
+ console = Console()
185
+ settings = load_settings()
186
+ namespace = settings.target(env).namespace
187
+ _modes.derive_mode(settings.cd, None) # an unknown cd value is a configuration error
188
+ resolved = _modes.resolve(settings, env, context)
189
+ target = Target(context=resolved.name, namespace=namespace)
190
+
191
+ if status:
192
+ _modes.announce(env, resolved, console=console)
193
+ _modes.require_known_context(env, resolved, dry_run=dry_run, console=console)
194
+ _show_status(
195
+ settings,
196
+ env,
197
+ target,
198
+ timeout=timeout or DEFAULT_STATUS_TIMEOUT,
199
+ dry_run=dry_run,
200
+ console=console,
201
+ )
202
+ return
203
+ timeout = timeout or DEFAULT_TIMEOUT
204
+
205
+ if env in PROTECTED_ENVS and not settings.auth_policy_implemented:
206
+ raise Refused(
207
+ f"Refusing to deploy to {env}: the manifest records auth_policy_implemented: false.\n"
208
+ f" Implement the {settings.auth_policy} auth policy (the custom stub lives in\n"
209
+ " <agent_directory>/policies/custom.py), then set auth_policy_implemented: true\n"
210
+ " in graph-agents-cli-manifest.yaml."
211
+ )
212
+
213
+ if restart:
214
+ _modes.announce(env, resolved, console=console)
215
+ _modes.confirm(env, resolved, yes=yes, dry_run=dry_run, console=console, action="restart")
216
+ _restart(settings, env, target, timeout=timeout, dry_run=dry_run, console=console)
217
+ return
218
+
219
+ console.print(f"Environment: {env} namespace: {namespace}")
220
+ options = _Options(
221
+ image=image,
222
+ env_file=env_file,
223
+ tag=tag,
224
+ yes=yes,
225
+ force_direct=force_direct,
226
+ dry_run=dry_run,
227
+ timeout=timeout,
228
+ atomic=atomic,
229
+ rotate_api_key=rotate_api_key,
230
+ )
231
+ if settings.cd == _modes.ARGOCD:
232
+ _deploy_argocd(settings, env, resolved, options, console=console)
233
+ elif settings.cd == _modes.HELM_PUSH:
234
+ _deploy_helm_push(settings, env, resolved, target, options, console=console)
235
+ else:
236
+ _deploy_direct(settings, env, resolved, target, options, console=console)
237
+
238
+
239
+ @dataclass(frozen=True)
240
+ class _Options:
241
+ image: str | None
242
+ env_file: str | None
243
+ tag: str | None
244
+ yes: bool
245
+ force_direct: bool
246
+ dry_run: bool
247
+ timeout: str
248
+ atomic: bool
249
+ rotate_api_key: bool
250
+
251
+
252
+ @dataclass(frozen=True)
253
+ class _ImagePlan:
254
+ repository: str
255
+ tag: str
256
+ build: bool
257
+
258
+ @property
259
+ def ref(self) -> str:
260
+ return f"{self.repository}:{self.tag}"
261
+
262
+
263
+ # --------------------------------------------------------------------------- modes
264
+
265
+
266
+ def _deploy_direct(
267
+ settings: DeploySettings,
268
+ env: str,
269
+ resolved: ResolvedContext,
270
+ target: Target,
271
+ opts: _Options,
272
+ *,
273
+ console: Console,
274
+ ) -> None:
275
+ # Project-only checks first: nothing below them may touch a cluster.
276
+ _require_chart(settings, env)
277
+ plan = _image_plan(settings, opts, console=console)
278
+ path = secrets_apply.resolve_env_file(env, opts.env_file)
279
+ chart_values = load_chart_values(settings.chart_dir, env)
280
+ _check_chart_env(settings, env, chart_values, console=console)
281
+ _check_peers(settings, env, chart_values, dry_run=opts.dry_run, console=console)
282
+ _check_jwt(settings, env, chart_values, None, dry_run=opts.dry_run, console=console, warn=False)
283
+ if path is None and not _modes.is_dev_env(env):
284
+ raise secrets_apply.missing_env_file_error(
285
+ env, _required.for_environment(settings, chart_values).secret_keys
286
+ )
287
+ values: dict[str, str] = {}
288
+ if path is not None:
289
+ values = secrets_apply.read_env_file(path)
290
+ # The allow-list for this environment (AUTH_JWT_SECRET joins it for HS* JWTs).
291
+ settings = _required.for_environment(settings, chart_values, values)
292
+ if path is not None:
293
+ secrets_apply.check_file_values(
294
+ values, settings.secret_keys, rotate_api_key=opts.rotate_api_key, source=path
295
+ )
296
+ elif opts.rotate_api_key:
297
+ raise ConfigError("--rotate-api-key needs an env file that sets API_KEY.")
298
+
299
+ _modes.announce(env, resolved, console=console)
300
+ _modes.confirm(
301
+ env, resolved, yes=opts.yes, dry_run=opts.dry_run, console=console, action="deploy to"
302
+ )
303
+
304
+ cluster: local_load.LocalCluster | None = None
305
+ if plan.build:
306
+ cluster, why = local_load.detect(target.context)
307
+ mode = _modes.LOCAL_LOAD if cluster else _modes.REGISTRY
308
+ detail = f": {cluster.describe()}" if cluster else f" ({why})" if why else ""
309
+ console.print(f"Mode: {_modes.describe(mode)}{detail}", markup=False)
310
+ else:
311
+ console.print("Mode: direct, image given (build, load and push skipped)")
312
+ console.print(f"Using image {plan.ref}.")
313
+
314
+ # Plan the Secret and check its required keys (read-only) before anything is
315
+ # built, pushed or changed: an incomplete Secret stops the deploy up front.
316
+ # --dry-run makes the same reads, so it refuses what the real run would.
317
+ secret_plan: secrets_apply.SecretPlan | None = None
318
+ live: secrets_apply.LiveSecret | None = None
319
+ if path is None:
320
+ keys = _verify_secret(
321
+ settings, env, target, fix_hint=None, dry_run=opts.dry_run, console=console
322
+ )
323
+ else:
324
+ secret_plan, live = secrets_apply.prepare(
325
+ name=settings.secret_name,
326
+ env=env,
327
+ target=target,
328
+ allowed=settings.secret_keys,
329
+ path=path,
330
+ values=values,
331
+ rotate_api_key=opts.rotate_api_key,
332
+ dry_run=opts.dry_run,
333
+ mint_api_key=_preflight.effective_auth_policy(settings, chart_values)
334
+ == "shared-bearer",
335
+ metrics_name=settings.metrics_secret_name,
336
+ read_live_on_dry_run=True,
337
+ )
338
+ keys = _verify_secret(
339
+ settings,
340
+ env,
341
+ target,
342
+ keys=set(secret_plan.data),
343
+ uncertain=bool(secret_plan.live_unread),
344
+ fix_hint=f"Add them to {path} and re-run deploy",
345
+ dry_run=opts.dry_run,
346
+ console=console,
347
+ )
348
+ _warn_dsn_without_tls(settings, env, chart_values, secret_plan.data, console=console)
349
+ if not secret_plan.data:
350
+ # Nothing to apply (a keyless project: the fake model, a keyless endpoint): like
351
+ # no env file, the Secret is left as it is, decided before anything is built.
352
+ _require_secret_to_leave(
353
+ settings, env, target, chart_values, live, path, dry_run=opts.dry_run
354
+ )
355
+ secret_plan = None
356
+ _check_jwt(settings, env, chart_values, keys, dry_run=opts.dry_run, console=console)
357
+ _check_release_idle(settings, target, dry_run=opts.dry_run, console=console)
358
+ before = _live_workload(settings, target)
359
+ _announce_same_image(settings, env, plan, before, console=console)
360
+
361
+ if plan.build:
362
+ _build(settings, plan, dry_run=opts.dry_run, console=console)
363
+ if cluster is not None:
364
+ _load(cluster, plan.ref, dry_run=opts.dry_run, console=console)
365
+ else:
366
+ _kube.run_cmd(
367
+ ["docker", "push", plan.ref], capture=False, dry_run=opts.dry_run, console=console
368
+ )
369
+
370
+ snaps: list[secrets_apply.Snapshot] = []
371
+ if secret_plan is None and path is not None:
372
+ console.print(
373
+ f" {path} sets none of the allow-listed keys ({', '.join(settings.secret_keys)}); "
374
+ f"the Secret {settings.secret_name} is left as is.",
375
+ style="yellow",
376
+ markup=False,
377
+ )
378
+ elif secret_plan is None:
379
+ console.print(
380
+ f" No env file found (.env.{env} or .env); the Secret {settings.secret_name} is "
381
+ "left as is.",
382
+ style="yellow",
383
+ )
384
+ else:
385
+ console.print(
386
+ f"Applying Secret {settings.secret_name} from {path} (allow-listed keys only)."
387
+ )
388
+ _required.print_unreached_hs_settings(
389
+ settings, env, chart_values, values, source=path, console=console
390
+ )
391
+ if not opts.dry_run:
392
+ # What the Secret(s) held before, to put back if the rollout fails.
393
+ snaps = secrets_apply.snapshot(secret_plan, live, managed=settings.secret_keys)
394
+ if opts.dry_run:
395
+ console.print(
396
+ " [dry-run] if the rollout fails and the release is rolled back, the Secret is "
397
+ "put back to its current values.",
398
+ style="cyan",
399
+ markup=False,
400
+ )
401
+ try:
402
+ if secret_plan is not None:
403
+ secrets_apply.apply_plan(secret_plan, dry_run=opts.dry_run, console=console, live=live)
404
+ _helm_upgrade(settings, env, target, plan, opts, console=console)
405
+ except _kube.DeployError as e:
406
+ # A failure while applying (the metrics Secret, say) leaves the release untouched too.
407
+ lines = _secret_outcome(snaps, env, e, console=console)
408
+ if not lines:
409
+ raise
410
+ raise type(e)("\n ".join([str(e.message), *lines])) from None
411
+ except KeyboardInterrupt:
412
+ # Interrupted mid-rollout: the release's state is unknown, so nothing is undone.
413
+ for line in secrets_apply.describe_unrestored(snaps, env, "the deploy was interrupted"):
414
+ console.print(f" {line}", style="yellow", markup=False)
415
+ raise
416
+ _report_rollout(settings, env, target, before, snaps, dry_run=opts.dry_run, console=console)
417
+ _print_done(settings, env, plan.ref, dry_run=opts.dry_run, console=console)
418
+
419
+
420
+ def _deploy_helm_push(
421
+ settings: DeploySettings,
422
+ env: str,
423
+ resolved: ResolvedContext,
424
+ target: Target,
425
+ opts: _Options,
426
+ *,
427
+ console: Console,
428
+ ) -> None:
429
+ _refuse_secrets(settings, env, opts, mode=_modes.HELM_PUSH, console=console)
430
+ _require_chart(settings, env)
431
+ chart_values = load_chart_values(settings.chart_dir, env)
432
+ _check_chart_env(settings, env, chart_values, console=console)
433
+ _check_peers(settings, env, chart_values, dry_run=opts.dry_run, console=console)
434
+ _check_jwt(settings, env, chart_values, None, dry_run=opts.dry_run, console=console, warn=False)
435
+ if env in PROTECTED_ENVS and not opts.force_direct and not _kube.in_ci():
436
+ raise Refused(
437
+ f"Refusing to deploy {env} from outside CI in helm-push mode (with or without "
438
+ "--image).\n"
439
+ f" The staging and promote-to-prod workflows run `deploy --env {env} --image <ref>` "
440
+ "on the self-hosted runner (GITHUB_ACTIONS=true); pass --force-direct to deploy "
441
+ "from here anyway."
442
+ )
443
+ plan = _image_plan(settings, opts, console=console)
444
+
445
+ _modes.announce(env, resolved, console=console)
446
+ _modes.confirm(
447
+ env, resolved, yes=opts.yes, dry_run=opts.dry_run, console=console, action="deploy to"
448
+ )
449
+ console.print(f"Mode: {_modes.describe(_modes.HELM_PUSH)}")
450
+ keys = _verify_secret(
451
+ settings,
452
+ env,
453
+ target,
454
+ fix_hint=f"The Secret owner provisions them with `graph-agents-cli secrets apply --env {env}`",
455
+ dry_run=opts.dry_run,
456
+ console=console,
457
+ )
458
+ _check_jwt(settings, env, chart_values, keys, dry_run=opts.dry_run, console=console)
459
+ _check_release_idle(settings, target, dry_run=opts.dry_run, console=console)
460
+ before = _live_workload(settings, target)
461
+ _announce_same_image(settings, env, plan, before, console=console)
462
+ if plan.build:
463
+ _build(settings, plan, dry_run=opts.dry_run, console=console)
464
+ _kube.run_cmd(
465
+ ["docker", "push", plan.ref], capture=False, dry_run=opts.dry_run, console=console
466
+ )
467
+ _helm_upgrade(settings, env, target, plan, opts, console=console)
468
+ _report_rollout(settings, env, target, before, [], dry_run=opts.dry_run, console=console)
469
+ _print_done(settings, env, plan.ref, dry_run=opts.dry_run, console=console)
470
+
471
+
472
+ def _deploy_argocd(
473
+ settings: DeploySettings,
474
+ env: str,
475
+ resolved: ResolvedContext,
476
+ opts: _Options,
477
+ *,
478
+ console: Console,
479
+ ) -> None:
480
+ console.print(f"Mode: {_modes.describe(_modes.ARGOCD)}")
481
+ console.print(
482
+ f" No cluster is contacted (kube context {resolved.name or '(none)'} is not used): "
483
+ "Argo CD applies the change after the pull request merges.",
484
+ style="dim",
485
+ markup=False,
486
+ )
487
+ _refuse_secrets(settings, env, opts, mode=_modes.ARGOCD, console=console)
488
+ values_path = settings.values_file(env)
489
+ if not values_path.is_file():
490
+ raise ConfigError(f"Values file not found: {values_path}")
491
+ chart_values = load_chart_values(settings.chart_dir, env)
492
+ # Argo CD renders these values as they are: check what its pods would get.
493
+ _check_chart_env(settings, env, chart_values, console=console)
494
+ _check_peers(settings, env, chart_values, dry_run=opts.dry_run, console=console)
495
+ _check_jwt(settings, env, chart_values, None, dry_run=opts.dry_run, console=console)
496
+ chart_repository = str((chart_values.get("image") or {}).get("repository") or "")
497
+ if _image.has_placeholder(chart_repository):
498
+ raise ConfigError(
499
+ f"image.repository in the chart values is still the placeholder {chart_repository!r}; "
500
+ "Argo CD would pull it. Set image.repository in "
501
+ f"{settings.chart_dir / 'values.yaml'} (and create_params.registry in the manifest)."
502
+ )
503
+ if opts.image:
504
+ repository, image_tag = split_image_ref(opts.image)
505
+ problem = _image.reference_problem(repository, image_tag)
506
+ if problem:
507
+ raise ConfigError(f"--image: {problem}")
508
+ if chart_repository and repository != chart_repository:
509
+ console.print(
510
+ f" argocd mode writes only image.tag: Argo CD pulls {chart_repository}:"
511
+ f"{image_tag}, not {repository}:{image_tag}. Change image.repository in the "
512
+ "chart values if the repository moved.",
513
+ style="yellow",
514
+ markup=False,
515
+ )
516
+ else:
517
+ repository = (settings.registry and settings.image_repository) or None
518
+ image_tag = opts.tag or gitops.short_sha() or _timestamp()
519
+ problem = _image.tag_problem(image_tag)
520
+ if problem:
521
+ raise ConfigError(f"--tag: {problem}.")
522
+ console.print(
523
+ f" No --image given; writing tag {image_tag!r}. The image must already be pushed "
524
+ "by CI for Argo CD to roll it out.",
525
+ style="yellow",
526
+ )
527
+ if opts.tag is None and gitops.worktree_dirty():
528
+ console.print(
529
+ " The working tree has uncommitted changes; they are not in the image CI "
530
+ f"built for {image_tag}.",
531
+ style="yellow",
532
+ )
533
+ result = gitops.write_desired_state(
534
+ env=env,
535
+ values_path=values_path,
536
+ image_repository=repository,
537
+ tag=image_tag,
538
+ project_name=settings.project_name,
539
+ dry_run=opts.dry_run,
540
+ console=console,
541
+ )
542
+ if not result.changed:
543
+ return
544
+ what = "Would open" if opts.dry_run else ("Opened" if result.created else "Updated")
545
+ where = f" {result.url}" if result.url else ""
546
+ console.print(
547
+ f"{what} pull request on branch {result.branch}.{where}", style="green", markup=False
548
+ )
549
+ if env == "prod":
550
+ console.print(
551
+ " The production change lands only when the PR merges after code-owner review."
552
+ )
553
+
554
+
555
+ # --------------------------------------------------------------------------- steps
556
+
557
+
558
+ def _require_chart(settings: DeploySettings, env: str) -> None:
559
+ chart = settings.chart_dir
560
+ if not (chart / "Chart.yaml").is_file():
561
+ raise ConfigError(f"Helm chart not found at {chart} (expected Chart.yaml).")
562
+ if not settings.values_file(env).is_file():
563
+ raise ConfigError(f"Values file not found: {settings.values_file(env)}")
564
+ _require_gateway_values(settings, env)
565
+
566
+
567
+ def _truthy(value: object) -> bool:
568
+ return value is True or (
569
+ isinstance(value, str) and value.strip().lower() in ("true", "yes", "1")
570
+ )
571
+
572
+
573
+ def _require_gateway_values(settings: DeploySettings, env: str) -> None:
574
+ """Mirror the chart's ``required`` on ``gateway.parentRef.name`` before any tool runs.
575
+
576
+ The scaffolded staging/prod values ship ``gateway.enabled: true`` with a blank
577
+ parentRef (the operator names the Gateway); helm would only refuse after the
578
+ image was built and pushed and the Secret applied, as a tool failure (exit
579
+ 2). This is configuration, so it is reported as such (exit 3) up front.
580
+ """
581
+ values = load_chart_values(settings.chart_dir, env)
582
+ gateway = values.get("gateway") if isinstance(values.get("gateway"), dict) else {}
583
+ if not _truthy(gateway.get("enabled")):
584
+ return
585
+ parent = gateway.get("parentRef") if isinstance(gateway.get("parentRef"), dict) else {}
586
+ if str(parent.get("name") or "").strip():
587
+ return
588
+ raise ConfigError(
589
+ f"gateway.parentRef.name is blank in {settings.values_file(env)} but gateway.enabled is "
590
+ "true: set it to the Gateway to attach to (or set gateway.enabled: false / "
591
+ "ingress.enabled: true)."
592
+ )
593
+
594
+
595
+ def _dockerfile(settings: DeploySettings) -> str:
596
+ for candidate in ("Dockerfile", f"Dockerfile.{settings.runtime}"):
597
+ if Path(candidate).is_file():
598
+ return candidate
599
+ raise ConfigError("No Dockerfile found in the project root.")
600
+
601
+
602
+ def _timestamp() -> str:
603
+ return _dt.datetime.now(_dt.UTC).strftime("%Y%m%d%H%M%S")
604
+
605
+
606
+ def _workstation_tag(console: Console) -> str:
607
+ """The short commit SHA; ``<sha>-dirty-<time>`` with uncommitted changes; a timestamp outside git.
608
+
609
+ Rebuilding at the same commit under the same tag renders an identical pod
610
+ spec (and ``IfNotPresent`` nodes keep the old image), so helm reports
611
+ success while nothing rolls out. A dirty tree therefore always gets a new tag.
612
+ """
613
+ sha = gitops.short_sha()
614
+ if not sha:
615
+ return _timestamp()
616
+ if gitops.worktree_dirty():
617
+ tag = f"{sha}-dirty-{_timestamp()}"
618
+ console.print(
619
+ f" The working tree has uncommitted changes: tagging the image {tag} so the "
620
+ "rollout picks them up. Commit first for a reproducible deploy.",
621
+ style="yellow",
622
+ markup=False,
623
+ )
624
+ return tag
625
+ return sha
626
+
627
+
628
+ def _image_plan(settings: DeploySettings, opts: _Options, *, console: Console) -> _ImagePlan:
629
+ """The image to deploy, validated before docker, kubectl or helm run (exit 3 when invalid)."""
630
+ if opts.image:
631
+ repository, image_tag = split_image_ref(opts.image)
632
+ problem = _image.reference_problem(repository, image_tag)
633
+ if problem:
634
+ raise ConfigError(f"--image: {problem}")
635
+ return _ImagePlan(repository, image_tag, build=False)
636
+ if _image.has_placeholder(settings.registry):
637
+ raise ConfigError(_image.placeholder_message(settings.registry))
638
+ repository = settings.image_repository
639
+ image_tag = opts.tag or _workstation_tag(console)
640
+ problem = _image.reference_problem(repository, image_tag, registry=settings.registry)
641
+ if problem:
642
+ raise ConfigError(problem)
643
+ return _ImagePlan(repository, image_tag, build=True)
644
+
645
+
646
+ def _build(settings: DeploySettings, plan: _ImagePlan, *, dry_run: bool, console: Console) -> None:
647
+ dockerfile = _dockerfile(settings)
648
+ _kube.run_cmd(
649
+ ["docker", "build", "-t", plan.ref, "-f", dockerfile, "."],
650
+ capture=False,
651
+ dry_run=dry_run,
652
+ console=console,
653
+ )
654
+
655
+
656
+ def _load(cluster: local_load.LocalCluster, image: str, *, dry_run: bool, console: Console) -> None:
657
+ commands = local_load.local_load_commands(cluster, image)
658
+ if not commands:
659
+ console.print(f" {cluster.describe()}: no image load needed.", markup=False)
660
+ for cmd in commands:
661
+ _kube.run_cmd(cmd, capture=False, dry_run=dry_run, console=console)
662
+
663
+
664
+ def _refuse_secrets(
665
+ settings: DeploySettings, env: str, opts: _Options, *, mode: str, console: Console
666
+ ) -> None:
667
+ procedure = secrets_apply.provisioning_procedure(
668
+ project=settings.project_name, env=env, owner=settings.secrets_owner, mode=mode
669
+ )
670
+ for flag, given in (("--env-file", opts.env_file), ("--rotate-api-key", opts.rotate_api_key)):
671
+ if given:
672
+ raise Refused(f"{flag} is not accepted in {mode} mode.\n {procedure}")
673
+ console.print(f" {procedure}", style="dim", markup=False)
674
+
675
+
676
+ def _first_line(error: object) -> str:
677
+ return next((line.strip() for line in str(error).splitlines() if line.strip()), "")
678
+
679
+
680
+ def _verify_secret(
681
+ settings: DeploySettings,
682
+ env: str,
683
+ target: Target,
684
+ *,
685
+ keys: set[str] | None = None,
686
+ uncertain: bool = False,
687
+ fix_hint: str | None,
688
+ dry_run: bool,
689
+ console: Console,
690
+ ) -> set[str] | None:
691
+ """Refuse (exit 1) before anything changes when the Secret lacks a key the pods need.
692
+
693
+ ``keys`` are the keys the Secret will hold after this deploy applies it; when
694
+ ``None`` the live Secret is read (read-only, so ``--dry-run`` reads it too and
695
+ refuses what the real run would). ``uncertain``: under ``--dry-run`` the live
696
+ Secret could not be read, so a key missing from ``keys`` may still be there.
697
+ Nothing has been built, pushed or applied when this refuses. Returns the keys
698
+ the Secret holds or will hold (``None`` when unknown).
699
+ """
700
+ required = _required.required_keys(settings, load_chart_values(settings.chart_dir, env))
701
+ name = settings.secret_name
702
+ absent = False
703
+ if keys is None:
704
+ try:
705
+ found = secrets_apply.secret_keys_present(name, target, console=console)
706
+ except _kube.ToolFailed as e:
707
+ if not dry_run:
708
+ raise
709
+ console.print(
710
+ f" [dry-run] could not read Secret {name} ({_first_line(e)}); the real run "
711
+ "refuses unless it holds: " + (", ".join(required) or "(no required key)"),
712
+ style="yellow",
713
+ markup=False,
714
+ )
715
+ return None
716
+ absent = found is None
717
+ present = found or set()
718
+ else:
719
+ present = keys
720
+ if not required:
721
+ return present
722
+ missing = [k for k in required if k not in present]
723
+ if not missing:
724
+ verb = "would hold" if dry_run else ("will hold" if keys is not None else "holds")
725
+ console.print(
726
+ f" Secret {name} {verb} the required key(s): {', '.join(required)}.", style="dim"
727
+ )
728
+ return present
729
+ if uncertain:
730
+ console.print(
731
+ f" [dry-run] Secret {name} would lack {', '.join(missing)} unless the live Secret "
732
+ "(not readable here) holds them: the real run refuses otherwise.",
733
+ style="yellow",
734
+ markup=False,
735
+ )
736
+ return present
737
+ fix = fix_hint or (
738
+ f"Put them in .env.{env} and re-run deploy, or provision them with "
739
+ f"`graph-agents-cli secrets apply --env {env}`"
740
+ )
741
+ why_jwt = (
742
+ f"\n {_required.JWT_SECRET_KEY} is needed because {settings.values_file(env)} (or "
743
+ "values.yaml) lists an HS* algorithm in AUTH_JWT_ALGORITHMS."
744
+ if _required.JWT_SECRET_KEY in missing
745
+ else ""
746
+ )
747
+ dry = "\n (--dry-run: the real deploy stops here the same way.)" if dry_run else ""
748
+ raise Refused(
749
+ f"Secret {name} in {target.namespace} {'would be' if dry_run else 'is'} missing required "
750
+ f"key(s): {', '.join(missing)}{' (the Secret does not exist)' if absent else ''}.\n"
751
+ " Without them the pods crash or answer every request with 503; nothing was built, "
752
+ "applied or deployed.\n"
753
+ f" {fix}, or remove a key from secrets.keys in graph-agents-cli-manifest.yaml if "
754
+ f"{env} does not need it.{why_jwt}{dry}"
755
+ )
756
+
757
+
758
+ def _require_secret_to_leave(
759
+ settings: DeploySettings,
760
+ env: str,
761
+ target: Target,
762
+ values: dict[str, Any],
763
+ live: secrets_apply.LiveSecret | None,
764
+ path: Path,
765
+ *,
766
+ dry_run: bool,
767
+ ) -> None:
768
+ """Refuse (exit 1) before anything changes when the Secret is left alone but the pods need it.
769
+
770
+ An env file with none of the allow-listed keys leaves the Secret as it is.
771
+ Outside dev the chart requires the Secret to exist (``secretOptional:
772
+ false``), so when there is none the pods would never start. ``live`` is
773
+ ``None`` under ``--dry-run`` when the Secret could not be read.
774
+ """
775
+ if _truthy(values.get("secretOptional")) or live is None or live.exists:
776
+ return
777
+ name = settings.secret_name
778
+ dry = "\n (--dry-run: the real deploy stops here the same way.)" if dry_run else ""
779
+ raise Refused(
780
+ f"Secret {name} does not exist in {target.namespace}, and the {env} chart values require "
781
+ "it (secretOptional: false): the pods would not start.\n"
782
+ f" {path} sets none of the allow-listed keys ({', '.join(settings.secret_keys)}), so "
783
+ "there is nothing to create it from; nothing was built, applied or deployed.\n"
784
+ f" Put the keys {env} needs in {path}, or set secretOptional: true in "
785
+ f"{settings.values_file(env)} if {env} needs no Secret.{dry}"
786
+ )
787
+
788
+
789
+ def _check_chart_env(
790
+ settings: DeploySettings, env: str, values: dict[str, Any], *, console: Console
791
+ ) -> None:
792
+ """Refuse (exit 3) a scaffold placeholder in the chart env outside dev; warn in dev.
793
+
794
+ ``api add`` writes ``<API>_BASE_URL: http://CHANGE-ME`` and an
795
+ ``openai-compatible`` project starts with ``OPENAI_BASE_URL`` at CHANGE-ME:
796
+ the pods would call that address, and every tool (or the model) would fail.
797
+ """
798
+ keys = _preflight.env_placeholders(values)
799
+ if not keys:
800
+ return
801
+ names = ", ".join(f"env.{k}" for k in keys)
802
+ where = f"{settings.values_file(env)} (or {settings.chart_dir / 'values.yaml'})"
803
+ if _modes.is_dev_env(env):
804
+ console.print(
805
+ f" Warning: {names} still hold(s) the placeholder CHANGE-ME: the {env} pods would "
806
+ f"call it, so those calls fail. Set the real value(s) in {where}.",
807
+ style="yellow",
808
+ markup=False,
809
+ )
810
+ return
811
+ raise ConfigError(
812
+ f"{names} still hold(s) the placeholder CHANGE-ME in the chart values for {env}: the "
813
+ f"pods would call it. Set the real value(s) in {where}."
814
+ )
815
+
816
+
817
+ def _check_peers(
818
+ settings: DeploySettings,
819
+ env: str,
820
+ values: dict[str, Any],
821
+ *,
822
+ dry_run: bool,
823
+ console: Console,
824
+ ) -> None:
825
+ """The peer rules (api-policy.yaml's protocol: a2a APIs): exit 3 outside dev, warnings.
826
+
827
+ Runs before anything is built. An invalid or absent policy is `lint`'s to report.
828
+ """
829
+ from graph_agents_cli._api_policy import POLICY_FILENAME, read_policy_document
830
+
831
+ document = read_policy_document(Path(POLICY_FILENAME))
832
+ findings = _preflight.peer_findings(settings, env, values, document)
833
+ error = findings.error()
834
+ if error:
835
+ raise ConfigError(
836
+ error + ("\n (--dry-run: the real deploy stops here the same way.)" if dry_run else "")
837
+ )
838
+ for note in findings.notes():
839
+ console.print(f" Warning: {note}", style="yellow", markup=False)
840
+
841
+
842
+ def _check_jwt(
843
+ settings: DeploySettings,
844
+ env: str,
845
+ values: dict[str, Any],
846
+ keys: set[str] | None,
847
+ *,
848
+ dry_run: bool,
849
+ console: Console,
850
+ warn: bool = True,
851
+ ) -> None:
852
+ """The ``jwt`` policy's verification settings: exit 3 outside dev, a warning in dev."""
853
+ findings = _preflight.jwt_findings(settings, env, values, keys)
854
+ error = findings.error()
855
+ if error:
856
+ raise ConfigError(
857
+ error + ("\n (--dry-run: the real deploy stops here the same way.)" if dry_run else "")
858
+ )
859
+ warning = findings.warning()
860
+ if warning and warn:
861
+ console.print(f" Warning: {warning}", style="yellow", markup=False)
862
+
863
+
864
+ def _warn_dsn_without_tls(
865
+ settings: DeploySettings,
866
+ env: str,
867
+ values: dict[str, Any],
868
+ data: dict[str, str],
869
+ *,
870
+ console: Console,
871
+ ) -> None:
872
+ """Outside dev, warn when the external database's connection string does not require TLS.
873
+
874
+ The value is inspected in memory and never printed.
875
+ """
876
+ for key in _preflight.dsn_keys(settings, values):
877
+ dsn = data.get(key) or ""
878
+ if _modes.is_dev_env(env) or not dsn or dsn == secrets_apply.PENDING_PLACEHOLDER:
879
+ continue
880
+ if _preflight.dsn_without_tls(values, dsn):
881
+ console.print(
882
+ f" Warning: {_preflight.dsn_tls_warning(key)}", style="yellow", markup=False
883
+ )
884
+
885
+
886
+ @dataclass(frozen=True)
887
+ class _Workload:
888
+ """The live Deployment as far as a deploy needs it (read-only)."""
889
+
890
+ image: str
891
+ generation: int
892
+
893
+
894
+ def _live_workload(settings: DeploySettings, target: Target) -> _Workload | None:
895
+ """The release's Deployment (its agent image and generation); ``None`` when absent or unreadable."""
896
+ try:
897
+ result = _kube.kubectl(
898
+ ["get", "deployment", settings.release, "-o", "json"],
899
+ target,
900
+ check=False,
901
+ quiet=True,
902
+ )
903
+ body = json.loads(result.stdout or "null") if result.returncode == 0 else None
904
+ except (_kube.ToolFailed, json.JSONDecodeError):
905
+ return None
906
+ if not isinstance(body, dict):
907
+ return None
908
+ containers = ((body.get("spec") or {}).get("template") or {}).get("spec", {}).get(
909
+ "containers"
910
+ ) or []
911
+ agent = next((c for c in containers if c.get("name") == "agent"), None) or (
912
+ containers[0] if containers else {}
913
+ )
914
+ try:
915
+ generation = int((body.get("metadata") or {}).get("generation") or 0)
916
+ except (TypeError, ValueError):
917
+ generation = 0
918
+ return _Workload(image=str(agent.get("image") or ""), generation=generation)
919
+
920
+
921
+ def _announce_same_image(
922
+ settings: DeploySettings,
923
+ env: str,
924
+ plan: _ImagePlan,
925
+ live: _Workload | None,
926
+ *,
927
+ console: Console,
928
+ ) -> None:
929
+ """Say up front when the release already runs this exact image reference."""
930
+ if live is None or live.image != plan.ref:
931
+ return
932
+ rebuilt = " The image is rebuilt and loaded under the same tag." if plan.build else ""
933
+ console.print(
934
+ f" {settings.release} already runs {plan.ref}: the image is unchanged.{rebuilt} helm "
935
+ "records a new revision, but the pods are replaced only if the chart values change, "
936
+ f"and a changed Secret reaches them only with `graph-agents-cli deploy --env {env} "
937
+ "--restart` (reported after the rollout).",
938
+ style="yellow",
939
+ markup=False,
940
+ )
941
+
942
+
943
+ def _secret_outcome(
944
+ snaps: list[secrets_apply.Snapshot], env: str, error: Exception, *, console: Console
945
+ ) -> list[str]:
946
+ """After a failed helm step: restore the Secret(s) when the release is as it was before."""
947
+ if not any(snap.touched for snap in snaps):
948
+ return []
949
+ if getattr(error, "release_unchanged", True):
950
+ return secrets_apply.restore(snaps, console=console)
951
+ return secrets_apply.describe_unrestored(
952
+ snaps, env, "the release was not put back to its previous revision"
953
+ )
954
+
955
+
956
+ def _report_rollout(
957
+ settings: DeploySettings,
958
+ env: str,
959
+ target: Target,
960
+ before: _Workload | None,
961
+ snaps: list[secrets_apply.Snapshot],
962
+ *,
963
+ dry_run: bool,
964
+ console: Console,
965
+ ) -> None:
966
+ """After a successful upgrade: say when no pod was replaced, and what that leaves stale."""
967
+ if dry_run or before is None:
968
+ return
969
+ after = _live_workload(settings, target)
970
+ if after is None or after.generation != before.generation:
971
+ return
972
+ console.print(
973
+ " No pods were replaced: the pod template (image and chart values) is unchanged.",
974
+ style="yellow",
975
+ )
976
+ changed = sorted({k for snap in snaps for k in snap.touched})
977
+ if changed:
978
+ console.print(
979
+ f" The Secret changed ({', '.join(changed)}), but running pods read it only when "
980
+ f"they start: run `graph-agents-cli deploy --env {env} --restart`.",
981
+ style="yellow",
982
+ markup=False,
983
+ )
984
+
985
+
986
+ def _helm_args(settings: DeploySettings, env: str, repository: str, tag: str) -> list[str]:
987
+ chart = settings.chart_dir
988
+ return [
989
+ str(chart),
990
+ "-f",
991
+ str(chart / "values.yaml"),
992
+ "-f",
993
+ str(settings.values_file(env)),
994
+ "--set",
995
+ f"image.repository={repository}",
996
+ "--set",
997
+ f"image.tag={tag}",
998
+ "--set",
999
+ f"existingSecret={settings.secret_name}",
1000
+ ]
1001
+
1002
+
1003
+ def _chart_dependencies(chart: Path) -> list[str]:
1004
+ """Names of the subcharts ``Chart.yaml`` declares (``[]`` when none or unreadable)."""
1005
+ import yaml
1006
+
1007
+ try:
1008
+ with open(chart / "Chart.yaml", encoding="utf-8") as f:
1009
+ data = yaml.safe_load(f) or {}
1010
+ except (OSError, yaml.YAMLError):
1011
+ return []
1012
+ deps = data.get("dependencies") if isinstance(data, dict) else None
1013
+ if not isinstance(deps, list):
1014
+ return []
1015
+ return [str(d.get("name")) for d in deps if isinstance(d, dict) and d.get("name")]
1016
+
1017
+
1018
+ def _missing_dependencies(chart: Path, names: list[str]) -> list[str]:
1019
+ """Declared subcharts with no archive or directory under ``charts/``."""
1020
+ charts_dir = chart / "charts"
1021
+ missing: list[str] = []
1022
+ for name in names:
1023
+ archives = list(charts_dir.glob(f"{name}-*.tgz")) if charts_dir.is_dir() else []
1024
+ if not archives and not (charts_dir / name).is_dir():
1025
+ missing.append(name)
1026
+ return missing
1027
+
1028
+
1029
+ def _helm_dependency_build(settings: DeploySettings, *, dry_run: bool, console: Console) -> None:
1030
+ """Fetch the subcharts ``Chart.yaml`` declares before ``helm upgrade`` or ``helm template``.
1031
+
1032
+ The scaffolded chart declares the bitnami ``postgresql`` and ``redis`` charts as
1033
+ conditional dependencies; helm refuses to render or install until they sit in
1034
+ ``charts/``. The build is skipped when every subchart is present and
1035
+ ``Chart.lock`` exists. Under ``--dry-run`` the command still runs: it only
1036
+ writes into the chart directory (no cluster access) and the dry-run render
1037
+ (``helm template``) cannot succeed without it.
1038
+ """
1039
+ chart = settings.chart_dir
1040
+ names = _chart_dependencies(chart)
1041
+ if not names:
1042
+ return
1043
+ missing = _missing_dependencies(chart, names)
1044
+ if not missing and (chart / "Chart.lock").is_file():
1045
+ return
1046
+ cmd = ["helm", "dependency", "build", str(chart)]
1047
+ _kube.echo_cmd(cmd, dry_run=dry_run, console=console)
1048
+ if dry_run:
1049
+ console.print(
1050
+ f" [dry-run] fetching subchart(s) {', '.join(missing or names)} into "
1051
+ f"{chart / 'charts'} so the render below can run (local only).",
1052
+ style="cyan",
1053
+ markup=False,
1054
+ )
1055
+ result = _kube.run_cmd(cmd, check=False, quiet=True, console=console)
1056
+ if result.returncode != 0:
1057
+ raise _kube.ToolFailed(
1058
+ f"helm dependency build failed (exit code {result.returncode}); the chart declares "
1059
+ f"subchart(s) {', '.join(names)} that must be fetched before helm can render it. "
1060
+ "Check network access to the chart repository or vendor the charts under "
1061
+ f"{chart / 'charts'}.\n{(result.stderr or result.stdout).strip()}"
1062
+ )
1063
+
1064
+
1065
+ def _helm_upgrade(
1066
+ settings: DeploySettings,
1067
+ env: str,
1068
+ target: Target,
1069
+ plan: _ImagePlan,
1070
+ opts: _Options,
1071
+ *,
1072
+ console: Console,
1073
+ ) -> None:
1074
+ """``helm upgrade --install --wait --timeout``; on failure print diagnostics, then roll back.
1075
+
1076
+ The rollback (``--atomic``, the default) is done here rather than with
1077
+ helm's own ``--atomic``: helm rolls back before it returns, which deletes
1078
+ the failed pods and their logs, the very output that explains the failure.
1079
+ It follows helm's rule: back to the newest deployed or superseded revision,
1080
+ or uninstall a first install that never succeeded.
1081
+
1082
+ It only ever undoes the revision this run created. The release's newest
1083
+ revision is recorded just before the upgrade; after a failure the CLI acts
1084
+ only when exactly one newer revision exists and it is ``failed``. When helm
1085
+ reports another operation in progress, or another deploy changed the
1086
+ release meanwhile, the release is left alone: rolling back would undo (or
1087
+ uninstall) someone else's rollout.
1088
+ """
1089
+ common = _helm_args(settings, env, plan.repository, plan.tag)
1090
+ upgrade = [
1091
+ "upgrade",
1092
+ "--install",
1093
+ settings.release,
1094
+ *common,
1095
+ "--create-namespace",
1096
+ "--wait",
1097
+ "--timeout",
1098
+ opts.timeout,
1099
+ ]
1100
+ _helm_dependency_build(settings, dry_run=opts.dry_run, console=console)
1101
+ if opts.dry_run:
1102
+ _kube.echo_cmd(_kube.helm_args(upgrade, target), dry_run=True, console=console)
1103
+ if opts.atomic:
1104
+ console.print(
1105
+ " [dry-run] a failed rollout prints pod diagnostics and is rolled back "
1106
+ "(--no-atomic keeps it).",
1107
+ style="cyan",
1108
+ markup=False,
1109
+ )
1110
+ console.print(
1111
+ " [dry-run] rendering with `helm template` instead:", style="cyan", markup=False
1112
+ )
1113
+ result = _kube.helm(
1114
+ ["template", settings.release, *common], target, check=False, console=console
1115
+ )
1116
+ if result.returncode != 0:
1117
+ raise _kube.ToolFailed(
1118
+ f"helm template failed (exit code {result.returncode}):\n{(result.stderr or result.stdout).strip()}"
1119
+ )
1120
+ if result.stdout:
1121
+ console.print(result.stdout, highlight=False, markup=False)
1122
+ return
1123
+ before = _release_history(settings, target, console=console)
1124
+ _refuse_if_busy(settings, target, before, changed="the release was not touched")
1125
+ console.print(f" helm waits up to {opts.timeout} for the rollout.", style="dim")
1126
+ started = _now()
1127
+ # Captured (helm prints nothing until the rollout ends under --wait) so that
1128
+ # helm's own "another operation is in progress" refusal can be recognised.
1129
+ result = _kube.helm(upgrade, target, check=False, console=console)
1130
+ for stream in (result.stdout, result.stderr):
1131
+ if (stream or "").strip():
1132
+ console.print(stream.rstrip(), highlight=False, markup=False)
1133
+ if result.returncode == 0:
1134
+ return
1135
+ after = _release_history(settings, target, console=console)
1136
+ if _HELM_BUSY in f"{result.stderr or ''}\n{result.stdout or ''}":
1137
+ raise RolloutFailed(
1138
+ _busy_message(settings, target, after, changed="the release was not touched"),
1139
+ release_unchanged=True,
1140
+ )
1141
+ outcome, unchanged = _failure_outcome(
1142
+ settings, target, before, after, opts, since=started, console=console
1143
+ )
1144
+ raise RolloutFailed(
1145
+ f"helm upgrade failed (exit code {result.returncode}) for {settings.release} in "
1146
+ f"{target.namespace}; {outcome}.",
1147
+ release_unchanged=unchanged,
1148
+ )
1149
+
1150
+
1151
+ class RolloutFailed(_kube.ToolFailed):
1152
+ """helm failed (exit 2); ``release_unchanged``: the release is back as it was before the run.
1153
+
1154
+ Only then is the Secret this run applied put back: a release left on the
1155
+ failed (or another deploy's) revision keeps the values it was rolled out with.
1156
+ """
1157
+
1158
+ def __init__(self, message: str, *, release_unchanged: bool = False) -> None:
1159
+ super().__init__(message)
1160
+ self.release_unchanged = release_unchanged
1161
+
1162
+
1163
+ def _now() -> _dt.datetime:
1164
+ return _dt.datetime.now(_dt.UTC)
1165
+
1166
+
1167
+ def _failure_outcome(
1168
+ settings: DeploySettings,
1169
+ target: Target,
1170
+ before: _History,
1171
+ after: _History,
1172
+ opts: _Options,
1173
+ *,
1174
+ since: _dt.datetime | None = None,
1175
+ console: Console,
1176
+ ) -> tuple[str, bool]:
1177
+ """Undo the failed revision this run created, if any; describe what happened to the release.
1178
+
1179
+ Returns the description and whether the release is back as it was before the
1180
+ run (no new revision, rolled back, or a failed first install uninstalled).
1181
+ """
1182
+ check = f"check `helm history {settings.release} -n {target.namespace}`"
1183
+ if not (before.readable and after.readable):
1184
+ return (
1185
+ "the release history could not be read, so the failure cannot be tied to a "
1186
+ f"revision of this run and nothing was rolled back; {check}"
1187
+ ), False
1188
+ newer = after.newer_than(before.latest)
1189
+ if not newer:
1190
+ return (
1191
+ "helm recorded no new revision (it failed before the rollout), so the release is "
1192
+ "unchanged and there is nothing to roll back"
1193
+ ), True
1194
+ newest = newer[-1]
1195
+ number, status = int(newest["revision"]), _status(newest)
1196
+ if status.startswith(_PENDING):
1197
+ if len(newer) > 1:
1198
+ return (
1199
+ f"another helm operation started after this one failed (revision {number} is "
1200
+ f"{status}), so nothing was rolled back; {check}"
1201
+ ), False
1202
+ # Either this run's helm stopped before finishing its revision (killed,
1203
+ # crashed, lost the connection) or another deploy holds the release.
1204
+ return (
1205
+ f"revision {number} is still {status} (helm stopped before finishing it, or "
1206
+ "another deploy is working on the release), so nothing was rolled back; if no "
1207
+ f"other deploy is running, {_clear_hint(settings, target, after)} clears it"
1208
+ ), False
1209
+ if len(newer) > 1:
1210
+ return (
1211
+ f"another deploy changed the release while this one ran (revisions "
1212
+ f"{int(newer[0]['revision'])} to {number} are new), so nothing was rolled back; "
1213
+ f"{check}"
1214
+ ), False
1215
+ if status != "failed":
1216
+ return (
1217
+ f"its new revision {number} is {status}, so nothing was rolled back; {check}",
1218
+ False,
1219
+ )
1220
+ # Exactly one new revision and it failed: the one this run created.
1221
+ # Read the diagnostics before the rollback: it removes the failed pods and their logs.
1222
+ _print_rollout_diagnostics(settings, target, since=since, console=console)
1223
+ if not opts.atomic:
1224
+ return (
1225
+ f"the failed revision {number} was left in place (--no-atomic); roll back with "
1226
+ f"`helm rollback {settings.release} -n {target.namespace}`"
1227
+ ), False
1228
+ return _roll_back(
1229
+ settings,
1230
+ target,
1231
+ after,
1232
+ number,
1233
+ existed=before.installed,
1234
+ timeout=opts.timeout,
1235
+ console=console,
1236
+ )
1237
+
1238
+
1239
+ def _diag(console: Console, args: list[str], target: Target, *, tail: int | None = None) -> str:
1240
+ """Run a read-only kubectl for diagnostics and print it; never raises. Returns stdout."""
1241
+ cmd = _kube.kubectl_args(args, target)
1242
+ try:
1243
+ result = _kube.run_cmd(cmd, check=False, quiet=True)
1244
+ except _kube.ToolFailed as e:
1245
+ console.print(f" $ {_kube.format_cmd(cmd)}\n {e}", markup=False, highlight=False)
1246
+ return ""
1247
+ out = (result.stdout or "").rstrip() or (result.stderr or "").rstrip() or "(no output)"
1248
+ lines = out.splitlines()
1249
+ if tail is not None and len(lines) > tail + 1:
1250
+ lines = lines[:1] + lines[-tail:] # keep the header row
1251
+ console.print(f" $ {_kube.format_cmd(cmd)}", style="dim", markup=False, highlight=False)
1252
+ for line in lines:
1253
+ console.print(f" {line}", markup=False, highlight=False)
1254
+ return result.stdout or ""
1255
+
1256
+
1257
+ def _pod_ready(pod: dict[str, Any]) -> bool:
1258
+ status = pod.get("status") or {}
1259
+ if status.get("phase") == "Succeeded":
1260
+ return True
1261
+ return any(
1262
+ c.get("type") == "Ready" and str(c.get("status")) == "True"
1263
+ for c in status.get("conditions") or []
1264
+ )
1265
+
1266
+
1267
+ def _terminating(pod: dict[str, Any]) -> bool:
1268
+ return bool((pod.get("metadata") or {}).get("deletionTimestamp"))
1269
+
1270
+
1271
+ def _container_notes(pod: dict[str, Any]) -> list[str]:
1272
+ status = pod.get("status") or {}
1273
+ notes: list[str] = []
1274
+ for c in (status.get("initContainerStatuses") or []) + (status.get("containerStatuses") or []):
1275
+ name = c.get("name", "?")
1276
+ state = c.get("state") or {}
1277
+ last = (c.get("lastState") or {}).get("terminated") or {}
1278
+ if "waiting" in state:
1279
+ w = state["waiting"] or {}
1280
+ notes.append(f"{name}: waiting {w.get('reason', '')} {w.get('message', '')}".rstrip())
1281
+ elif "terminated" in state:
1282
+ t = state["terminated"] or {}
1283
+ notes.append(f"{name}: terminated {t.get('reason', '')} exit {t.get('exitCode')}")
1284
+ if last:
1285
+ notes.append(
1286
+ f"{name}: last run {last.get('reason', '')} exit {last.get('exitCode')} "
1287
+ f"(restarts {c.get('restartCount', 0)})"
1288
+ )
1289
+ return notes
1290
+
1291
+
1292
+ def _release_pods(settings: DeploySettings, target: Target) -> list[dict[str, Any]]:
1293
+ """The release's pods (``-l app.kubernetes.io/instance=<release>``); ``[]`` when unreadable."""
1294
+ selector = f"app.kubernetes.io/instance={settings.release}"
1295
+ try:
1296
+ listing = _kube.run_cmd(
1297
+ _kube.kubectl_args(["get", "pods", "-l", selector, "-o", "json"], target),
1298
+ check=False,
1299
+ quiet=True,
1300
+ )
1301
+ pods = json.loads(listing.stdout or "{}").get("items") or []
1302
+ except (_kube.ToolFailed, json.JSONDecodeError, AttributeError):
1303
+ return []
1304
+ return [p for p in pods if isinstance(p, dict)]
1305
+
1306
+
1307
+ def _release_objects(settings: DeploySettings, target: Target) -> set[tuple[str, str]]:
1308
+ """``(kind, name)`` of the release's objects that warning events are about.
1309
+
1310
+ The Deployment, its ReplicaSets and pods, and the bundled database's
1311
+ StatefulSet, pods and volume claims: everything labelled with the release.
1312
+ """
1313
+ selector = f"app.kubernetes.io/instance={settings.release}"
1314
+ objects = {("Deployment", settings.release)}
1315
+ try:
1316
+ listing = _kube.run_cmd(
1317
+ _kube.kubectl_args(
1318
+ [
1319
+ "get",
1320
+ "pods,replicasets,deployments,statefulsets,persistentvolumeclaims",
1321
+ "-l",
1322
+ selector,
1323
+ "-o",
1324
+ "json",
1325
+ ],
1326
+ target,
1327
+ ),
1328
+ check=False,
1329
+ quiet=True,
1330
+ )
1331
+ items = json.loads(listing.stdout or "{}").get("items") or []
1332
+ except (_kube.ToolFailed, json.JSONDecodeError, AttributeError):
1333
+ items = []
1334
+ for item in items:
1335
+ if isinstance(item, dict):
1336
+ name = (item.get("metadata") or {}).get("name")
1337
+ if item.get("kind") and name:
1338
+ objects.add((str(item["kind"]), str(name)))
1339
+ return objects
1340
+
1341
+
1342
+ def _parse_time(value: object) -> _dt.datetime | None:
1343
+ if not value:
1344
+ return None
1345
+ try:
1346
+ parsed = _dt.datetime.fromisoformat(str(value).replace("Z", "+00:00"))
1347
+ except ValueError:
1348
+ return None
1349
+ return parsed if parsed.tzinfo else parsed.replace(tzinfo=_dt.UTC)
1350
+
1351
+
1352
+ def _event_time(event: dict[str, Any]) -> _dt.datetime | None:
1353
+ """When a (possibly repeated) event was last seen."""
1354
+ series = event.get("series") or {}
1355
+ for value in (
1356
+ event.get("lastTimestamp"),
1357
+ series.get("lastObservedTime") if isinstance(series, dict) else None,
1358
+ event.get("eventTime"),
1359
+ event.get("firstTimestamp"),
1360
+ (event.get("metadata") or {}).get("creationTimestamp"),
1361
+ ):
1362
+ parsed = _parse_time(value)
1363
+ if parsed is not None:
1364
+ return parsed
1365
+ return None
1366
+
1367
+
1368
+ def _age(when: _dt.datetime | None) -> str:
1369
+ if when is None:
1370
+ return "?"
1371
+ seconds = max(0, int((_now() - when).total_seconds()))
1372
+ if seconds < 120:
1373
+ return f"{seconds}s"
1374
+ if seconds < 7200:
1375
+ return f"{seconds // 60}m"
1376
+ return f"{seconds // 3600}h"
1377
+
1378
+
1379
+ def _print_warning_events(
1380
+ settings: DeploySettings,
1381
+ target: Target,
1382
+ *,
1383
+ since: _dt.datetime | None,
1384
+ console: Console,
1385
+ ) -> None:
1386
+ """Warning events about this release's objects, newer than ``since`` (less clock skew).
1387
+
1388
+ The namespace may hold other workloads and events from earlier rollouts
1389
+ (events live for an hour): only this release's objects are shown, and with
1390
+ ``since`` only what happened during this run.
1391
+ """
1392
+ cmd = _kube.kubectl_args(
1393
+ ["get", "events", "--field-selector", "type=Warning", "-o", "json"], target
1394
+ )
1395
+ console.print(f" $ {_kube.format_cmd(cmd)}", style="dim", markup=False, highlight=False)
1396
+ try:
1397
+ result = _kube.run_cmd(cmd, check=False, quiet=True)
1398
+ events = json.loads(result.stdout or "{}").get("items") or []
1399
+ except (_kube.ToolFailed, json.JSONDecodeError, AttributeError) as e:
1400
+ console.print(f" (could not list events: {_first_line(e) or 'no output'})", markup=False)
1401
+ return
1402
+ objects = _release_objects(settings, target)
1403
+ cutoff = since - _dt.timedelta(seconds=EVENT_CLOCK_SKEW_S) if since else None
1404
+ mine: list[tuple[_dt.datetime | None, dict[str, Any]]] = []
1405
+ older = 0
1406
+ for event in events:
1407
+ if not isinstance(event, dict):
1408
+ continue
1409
+ involved = event.get("involvedObject") or event.get("regarding") or {}
1410
+ if (str(involved.get("kind")), str(involved.get("name"))) not in objects:
1411
+ continue
1412
+ when = _event_time(event)
1413
+ if cutoff is not None and (when is None or when < cutoff):
1414
+ older += 1
1415
+ continue
1416
+ mine.append((when, event))
1417
+ mine.sort(key=lambda pair: pair[0] or _dt.datetime.min.replace(tzinfo=_dt.UTC))
1418
+ if not mine:
1419
+ console.print(" (no warning events for this release)", markup=False)
1420
+ for when, event in mine[-DIAGNOSTIC_EVENTS:]:
1421
+ involved = event.get("involvedObject") or event.get("regarding") or {}
1422
+ count = event.get("count") or ((event.get("series") or {}).get("count"))
1423
+ times = f" (x{count})" if isinstance(count, int) and count > 1 else ""
1424
+ message = " ".join(str(event.get("message") or event.get("note") or "").split())
1425
+ console.print(
1426
+ f" {_age(when):>4} ago {involved.get('kind')}/{involved.get('name')} "
1427
+ f"{event.get('reason', '')}: {message}{times}",
1428
+ markup=False,
1429
+ highlight=False,
1430
+ )
1431
+ if older:
1432
+ console.print(
1433
+ f" ({older} older warning event(s) for this release, from before this run, not "
1434
+ "shown)",
1435
+ style="dim",
1436
+ markup=False,
1437
+ )
1438
+
1439
+
1440
+ def _print_rollout_diagnostics(
1441
+ settings: DeploySettings,
1442
+ target: Target,
1443
+ *,
1444
+ since: _dt.datetime | None = None,
1445
+ headline: str | None = None,
1446
+ console: Console,
1447
+ ) -> None:
1448
+ """Pods, container states, this release's warning events and recent logs (best effort)."""
1449
+ selector = f"app.kubernetes.io/instance={settings.release}"
1450
+ console.print(
1451
+ headline or f"Rollout of {settings.release} in {target.namespace} failed; diagnostics:",
1452
+ style="yellow",
1453
+ )
1454
+ _diag(console, ["get", "pods", "-l", selector, "-o", "wide"], target)
1455
+ pods = _release_pods(settings, target)
1456
+ # A terminating pod belongs to the revision being replaced: not what failed.
1457
+ failing = [p for p in pods if not _pod_ready(p) and not _terminating(p)]
1458
+ for pod in failing[:DIAGNOSTIC_PODS]:
1459
+ name = (pod.get("metadata") or {}).get("name", "?")
1460
+ for note in _container_notes(pod):
1461
+ console.print(f" {name}: {note}", markup=False, highlight=False)
1462
+ _print_warning_events(settings, target, since=since, console=console)
1463
+ for pod in failing[:DIAGNOSTIC_PODS]:
1464
+ name = (pod.get("metadata") or {}).get("name", "?")
1465
+ logs = _diag(
1466
+ console,
1467
+ ["logs", name, "--all-containers", f"--tail={DIAGNOSTIC_LOG_LINES}"],
1468
+ target,
1469
+ )
1470
+ restarted = any(
1471
+ int(c.get("restartCount") or 0) > 0
1472
+ for c in (pod.get("status") or {}).get("containerStatuses") or []
1473
+ )
1474
+ if restarted and not logs.strip():
1475
+ _diag(
1476
+ console,
1477
+ ["logs", name, "--all-containers", "--previous", f"--tail={DIAGNOSTIC_LOG_LINES}"],
1478
+ target,
1479
+ )
1480
+
1481
+
1482
+ _HEALTHY = ("deployed", "superseded")
1483
+ _PENDING = "pending-"
1484
+ # helm's refusal while the release's newest revision is pending-install/-upgrade/-rollback.
1485
+ _HELM_BUSY = "another operation (install/upgrade/rollback) is in progress"
1486
+
1487
+
1488
+ def _status(revision: dict[str, Any]) -> str:
1489
+ return str(revision.get("status", "")).strip().lower()
1490
+
1491
+
1492
+ @dataclass(frozen=True)
1493
+ class _History:
1494
+ """``helm history`` of the release; ``readable`` is False when helm could not read it."""
1495
+
1496
+ revisions: tuple[dict[str, Any], ...] = ()
1497
+ readable: bool = True
1498
+
1499
+ @property
1500
+ def latest(self) -> int:
1501
+ """The newest revision number (0 when the release has none)."""
1502
+ return max((int(r["revision"]) for r in self.revisions), default=0)
1503
+
1504
+ def newer_than(self, revision: int) -> list[dict[str, Any]]:
1505
+ """Revisions after ``revision``, oldest first."""
1506
+ newer = [r for r in self.revisions if int(r["revision"]) > revision]
1507
+ return sorted(newer, key=lambda r: int(r["revision"]))
1508
+
1509
+ @property
1510
+ def pending(self) -> dict[str, Any] | None:
1511
+ """The newest revision when helm is (or was, if interrupted) still working on it."""
1512
+ newest = max(self.revisions, key=lambda r: int(r["revision"]), default=None)
1513
+ return newest if newest is not None and _status(newest).startswith(_PENDING) else None
1514
+
1515
+ @property
1516
+ def installed(self) -> bool:
1517
+ """Whether the release exists (``uninstall --keep-history`` leaves uninstalled ones)."""
1518
+ return any(_status(r) != "uninstalled" for r in self.revisions)
1519
+
1520
+
1521
+ def _release_history(settings: DeploySettings, target: Target, *, console: Console) -> _History:
1522
+ """``helm history`` of the release: empty when it does not exist, unreadable on other errors."""
1523
+ try:
1524
+ history = _kube.helm(
1525
+ ["history", settings.release, "-o", "json"],
1526
+ target,
1527
+ check=False,
1528
+ quiet=True,
1529
+ console=console,
1530
+ )
1531
+ except _kube.ToolFailed:
1532
+ return _History(readable=False)
1533
+ if history.returncode != 0:
1534
+ # `Error: release: not found` means no release yet; anything else is unknown.
1535
+ return _History(readable="release: not found" in (history.stderr or ""))
1536
+ try:
1537
+ revisions = json.loads(history.stdout or "[]")
1538
+ except json.JSONDecodeError:
1539
+ return _History(readable=False)
1540
+ if not isinstance(revisions, list):
1541
+ return _History(readable=False)
1542
+ return _History(
1543
+ tuple(r for r in revisions if isinstance(r, dict) and str(r.get("revision", "")).isdigit())
1544
+ )
1545
+
1546
+
1547
+ def _clear_hint(settings: DeploySettings, target: Target, history: _History) -> str:
1548
+ """The command that clears a release left pending by an interrupted helm."""
1549
+ release, ns = settings.release, target.namespace
1550
+ pending = history.pending
1551
+ below = int(pending["revision"]) if pending is not None else history.latest + 1
1552
+ good = [
1553
+ int(r["revision"])
1554
+ for r in history.revisions
1555
+ if _status(r) in _HEALTHY and int(r["revision"]) < below
1556
+ ]
1557
+ if good:
1558
+ return f"`helm rollback {release} {max(good)} -n {ns}`"
1559
+ if pending is not None and _status(pending) == "pending-install":
1560
+ return f"`helm uninstall {release} -n {ns}` (the first install never finished)"
1561
+ return f"`helm rollback {release} <last good revision> -n {ns}`"
1562
+
1563
+
1564
+ def _busy_message(
1565
+ settings: DeploySettings, target: Target, history: _History, *, changed: str
1566
+ ) -> str:
1567
+ release, ns = settings.release, target.namespace
1568
+ pending = history.pending
1569
+ what = f" (revision {pending['revision']} is {_status(pending)})" if pending else ""
1570
+ return (
1571
+ f"Another helm operation (install/upgrade/rollback) is in progress on {release} in "
1572
+ f"{ns}{what}; {changed}, and nothing was rolled back.\n"
1573
+ f" Wait for it to finish (`helm history {release} -n {ns}`) and deploy again. If no "
1574
+ "other deploy is running, an interrupted helm left the release pending: "
1575
+ f"{_clear_hint(settings, target, history)} clears it."
1576
+ )
1577
+
1578
+
1579
+ def _refuse_if_busy(
1580
+ settings: DeploySettings, target: Target, history: _History, *, changed: str
1581
+ ) -> None:
1582
+ """Exit 2 when another helm operation holds the release (helm itself would refuse)."""
1583
+ if history.pending is not None:
1584
+ raise _kube.ToolFailed(_busy_message(settings, target, history, changed=changed))
1585
+
1586
+
1587
+ def _check_release_idle(
1588
+ settings: DeploySettings, target: Target, *, dry_run: bool, console: Console
1589
+ ) -> None:
1590
+ """Stop before anything is built or applied while another deploy is rolling out."""
1591
+ if dry_run:
1592
+ return
1593
+ history = _release_history(settings, target, console=console)
1594
+ _refuse_if_busy(settings, target, history, changed="nothing was built, applied or deployed")
1595
+
1596
+
1597
+ def _roll_back(
1598
+ settings: DeploySettings,
1599
+ target: Target,
1600
+ history: _History,
1601
+ failed: int,
1602
+ *,
1603
+ existed: bool,
1604
+ timeout: str,
1605
+ console: Console,
1606
+ ) -> tuple[str, bool]:
1607
+ """Undo this run's failed revision ``failed`` the way helm's ``--atomic`` would; describe it.
1608
+
1609
+ Back to the newest deployed or superseded revision before it; with none, a
1610
+ release this run installed (``existed`` False) is uninstalled, and one that
1611
+ existed before is left in place (helm's ``--atomic`` does the same).
1612
+ """
1613
+ release = settings.release
1614
+ good = [r for r in history.revisions if _status(r) in _HEALTHY and int(r["revision"]) < failed]
1615
+ if good:
1616
+ revision = str(max(int(r["revision"]) for r in good))
1617
+ console.print(f"Rolling back {release} to revision {revision} (--atomic).", style="yellow")
1618
+ result = _kube.helm(
1619
+ ["rollback", release, revision, "--wait", "--timeout", timeout],
1620
+ target,
1621
+ capture=False,
1622
+ check=False,
1623
+ console=console,
1624
+ )
1625
+ if result.returncode == 0:
1626
+ return f"rolled back to revision {revision}", True
1627
+ return (
1628
+ f"the rollback to revision {revision} also failed (exit code {result.returncode}); "
1629
+ f"check `helm history {release} -n {target.namespace}`"
1630
+ ), False
1631
+ if existed:
1632
+ return (
1633
+ f"there is no earlier successful revision to roll back to, so the failed revision "
1634
+ f"{failed} was left in place; fix the cause and deploy again, or remove it with "
1635
+ f"`helm uninstall {release} -n {target.namespace}`"
1636
+ ), False
1637
+ console.print(
1638
+ f"Uninstalling {release}: the first install never succeeded (--atomic).", style="yellow"
1639
+ )
1640
+ result = _kube.helm(
1641
+ ["uninstall", release, "--wait", "--timeout", timeout],
1642
+ target,
1643
+ capture=False,
1644
+ check=False,
1645
+ console=console,
1646
+ )
1647
+ if result.returncode == 0:
1648
+ return "the failed first install was uninstalled (there was no earlier revision)", True
1649
+ return (
1650
+ f"uninstalling the failed first install also failed (exit code {result.returncode}); "
1651
+ f"check `helm status {release} -n {target.namespace}`"
1652
+ ), False
1653
+
1654
+
1655
+ def _print_done(
1656
+ settings: DeploySettings, env: str, image: str, *, dry_run: bool, console: Console
1657
+ ) -> None:
1658
+ verb = "Would deploy" if dry_run else "Deployed"
1659
+ console.print(f"{verb} {settings.release} ({image}) to {env}.", style="green")
1660
+
1661
+
1662
+ # --------------------------------------------------------------------------- status / restart
1663
+
1664
+
1665
+ class NotReady(_kube.DeployError):
1666
+ """``deploy --status``: the rollout is not complete within ``--timeout`` (exit 1)."""
1667
+
1668
+ exit_code = 1
1669
+
1670
+
1671
+ _ROLLOUT_OK = "ok"
1672
+ _ROLLOUT_NOT_READY = "not-ready"
1673
+ _ROLLOUT_ABSENT = "absent"
1674
+
1675
+
1676
+ def _rollout_status_cmd(settings: DeploySettings, target: Target, timeout: str) -> list[str]:
1677
+ return _kube.kubectl_args(
1678
+ ["rollout", "status", f"deployment/{settings.release}", f"--timeout={timeout}"], target
1679
+ )
1680
+
1681
+
1682
+ def _wait_for_rollout(
1683
+ settings: DeploySettings, target: Target, *, timeout: str, console: Console
1684
+ ) -> tuple[str, str]:
1685
+ """``kubectl rollout status --timeout``: ``(state, kubectl's last line)``, never unbounded.
1686
+
1687
+ A timeout or an exceeded progress deadline is ``not-ready`` and a missing
1688
+ Deployment ``absent``; any other kubectl failure (cluster unreachable,
1689
+ credentials) is a tool failure (exit 2).
1690
+ """
1691
+ cmd = _rollout_status_cmd(settings, target, timeout)
1692
+ _kube.echo_cmd(cmd, console=console)
1693
+ console.print(
1694
+ f" Waiting up to {timeout} for deployment/{settings.release} to roll out.", style="dim"
1695
+ )
1696
+ result = _kube.run_cmd(cmd, check=False, quiet=True)
1697
+ out = (result.stdout or "").strip()
1698
+ err = (result.stderr or "").strip()
1699
+ last = next((line for line in reversed((err or out).splitlines()) if line.strip()), "")
1700
+ if result.returncode == 0:
1701
+ return _ROLLOUT_OK, last
1702
+ if "(NotFound)" in err or "not found" in err.lower():
1703
+ return _ROLLOUT_ABSENT, last
1704
+ if "timed out" in err or "progress deadline" in err:
1705
+ return _ROLLOUT_NOT_READY, last
1706
+ raise _kube.ToolFailed(
1707
+ f"Command failed (exit code {result.returncode}): {_kube.format_cmd(cmd)}"
1708
+ + (f"\n{err or out}" if err or out else "")
1709
+ )
1710
+
1711
+
1712
+ def _print_workload_summary(settings: DeploySettings, target: Target, *, console: Console) -> None:
1713
+ """The Deployment's replicas and image, the helm revision, and every pod's readiness."""
1714
+ release = settings.release
1715
+ try:
1716
+ result = _kube.kubectl(
1717
+ ["get", "deployment", release, "-o", "json"], target, check=False, quiet=True
1718
+ )
1719
+ deployment = json.loads(result.stdout or "null") if result.returncode == 0 else None
1720
+ except (_kube.ToolFailed, json.JSONDecodeError):
1721
+ deployment = None
1722
+ if isinstance(deployment, dict):
1723
+ spec = deployment.get("spec") or {}
1724
+ status = deployment.get("status") or {}
1725
+ containers = ((spec.get("template") or {}).get("spec") or {}).get("containers") or []
1726
+ agent = next((c for c in containers if c.get("name") == "agent"), None) or (
1727
+ containers[0] if containers else {}
1728
+ )
1729
+ wanted = spec.get("replicas", 1)
1730
+ console.print(
1731
+ f"deployment/{release}: {status.get('readyReplicas') or 0}/{wanted} ready, "
1732
+ f"{status.get('updatedReplicas') or 0} up to date, image {agent.get('image', '?')}",
1733
+ markup=False,
1734
+ highlight=False,
1735
+ )
1736
+ for condition in status.get("conditions") or []:
1737
+ if str(condition.get("status")) != "True":
1738
+ console.print(
1739
+ f" {condition.get('type')}: {condition.get('reason', '')} "
1740
+ f"{condition.get('message', '')}".rstrip(),
1741
+ markup=False,
1742
+ highlight=False,
1743
+ )
1744
+ if settings.cd != _modes.ARGOCD:
1745
+ history = _release_history(settings, target, console=console)
1746
+ if history.readable and history.revisions:
1747
+ newest = max(history.revisions, key=lambda r: int(r["revision"]))
1748
+ console.print(
1749
+ f"helm release {release}: revision {newest['revision']} ({_status(newest)})",
1750
+ markup=False,
1751
+ )
1752
+ for pod in _release_pods(settings, target):
1753
+ name = (pod.get("metadata") or {}).get("name", "?")
1754
+ statuses = (pod.get("status") or {}).get("containerStatuses") or []
1755
+ restarts = sum(int(c.get("restartCount") or 0) for c in statuses)
1756
+ last = next(
1757
+ (
1758
+ (c.get("lastState") or {}).get("terminated")
1759
+ for c in statuses
1760
+ if (c.get("lastState") or {}).get("terminated")
1761
+ ),
1762
+ None,
1763
+ )
1764
+ why = f" (last exit: {last.get('reason', '')} {last.get('exitCode')})" if last else ""
1765
+ if _terminating(pod):
1766
+ state = "terminating"
1767
+ else:
1768
+ state = "ready" if _pod_ready(pod) else "NOT ready"
1769
+ console.print(
1770
+ f" pod {name}: {state}, restarts {restarts}{why}", markup=False, highlight=False
1771
+ )
1772
+
1773
+
1774
+ def _show_status(
1775
+ settings: DeploySettings,
1776
+ env: str,
1777
+ target: Target,
1778
+ *,
1779
+ timeout: str,
1780
+ dry_run: bool,
1781
+ console: Console,
1782
+ ) -> None:
1783
+ """Report the rollout within ``timeout``: exit 1 (with diagnostics) when it is not complete."""
1784
+ if settings.cd == _modes.ARGOCD and _kube.tool_available("argocd"):
1785
+ _kube.run_cmd(
1786
+ ["argocd", "app", "get", f"{settings.project_name}-{env}"],
1787
+ capture=False,
1788
+ dry_run=dry_run,
1789
+ console=console,
1790
+ )
1791
+ return
1792
+ if dry_run:
1793
+ _kube.echo_cmd(
1794
+ _rollout_status_cmd(settings, target, timeout), dry_run=True, console=console
1795
+ )
1796
+ return
1797
+ state, detail = _wait_for_rollout(settings, target, timeout=timeout, console=console)
1798
+ where = f"deployment/{settings.release} in {target.namespace}"
1799
+ if state == _ROLLOUT_ABSENT:
1800
+ raise NotReady(f"{where} does not exist (not deployed yet?): {detail}")
1801
+ _print_workload_summary(settings, target, console=console)
1802
+ if state == _ROLLOUT_OK:
1803
+ console.print(f"{where}: rollout complete.", style="green", markup=False)
1804
+ return
1805
+ _print_rollout_diagnostics(
1806
+ settings,
1807
+ target,
1808
+ headline=f"{where} is not ready after {timeout}; diagnostics:",
1809
+ console=console,
1810
+ )
1811
+ raise NotReady(
1812
+ f"The rollout of {where} did not complete within {timeout} ({detail}). Pass a longer "
1813
+ "--timeout to wait more; the diagnostics above show why the pods are not ready."
1814
+ )
1815
+
1816
+
1817
+ def _restart(
1818
+ settings: DeploySettings,
1819
+ env: str,
1820
+ target: Target,
1821
+ *,
1822
+ timeout: str,
1823
+ dry_run: bool,
1824
+ console: Console,
1825
+ ) -> None:
1826
+ """``kubectl rollout restart``, then wait (bounded) until the new pods are ready."""
1827
+ if settings.cd == _modes.ARGOCD:
1828
+ console.print(
1829
+ " Warning: this environment is reconciled by Argo CD with self-heal; the restart "
1830
+ "annotation may be reverted. Prefer an Argo resource action (restart) on the Deployment.",
1831
+ style="yellow",
1832
+ )
1833
+ where = f"deployment/{settings.release} in {target.namespace}"
1834
+ _kube.kubectl(
1835
+ ["rollout", "restart", f"deployment/{settings.release}"],
1836
+ target,
1837
+ capture=False,
1838
+ dry_run=dry_run,
1839
+ console=console,
1840
+ )
1841
+ if dry_run:
1842
+ _kube.echo_cmd(
1843
+ _rollout_status_cmd(settings, target, timeout), dry_run=True, console=console
1844
+ )
1845
+ console.print(f"Would restart {where} and wait up to {timeout} for the new pods.")
1846
+ return
1847
+ started = _now()
1848
+ state, detail = _wait_for_rollout(settings, target, timeout=timeout, console=console)
1849
+ _print_workload_summary(settings, target, console=console)
1850
+ if state == _ROLLOUT_OK:
1851
+ console.print(f"Restarted {where}: the new pods are ready.", style="green", markup=False)
1852
+ return
1853
+ _print_rollout_diagnostics(
1854
+ settings,
1855
+ target,
1856
+ since=started,
1857
+ headline=f"The restart of {where} did not finish within {timeout}; diagnostics:",
1858
+ console=console,
1859
+ )
1860
+ raise _kube.ToolFailed(
1861
+ f"The restarted pods of {where} did not become ready within {timeout} ({detail}).\n"
1862
+ " The pods that were running keep serving until new ones are ready. Fix the cause "
1863
+ f"(often a Secret value: `graph-agents-cli secrets status --env {env}`) and restart "
1864
+ f"again, or go back to the previous pods with `kubectl rollout undo "
1865
+ f"deployment/{settings.release} -n {target.namespace}`."
1866
+ )