graph-agents-cli 0.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_agents_cli/__init__.py +26 -0
- graph_agents_cli/_api_policy.py +2145 -0
- graph_agents_cli/_approvals.py +400 -0
- graph_agents_cli/_build.py +186 -0
- graph_agents_cli/_build_info.json +7 -0
- graph_agents_cli/_chat_client.py +462 -0
- graph_agents_cli/_click.py +157 -0
- graph_agents_cli/_defaults.py +139 -0
- graph_agents_cli/_experiments.py +64 -0
- graph_agents_cli/_http.py +192 -0
- graph_agents_cli/_output.py +83 -0
- graph_agents_cli/_project.py +462 -0
- graph_agents_cli/_remote.py +220 -0
- graph_agents_cli/_response_schema.py +264 -0
- graph_agents_cli/_runner.py +319 -0
- graph_agents_cli/_skills_check.py +274 -0
- graph_agents_cli/_tools.py +189 -0
- graph_agents_cli/_trust.py +66 -0
- graph_agents_cli/api/__init__.py +15 -0
- graph_agents_cli/api/_changes.py +506 -0
- graph_agents_cli/api/_files.py +658 -0
- graph_agents_cli/api/cmd_api.py +2480 -0
- graph_agents_cli/deploy/__init__.py +15 -0
- graph_agents_cli/deploy/_config.py +171 -0
- graph_agents_cli/deploy/_image.py +128 -0
- graph_agents_cli/deploy/_kube.py +286 -0
- graph_agents_cli/deploy/_modes.py +234 -0
- graph_agents_cli/deploy/_preflight.py +370 -0
- graph_agents_cli/deploy/_values.py +168 -0
- graph_agents_cli/deploy/cmd_deploy.py +1866 -0
- graph_agents_cli/deploy/gitops.py +562 -0
- graph_agents_cli/deploy/local_load.py +273 -0
- graph_agents_cli/dev/__init__.py +13 -0
- graph_agents_cli/dev/cmd_build.py +131 -0
- graph_agents_cli/dev/cmd_install.py +78 -0
- graph_agents_cli/dev/cmd_lint.py +119 -0
- graph_agents_cli/dev/cmd_playground.py +297 -0
- graph_agents_cli/dev/policy_check.py +1287 -0
- graph_agents_cli/eval/__init__.py +22 -0
- graph_agents_cli/eval/_client.py +670 -0
- graph_agents_cli/eval/_common.py +177 -0
- graph_agents_cli/eval/_judge.py +168 -0
- graph_agents_cli/eval/_judge_runner.py +238 -0
- graph_agents_cli/eval/_paths.py +212 -0
- graph_agents_cli/eval/checks.py +581 -0
- graph_agents_cli/eval/cmd_analyze.py +278 -0
- graph_agents_cli/eval/cmd_compare.py +284 -0
- graph_agents_cli/eval/cmd_eval_group.py +80 -0
- graph_agents_cli/eval/cmd_generate.py +558 -0
- graph_agents_cli/eval/cmd_grade.py +466 -0
- graph_agents_cli/eval/cmd_metric.py +156 -0
- graph_agents_cli/eval/cmd_run.py +370 -0
- graph_agents_cli/eval/cmd_submit.py +400 -0
- graph_agents_cli/eval/config.py +435 -0
- graph_agents_cli/eval/dataset.py +350 -0
- graph_agents_cli/eval/gate.py +420 -0
- graph_agents_cli/eval/transcript.py +192 -0
- graph_agents_cli/extension/__init__.py +13 -0
- graph_agents_cli/extension/_compat.py +86 -0
- graph_agents_cli/extension/_loader.py +293 -0
- graph_agents_cli/extension/_manifest.py +135 -0
- graph_agents_cli/extension/_overrides.py +195 -0
- graph_agents_cli/extension/_paths.py +91 -0
- graph_agents_cli/extension/_refs.py +193 -0
- graph_agents_cli/extension/_resolver.py +453 -0
- graph_agents_cli/extension/_schema.py +106 -0
- graph_agents_cli/extension/_spec.py +253 -0
- graph_agents_cli/extension/_sync.py +102 -0
- graph_agents_cli/extension/_trust.py +58 -0
- graph_agents_cli/extension/cmd_extension_add.py +259 -0
- graph_agents_cli/extension/cmd_extension_group.py +57 -0
- graph_agents_cli/extension/cmd_extension_list.py +56 -0
- graph_agents_cli/extension/cmd_extension_remove.py +61 -0
- graph_agents_cli/extension/cmd_extension_update.py +195 -0
- graph_agents_cli/info/__init__.py +13 -0
- graph_agents_cli/info/cmd_info.py +222 -0
- graph_agents_cli/infra/__init__.py +15 -0
- graph_agents_cli/infra/checks.py +1169 -0
- graph_agents_cli/infra/cmd_infra.py +103 -0
- graph_agents_cli/main.py +591 -0
- graph_agents_cli/peer/__init__.py +15 -0
- graph_agents_cli/peer/_generate.py +254 -0
- graph_agents_cli/peer/cmd_peer.py +1151 -0
- graph_agents_cli/run/__init__.py +13 -0
- graph_agents_cli/run/_local_server.py +1157 -0
- graph_agents_cli/run/_signals.py +141 -0
- graph_agents_cli/run/cmd_approvals.py +530 -0
- graph_agents_cli/run/cmd_run.py +1421 -0
- graph_agents_cli/scaffold/__init__.py +19 -0
- graph_agents_cli/scaffold/agents/README.md +24 -0
- graph_agents_cli/scaffold/agents/empty_py/.template/templateconfig.yaml +22 -0
- graph_agents_cli/scaffold/agents/langgraph/.env.example +292 -0
- graph_agents_cli/scaffold/agents/langgraph/.template/templateconfig.yaml +28 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile +59 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile.langgraph-server +59 -0
- graph_agents_cli/scaffold/agents/langgraph/README.md +571 -0
- graph_agents_cli/scaffold/agents/langgraph/api-policy.yaml +60 -0
- graph_agents_cli/scaffold/agents/langgraph/app/__init__.py +20 -0
- graph_agents_cli/scaffold/agents/langgraph/app/agent.py +174 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/__init__.py +15 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a.py +2162 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a_client.py +1167 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/api_client.py +4220 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/approvals.py +1349 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/auth.py +1986 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/chat.py +2962 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/checkpointer.py +432 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/content.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/db.py +580 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/limits.py +203 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/metrics.py +231 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/middleware.py +361 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/model.py +611 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/playground.py +230 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/run_locks.py +459 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/structured.py +755 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/telemetry.py +681 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/threads.py +493 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/token_exchange.py +959 -0
- graph_agents_cli/scaffold/agents/langgraph/app/fast_api_app.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/__init__.py +55 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/custom.py +97 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/__init__.py +46 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/example_api.py +92 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/weather.py +33 -0
- graph_agents_cli/scaffold/agents/langgraph/langgraph.json +14 -0
- graph_agents_cli/scaffold/agents/langgraph/pyproject.toml +78 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/conftest.py +376 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/datasets/basic-dataset.json +53 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/eval_config.yaml +32 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/approval_graph.py +137 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_issuer.py +216 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_openai.py +357 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_outcomes.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_relay.py +479 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_api_surface.py +812 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals.py +1367 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals_server.py +794 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor.py +497 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor_server.py +247 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_history_repair.py +278 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_model_apis.py +242 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_postgres.py +637 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_resilience_postgres.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_runtime_guardrails.py +854 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_e2e.py +340 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_runtime.py +989 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_answers.py +584 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_server.py +222 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_token_exchange_issuer.py +650 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/.results/.placeholder +0 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/README.md +22 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/conftest.py +21 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/load_test.py +81 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_client.py +824 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_scoping.py +724 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client.py +1214 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client_hardening.py +716 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_policy_rpc.py +767 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_approval_ledger.py +1536 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_fake_model.py +115 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_jwt_policy.py +991 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_limits.py +310 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging.py +148 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging_hardening.py +271 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_policy.py +378 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_resilience.py +610 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_server_auth.py +702 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_structured.py +673 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_telemetry.py +404 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_thread_listing.py +255 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_threads.py +268 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_token_exchange.py +1320 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_untrusted_content.py +393 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-fastapi.lock +2084 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-langgraph-server.lock +2106 -0
- graph_agents_cli/scaffold/agents/langgraph/{{cookiecutter.agent_guidance_filename}} +129 -0
- graph_agents_cli/scaffold/base_templates/_shared/graph-agents-cli-manifest.yaml +36 -0
- graph_agents_cli/scaffold/base_templates/python/.dockerignore +32 -0
- graph_agents_cli/scaffold/base_templates/python/.github/CODEOWNERS +30 -0
- graph_agents_cli/scaffold/base_templates/python/.github/agent.env +7 -0
- graph_agents_cli/scaffold/base_templates/python/.github/workflows/pr_checks.yaml +214 -0
- graph_agents_cli/scaffold/base_templates/python/.gitignore +209 -0
- graph_agents_cli/scaffold/base_templates/python/tests/unit/test_dummy.py +23 -0
- graph_agents_cli/scaffold/base_templates/python/{{cookiecutter.agent_guidance_filename}} +35 -0
- graph_agents_cli/scaffold/cmd_scaffold_group.py +49 -0
- graph_agents_cli/scaffold/commands/__init__.py +13 -0
- graph_agents_cli/scaffold/commands/create.py +1424 -0
- graph_agents_cli/scaffold/commands/enhance.py +1652 -0
- graph_agents_cli/scaffold/commands/upgrade.py +570 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/agent.env +12 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/promote-to-prod.yaml +371 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/staging.yaml +450 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-dev.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-prod.yaml +41 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-staging.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/.helmignore +14 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/Chart.yaml +21 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/examples/networkpolicy.yaml +103 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/NOTES.txt +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/_helpers.tpl +189 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/certificate.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/configmap.yaml +10 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/deployment.yaml +199 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/hpa.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/httproute.yaml +30 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/ingress.yaml +39 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/networkpolicy.yaml +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/pdb.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/postgresql-secret.yaml +37 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/service.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/serviceaccount.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/servicemonitor.yaml +42 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-dev.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-prod.yaml +45 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-staging.yaml +29 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values.yaml +396 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/tests/integration/test_chart.py +269 -0
- graph_agents_cli/scaffold/deployment_targets/none/README.md +5 -0
- graph_agents_cli/scaffold/deployment_targets/none/python/README.md +6 -0
- graph_agents_cli/scaffold/utils/__init__.py +13 -0
- graph_agents_cli/scaffold/utils/backup.py +212 -0
- graph_agents_cli/scaffold/utils/build_record.py +257 -0
- graph_agents_cli/scaffold/utils/cli_options.py +184 -0
- graph_agents_cli/scaffold/utils/fs.py +83 -0
- graph_agents_cli/scaffold/utils/generate_locks.py +214 -0
- graph_agents_cli/scaffold/utils/generation_metadata.py +88 -0
- graph_agents_cli/scaffold/utils/keyedit.py +768 -0
- graph_agents_cli/scaffold/utils/keymerge.py +537 -0
- graph_agents_cli/scaffold/utils/language.py +138 -0
- graph_agents_cli/scaffold/utils/lock_utils.py +94 -0
- graph_agents_cli/scaffold/utils/logging.py +77 -0
- graph_agents_cli/scaffold/utils/manifest.py +292 -0
- graph_agents_cli/scaffold/utils/merge.py +970 -0
- graph_agents_cli/scaffold/utils/merge3.py +216 -0
- graph_agents_cli/scaffold/utils/openapi_seed.py +199 -0
- graph_agents_cli/scaffold/utils/remote_template.py +376 -0
- graph_agents_cli/scaffold/utils/template.py +1352 -0
- graph_agents_cli/scaffold/utils/upgrade.py +894 -0
- graph_agents_cli/scaffold/utils/version.py +438 -0
- graph_agents_cli/secrets/__init__.py +15 -0
- graph_agents_cli/secrets/_apply.py +954 -0
- graph_agents_cli/secrets/_required.py +188 -0
- graph_agents_cli/secrets/cmd_secrets.py +211 -0
- graph_agents_cli/setup/__init__.py +13 -0
- graph_agents_cli/setup/_antigravity.py +221 -0
- graph_agents_cli/setup/cmd_auth.py +1030 -0
- graph_agents_cli/setup/cmd_dev_token.py +513 -0
- graph_agents_cli/setup/cmd_setup.py +428 -0
- graph_agents_cli/setup/cmd_update.py +140 -0
- graph_agents_cli/skills/__init__.py +13 -0
- graph_agents_cli/skills/_bundle.py +65 -0
- graph_agents_cli/skills/data/README.md +19 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/SKILL.md +357 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/github-settings.md +113 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/gitops.md +137 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/kubernetes.md +315 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/secrets.md +160 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/SKILL.md +303 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/dataset_schema.md +282 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/metrics-guide.md +143 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/SKILL.md +659 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langchain-models.md +124 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langgraph.md +235 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/template-contract.md +477 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/SKILL.md +231 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/langsmith.md +46 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/otel.md +59 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/SKILL.md +414 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/references/flags.md +134 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/SKILL.md +478 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/brainstorming.md +118 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/commands.md +419 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/extension.md +156 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/internals.md +67 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/spec-template.md +56 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/terminology.md +119 -0
- graph_agents_cli/system/__init__.py +15 -0
- graph_agents_cli/system/_apply.py +519 -0
- graph_agents_cli/system/_checks.py +1023 -0
- graph_agents_cli/system/_deploy.py +215 -0
- graph_agents_cli/system/_model.py +363 -0
- graph_agents_cli/system/_system.py +664 -0
- graph_agents_cli/system/_views.py +208 -0
- graph_agents_cli/system/cmd_system.py +423 -0
- graph_agents_cli-0.3.1.dist-info/METADATA +162 -0
- graph_agents_cli-0.3.1.dist-info/RECORD +291 -0
- graph_agents_cli-0.3.1.dist-info/WHEEL +4 -0
- graph_agents_cli-0.3.1.dist-info/entry_points.txt +2 -0
- graph_agents_cli-0.3.1.dist-info/licenses/LICENSE +201 -0
- graph_agents_cli-0.3.1.dist-info/licenses/NOTICE +19 -0
|
@@ -0,0 +1,1866 @@
|
|
|
1
|
+
# Copyright 2026 graph-agents-cli contributors
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# https://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
"""graph-agents-cli deploy command — deploy the agent to Kubernetes.
|
|
15
|
+
|
|
16
|
+
Every check that needs only the project (chart and values, image reference,
|
|
17
|
+
env file) runs before anything touches a cluster, so a configuration error
|
|
18
|
+
never leaves a half-done deploy behind. Then the kube context is printed (and
|
|
19
|
+
confirmed outside ``dev``), and only then are images built and loaded or
|
|
20
|
+
pushed, the Secret applied, its required keys verified, and helm run.
|
|
21
|
+
|
|
22
|
+
A failed rollout leaves the environment as it was: the release is rolled
|
|
23
|
+
back to its last good revision, and the app Secret this run applied is put
|
|
24
|
+
back to its previous values (only while nobody else changed it since).
|
|
25
|
+
``--status`` and ``--restart`` wait for the rollout for a bounded time and
|
|
26
|
+
explain a rollout that does not finish.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
import datetime as _dt
|
|
32
|
+
import json
|
|
33
|
+
import re
|
|
34
|
+
from dataclasses import dataclass
|
|
35
|
+
from pathlib import Path
|
|
36
|
+
from typing import Any
|
|
37
|
+
|
|
38
|
+
import click
|
|
39
|
+
|
|
40
|
+
from graph_agents_cli._output import Console
|
|
41
|
+
from graph_agents_cli.deploy import _image, _kube, _modes, _preflight, gitops, local_load
|
|
42
|
+
from graph_agents_cli.deploy._config import DeploySettings, load_settings
|
|
43
|
+
from graph_agents_cli.deploy._kube import ConfigError, Refused, Target
|
|
44
|
+
from graph_agents_cli.deploy._modes import ResolvedContext
|
|
45
|
+
from graph_agents_cli.deploy._values import load_chart_values, split_image_ref
|
|
46
|
+
from graph_agents_cli.secrets import _apply as secrets_apply
|
|
47
|
+
from graph_agents_cli.secrets import _required
|
|
48
|
+
|
|
49
|
+
PROTECTED_ENVS = _modes.PROTECTED_ENVS
|
|
50
|
+
DEFAULT_TIMEOUT = "5m"
|
|
51
|
+
# `--status` only reads: it reports within a minute unless told otherwise.
|
|
52
|
+
DEFAULT_STATUS_TIMEOUT = "60s"
|
|
53
|
+
# Warning events older than the start of this run (less this clock-skew
|
|
54
|
+
# allowance) belong to earlier rollouts and are not printed.
|
|
55
|
+
EVENT_CLOCK_SKEW_S = 30
|
|
56
|
+
_DURATION = re.compile(r"^(?=\d)(?:(\d+)h)?(?:(\d+)m)?(?:(\d+)s)?$")
|
|
57
|
+
DIAGNOSTIC_PODS = 3
|
|
58
|
+
DIAGNOSTIC_LOG_LINES = 40
|
|
59
|
+
DIAGNOSTIC_EVENTS = 15
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _timeout_option(_ctx: click.Context, _param: click.Parameter, value: str | None) -> str | None:
|
|
63
|
+
"""A helm duration: ``300`` (seconds), ``300s``, ``10m`` or ``1h30m``; never zero."""
|
|
64
|
+
if value is None:
|
|
65
|
+
return None
|
|
66
|
+
raw = value.strip().lower()
|
|
67
|
+
if raw.isdigit():
|
|
68
|
+
raw += "s"
|
|
69
|
+
match = _DURATION.match(raw)
|
|
70
|
+
if not match or not any(match.groups()) or not any(int(g or 0) for g in match.groups()):
|
|
71
|
+
raise click.BadParameter(
|
|
72
|
+
f"{value!r} is not a duration; use for example 300s, 10m or 1h30m."
|
|
73
|
+
)
|
|
74
|
+
return raw
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@click.command("deploy")
|
|
78
|
+
@click.option("--env", "env", required=True, help="Target environment (dev, staging, prod).")
|
|
79
|
+
@click.option(
|
|
80
|
+
"--image",
|
|
81
|
+
"image",
|
|
82
|
+
default=None,
|
|
83
|
+
help="Image reference to deploy instead of building one (CI mode).",
|
|
84
|
+
)
|
|
85
|
+
@click.option(
|
|
86
|
+
"--env-file",
|
|
87
|
+
"env_file",
|
|
88
|
+
default=None,
|
|
89
|
+
help="Env file for the Secret; defaults to .env.<env> (dev also falls back to .env).",
|
|
90
|
+
)
|
|
91
|
+
@click.option(
|
|
92
|
+
"--context",
|
|
93
|
+
"context",
|
|
94
|
+
default=None,
|
|
95
|
+
help="Kube context to use instead of environments.<env>.context.",
|
|
96
|
+
)
|
|
97
|
+
@click.option(
|
|
98
|
+
"--yes",
|
|
99
|
+
"-y",
|
|
100
|
+
"yes",
|
|
101
|
+
is_flag=True,
|
|
102
|
+
help="Accept the kubeconfig's current context outside dev without prompting.",
|
|
103
|
+
)
|
|
104
|
+
@click.option(
|
|
105
|
+
"--status",
|
|
106
|
+
"status",
|
|
107
|
+
is_flag=True,
|
|
108
|
+
help="Report the rollout, pods and warning events instead of deploying (exit 1 when not "
|
|
109
|
+
"ready within --timeout, default 60s).",
|
|
110
|
+
)
|
|
111
|
+
@click.option(
|
|
112
|
+
"--restart",
|
|
113
|
+
"restart",
|
|
114
|
+
is_flag=True,
|
|
115
|
+
help="Rollout-restart the Deployment (after a Secret rotation) and wait for the new pods.",
|
|
116
|
+
)
|
|
117
|
+
@click.option(
|
|
118
|
+
"--force-direct",
|
|
119
|
+
"force_direct",
|
|
120
|
+
is_flag=True,
|
|
121
|
+
help="helm-push mode: allow a deploy to staging/prod from outside CI.",
|
|
122
|
+
)
|
|
123
|
+
@click.option(
|
|
124
|
+
"--dry-run",
|
|
125
|
+
"dry_run",
|
|
126
|
+
is_flag=True,
|
|
127
|
+
help="Print every command; run `helm template` instead of upgrade.",
|
|
128
|
+
)
|
|
129
|
+
@click.option(
|
|
130
|
+
"--tag",
|
|
131
|
+
"tag",
|
|
132
|
+
default=None,
|
|
133
|
+
help="Image tag for a local build (default: short git sha, plus -dirty-<time> for "
|
|
134
|
+
"uncommitted changes; a timestamp outside git).",
|
|
135
|
+
)
|
|
136
|
+
@click.option(
|
|
137
|
+
"--timeout",
|
|
138
|
+
"timeout",
|
|
139
|
+
default=None,
|
|
140
|
+
callback=_timeout_option,
|
|
141
|
+
help=f"How long to wait for the rollout (e.g. 300s, 10m; default {DEFAULT_TIMEOUT}, "
|
|
142
|
+
f"{DEFAULT_STATUS_TIMEOUT} for --status).",
|
|
143
|
+
)
|
|
144
|
+
@click.option(
|
|
145
|
+
"--atomic/--no-atomic",
|
|
146
|
+
"atomic",
|
|
147
|
+
default=True,
|
|
148
|
+
show_default=True,
|
|
149
|
+
help="Roll back a failed rollout (after printing pod diagnostics).",
|
|
150
|
+
)
|
|
151
|
+
@click.option(
|
|
152
|
+
"--rotate-api-key",
|
|
153
|
+
"rotate_api_key",
|
|
154
|
+
is_flag=True,
|
|
155
|
+
help="Replace the live API_KEY with the one in the env file (otherwise the live key wins).",
|
|
156
|
+
)
|
|
157
|
+
def cmd_deploy(
|
|
158
|
+
env: str,
|
|
159
|
+
image: str | None,
|
|
160
|
+
env_file: str | None,
|
|
161
|
+
context: str | None,
|
|
162
|
+
yes: bool,
|
|
163
|
+
status: bool,
|
|
164
|
+
restart: bool,
|
|
165
|
+
force_direct: bool,
|
|
166
|
+
dry_run: bool,
|
|
167
|
+
tag: str | None,
|
|
168
|
+
timeout: str | None,
|
|
169
|
+
atomic: bool,
|
|
170
|
+
rotate_api_key: bool,
|
|
171
|
+
) -> None:
|
|
172
|
+
"""Deploy the agent to Kubernetes (mode depends on the project's CD setting).
|
|
173
|
+
|
|
174
|
+
\b
|
|
175
|
+
Modes (from create_params.cd in the manifest):
|
|
176
|
+
skip direct: build, local-load or push, apply the Secret, helm upgrade
|
|
177
|
+
helm-push CI builds and pushes; deploy --image runs helm only
|
|
178
|
+
argocd never runs helm: writes image.tag into values-<env>.yaml and opens a PR
|
|
179
|
+
\b
|
|
180
|
+
Outside dev the kube context must be recorded in the manifest
|
|
181
|
+
(environments.<env>.context) or passed with --context; the kubeconfig's
|
|
182
|
+
current context is used only after a confirmation (or --yes).
|
|
183
|
+
"""
|
|
184
|
+
console = Console()
|
|
185
|
+
settings = load_settings()
|
|
186
|
+
namespace = settings.target(env).namespace
|
|
187
|
+
_modes.derive_mode(settings.cd, None) # an unknown cd value is a configuration error
|
|
188
|
+
resolved = _modes.resolve(settings, env, context)
|
|
189
|
+
target = Target(context=resolved.name, namespace=namespace)
|
|
190
|
+
|
|
191
|
+
if status:
|
|
192
|
+
_modes.announce(env, resolved, console=console)
|
|
193
|
+
_modes.require_known_context(env, resolved, dry_run=dry_run, console=console)
|
|
194
|
+
_show_status(
|
|
195
|
+
settings,
|
|
196
|
+
env,
|
|
197
|
+
target,
|
|
198
|
+
timeout=timeout or DEFAULT_STATUS_TIMEOUT,
|
|
199
|
+
dry_run=dry_run,
|
|
200
|
+
console=console,
|
|
201
|
+
)
|
|
202
|
+
return
|
|
203
|
+
timeout = timeout or DEFAULT_TIMEOUT
|
|
204
|
+
|
|
205
|
+
if env in PROTECTED_ENVS and not settings.auth_policy_implemented:
|
|
206
|
+
raise Refused(
|
|
207
|
+
f"Refusing to deploy to {env}: the manifest records auth_policy_implemented: false.\n"
|
|
208
|
+
f" Implement the {settings.auth_policy} auth policy (the custom stub lives in\n"
|
|
209
|
+
" <agent_directory>/policies/custom.py), then set auth_policy_implemented: true\n"
|
|
210
|
+
" in graph-agents-cli-manifest.yaml."
|
|
211
|
+
)
|
|
212
|
+
|
|
213
|
+
if restart:
|
|
214
|
+
_modes.announce(env, resolved, console=console)
|
|
215
|
+
_modes.confirm(env, resolved, yes=yes, dry_run=dry_run, console=console, action="restart")
|
|
216
|
+
_restart(settings, env, target, timeout=timeout, dry_run=dry_run, console=console)
|
|
217
|
+
return
|
|
218
|
+
|
|
219
|
+
console.print(f"Environment: {env} namespace: {namespace}")
|
|
220
|
+
options = _Options(
|
|
221
|
+
image=image,
|
|
222
|
+
env_file=env_file,
|
|
223
|
+
tag=tag,
|
|
224
|
+
yes=yes,
|
|
225
|
+
force_direct=force_direct,
|
|
226
|
+
dry_run=dry_run,
|
|
227
|
+
timeout=timeout,
|
|
228
|
+
atomic=atomic,
|
|
229
|
+
rotate_api_key=rotate_api_key,
|
|
230
|
+
)
|
|
231
|
+
if settings.cd == _modes.ARGOCD:
|
|
232
|
+
_deploy_argocd(settings, env, resolved, options, console=console)
|
|
233
|
+
elif settings.cd == _modes.HELM_PUSH:
|
|
234
|
+
_deploy_helm_push(settings, env, resolved, target, options, console=console)
|
|
235
|
+
else:
|
|
236
|
+
_deploy_direct(settings, env, resolved, target, options, console=console)
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
@dataclass(frozen=True)
|
|
240
|
+
class _Options:
|
|
241
|
+
image: str | None
|
|
242
|
+
env_file: str | None
|
|
243
|
+
tag: str | None
|
|
244
|
+
yes: bool
|
|
245
|
+
force_direct: bool
|
|
246
|
+
dry_run: bool
|
|
247
|
+
timeout: str
|
|
248
|
+
atomic: bool
|
|
249
|
+
rotate_api_key: bool
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
@dataclass(frozen=True)
|
|
253
|
+
class _ImagePlan:
|
|
254
|
+
repository: str
|
|
255
|
+
tag: str
|
|
256
|
+
build: bool
|
|
257
|
+
|
|
258
|
+
@property
|
|
259
|
+
def ref(self) -> str:
|
|
260
|
+
return f"{self.repository}:{self.tag}"
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
# --------------------------------------------------------------------------- modes
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _deploy_direct(
|
|
267
|
+
settings: DeploySettings,
|
|
268
|
+
env: str,
|
|
269
|
+
resolved: ResolvedContext,
|
|
270
|
+
target: Target,
|
|
271
|
+
opts: _Options,
|
|
272
|
+
*,
|
|
273
|
+
console: Console,
|
|
274
|
+
) -> None:
|
|
275
|
+
# Project-only checks first: nothing below them may touch a cluster.
|
|
276
|
+
_require_chart(settings, env)
|
|
277
|
+
plan = _image_plan(settings, opts, console=console)
|
|
278
|
+
path = secrets_apply.resolve_env_file(env, opts.env_file)
|
|
279
|
+
chart_values = load_chart_values(settings.chart_dir, env)
|
|
280
|
+
_check_chart_env(settings, env, chart_values, console=console)
|
|
281
|
+
_check_peers(settings, env, chart_values, dry_run=opts.dry_run, console=console)
|
|
282
|
+
_check_jwt(settings, env, chart_values, None, dry_run=opts.dry_run, console=console, warn=False)
|
|
283
|
+
if path is None and not _modes.is_dev_env(env):
|
|
284
|
+
raise secrets_apply.missing_env_file_error(
|
|
285
|
+
env, _required.for_environment(settings, chart_values).secret_keys
|
|
286
|
+
)
|
|
287
|
+
values: dict[str, str] = {}
|
|
288
|
+
if path is not None:
|
|
289
|
+
values = secrets_apply.read_env_file(path)
|
|
290
|
+
# The allow-list for this environment (AUTH_JWT_SECRET joins it for HS* JWTs).
|
|
291
|
+
settings = _required.for_environment(settings, chart_values, values)
|
|
292
|
+
if path is not None:
|
|
293
|
+
secrets_apply.check_file_values(
|
|
294
|
+
values, settings.secret_keys, rotate_api_key=opts.rotate_api_key, source=path
|
|
295
|
+
)
|
|
296
|
+
elif opts.rotate_api_key:
|
|
297
|
+
raise ConfigError("--rotate-api-key needs an env file that sets API_KEY.")
|
|
298
|
+
|
|
299
|
+
_modes.announce(env, resolved, console=console)
|
|
300
|
+
_modes.confirm(
|
|
301
|
+
env, resolved, yes=opts.yes, dry_run=opts.dry_run, console=console, action="deploy to"
|
|
302
|
+
)
|
|
303
|
+
|
|
304
|
+
cluster: local_load.LocalCluster | None = None
|
|
305
|
+
if plan.build:
|
|
306
|
+
cluster, why = local_load.detect(target.context)
|
|
307
|
+
mode = _modes.LOCAL_LOAD if cluster else _modes.REGISTRY
|
|
308
|
+
detail = f": {cluster.describe()}" if cluster else f" ({why})" if why else ""
|
|
309
|
+
console.print(f"Mode: {_modes.describe(mode)}{detail}", markup=False)
|
|
310
|
+
else:
|
|
311
|
+
console.print("Mode: direct, image given (build, load and push skipped)")
|
|
312
|
+
console.print(f"Using image {plan.ref}.")
|
|
313
|
+
|
|
314
|
+
# Plan the Secret and check its required keys (read-only) before anything is
|
|
315
|
+
# built, pushed or changed: an incomplete Secret stops the deploy up front.
|
|
316
|
+
# --dry-run makes the same reads, so it refuses what the real run would.
|
|
317
|
+
secret_plan: secrets_apply.SecretPlan | None = None
|
|
318
|
+
live: secrets_apply.LiveSecret | None = None
|
|
319
|
+
if path is None:
|
|
320
|
+
keys = _verify_secret(
|
|
321
|
+
settings, env, target, fix_hint=None, dry_run=opts.dry_run, console=console
|
|
322
|
+
)
|
|
323
|
+
else:
|
|
324
|
+
secret_plan, live = secrets_apply.prepare(
|
|
325
|
+
name=settings.secret_name,
|
|
326
|
+
env=env,
|
|
327
|
+
target=target,
|
|
328
|
+
allowed=settings.secret_keys,
|
|
329
|
+
path=path,
|
|
330
|
+
values=values,
|
|
331
|
+
rotate_api_key=opts.rotate_api_key,
|
|
332
|
+
dry_run=opts.dry_run,
|
|
333
|
+
mint_api_key=_preflight.effective_auth_policy(settings, chart_values)
|
|
334
|
+
== "shared-bearer",
|
|
335
|
+
metrics_name=settings.metrics_secret_name,
|
|
336
|
+
read_live_on_dry_run=True,
|
|
337
|
+
)
|
|
338
|
+
keys = _verify_secret(
|
|
339
|
+
settings,
|
|
340
|
+
env,
|
|
341
|
+
target,
|
|
342
|
+
keys=set(secret_plan.data),
|
|
343
|
+
uncertain=bool(secret_plan.live_unread),
|
|
344
|
+
fix_hint=f"Add them to {path} and re-run deploy",
|
|
345
|
+
dry_run=opts.dry_run,
|
|
346
|
+
console=console,
|
|
347
|
+
)
|
|
348
|
+
_warn_dsn_without_tls(settings, env, chart_values, secret_plan.data, console=console)
|
|
349
|
+
if not secret_plan.data:
|
|
350
|
+
# Nothing to apply (a keyless project: the fake model, a keyless endpoint): like
|
|
351
|
+
# no env file, the Secret is left as it is, decided before anything is built.
|
|
352
|
+
_require_secret_to_leave(
|
|
353
|
+
settings, env, target, chart_values, live, path, dry_run=opts.dry_run
|
|
354
|
+
)
|
|
355
|
+
secret_plan = None
|
|
356
|
+
_check_jwt(settings, env, chart_values, keys, dry_run=opts.dry_run, console=console)
|
|
357
|
+
_check_release_idle(settings, target, dry_run=opts.dry_run, console=console)
|
|
358
|
+
before = _live_workload(settings, target)
|
|
359
|
+
_announce_same_image(settings, env, plan, before, console=console)
|
|
360
|
+
|
|
361
|
+
if plan.build:
|
|
362
|
+
_build(settings, plan, dry_run=opts.dry_run, console=console)
|
|
363
|
+
if cluster is not None:
|
|
364
|
+
_load(cluster, plan.ref, dry_run=opts.dry_run, console=console)
|
|
365
|
+
else:
|
|
366
|
+
_kube.run_cmd(
|
|
367
|
+
["docker", "push", plan.ref], capture=False, dry_run=opts.dry_run, console=console
|
|
368
|
+
)
|
|
369
|
+
|
|
370
|
+
snaps: list[secrets_apply.Snapshot] = []
|
|
371
|
+
if secret_plan is None and path is not None:
|
|
372
|
+
console.print(
|
|
373
|
+
f" {path} sets none of the allow-listed keys ({', '.join(settings.secret_keys)}); "
|
|
374
|
+
f"the Secret {settings.secret_name} is left as is.",
|
|
375
|
+
style="yellow",
|
|
376
|
+
markup=False,
|
|
377
|
+
)
|
|
378
|
+
elif secret_plan is None:
|
|
379
|
+
console.print(
|
|
380
|
+
f" No env file found (.env.{env} or .env); the Secret {settings.secret_name} is "
|
|
381
|
+
"left as is.",
|
|
382
|
+
style="yellow",
|
|
383
|
+
)
|
|
384
|
+
else:
|
|
385
|
+
console.print(
|
|
386
|
+
f"Applying Secret {settings.secret_name} from {path} (allow-listed keys only)."
|
|
387
|
+
)
|
|
388
|
+
_required.print_unreached_hs_settings(
|
|
389
|
+
settings, env, chart_values, values, source=path, console=console
|
|
390
|
+
)
|
|
391
|
+
if not opts.dry_run:
|
|
392
|
+
# What the Secret(s) held before, to put back if the rollout fails.
|
|
393
|
+
snaps = secrets_apply.snapshot(secret_plan, live, managed=settings.secret_keys)
|
|
394
|
+
if opts.dry_run:
|
|
395
|
+
console.print(
|
|
396
|
+
" [dry-run] if the rollout fails and the release is rolled back, the Secret is "
|
|
397
|
+
"put back to its current values.",
|
|
398
|
+
style="cyan",
|
|
399
|
+
markup=False,
|
|
400
|
+
)
|
|
401
|
+
try:
|
|
402
|
+
if secret_plan is not None:
|
|
403
|
+
secrets_apply.apply_plan(secret_plan, dry_run=opts.dry_run, console=console, live=live)
|
|
404
|
+
_helm_upgrade(settings, env, target, plan, opts, console=console)
|
|
405
|
+
except _kube.DeployError as e:
|
|
406
|
+
# A failure while applying (the metrics Secret, say) leaves the release untouched too.
|
|
407
|
+
lines = _secret_outcome(snaps, env, e, console=console)
|
|
408
|
+
if not lines:
|
|
409
|
+
raise
|
|
410
|
+
raise type(e)("\n ".join([str(e.message), *lines])) from None
|
|
411
|
+
except KeyboardInterrupt:
|
|
412
|
+
# Interrupted mid-rollout: the release's state is unknown, so nothing is undone.
|
|
413
|
+
for line in secrets_apply.describe_unrestored(snaps, env, "the deploy was interrupted"):
|
|
414
|
+
console.print(f" {line}", style="yellow", markup=False)
|
|
415
|
+
raise
|
|
416
|
+
_report_rollout(settings, env, target, before, snaps, dry_run=opts.dry_run, console=console)
|
|
417
|
+
_print_done(settings, env, plan.ref, dry_run=opts.dry_run, console=console)
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def _deploy_helm_push(
|
|
421
|
+
settings: DeploySettings,
|
|
422
|
+
env: str,
|
|
423
|
+
resolved: ResolvedContext,
|
|
424
|
+
target: Target,
|
|
425
|
+
opts: _Options,
|
|
426
|
+
*,
|
|
427
|
+
console: Console,
|
|
428
|
+
) -> None:
|
|
429
|
+
_refuse_secrets(settings, env, opts, mode=_modes.HELM_PUSH, console=console)
|
|
430
|
+
_require_chart(settings, env)
|
|
431
|
+
chart_values = load_chart_values(settings.chart_dir, env)
|
|
432
|
+
_check_chart_env(settings, env, chart_values, console=console)
|
|
433
|
+
_check_peers(settings, env, chart_values, dry_run=opts.dry_run, console=console)
|
|
434
|
+
_check_jwt(settings, env, chart_values, None, dry_run=opts.dry_run, console=console, warn=False)
|
|
435
|
+
if env in PROTECTED_ENVS and not opts.force_direct and not _kube.in_ci():
|
|
436
|
+
raise Refused(
|
|
437
|
+
f"Refusing to deploy {env} from outside CI in helm-push mode (with or without "
|
|
438
|
+
"--image).\n"
|
|
439
|
+
f" The staging and promote-to-prod workflows run `deploy --env {env} --image <ref>` "
|
|
440
|
+
"on the self-hosted runner (GITHUB_ACTIONS=true); pass --force-direct to deploy "
|
|
441
|
+
"from here anyway."
|
|
442
|
+
)
|
|
443
|
+
plan = _image_plan(settings, opts, console=console)
|
|
444
|
+
|
|
445
|
+
_modes.announce(env, resolved, console=console)
|
|
446
|
+
_modes.confirm(
|
|
447
|
+
env, resolved, yes=opts.yes, dry_run=opts.dry_run, console=console, action="deploy to"
|
|
448
|
+
)
|
|
449
|
+
console.print(f"Mode: {_modes.describe(_modes.HELM_PUSH)}")
|
|
450
|
+
keys = _verify_secret(
|
|
451
|
+
settings,
|
|
452
|
+
env,
|
|
453
|
+
target,
|
|
454
|
+
fix_hint=f"The Secret owner provisions them with `graph-agents-cli secrets apply --env {env}`",
|
|
455
|
+
dry_run=opts.dry_run,
|
|
456
|
+
console=console,
|
|
457
|
+
)
|
|
458
|
+
_check_jwt(settings, env, chart_values, keys, dry_run=opts.dry_run, console=console)
|
|
459
|
+
_check_release_idle(settings, target, dry_run=opts.dry_run, console=console)
|
|
460
|
+
before = _live_workload(settings, target)
|
|
461
|
+
_announce_same_image(settings, env, plan, before, console=console)
|
|
462
|
+
if plan.build:
|
|
463
|
+
_build(settings, plan, dry_run=opts.dry_run, console=console)
|
|
464
|
+
_kube.run_cmd(
|
|
465
|
+
["docker", "push", plan.ref], capture=False, dry_run=opts.dry_run, console=console
|
|
466
|
+
)
|
|
467
|
+
_helm_upgrade(settings, env, target, plan, opts, console=console)
|
|
468
|
+
_report_rollout(settings, env, target, before, [], dry_run=opts.dry_run, console=console)
|
|
469
|
+
_print_done(settings, env, plan.ref, dry_run=opts.dry_run, console=console)
|
|
470
|
+
|
|
471
|
+
|
|
472
|
+
def _deploy_argocd(
|
|
473
|
+
settings: DeploySettings,
|
|
474
|
+
env: str,
|
|
475
|
+
resolved: ResolvedContext,
|
|
476
|
+
opts: _Options,
|
|
477
|
+
*,
|
|
478
|
+
console: Console,
|
|
479
|
+
) -> None:
|
|
480
|
+
console.print(f"Mode: {_modes.describe(_modes.ARGOCD)}")
|
|
481
|
+
console.print(
|
|
482
|
+
f" No cluster is contacted (kube context {resolved.name or '(none)'} is not used): "
|
|
483
|
+
"Argo CD applies the change after the pull request merges.",
|
|
484
|
+
style="dim",
|
|
485
|
+
markup=False,
|
|
486
|
+
)
|
|
487
|
+
_refuse_secrets(settings, env, opts, mode=_modes.ARGOCD, console=console)
|
|
488
|
+
values_path = settings.values_file(env)
|
|
489
|
+
if not values_path.is_file():
|
|
490
|
+
raise ConfigError(f"Values file not found: {values_path}")
|
|
491
|
+
chart_values = load_chart_values(settings.chart_dir, env)
|
|
492
|
+
# Argo CD renders these values as they are: check what its pods would get.
|
|
493
|
+
_check_chart_env(settings, env, chart_values, console=console)
|
|
494
|
+
_check_peers(settings, env, chart_values, dry_run=opts.dry_run, console=console)
|
|
495
|
+
_check_jwt(settings, env, chart_values, None, dry_run=opts.dry_run, console=console)
|
|
496
|
+
chart_repository = str((chart_values.get("image") or {}).get("repository") or "")
|
|
497
|
+
if _image.has_placeholder(chart_repository):
|
|
498
|
+
raise ConfigError(
|
|
499
|
+
f"image.repository in the chart values is still the placeholder {chart_repository!r}; "
|
|
500
|
+
"Argo CD would pull it. Set image.repository in "
|
|
501
|
+
f"{settings.chart_dir / 'values.yaml'} (and create_params.registry in the manifest)."
|
|
502
|
+
)
|
|
503
|
+
if opts.image:
|
|
504
|
+
repository, image_tag = split_image_ref(opts.image)
|
|
505
|
+
problem = _image.reference_problem(repository, image_tag)
|
|
506
|
+
if problem:
|
|
507
|
+
raise ConfigError(f"--image: {problem}")
|
|
508
|
+
if chart_repository and repository != chart_repository:
|
|
509
|
+
console.print(
|
|
510
|
+
f" argocd mode writes only image.tag: Argo CD pulls {chart_repository}:"
|
|
511
|
+
f"{image_tag}, not {repository}:{image_tag}. Change image.repository in the "
|
|
512
|
+
"chart values if the repository moved.",
|
|
513
|
+
style="yellow",
|
|
514
|
+
markup=False,
|
|
515
|
+
)
|
|
516
|
+
else:
|
|
517
|
+
repository = (settings.registry and settings.image_repository) or None
|
|
518
|
+
image_tag = opts.tag or gitops.short_sha() or _timestamp()
|
|
519
|
+
problem = _image.tag_problem(image_tag)
|
|
520
|
+
if problem:
|
|
521
|
+
raise ConfigError(f"--tag: {problem}.")
|
|
522
|
+
console.print(
|
|
523
|
+
f" No --image given; writing tag {image_tag!r}. The image must already be pushed "
|
|
524
|
+
"by CI for Argo CD to roll it out.",
|
|
525
|
+
style="yellow",
|
|
526
|
+
)
|
|
527
|
+
if opts.tag is None and gitops.worktree_dirty():
|
|
528
|
+
console.print(
|
|
529
|
+
" The working tree has uncommitted changes; they are not in the image CI "
|
|
530
|
+
f"built for {image_tag}.",
|
|
531
|
+
style="yellow",
|
|
532
|
+
)
|
|
533
|
+
result = gitops.write_desired_state(
|
|
534
|
+
env=env,
|
|
535
|
+
values_path=values_path,
|
|
536
|
+
image_repository=repository,
|
|
537
|
+
tag=image_tag,
|
|
538
|
+
project_name=settings.project_name,
|
|
539
|
+
dry_run=opts.dry_run,
|
|
540
|
+
console=console,
|
|
541
|
+
)
|
|
542
|
+
if not result.changed:
|
|
543
|
+
return
|
|
544
|
+
what = "Would open" if opts.dry_run else ("Opened" if result.created else "Updated")
|
|
545
|
+
where = f" {result.url}" if result.url else ""
|
|
546
|
+
console.print(
|
|
547
|
+
f"{what} pull request on branch {result.branch}.{where}", style="green", markup=False
|
|
548
|
+
)
|
|
549
|
+
if env == "prod":
|
|
550
|
+
console.print(
|
|
551
|
+
" The production change lands only when the PR merges after code-owner review."
|
|
552
|
+
)
|
|
553
|
+
|
|
554
|
+
|
|
555
|
+
# --------------------------------------------------------------------------- steps
|
|
556
|
+
|
|
557
|
+
|
|
558
|
+
def _require_chart(settings: DeploySettings, env: str) -> None:
|
|
559
|
+
chart = settings.chart_dir
|
|
560
|
+
if not (chart / "Chart.yaml").is_file():
|
|
561
|
+
raise ConfigError(f"Helm chart not found at {chart} (expected Chart.yaml).")
|
|
562
|
+
if not settings.values_file(env).is_file():
|
|
563
|
+
raise ConfigError(f"Values file not found: {settings.values_file(env)}")
|
|
564
|
+
_require_gateway_values(settings, env)
|
|
565
|
+
|
|
566
|
+
|
|
567
|
+
def _truthy(value: object) -> bool:
|
|
568
|
+
return value is True or (
|
|
569
|
+
isinstance(value, str) and value.strip().lower() in ("true", "yes", "1")
|
|
570
|
+
)
|
|
571
|
+
|
|
572
|
+
|
|
573
|
+
def _require_gateway_values(settings: DeploySettings, env: str) -> None:
|
|
574
|
+
"""Mirror the chart's ``required`` on ``gateway.parentRef.name`` before any tool runs.
|
|
575
|
+
|
|
576
|
+
The scaffolded staging/prod values ship ``gateway.enabled: true`` with a blank
|
|
577
|
+
parentRef (the operator names the Gateway); helm would only refuse after the
|
|
578
|
+
image was built and pushed and the Secret applied, as a tool failure (exit
|
|
579
|
+
2). This is configuration, so it is reported as such (exit 3) up front.
|
|
580
|
+
"""
|
|
581
|
+
values = load_chart_values(settings.chart_dir, env)
|
|
582
|
+
gateway = values.get("gateway") if isinstance(values.get("gateway"), dict) else {}
|
|
583
|
+
if not _truthy(gateway.get("enabled")):
|
|
584
|
+
return
|
|
585
|
+
parent = gateway.get("parentRef") if isinstance(gateway.get("parentRef"), dict) else {}
|
|
586
|
+
if str(parent.get("name") or "").strip():
|
|
587
|
+
return
|
|
588
|
+
raise ConfigError(
|
|
589
|
+
f"gateway.parentRef.name is blank in {settings.values_file(env)} but gateway.enabled is "
|
|
590
|
+
"true: set it to the Gateway to attach to (or set gateway.enabled: false / "
|
|
591
|
+
"ingress.enabled: true)."
|
|
592
|
+
)
|
|
593
|
+
|
|
594
|
+
|
|
595
|
+
def _dockerfile(settings: DeploySettings) -> str:
|
|
596
|
+
for candidate in ("Dockerfile", f"Dockerfile.{settings.runtime}"):
|
|
597
|
+
if Path(candidate).is_file():
|
|
598
|
+
return candidate
|
|
599
|
+
raise ConfigError("No Dockerfile found in the project root.")
|
|
600
|
+
|
|
601
|
+
|
|
602
|
+
def _timestamp() -> str:
|
|
603
|
+
return _dt.datetime.now(_dt.UTC).strftime("%Y%m%d%H%M%S")
|
|
604
|
+
|
|
605
|
+
|
|
606
|
+
def _workstation_tag(console: Console) -> str:
|
|
607
|
+
"""The short commit SHA; ``<sha>-dirty-<time>`` with uncommitted changes; a timestamp outside git.
|
|
608
|
+
|
|
609
|
+
Rebuilding at the same commit under the same tag renders an identical pod
|
|
610
|
+
spec (and ``IfNotPresent`` nodes keep the old image), so helm reports
|
|
611
|
+
success while nothing rolls out. A dirty tree therefore always gets a new tag.
|
|
612
|
+
"""
|
|
613
|
+
sha = gitops.short_sha()
|
|
614
|
+
if not sha:
|
|
615
|
+
return _timestamp()
|
|
616
|
+
if gitops.worktree_dirty():
|
|
617
|
+
tag = f"{sha}-dirty-{_timestamp()}"
|
|
618
|
+
console.print(
|
|
619
|
+
f" The working tree has uncommitted changes: tagging the image {tag} so the "
|
|
620
|
+
"rollout picks them up. Commit first for a reproducible deploy.",
|
|
621
|
+
style="yellow",
|
|
622
|
+
markup=False,
|
|
623
|
+
)
|
|
624
|
+
return tag
|
|
625
|
+
return sha
|
|
626
|
+
|
|
627
|
+
|
|
628
|
+
def _image_plan(settings: DeploySettings, opts: _Options, *, console: Console) -> _ImagePlan:
|
|
629
|
+
"""The image to deploy, validated before docker, kubectl or helm run (exit 3 when invalid)."""
|
|
630
|
+
if opts.image:
|
|
631
|
+
repository, image_tag = split_image_ref(opts.image)
|
|
632
|
+
problem = _image.reference_problem(repository, image_tag)
|
|
633
|
+
if problem:
|
|
634
|
+
raise ConfigError(f"--image: {problem}")
|
|
635
|
+
return _ImagePlan(repository, image_tag, build=False)
|
|
636
|
+
if _image.has_placeholder(settings.registry):
|
|
637
|
+
raise ConfigError(_image.placeholder_message(settings.registry))
|
|
638
|
+
repository = settings.image_repository
|
|
639
|
+
image_tag = opts.tag or _workstation_tag(console)
|
|
640
|
+
problem = _image.reference_problem(repository, image_tag, registry=settings.registry)
|
|
641
|
+
if problem:
|
|
642
|
+
raise ConfigError(problem)
|
|
643
|
+
return _ImagePlan(repository, image_tag, build=True)
|
|
644
|
+
|
|
645
|
+
|
|
646
|
+
def _build(settings: DeploySettings, plan: _ImagePlan, *, dry_run: bool, console: Console) -> None:
|
|
647
|
+
dockerfile = _dockerfile(settings)
|
|
648
|
+
_kube.run_cmd(
|
|
649
|
+
["docker", "build", "-t", plan.ref, "-f", dockerfile, "."],
|
|
650
|
+
capture=False,
|
|
651
|
+
dry_run=dry_run,
|
|
652
|
+
console=console,
|
|
653
|
+
)
|
|
654
|
+
|
|
655
|
+
|
|
656
|
+
def _load(cluster: local_load.LocalCluster, image: str, *, dry_run: bool, console: Console) -> None:
|
|
657
|
+
commands = local_load.local_load_commands(cluster, image)
|
|
658
|
+
if not commands:
|
|
659
|
+
console.print(f" {cluster.describe()}: no image load needed.", markup=False)
|
|
660
|
+
for cmd in commands:
|
|
661
|
+
_kube.run_cmd(cmd, capture=False, dry_run=dry_run, console=console)
|
|
662
|
+
|
|
663
|
+
|
|
664
|
+
def _refuse_secrets(
|
|
665
|
+
settings: DeploySettings, env: str, opts: _Options, *, mode: str, console: Console
|
|
666
|
+
) -> None:
|
|
667
|
+
procedure = secrets_apply.provisioning_procedure(
|
|
668
|
+
project=settings.project_name, env=env, owner=settings.secrets_owner, mode=mode
|
|
669
|
+
)
|
|
670
|
+
for flag, given in (("--env-file", opts.env_file), ("--rotate-api-key", opts.rotate_api_key)):
|
|
671
|
+
if given:
|
|
672
|
+
raise Refused(f"{flag} is not accepted in {mode} mode.\n {procedure}")
|
|
673
|
+
console.print(f" {procedure}", style="dim", markup=False)
|
|
674
|
+
|
|
675
|
+
|
|
676
|
+
def _first_line(error: object) -> str:
|
|
677
|
+
return next((line.strip() for line in str(error).splitlines() if line.strip()), "")
|
|
678
|
+
|
|
679
|
+
|
|
680
|
+
def _verify_secret(
|
|
681
|
+
settings: DeploySettings,
|
|
682
|
+
env: str,
|
|
683
|
+
target: Target,
|
|
684
|
+
*,
|
|
685
|
+
keys: set[str] | None = None,
|
|
686
|
+
uncertain: bool = False,
|
|
687
|
+
fix_hint: str | None,
|
|
688
|
+
dry_run: bool,
|
|
689
|
+
console: Console,
|
|
690
|
+
) -> set[str] | None:
|
|
691
|
+
"""Refuse (exit 1) before anything changes when the Secret lacks a key the pods need.
|
|
692
|
+
|
|
693
|
+
``keys`` are the keys the Secret will hold after this deploy applies it; when
|
|
694
|
+
``None`` the live Secret is read (read-only, so ``--dry-run`` reads it too and
|
|
695
|
+
refuses what the real run would). ``uncertain``: under ``--dry-run`` the live
|
|
696
|
+
Secret could not be read, so a key missing from ``keys`` may still be there.
|
|
697
|
+
Nothing has been built, pushed or applied when this refuses. Returns the keys
|
|
698
|
+
the Secret holds or will hold (``None`` when unknown).
|
|
699
|
+
"""
|
|
700
|
+
required = _required.required_keys(settings, load_chart_values(settings.chart_dir, env))
|
|
701
|
+
name = settings.secret_name
|
|
702
|
+
absent = False
|
|
703
|
+
if keys is None:
|
|
704
|
+
try:
|
|
705
|
+
found = secrets_apply.secret_keys_present(name, target, console=console)
|
|
706
|
+
except _kube.ToolFailed as e:
|
|
707
|
+
if not dry_run:
|
|
708
|
+
raise
|
|
709
|
+
console.print(
|
|
710
|
+
f" [dry-run] could not read Secret {name} ({_first_line(e)}); the real run "
|
|
711
|
+
"refuses unless it holds: " + (", ".join(required) or "(no required key)"),
|
|
712
|
+
style="yellow",
|
|
713
|
+
markup=False,
|
|
714
|
+
)
|
|
715
|
+
return None
|
|
716
|
+
absent = found is None
|
|
717
|
+
present = found or set()
|
|
718
|
+
else:
|
|
719
|
+
present = keys
|
|
720
|
+
if not required:
|
|
721
|
+
return present
|
|
722
|
+
missing = [k for k in required if k not in present]
|
|
723
|
+
if not missing:
|
|
724
|
+
verb = "would hold" if dry_run else ("will hold" if keys is not None else "holds")
|
|
725
|
+
console.print(
|
|
726
|
+
f" Secret {name} {verb} the required key(s): {', '.join(required)}.", style="dim"
|
|
727
|
+
)
|
|
728
|
+
return present
|
|
729
|
+
if uncertain:
|
|
730
|
+
console.print(
|
|
731
|
+
f" [dry-run] Secret {name} would lack {', '.join(missing)} unless the live Secret "
|
|
732
|
+
"(not readable here) holds them: the real run refuses otherwise.",
|
|
733
|
+
style="yellow",
|
|
734
|
+
markup=False,
|
|
735
|
+
)
|
|
736
|
+
return present
|
|
737
|
+
fix = fix_hint or (
|
|
738
|
+
f"Put them in .env.{env} and re-run deploy, or provision them with "
|
|
739
|
+
f"`graph-agents-cli secrets apply --env {env}`"
|
|
740
|
+
)
|
|
741
|
+
why_jwt = (
|
|
742
|
+
f"\n {_required.JWT_SECRET_KEY} is needed because {settings.values_file(env)} (or "
|
|
743
|
+
"values.yaml) lists an HS* algorithm in AUTH_JWT_ALGORITHMS."
|
|
744
|
+
if _required.JWT_SECRET_KEY in missing
|
|
745
|
+
else ""
|
|
746
|
+
)
|
|
747
|
+
dry = "\n (--dry-run: the real deploy stops here the same way.)" if dry_run else ""
|
|
748
|
+
raise Refused(
|
|
749
|
+
f"Secret {name} in {target.namespace} {'would be' if dry_run else 'is'} missing required "
|
|
750
|
+
f"key(s): {', '.join(missing)}{' (the Secret does not exist)' if absent else ''}.\n"
|
|
751
|
+
" Without them the pods crash or answer every request with 503; nothing was built, "
|
|
752
|
+
"applied or deployed.\n"
|
|
753
|
+
f" {fix}, or remove a key from secrets.keys in graph-agents-cli-manifest.yaml if "
|
|
754
|
+
f"{env} does not need it.{why_jwt}{dry}"
|
|
755
|
+
)
|
|
756
|
+
|
|
757
|
+
|
|
758
|
+
def _require_secret_to_leave(
|
|
759
|
+
settings: DeploySettings,
|
|
760
|
+
env: str,
|
|
761
|
+
target: Target,
|
|
762
|
+
values: dict[str, Any],
|
|
763
|
+
live: secrets_apply.LiveSecret | None,
|
|
764
|
+
path: Path,
|
|
765
|
+
*,
|
|
766
|
+
dry_run: bool,
|
|
767
|
+
) -> None:
|
|
768
|
+
"""Refuse (exit 1) before anything changes when the Secret is left alone but the pods need it.
|
|
769
|
+
|
|
770
|
+
An env file with none of the allow-listed keys leaves the Secret as it is.
|
|
771
|
+
Outside dev the chart requires the Secret to exist (``secretOptional:
|
|
772
|
+
false``), so when there is none the pods would never start. ``live`` is
|
|
773
|
+
``None`` under ``--dry-run`` when the Secret could not be read.
|
|
774
|
+
"""
|
|
775
|
+
if _truthy(values.get("secretOptional")) or live is None or live.exists:
|
|
776
|
+
return
|
|
777
|
+
name = settings.secret_name
|
|
778
|
+
dry = "\n (--dry-run: the real deploy stops here the same way.)" if dry_run else ""
|
|
779
|
+
raise Refused(
|
|
780
|
+
f"Secret {name} does not exist in {target.namespace}, and the {env} chart values require "
|
|
781
|
+
"it (secretOptional: false): the pods would not start.\n"
|
|
782
|
+
f" {path} sets none of the allow-listed keys ({', '.join(settings.secret_keys)}), so "
|
|
783
|
+
"there is nothing to create it from; nothing was built, applied or deployed.\n"
|
|
784
|
+
f" Put the keys {env} needs in {path}, or set secretOptional: true in "
|
|
785
|
+
f"{settings.values_file(env)} if {env} needs no Secret.{dry}"
|
|
786
|
+
)
|
|
787
|
+
|
|
788
|
+
|
|
789
|
+
def _check_chart_env(
|
|
790
|
+
settings: DeploySettings, env: str, values: dict[str, Any], *, console: Console
|
|
791
|
+
) -> None:
|
|
792
|
+
"""Refuse (exit 3) a scaffold placeholder in the chart env outside dev; warn in dev.
|
|
793
|
+
|
|
794
|
+
``api add`` writes ``<API>_BASE_URL: http://CHANGE-ME`` and an
|
|
795
|
+
``openai-compatible`` project starts with ``OPENAI_BASE_URL`` at CHANGE-ME:
|
|
796
|
+
the pods would call that address, and every tool (or the model) would fail.
|
|
797
|
+
"""
|
|
798
|
+
keys = _preflight.env_placeholders(values)
|
|
799
|
+
if not keys:
|
|
800
|
+
return
|
|
801
|
+
names = ", ".join(f"env.{k}" for k in keys)
|
|
802
|
+
where = f"{settings.values_file(env)} (or {settings.chart_dir / 'values.yaml'})"
|
|
803
|
+
if _modes.is_dev_env(env):
|
|
804
|
+
console.print(
|
|
805
|
+
f" Warning: {names} still hold(s) the placeholder CHANGE-ME: the {env} pods would "
|
|
806
|
+
f"call it, so those calls fail. Set the real value(s) in {where}.",
|
|
807
|
+
style="yellow",
|
|
808
|
+
markup=False,
|
|
809
|
+
)
|
|
810
|
+
return
|
|
811
|
+
raise ConfigError(
|
|
812
|
+
f"{names} still hold(s) the placeholder CHANGE-ME in the chart values for {env}: the "
|
|
813
|
+
f"pods would call it. Set the real value(s) in {where}."
|
|
814
|
+
)
|
|
815
|
+
|
|
816
|
+
|
|
817
|
+
def _check_peers(
|
|
818
|
+
settings: DeploySettings,
|
|
819
|
+
env: str,
|
|
820
|
+
values: dict[str, Any],
|
|
821
|
+
*,
|
|
822
|
+
dry_run: bool,
|
|
823
|
+
console: Console,
|
|
824
|
+
) -> None:
|
|
825
|
+
"""The peer rules (api-policy.yaml's protocol: a2a APIs): exit 3 outside dev, warnings.
|
|
826
|
+
|
|
827
|
+
Runs before anything is built. An invalid or absent policy is `lint`'s to report.
|
|
828
|
+
"""
|
|
829
|
+
from graph_agents_cli._api_policy import POLICY_FILENAME, read_policy_document
|
|
830
|
+
|
|
831
|
+
document = read_policy_document(Path(POLICY_FILENAME))
|
|
832
|
+
findings = _preflight.peer_findings(settings, env, values, document)
|
|
833
|
+
error = findings.error()
|
|
834
|
+
if error:
|
|
835
|
+
raise ConfigError(
|
|
836
|
+
error + ("\n (--dry-run: the real deploy stops here the same way.)" if dry_run else "")
|
|
837
|
+
)
|
|
838
|
+
for note in findings.notes():
|
|
839
|
+
console.print(f" Warning: {note}", style="yellow", markup=False)
|
|
840
|
+
|
|
841
|
+
|
|
842
|
+
def _check_jwt(
|
|
843
|
+
settings: DeploySettings,
|
|
844
|
+
env: str,
|
|
845
|
+
values: dict[str, Any],
|
|
846
|
+
keys: set[str] | None,
|
|
847
|
+
*,
|
|
848
|
+
dry_run: bool,
|
|
849
|
+
console: Console,
|
|
850
|
+
warn: bool = True,
|
|
851
|
+
) -> None:
|
|
852
|
+
"""The ``jwt`` policy's verification settings: exit 3 outside dev, a warning in dev."""
|
|
853
|
+
findings = _preflight.jwt_findings(settings, env, values, keys)
|
|
854
|
+
error = findings.error()
|
|
855
|
+
if error:
|
|
856
|
+
raise ConfigError(
|
|
857
|
+
error + ("\n (--dry-run: the real deploy stops here the same way.)" if dry_run else "")
|
|
858
|
+
)
|
|
859
|
+
warning = findings.warning()
|
|
860
|
+
if warning and warn:
|
|
861
|
+
console.print(f" Warning: {warning}", style="yellow", markup=False)
|
|
862
|
+
|
|
863
|
+
|
|
864
|
+
def _warn_dsn_without_tls(
|
|
865
|
+
settings: DeploySettings,
|
|
866
|
+
env: str,
|
|
867
|
+
values: dict[str, Any],
|
|
868
|
+
data: dict[str, str],
|
|
869
|
+
*,
|
|
870
|
+
console: Console,
|
|
871
|
+
) -> None:
|
|
872
|
+
"""Outside dev, warn when the external database's connection string does not require TLS.
|
|
873
|
+
|
|
874
|
+
The value is inspected in memory and never printed.
|
|
875
|
+
"""
|
|
876
|
+
for key in _preflight.dsn_keys(settings, values):
|
|
877
|
+
dsn = data.get(key) or ""
|
|
878
|
+
if _modes.is_dev_env(env) or not dsn or dsn == secrets_apply.PENDING_PLACEHOLDER:
|
|
879
|
+
continue
|
|
880
|
+
if _preflight.dsn_without_tls(values, dsn):
|
|
881
|
+
console.print(
|
|
882
|
+
f" Warning: {_preflight.dsn_tls_warning(key)}", style="yellow", markup=False
|
|
883
|
+
)
|
|
884
|
+
|
|
885
|
+
|
|
886
|
+
@dataclass(frozen=True)
|
|
887
|
+
class _Workload:
|
|
888
|
+
"""The live Deployment as far as a deploy needs it (read-only)."""
|
|
889
|
+
|
|
890
|
+
image: str
|
|
891
|
+
generation: int
|
|
892
|
+
|
|
893
|
+
|
|
894
|
+
def _live_workload(settings: DeploySettings, target: Target) -> _Workload | None:
|
|
895
|
+
"""The release's Deployment (its agent image and generation); ``None`` when absent or unreadable."""
|
|
896
|
+
try:
|
|
897
|
+
result = _kube.kubectl(
|
|
898
|
+
["get", "deployment", settings.release, "-o", "json"],
|
|
899
|
+
target,
|
|
900
|
+
check=False,
|
|
901
|
+
quiet=True,
|
|
902
|
+
)
|
|
903
|
+
body = json.loads(result.stdout or "null") if result.returncode == 0 else None
|
|
904
|
+
except (_kube.ToolFailed, json.JSONDecodeError):
|
|
905
|
+
return None
|
|
906
|
+
if not isinstance(body, dict):
|
|
907
|
+
return None
|
|
908
|
+
containers = ((body.get("spec") or {}).get("template") or {}).get("spec", {}).get(
|
|
909
|
+
"containers"
|
|
910
|
+
) or []
|
|
911
|
+
agent = next((c for c in containers if c.get("name") == "agent"), None) or (
|
|
912
|
+
containers[0] if containers else {}
|
|
913
|
+
)
|
|
914
|
+
try:
|
|
915
|
+
generation = int((body.get("metadata") or {}).get("generation") or 0)
|
|
916
|
+
except (TypeError, ValueError):
|
|
917
|
+
generation = 0
|
|
918
|
+
return _Workload(image=str(agent.get("image") or ""), generation=generation)
|
|
919
|
+
|
|
920
|
+
|
|
921
|
+
def _announce_same_image(
|
|
922
|
+
settings: DeploySettings,
|
|
923
|
+
env: str,
|
|
924
|
+
plan: _ImagePlan,
|
|
925
|
+
live: _Workload | None,
|
|
926
|
+
*,
|
|
927
|
+
console: Console,
|
|
928
|
+
) -> None:
|
|
929
|
+
"""Say up front when the release already runs this exact image reference."""
|
|
930
|
+
if live is None or live.image != plan.ref:
|
|
931
|
+
return
|
|
932
|
+
rebuilt = " The image is rebuilt and loaded under the same tag." if plan.build else ""
|
|
933
|
+
console.print(
|
|
934
|
+
f" {settings.release} already runs {plan.ref}: the image is unchanged.{rebuilt} helm "
|
|
935
|
+
"records a new revision, but the pods are replaced only if the chart values change, "
|
|
936
|
+
f"and a changed Secret reaches them only with `graph-agents-cli deploy --env {env} "
|
|
937
|
+
"--restart` (reported after the rollout).",
|
|
938
|
+
style="yellow",
|
|
939
|
+
markup=False,
|
|
940
|
+
)
|
|
941
|
+
|
|
942
|
+
|
|
943
|
+
def _secret_outcome(
|
|
944
|
+
snaps: list[secrets_apply.Snapshot], env: str, error: Exception, *, console: Console
|
|
945
|
+
) -> list[str]:
|
|
946
|
+
"""After a failed helm step: restore the Secret(s) when the release is as it was before."""
|
|
947
|
+
if not any(snap.touched for snap in snaps):
|
|
948
|
+
return []
|
|
949
|
+
if getattr(error, "release_unchanged", True):
|
|
950
|
+
return secrets_apply.restore(snaps, console=console)
|
|
951
|
+
return secrets_apply.describe_unrestored(
|
|
952
|
+
snaps, env, "the release was not put back to its previous revision"
|
|
953
|
+
)
|
|
954
|
+
|
|
955
|
+
|
|
956
|
+
def _report_rollout(
|
|
957
|
+
settings: DeploySettings,
|
|
958
|
+
env: str,
|
|
959
|
+
target: Target,
|
|
960
|
+
before: _Workload | None,
|
|
961
|
+
snaps: list[secrets_apply.Snapshot],
|
|
962
|
+
*,
|
|
963
|
+
dry_run: bool,
|
|
964
|
+
console: Console,
|
|
965
|
+
) -> None:
|
|
966
|
+
"""After a successful upgrade: say when no pod was replaced, and what that leaves stale."""
|
|
967
|
+
if dry_run or before is None:
|
|
968
|
+
return
|
|
969
|
+
after = _live_workload(settings, target)
|
|
970
|
+
if after is None or after.generation != before.generation:
|
|
971
|
+
return
|
|
972
|
+
console.print(
|
|
973
|
+
" No pods were replaced: the pod template (image and chart values) is unchanged.",
|
|
974
|
+
style="yellow",
|
|
975
|
+
)
|
|
976
|
+
changed = sorted({k for snap in snaps for k in snap.touched})
|
|
977
|
+
if changed:
|
|
978
|
+
console.print(
|
|
979
|
+
f" The Secret changed ({', '.join(changed)}), but running pods read it only when "
|
|
980
|
+
f"they start: run `graph-agents-cli deploy --env {env} --restart`.",
|
|
981
|
+
style="yellow",
|
|
982
|
+
markup=False,
|
|
983
|
+
)
|
|
984
|
+
|
|
985
|
+
|
|
986
|
+
def _helm_args(settings: DeploySettings, env: str, repository: str, tag: str) -> list[str]:
|
|
987
|
+
chart = settings.chart_dir
|
|
988
|
+
return [
|
|
989
|
+
str(chart),
|
|
990
|
+
"-f",
|
|
991
|
+
str(chart / "values.yaml"),
|
|
992
|
+
"-f",
|
|
993
|
+
str(settings.values_file(env)),
|
|
994
|
+
"--set",
|
|
995
|
+
f"image.repository={repository}",
|
|
996
|
+
"--set",
|
|
997
|
+
f"image.tag={tag}",
|
|
998
|
+
"--set",
|
|
999
|
+
f"existingSecret={settings.secret_name}",
|
|
1000
|
+
]
|
|
1001
|
+
|
|
1002
|
+
|
|
1003
|
+
def _chart_dependencies(chart: Path) -> list[str]:
|
|
1004
|
+
"""Names of the subcharts ``Chart.yaml`` declares (``[]`` when none or unreadable)."""
|
|
1005
|
+
import yaml
|
|
1006
|
+
|
|
1007
|
+
try:
|
|
1008
|
+
with open(chart / "Chart.yaml", encoding="utf-8") as f:
|
|
1009
|
+
data = yaml.safe_load(f) or {}
|
|
1010
|
+
except (OSError, yaml.YAMLError):
|
|
1011
|
+
return []
|
|
1012
|
+
deps = data.get("dependencies") if isinstance(data, dict) else None
|
|
1013
|
+
if not isinstance(deps, list):
|
|
1014
|
+
return []
|
|
1015
|
+
return [str(d.get("name")) for d in deps if isinstance(d, dict) and d.get("name")]
|
|
1016
|
+
|
|
1017
|
+
|
|
1018
|
+
def _missing_dependencies(chart: Path, names: list[str]) -> list[str]:
|
|
1019
|
+
"""Declared subcharts with no archive or directory under ``charts/``."""
|
|
1020
|
+
charts_dir = chart / "charts"
|
|
1021
|
+
missing: list[str] = []
|
|
1022
|
+
for name in names:
|
|
1023
|
+
archives = list(charts_dir.glob(f"{name}-*.tgz")) if charts_dir.is_dir() else []
|
|
1024
|
+
if not archives and not (charts_dir / name).is_dir():
|
|
1025
|
+
missing.append(name)
|
|
1026
|
+
return missing
|
|
1027
|
+
|
|
1028
|
+
|
|
1029
|
+
def _helm_dependency_build(settings: DeploySettings, *, dry_run: bool, console: Console) -> None:
|
|
1030
|
+
"""Fetch the subcharts ``Chart.yaml`` declares before ``helm upgrade`` or ``helm template``.
|
|
1031
|
+
|
|
1032
|
+
The scaffolded chart declares the bitnami ``postgresql`` and ``redis`` charts as
|
|
1033
|
+
conditional dependencies; helm refuses to render or install until they sit in
|
|
1034
|
+
``charts/``. The build is skipped when every subchart is present and
|
|
1035
|
+
``Chart.lock`` exists. Under ``--dry-run`` the command still runs: it only
|
|
1036
|
+
writes into the chart directory (no cluster access) and the dry-run render
|
|
1037
|
+
(``helm template``) cannot succeed without it.
|
|
1038
|
+
"""
|
|
1039
|
+
chart = settings.chart_dir
|
|
1040
|
+
names = _chart_dependencies(chart)
|
|
1041
|
+
if not names:
|
|
1042
|
+
return
|
|
1043
|
+
missing = _missing_dependencies(chart, names)
|
|
1044
|
+
if not missing and (chart / "Chart.lock").is_file():
|
|
1045
|
+
return
|
|
1046
|
+
cmd = ["helm", "dependency", "build", str(chart)]
|
|
1047
|
+
_kube.echo_cmd(cmd, dry_run=dry_run, console=console)
|
|
1048
|
+
if dry_run:
|
|
1049
|
+
console.print(
|
|
1050
|
+
f" [dry-run] fetching subchart(s) {', '.join(missing or names)} into "
|
|
1051
|
+
f"{chart / 'charts'} so the render below can run (local only).",
|
|
1052
|
+
style="cyan",
|
|
1053
|
+
markup=False,
|
|
1054
|
+
)
|
|
1055
|
+
result = _kube.run_cmd(cmd, check=False, quiet=True, console=console)
|
|
1056
|
+
if result.returncode != 0:
|
|
1057
|
+
raise _kube.ToolFailed(
|
|
1058
|
+
f"helm dependency build failed (exit code {result.returncode}); the chart declares "
|
|
1059
|
+
f"subchart(s) {', '.join(names)} that must be fetched before helm can render it. "
|
|
1060
|
+
"Check network access to the chart repository or vendor the charts under "
|
|
1061
|
+
f"{chart / 'charts'}.\n{(result.stderr or result.stdout).strip()}"
|
|
1062
|
+
)
|
|
1063
|
+
|
|
1064
|
+
|
|
1065
|
+
def _helm_upgrade(
|
|
1066
|
+
settings: DeploySettings,
|
|
1067
|
+
env: str,
|
|
1068
|
+
target: Target,
|
|
1069
|
+
plan: _ImagePlan,
|
|
1070
|
+
opts: _Options,
|
|
1071
|
+
*,
|
|
1072
|
+
console: Console,
|
|
1073
|
+
) -> None:
|
|
1074
|
+
"""``helm upgrade --install --wait --timeout``; on failure print diagnostics, then roll back.
|
|
1075
|
+
|
|
1076
|
+
The rollback (``--atomic``, the default) is done here rather than with
|
|
1077
|
+
helm's own ``--atomic``: helm rolls back before it returns, which deletes
|
|
1078
|
+
the failed pods and their logs, the very output that explains the failure.
|
|
1079
|
+
It follows helm's rule: back to the newest deployed or superseded revision,
|
|
1080
|
+
or uninstall a first install that never succeeded.
|
|
1081
|
+
|
|
1082
|
+
It only ever undoes the revision this run created. The release's newest
|
|
1083
|
+
revision is recorded just before the upgrade; after a failure the CLI acts
|
|
1084
|
+
only when exactly one newer revision exists and it is ``failed``. When helm
|
|
1085
|
+
reports another operation in progress, or another deploy changed the
|
|
1086
|
+
release meanwhile, the release is left alone: rolling back would undo (or
|
|
1087
|
+
uninstall) someone else's rollout.
|
|
1088
|
+
"""
|
|
1089
|
+
common = _helm_args(settings, env, plan.repository, plan.tag)
|
|
1090
|
+
upgrade = [
|
|
1091
|
+
"upgrade",
|
|
1092
|
+
"--install",
|
|
1093
|
+
settings.release,
|
|
1094
|
+
*common,
|
|
1095
|
+
"--create-namespace",
|
|
1096
|
+
"--wait",
|
|
1097
|
+
"--timeout",
|
|
1098
|
+
opts.timeout,
|
|
1099
|
+
]
|
|
1100
|
+
_helm_dependency_build(settings, dry_run=opts.dry_run, console=console)
|
|
1101
|
+
if opts.dry_run:
|
|
1102
|
+
_kube.echo_cmd(_kube.helm_args(upgrade, target), dry_run=True, console=console)
|
|
1103
|
+
if opts.atomic:
|
|
1104
|
+
console.print(
|
|
1105
|
+
" [dry-run] a failed rollout prints pod diagnostics and is rolled back "
|
|
1106
|
+
"(--no-atomic keeps it).",
|
|
1107
|
+
style="cyan",
|
|
1108
|
+
markup=False,
|
|
1109
|
+
)
|
|
1110
|
+
console.print(
|
|
1111
|
+
" [dry-run] rendering with `helm template` instead:", style="cyan", markup=False
|
|
1112
|
+
)
|
|
1113
|
+
result = _kube.helm(
|
|
1114
|
+
["template", settings.release, *common], target, check=False, console=console
|
|
1115
|
+
)
|
|
1116
|
+
if result.returncode != 0:
|
|
1117
|
+
raise _kube.ToolFailed(
|
|
1118
|
+
f"helm template failed (exit code {result.returncode}):\n{(result.stderr or result.stdout).strip()}"
|
|
1119
|
+
)
|
|
1120
|
+
if result.stdout:
|
|
1121
|
+
console.print(result.stdout, highlight=False, markup=False)
|
|
1122
|
+
return
|
|
1123
|
+
before = _release_history(settings, target, console=console)
|
|
1124
|
+
_refuse_if_busy(settings, target, before, changed="the release was not touched")
|
|
1125
|
+
console.print(f" helm waits up to {opts.timeout} for the rollout.", style="dim")
|
|
1126
|
+
started = _now()
|
|
1127
|
+
# Captured (helm prints nothing until the rollout ends under --wait) so that
|
|
1128
|
+
# helm's own "another operation is in progress" refusal can be recognised.
|
|
1129
|
+
result = _kube.helm(upgrade, target, check=False, console=console)
|
|
1130
|
+
for stream in (result.stdout, result.stderr):
|
|
1131
|
+
if (stream or "").strip():
|
|
1132
|
+
console.print(stream.rstrip(), highlight=False, markup=False)
|
|
1133
|
+
if result.returncode == 0:
|
|
1134
|
+
return
|
|
1135
|
+
after = _release_history(settings, target, console=console)
|
|
1136
|
+
if _HELM_BUSY in f"{result.stderr or ''}\n{result.stdout or ''}":
|
|
1137
|
+
raise RolloutFailed(
|
|
1138
|
+
_busy_message(settings, target, after, changed="the release was not touched"),
|
|
1139
|
+
release_unchanged=True,
|
|
1140
|
+
)
|
|
1141
|
+
outcome, unchanged = _failure_outcome(
|
|
1142
|
+
settings, target, before, after, opts, since=started, console=console
|
|
1143
|
+
)
|
|
1144
|
+
raise RolloutFailed(
|
|
1145
|
+
f"helm upgrade failed (exit code {result.returncode}) for {settings.release} in "
|
|
1146
|
+
f"{target.namespace}; {outcome}.",
|
|
1147
|
+
release_unchanged=unchanged,
|
|
1148
|
+
)
|
|
1149
|
+
|
|
1150
|
+
|
|
1151
|
+
class RolloutFailed(_kube.ToolFailed):
|
|
1152
|
+
"""helm failed (exit 2); ``release_unchanged``: the release is back as it was before the run.
|
|
1153
|
+
|
|
1154
|
+
Only then is the Secret this run applied put back: a release left on the
|
|
1155
|
+
failed (or another deploy's) revision keeps the values it was rolled out with.
|
|
1156
|
+
"""
|
|
1157
|
+
|
|
1158
|
+
def __init__(self, message: str, *, release_unchanged: bool = False) -> None:
|
|
1159
|
+
super().__init__(message)
|
|
1160
|
+
self.release_unchanged = release_unchanged
|
|
1161
|
+
|
|
1162
|
+
|
|
1163
|
+
def _now() -> _dt.datetime:
|
|
1164
|
+
return _dt.datetime.now(_dt.UTC)
|
|
1165
|
+
|
|
1166
|
+
|
|
1167
|
+
def _failure_outcome(
|
|
1168
|
+
settings: DeploySettings,
|
|
1169
|
+
target: Target,
|
|
1170
|
+
before: _History,
|
|
1171
|
+
after: _History,
|
|
1172
|
+
opts: _Options,
|
|
1173
|
+
*,
|
|
1174
|
+
since: _dt.datetime | None = None,
|
|
1175
|
+
console: Console,
|
|
1176
|
+
) -> tuple[str, bool]:
|
|
1177
|
+
"""Undo the failed revision this run created, if any; describe what happened to the release.
|
|
1178
|
+
|
|
1179
|
+
Returns the description and whether the release is back as it was before the
|
|
1180
|
+
run (no new revision, rolled back, or a failed first install uninstalled).
|
|
1181
|
+
"""
|
|
1182
|
+
check = f"check `helm history {settings.release} -n {target.namespace}`"
|
|
1183
|
+
if not (before.readable and after.readable):
|
|
1184
|
+
return (
|
|
1185
|
+
"the release history could not be read, so the failure cannot be tied to a "
|
|
1186
|
+
f"revision of this run and nothing was rolled back; {check}"
|
|
1187
|
+
), False
|
|
1188
|
+
newer = after.newer_than(before.latest)
|
|
1189
|
+
if not newer:
|
|
1190
|
+
return (
|
|
1191
|
+
"helm recorded no new revision (it failed before the rollout), so the release is "
|
|
1192
|
+
"unchanged and there is nothing to roll back"
|
|
1193
|
+
), True
|
|
1194
|
+
newest = newer[-1]
|
|
1195
|
+
number, status = int(newest["revision"]), _status(newest)
|
|
1196
|
+
if status.startswith(_PENDING):
|
|
1197
|
+
if len(newer) > 1:
|
|
1198
|
+
return (
|
|
1199
|
+
f"another helm operation started after this one failed (revision {number} is "
|
|
1200
|
+
f"{status}), so nothing was rolled back; {check}"
|
|
1201
|
+
), False
|
|
1202
|
+
# Either this run's helm stopped before finishing its revision (killed,
|
|
1203
|
+
# crashed, lost the connection) or another deploy holds the release.
|
|
1204
|
+
return (
|
|
1205
|
+
f"revision {number} is still {status} (helm stopped before finishing it, or "
|
|
1206
|
+
"another deploy is working on the release), so nothing was rolled back; if no "
|
|
1207
|
+
f"other deploy is running, {_clear_hint(settings, target, after)} clears it"
|
|
1208
|
+
), False
|
|
1209
|
+
if len(newer) > 1:
|
|
1210
|
+
return (
|
|
1211
|
+
f"another deploy changed the release while this one ran (revisions "
|
|
1212
|
+
f"{int(newer[0]['revision'])} to {number} are new), so nothing was rolled back; "
|
|
1213
|
+
f"{check}"
|
|
1214
|
+
), False
|
|
1215
|
+
if status != "failed":
|
|
1216
|
+
return (
|
|
1217
|
+
f"its new revision {number} is {status}, so nothing was rolled back; {check}",
|
|
1218
|
+
False,
|
|
1219
|
+
)
|
|
1220
|
+
# Exactly one new revision and it failed: the one this run created.
|
|
1221
|
+
# Read the diagnostics before the rollback: it removes the failed pods and their logs.
|
|
1222
|
+
_print_rollout_diagnostics(settings, target, since=since, console=console)
|
|
1223
|
+
if not opts.atomic:
|
|
1224
|
+
return (
|
|
1225
|
+
f"the failed revision {number} was left in place (--no-atomic); roll back with "
|
|
1226
|
+
f"`helm rollback {settings.release} -n {target.namespace}`"
|
|
1227
|
+
), False
|
|
1228
|
+
return _roll_back(
|
|
1229
|
+
settings,
|
|
1230
|
+
target,
|
|
1231
|
+
after,
|
|
1232
|
+
number,
|
|
1233
|
+
existed=before.installed,
|
|
1234
|
+
timeout=opts.timeout,
|
|
1235
|
+
console=console,
|
|
1236
|
+
)
|
|
1237
|
+
|
|
1238
|
+
|
|
1239
|
+
def _diag(console: Console, args: list[str], target: Target, *, tail: int | None = None) -> str:
|
|
1240
|
+
"""Run a read-only kubectl for diagnostics and print it; never raises. Returns stdout."""
|
|
1241
|
+
cmd = _kube.kubectl_args(args, target)
|
|
1242
|
+
try:
|
|
1243
|
+
result = _kube.run_cmd(cmd, check=False, quiet=True)
|
|
1244
|
+
except _kube.ToolFailed as e:
|
|
1245
|
+
console.print(f" $ {_kube.format_cmd(cmd)}\n {e}", markup=False, highlight=False)
|
|
1246
|
+
return ""
|
|
1247
|
+
out = (result.stdout or "").rstrip() or (result.stderr or "").rstrip() or "(no output)"
|
|
1248
|
+
lines = out.splitlines()
|
|
1249
|
+
if tail is not None and len(lines) > tail + 1:
|
|
1250
|
+
lines = lines[:1] + lines[-tail:] # keep the header row
|
|
1251
|
+
console.print(f" $ {_kube.format_cmd(cmd)}", style="dim", markup=False, highlight=False)
|
|
1252
|
+
for line in lines:
|
|
1253
|
+
console.print(f" {line}", markup=False, highlight=False)
|
|
1254
|
+
return result.stdout or ""
|
|
1255
|
+
|
|
1256
|
+
|
|
1257
|
+
def _pod_ready(pod: dict[str, Any]) -> bool:
|
|
1258
|
+
status = pod.get("status") or {}
|
|
1259
|
+
if status.get("phase") == "Succeeded":
|
|
1260
|
+
return True
|
|
1261
|
+
return any(
|
|
1262
|
+
c.get("type") == "Ready" and str(c.get("status")) == "True"
|
|
1263
|
+
for c in status.get("conditions") or []
|
|
1264
|
+
)
|
|
1265
|
+
|
|
1266
|
+
|
|
1267
|
+
def _terminating(pod: dict[str, Any]) -> bool:
|
|
1268
|
+
return bool((pod.get("metadata") or {}).get("deletionTimestamp"))
|
|
1269
|
+
|
|
1270
|
+
|
|
1271
|
+
def _container_notes(pod: dict[str, Any]) -> list[str]:
|
|
1272
|
+
status = pod.get("status") or {}
|
|
1273
|
+
notes: list[str] = []
|
|
1274
|
+
for c in (status.get("initContainerStatuses") or []) + (status.get("containerStatuses") or []):
|
|
1275
|
+
name = c.get("name", "?")
|
|
1276
|
+
state = c.get("state") or {}
|
|
1277
|
+
last = (c.get("lastState") or {}).get("terminated") or {}
|
|
1278
|
+
if "waiting" in state:
|
|
1279
|
+
w = state["waiting"] or {}
|
|
1280
|
+
notes.append(f"{name}: waiting {w.get('reason', '')} {w.get('message', '')}".rstrip())
|
|
1281
|
+
elif "terminated" in state:
|
|
1282
|
+
t = state["terminated"] or {}
|
|
1283
|
+
notes.append(f"{name}: terminated {t.get('reason', '')} exit {t.get('exitCode')}")
|
|
1284
|
+
if last:
|
|
1285
|
+
notes.append(
|
|
1286
|
+
f"{name}: last run {last.get('reason', '')} exit {last.get('exitCode')} "
|
|
1287
|
+
f"(restarts {c.get('restartCount', 0)})"
|
|
1288
|
+
)
|
|
1289
|
+
return notes
|
|
1290
|
+
|
|
1291
|
+
|
|
1292
|
+
def _release_pods(settings: DeploySettings, target: Target) -> list[dict[str, Any]]:
|
|
1293
|
+
"""The release's pods (``-l app.kubernetes.io/instance=<release>``); ``[]`` when unreadable."""
|
|
1294
|
+
selector = f"app.kubernetes.io/instance={settings.release}"
|
|
1295
|
+
try:
|
|
1296
|
+
listing = _kube.run_cmd(
|
|
1297
|
+
_kube.kubectl_args(["get", "pods", "-l", selector, "-o", "json"], target),
|
|
1298
|
+
check=False,
|
|
1299
|
+
quiet=True,
|
|
1300
|
+
)
|
|
1301
|
+
pods = json.loads(listing.stdout or "{}").get("items") or []
|
|
1302
|
+
except (_kube.ToolFailed, json.JSONDecodeError, AttributeError):
|
|
1303
|
+
return []
|
|
1304
|
+
return [p for p in pods if isinstance(p, dict)]
|
|
1305
|
+
|
|
1306
|
+
|
|
1307
|
+
def _release_objects(settings: DeploySettings, target: Target) -> set[tuple[str, str]]:
|
|
1308
|
+
"""``(kind, name)`` of the release's objects that warning events are about.
|
|
1309
|
+
|
|
1310
|
+
The Deployment, its ReplicaSets and pods, and the bundled database's
|
|
1311
|
+
StatefulSet, pods and volume claims: everything labelled with the release.
|
|
1312
|
+
"""
|
|
1313
|
+
selector = f"app.kubernetes.io/instance={settings.release}"
|
|
1314
|
+
objects = {("Deployment", settings.release)}
|
|
1315
|
+
try:
|
|
1316
|
+
listing = _kube.run_cmd(
|
|
1317
|
+
_kube.kubectl_args(
|
|
1318
|
+
[
|
|
1319
|
+
"get",
|
|
1320
|
+
"pods,replicasets,deployments,statefulsets,persistentvolumeclaims",
|
|
1321
|
+
"-l",
|
|
1322
|
+
selector,
|
|
1323
|
+
"-o",
|
|
1324
|
+
"json",
|
|
1325
|
+
],
|
|
1326
|
+
target,
|
|
1327
|
+
),
|
|
1328
|
+
check=False,
|
|
1329
|
+
quiet=True,
|
|
1330
|
+
)
|
|
1331
|
+
items = json.loads(listing.stdout or "{}").get("items") or []
|
|
1332
|
+
except (_kube.ToolFailed, json.JSONDecodeError, AttributeError):
|
|
1333
|
+
items = []
|
|
1334
|
+
for item in items:
|
|
1335
|
+
if isinstance(item, dict):
|
|
1336
|
+
name = (item.get("metadata") or {}).get("name")
|
|
1337
|
+
if item.get("kind") and name:
|
|
1338
|
+
objects.add((str(item["kind"]), str(name)))
|
|
1339
|
+
return objects
|
|
1340
|
+
|
|
1341
|
+
|
|
1342
|
+
def _parse_time(value: object) -> _dt.datetime | None:
|
|
1343
|
+
if not value:
|
|
1344
|
+
return None
|
|
1345
|
+
try:
|
|
1346
|
+
parsed = _dt.datetime.fromisoformat(str(value).replace("Z", "+00:00"))
|
|
1347
|
+
except ValueError:
|
|
1348
|
+
return None
|
|
1349
|
+
return parsed if parsed.tzinfo else parsed.replace(tzinfo=_dt.UTC)
|
|
1350
|
+
|
|
1351
|
+
|
|
1352
|
+
def _event_time(event: dict[str, Any]) -> _dt.datetime | None:
|
|
1353
|
+
"""When a (possibly repeated) event was last seen."""
|
|
1354
|
+
series = event.get("series") or {}
|
|
1355
|
+
for value in (
|
|
1356
|
+
event.get("lastTimestamp"),
|
|
1357
|
+
series.get("lastObservedTime") if isinstance(series, dict) else None,
|
|
1358
|
+
event.get("eventTime"),
|
|
1359
|
+
event.get("firstTimestamp"),
|
|
1360
|
+
(event.get("metadata") or {}).get("creationTimestamp"),
|
|
1361
|
+
):
|
|
1362
|
+
parsed = _parse_time(value)
|
|
1363
|
+
if parsed is not None:
|
|
1364
|
+
return parsed
|
|
1365
|
+
return None
|
|
1366
|
+
|
|
1367
|
+
|
|
1368
|
+
def _age(when: _dt.datetime | None) -> str:
|
|
1369
|
+
if when is None:
|
|
1370
|
+
return "?"
|
|
1371
|
+
seconds = max(0, int((_now() - when).total_seconds()))
|
|
1372
|
+
if seconds < 120:
|
|
1373
|
+
return f"{seconds}s"
|
|
1374
|
+
if seconds < 7200:
|
|
1375
|
+
return f"{seconds // 60}m"
|
|
1376
|
+
return f"{seconds // 3600}h"
|
|
1377
|
+
|
|
1378
|
+
|
|
1379
|
+
def _print_warning_events(
|
|
1380
|
+
settings: DeploySettings,
|
|
1381
|
+
target: Target,
|
|
1382
|
+
*,
|
|
1383
|
+
since: _dt.datetime | None,
|
|
1384
|
+
console: Console,
|
|
1385
|
+
) -> None:
|
|
1386
|
+
"""Warning events about this release's objects, newer than ``since`` (less clock skew).
|
|
1387
|
+
|
|
1388
|
+
The namespace may hold other workloads and events from earlier rollouts
|
|
1389
|
+
(events live for an hour): only this release's objects are shown, and with
|
|
1390
|
+
``since`` only what happened during this run.
|
|
1391
|
+
"""
|
|
1392
|
+
cmd = _kube.kubectl_args(
|
|
1393
|
+
["get", "events", "--field-selector", "type=Warning", "-o", "json"], target
|
|
1394
|
+
)
|
|
1395
|
+
console.print(f" $ {_kube.format_cmd(cmd)}", style="dim", markup=False, highlight=False)
|
|
1396
|
+
try:
|
|
1397
|
+
result = _kube.run_cmd(cmd, check=False, quiet=True)
|
|
1398
|
+
events = json.loads(result.stdout or "{}").get("items") or []
|
|
1399
|
+
except (_kube.ToolFailed, json.JSONDecodeError, AttributeError) as e:
|
|
1400
|
+
console.print(f" (could not list events: {_first_line(e) or 'no output'})", markup=False)
|
|
1401
|
+
return
|
|
1402
|
+
objects = _release_objects(settings, target)
|
|
1403
|
+
cutoff = since - _dt.timedelta(seconds=EVENT_CLOCK_SKEW_S) if since else None
|
|
1404
|
+
mine: list[tuple[_dt.datetime | None, dict[str, Any]]] = []
|
|
1405
|
+
older = 0
|
|
1406
|
+
for event in events:
|
|
1407
|
+
if not isinstance(event, dict):
|
|
1408
|
+
continue
|
|
1409
|
+
involved = event.get("involvedObject") or event.get("regarding") or {}
|
|
1410
|
+
if (str(involved.get("kind")), str(involved.get("name"))) not in objects:
|
|
1411
|
+
continue
|
|
1412
|
+
when = _event_time(event)
|
|
1413
|
+
if cutoff is not None and (when is None or when < cutoff):
|
|
1414
|
+
older += 1
|
|
1415
|
+
continue
|
|
1416
|
+
mine.append((when, event))
|
|
1417
|
+
mine.sort(key=lambda pair: pair[0] or _dt.datetime.min.replace(tzinfo=_dt.UTC))
|
|
1418
|
+
if not mine:
|
|
1419
|
+
console.print(" (no warning events for this release)", markup=False)
|
|
1420
|
+
for when, event in mine[-DIAGNOSTIC_EVENTS:]:
|
|
1421
|
+
involved = event.get("involvedObject") or event.get("regarding") or {}
|
|
1422
|
+
count = event.get("count") or ((event.get("series") or {}).get("count"))
|
|
1423
|
+
times = f" (x{count})" if isinstance(count, int) and count > 1 else ""
|
|
1424
|
+
message = " ".join(str(event.get("message") or event.get("note") or "").split())
|
|
1425
|
+
console.print(
|
|
1426
|
+
f" {_age(when):>4} ago {involved.get('kind')}/{involved.get('name')} "
|
|
1427
|
+
f"{event.get('reason', '')}: {message}{times}",
|
|
1428
|
+
markup=False,
|
|
1429
|
+
highlight=False,
|
|
1430
|
+
)
|
|
1431
|
+
if older:
|
|
1432
|
+
console.print(
|
|
1433
|
+
f" ({older} older warning event(s) for this release, from before this run, not "
|
|
1434
|
+
"shown)",
|
|
1435
|
+
style="dim",
|
|
1436
|
+
markup=False,
|
|
1437
|
+
)
|
|
1438
|
+
|
|
1439
|
+
|
|
1440
|
+
def _print_rollout_diagnostics(
|
|
1441
|
+
settings: DeploySettings,
|
|
1442
|
+
target: Target,
|
|
1443
|
+
*,
|
|
1444
|
+
since: _dt.datetime | None = None,
|
|
1445
|
+
headline: str | None = None,
|
|
1446
|
+
console: Console,
|
|
1447
|
+
) -> None:
|
|
1448
|
+
"""Pods, container states, this release's warning events and recent logs (best effort)."""
|
|
1449
|
+
selector = f"app.kubernetes.io/instance={settings.release}"
|
|
1450
|
+
console.print(
|
|
1451
|
+
headline or f"Rollout of {settings.release} in {target.namespace} failed; diagnostics:",
|
|
1452
|
+
style="yellow",
|
|
1453
|
+
)
|
|
1454
|
+
_diag(console, ["get", "pods", "-l", selector, "-o", "wide"], target)
|
|
1455
|
+
pods = _release_pods(settings, target)
|
|
1456
|
+
# A terminating pod belongs to the revision being replaced: not what failed.
|
|
1457
|
+
failing = [p for p in pods if not _pod_ready(p) and not _terminating(p)]
|
|
1458
|
+
for pod in failing[:DIAGNOSTIC_PODS]:
|
|
1459
|
+
name = (pod.get("metadata") or {}).get("name", "?")
|
|
1460
|
+
for note in _container_notes(pod):
|
|
1461
|
+
console.print(f" {name}: {note}", markup=False, highlight=False)
|
|
1462
|
+
_print_warning_events(settings, target, since=since, console=console)
|
|
1463
|
+
for pod in failing[:DIAGNOSTIC_PODS]:
|
|
1464
|
+
name = (pod.get("metadata") or {}).get("name", "?")
|
|
1465
|
+
logs = _diag(
|
|
1466
|
+
console,
|
|
1467
|
+
["logs", name, "--all-containers", f"--tail={DIAGNOSTIC_LOG_LINES}"],
|
|
1468
|
+
target,
|
|
1469
|
+
)
|
|
1470
|
+
restarted = any(
|
|
1471
|
+
int(c.get("restartCount") or 0) > 0
|
|
1472
|
+
for c in (pod.get("status") or {}).get("containerStatuses") or []
|
|
1473
|
+
)
|
|
1474
|
+
if restarted and not logs.strip():
|
|
1475
|
+
_diag(
|
|
1476
|
+
console,
|
|
1477
|
+
["logs", name, "--all-containers", "--previous", f"--tail={DIAGNOSTIC_LOG_LINES}"],
|
|
1478
|
+
target,
|
|
1479
|
+
)
|
|
1480
|
+
|
|
1481
|
+
|
|
1482
|
+
_HEALTHY = ("deployed", "superseded")
|
|
1483
|
+
_PENDING = "pending-"
|
|
1484
|
+
# helm's refusal while the release's newest revision is pending-install/-upgrade/-rollback.
|
|
1485
|
+
_HELM_BUSY = "another operation (install/upgrade/rollback) is in progress"
|
|
1486
|
+
|
|
1487
|
+
|
|
1488
|
+
def _status(revision: dict[str, Any]) -> str:
|
|
1489
|
+
return str(revision.get("status", "")).strip().lower()
|
|
1490
|
+
|
|
1491
|
+
|
|
1492
|
+
@dataclass(frozen=True)
|
|
1493
|
+
class _History:
|
|
1494
|
+
"""``helm history`` of the release; ``readable`` is False when helm could not read it."""
|
|
1495
|
+
|
|
1496
|
+
revisions: tuple[dict[str, Any], ...] = ()
|
|
1497
|
+
readable: bool = True
|
|
1498
|
+
|
|
1499
|
+
@property
|
|
1500
|
+
def latest(self) -> int:
|
|
1501
|
+
"""The newest revision number (0 when the release has none)."""
|
|
1502
|
+
return max((int(r["revision"]) for r in self.revisions), default=0)
|
|
1503
|
+
|
|
1504
|
+
def newer_than(self, revision: int) -> list[dict[str, Any]]:
|
|
1505
|
+
"""Revisions after ``revision``, oldest first."""
|
|
1506
|
+
newer = [r for r in self.revisions if int(r["revision"]) > revision]
|
|
1507
|
+
return sorted(newer, key=lambda r: int(r["revision"]))
|
|
1508
|
+
|
|
1509
|
+
@property
|
|
1510
|
+
def pending(self) -> dict[str, Any] | None:
|
|
1511
|
+
"""The newest revision when helm is (or was, if interrupted) still working on it."""
|
|
1512
|
+
newest = max(self.revisions, key=lambda r: int(r["revision"]), default=None)
|
|
1513
|
+
return newest if newest is not None and _status(newest).startswith(_PENDING) else None
|
|
1514
|
+
|
|
1515
|
+
@property
|
|
1516
|
+
def installed(self) -> bool:
|
|
1517
|
+
"""Whether the release exists (``uninstall --keep-history`` leaves uninstalled ones)."""
|
|
1518
|
+
return any(_status(r) != "uninstalled" for r in self.revisions)
|
|
1519
|
+
|
|
1520
|
+
|
|
1521
|
+
def _release_history(settings: DeploySettings, target: Target, *, console: Console) -> _History:
|
|
1522
|
+
"""``helm history`` of the release: empty when it does not exist, unreadable on other errors."""
|
|
1523
|
+
try:
|
|
1524
|
+
history = _kube.helm(
|
|
1525
|
+
["history", settings.release, "-o", "json"],
|
|
1526
|
+
target,
|
|
1527
|
+
check=False,
|
|
1528
|
+
quiet=True,
|
|
1529
|
+
console=console,
|
|
1530
|
+
)
|
|
1531
|
+
except _kube.ToolFailed:
|
|
1532
|
+
return _History(readable=False)
|
|
1533
|
+
if history.returncode != 0:
|
|
1534
|
+
# `Error: release: not found` means no release yet; anything else is unknown.
|
|
1535
|
+
return _History(readable="release: not found" in (history.stderr or ""))
|
|
1536
|
+
try:
|
|
1537
|
+
revisions = json.loads(history.stdout or "[]")
|
|
1538
|
+
except json.JSONDecodeError:
|
|
1539
|
+
return _History(readable=False)
|
|
1540
|
+
if not isinstance(revisions, list):
|
|
1541
|
+
return _History(readable=False)
|
|
1542
|
+
return _History(
|
|
1543
|
+
tuple(r for r in revisions if isinstance(r, dict) and str(r.get("revision", "")).isdigit())
|
|
1544
|
+
)
|
|
1545
|
+
|
|
1546
|
+
|
|
1547
|
+
def _clear_hint(settings: DeploySettings, target: Target, history: _History) -> str:
|
|
1548
|
+
"""The command that clears a release left pending by an interrupted helm."""
|
|
1549
|
+
release, ns = settings.release, target.namespace
|
|
1550
|
+
pending = history.pending
|
|
1551
|
+
below = int(pending["revision"]) if pending is not None else history.latest + 1
|
|
1552
|
+
good = [
|
|
1553
|
+
int(r["revision"])
|
|
1554
|
+
for r in history.revisions
|
|
1555
|
+
if _status(r) in _HEALTHY and int(r["revision"]) < below
|
|
1556
|
+
]
|
|
1557
|
+
if good:
|
|
1558
|
+
return f"`helm rollback {release} {max(good)} -n {ns}`"
|
|
1559
|
+
if pending is not None and _status(pending) == "pending-install":
|
|
1560
|
+
return f"`helm uninstall {release} -n {ns}` (the first install never finished)"
|
|
1561
|
+
return f"`helm rollback {release} <last good revision> -n {ns}`"
|
|
1562
|
+
|
|
1563
|
+
|
|
1564
|
+
def _busy_message(
|
|
1565
|
+
settings: DeploySettings, target: Target, history: _History, *, changed: str
|
|
1566
|
+
) -> str:
|
|
1567
|
+
release, ns = settings.release, target.namespace
|
|
1568
|
+
pending = history.pending
|
|
1569
|
+
what = f" (revision {pending['revision']} is {_status(pending)})" if pending else ""
|
|
1570
|
+
return (
|
|
1571
|
+
f"Another helm operation (install/upgrade/rollback) is in progress on {release} in "
|
|
1572
|
+
f"{ns}{what}; {changed}, and nothing was rolled back.\n"
|
|
1573
|
+
f" Wait for it to finish (`helm history {release} -n {ns}`) and deploy again. If no "
|
|
1574
|
+
"other deploy is running, an interrupted helm left the release pending: "
|
|
1575
|
+
f"{_clear_hint(settings, target, history)} clears it."
|
|
1576
|
+
)
|
|
1577
|
+
|
|
1578
|
+
|
|
1579
|
+
def _refuse_if_busy(
|
|
1580
|
+
settings: DeploySettings, target: Target, history: _History, *, changed: str
|
|
1581
|
+
) -> None:
|
|
1582
|
+
"""Exit 2 when another helm operation holds the release (helm itself would refuse)."""
|
|
1583
|
+
if history.pending is not None:
|
|
1584
|
+
raise _kube.ToolFailed(_busy_message(settings, target, history, changed=changed))
|
|
1585
|
+
|
|
1586
|
+
|
|
1587
|
+
def _check_release_idle(
|
|
1588
|
+
settings: DeploySettings, target: Target, *, dry_run: bool, console: Console
|
|
1589
|
+
) -> None:
|
|
1590
|
+
"""Stop before anything is built or applied while another deploy is rolling out."""
|
|
1591
|
+
if dry_run:
|
|
1592
|
+
return
|
|
1593
|
+
history = _release_history(settings, target, console=console)
|
|
1594
|
+
_refuse_if_busy(settings, target, history, changed="nothing was built, applied or deployed")
|
|
1595
|
+
|
|
1596
|
+
|
|
1597
|
+
def _roll_back(
|
|
1598
|
+
settings: DeploySettings,
|
|
1599
|
+
target: Target,
|
|
1600
|
+
history: _History,
|
|
1601
|
+
failed: int,
|
|
1602
|
+
*,
|
|
1603
|
+
existed: bool,
|
|
1604
|
+
timeout: str,
|
|
1605
|
+
console: Console,
|
|
1606
|
+
) -> tuple[str, bool]:
|
|
1607
|
+
"""Undo this run's failed revision ``failed`` the way helm's ``--atomic`` would; describe it.
|
|
1608
|
+
|
|
1609
|
+
Back to the newest deployed or superseded revision before it; with none, a
|
|
1610
|
+
release this run installed (``existed`` False) is uninstalled, and one that
|
|
1611
|
+
existed before is left in place (helm's ``--atomic`` does the same).
|
|
1612
|
+
"""
|
|
1613
|
+
release = settings.release
|
|
1614
|
+
good = [r for r in history.revisions if _status(r) in _HEALTHY and int(r["revision"]) < failed]
|
|
1615
|
+
if good:
|
|
1616
|
+
revision = str(max(int(r["revision"]) for r in good))
|
|
1617
|
+
console.print(f"Rolling back {release} to revision {revision} (--atomic).", style="yellow")
|
|
1618
|
+
result = _kube.helm(
|
|
1619
|
+
["rollback", release, revision, "--wait", "--timeout", timeout],
|
|
1620
|
+
target,
|
|
1621
|
+
capture=False,
|
|
1622
|
+
check=False,
|
|
1623
|
+
console=console,
|
|
1624
|
+
)
|
|
1625
|
+
if result.returncode == 0:
|
|
1626
|
+
return f"rolled back to revision {revision}", True
|
|
1627
|
+
return (
|
|
1628
|
+
f"the rollback to revision {revision} also failed (exit code {result.returncode}); "
|
|
1629
|
+
f"check `helm history {release} -n {target.namespace}`"
|
|
1630
|
+
), False
|
|
1631
|
+
if existed:
|
|
1632
|
+
return (
|
|
1633
|
+
f"there is no earlier successful revision to roll back to, so the failed revision "
|
|
1634
|
+
f"{failed} was left in place; fix the cause and deploy again, or remove it with "
|
|
1635
|
+
f"`helm uninstall {release} -n {target.namespace}`"
|
|
1636
|
+
), False
|
|
1637
|
+
console.print(
|
|
1638
|
+
f"Uninstalling {release}: the first install never succeeded (--atomic).", style="yellow"
|
|
1639
|
+
)
|
|
1640
|
+
result = _kube.helm(
|
|
1641
|
+
["uninstall", release, "--wait", "--timeout", timeout],
|
|
1642
|
+
target,
|
|
1643
|
+
capture=False,
|
|
1644
|
+
check=False,
|
|
1645
|
+
console=console,
|
|
1646
|
+
)
|
|
1647
|
+
if result.returncode == 0:
|
|
1648
|
+
return "the failed first install was uninstalled (there was no earlier revision)", True
|
|
1649
|
+
return (
|
|
1650
|
+
f"uninstalling the failed first install also failed (exit code {result.returncode}); "
|
|
1651
|
+
f"check `helm status {release} -n {target.namespace}`"
|
|
1652
|
+
), False
|
|
1653
|
+
|
|
1654
|
+
|
|
1655
|
+
def _print_done(
|
|
1656
|
+
settings: DeploySettings, env: str, image: str, *, dry_run: bool, console: Console
|
|
1657
|
+
) -> None:
|
|
1658
|
+
verb = "Would deploy" if dry_run else "Deployed"
|
|
1659
|
+
console.print(f"{verb} {settings.release} ({image}) to {env}.", style="green")
|
|
1660
|
+
|
|
1661
|
+
|
|
1662
|
+
# --------------------------------------------------------------------------- status / restart
|
|
1663
|
+
|
|
1664
|
+
|
|
1665
|
+
class NotReady(_kube.DeployError):
|
|
1666
|
+
"""``deploy --status``: the rollout is not complete within ``--timeout`` (exit 1)."""
|
|
1667
|
+
|
|
1668
|
+
exit_code = 1
|
|
1669
|
+
|
|
1670
|
+
|
|
1671
|
+
_ROLLOUT_OK = "ok"
|
|
1672
|
+
_ROLLOUT_NOT_READY = "not-ready"
|
|
1673
|
+
_ROLLOUT_ABSENT = "absent"
|
|
1674
|
+
|
|
1675
|
+
|
|
1676
|
+
def _rollout_status_cmd(settings: DeploySettings, target: Target, timeout: str) -> list[str]:
|
|
1677
|
+
return _kube.kubectl_args(
|
|
1678
|
+
["rollout", "status", f"deployment/{settings.release}", f"--timeout={timeout}"], target
|
|
1679
|
+
)
|
|
1680
|
+
|
|
1681
|
+
|
|
1682
|
+
def _wait_for_rollout(
|
|
1683
|
+
settings: DeploySettings, target: Target, *, timeout: str, console: Console
|
|
1684
|
+
) -> tuple[str, str]:
|
|
1685
|
+
"""``kubectl rollout status --timeout``: ``(state, kubectl's last line)``, never unbounded.
|
|
1686
|
+
|
|
1687
|
+
A timeout or an exceeded progress deadline is ``not-ready`` and a missing
|
|
1688
|
+
Deployment ``absent``; any other kubectl failure (cluster unreachable,
|
|
1689
|
+
credentials) is a tool failure (exit 2).
|
|
1690
|
+
"""
|
|
1691
|
+
cmd = _rollout_status_cmd(settings, target, timeout)
|
|
1692
|
+
_kube.echo_cmd(cmd, console=console)
|
|
1693
|
+
console.print(
|
|
1694
|
+
f" Waiting up to {timeout} for deployment/{settings.release} to roll out.", style="dim"
|
|
1695
|
+
)
|
|
1696
|
+
result = _kube.run_cmd(cmd, check=False, quiet=True)
|
|
1697
|
+
out = (result.stdout or "").strip()
|
|
1698
|
+
err = (result.stderr or "").strip()
|
|
1699
|
+
last = next((line for line in reversed((err or out).splitlines()) if line.strip()), "")
|
|
1700
|
+
if result.returncode == 0:
|
|
1701
|
+
return _ROLLOUT_OK, last
|
|
1702
|
+
if "(NotFound)" in err or "not found" in err.lower():
|
|
1703
|
+
return _ROLLOUT_ABSENT, last
|
|
1704
|
+
if "timed out" in err or "progress deadline" in err:
|
|
1705
|
+
return _ROLLOUT_NOT_READY, last
|
|
1706
|
+
raise _kube.ToolFailed(
|
|
1707
|
+
f"Command failed (exit code {result.returncode}): {_kube.format_cmd(cmd)}"
|
|
1708
|
+
+ (f"\n{err or out}" if err or out else "")
|
|
1709
|
+
)
|
|
1710
|
+
|
|
1711
|
+
|
|
1712
|
+
def _print_workload_summary(settings: DeploySettings, target: Target, *, console: Console) -> None:
|
|
1713
|
+
"""The Deployment's replicas and image, the helm revision, and every pod's readiness."""
|
|
1714
|
+
release = settings.release
|
|
1715
|
+
try:
|
|
1716
|
+
result = _kube.kubectl(
|
|
1717
|
+
["get", "deployment", release, "-o", "json"], target, check=False, quiet=True
|
|
1718
|
+
)
|
|
1719
|
+
deployment = json.loads(result.stdout or "null") if result.returncode == 0 else None
|
|
1720
|
+
except (_kube.ToolFailed, json.JSONDecodeError):
|
|
1721
|
+
deployment = None
|
|
1722
|
+
if isinstance(deployment, dict):
|
|
1723
|
+
spec = deployment.get("spec") or {}
|
|
1724
|
+
status = deployment.get("status") or {}
|
|
1725
|
+
containers = ((spec.get("template") or {}).get("spec") or {}).get("containers") or []
|
|
1726
|
+
agent = next((c for c in containers if c.get("name") == "agent"), None) or (
|
|
1727
|
+
containers[0] if containers else {}
|
|
1728
|
+
)
|
|
1729
|
+
wanted = spec.get("replicas", 1)
|
|
1730
|
+
console.print(
|
|
1731
|
+
f"deployment/{release}: {status.get('readyReplicas') or 0}/{wanted} ready, "
|
|
1732
|
+
f"{status.get('updatedReplicas') or 0} up to date, image {agent.get('image', '?')}",
|
|
1733
|
+
markup=False,
|
|
1734
|
+
highlight=False,
|
|
1735
|
+
)
|
|
1736
|
+
for condition in status.get("conditions") or []:
|
|
1737
|
+
if str(condition.get("status")) != "True":
|
|
1738
|
+
console.print(
|
|
1739
|
+
f" {condition.get('type')}: {condition.get('reason', '')} "
|
|
1740
|
+
f"{condition.get('message', '')}".rstrip(),
|
|
1741
|
+
markup=False,
|
|
1742
|
+
highlight=False,
|
|
1743
|
+
)
|
|
1744
|
+
if settings.cd != _modes.ARGOCD:
|
|
1745
|
+
history = _release_history(settings, target, console=console)
|
|
1746
|
+
if history.readable and history.revisions:
|
|
1747
|
+
newest = max(history.revisions, key=lambda r: int(r["revision"]))
|
|
1748
|
+
console.print(
|
|
1749
|
+
f"helm release {release}: revision {newest['revision']} ({_status(newest)})",
|
|
1750
|
+
markup=False,
|
|
1751
|
+
)
|
|
1752
|
+
for pod in _release_pods(settings, target):
|
|
1753
|
+
name = (pod.get("metadata") or {}).get("name", "?")
|
|
1754
|
+
statuses = (pod.get("status") or {}).get("containerStatuses") or []
|
|
1755
|
+
restarts = sum(int(c.get("restartCount") or 0) for c in statuses)
|
|
1756
|
+
last = next(
|
|
1757
|
+
(
|
|
1758
|
+
(c.get("lastState") or {}).get("terminated")
|
|
1759
|
+
for c in statuses
|
|
1760
|
+
if (c.get("lastState") or {}).get("terminated")
|
|
1761
|
+
),
|
|
1762
|
+
None,
|
|
1763
|
+
)
|
|
1764
|
+
why = f" (last exit: {last.get('reason', '')} {last.get('exitCode')})" if last else ""
|
|
1765
|
+
if _terminating(pod):
|
|
1766
|
+
state = "terminating"
|
|
1767
|
+
else:
|
|
1768
|
+
state = "ready" if _pod_ready(pod) else "NOT ready"
|
|
1769
|
+
console.print(
|
|
1770
|
+
f" pod {name}: {state}, restarts {restarts}{why}", markup=False, highlight=False
|
|
1771
|
+
)
|
|
1772
|
+
|
|
1773
|
+
|
|
1774
|
+
def _show_status(
|
|
1775
|
+
settings: DeploySettings,
|
|
1776
|
+
env: str,
|
|
1777
|
+
target: Target,
|
|
1778
|
+
*,
|
|
1779
|
+
timeout: str,
|
|
1780
|
+
dry_run: bool,
|
|
1781
|
+
console: Console,
|
|
1782
|
+
) -> None:
|
|
1783
|
+
"""Report the rollout within ``timeout``: exit 1 (with diagnostics) when it is not complete."""
|
|
1784
|
+
if settings.cd == _modes.ARGOCD and _kube.tool_available("argocd"):
|
|
1785
|
+
_kube.run_cmd(
|
|
1786
|
+
["argocd", "app", "get", f"{settings.project_name}-{env}"],
|
|
1787
|
+
capture=False,
|
|
1788
|
+
dry_run=dry_run,
|
|
1789
|
+
console=console,
|
|
1790
|
+
)
|
|
1791
|
+
return
|
|
1792
|
+
if dry_run:
|
|
1793
|
+
_kube.echo_cmd(
|
|
1794
|
+
_rollout_status_cmd(settings, target, timeout), dry_run=True, console=console
|
|
1795
|
+
)
|
|
1796
|
+
return
|
|
1797
|
+
state, detail = _wait_for_rollout(settings, target, timeout=timeout, console=console)
|
|
1798
|
+
where = f"deployment/{settings.release} in {target.namespace}"
|
|
1799
|
+
if state == _ROLLOUT_ABSENT:
|
|
1800
|
+
raise NotReady(f"{where} does not exist (not deployed yet?): {detail}")
|
|
1801
|
+
_print_workload_summary(settings, target, console=console)
|
|
1802
|
+
if state == _ROLLOUT_OK:
|
|
1803
|
+
console.print(f"{where}: rollout complete.", style="green", markup=False)
|
|
1804
|
+
return
|
|
1805
|
+
_print_rollout_diagnostics(
|
|
1806
|
+
settings,
|
|
1807
|
+
target,
|
|
1808
|
+
headline=f"{where} is not ready after {timeout}; diagnostics:",
|
|
1809
|
+
console=console,
|
|
1810
|
+
)
|
|
1811
|
+
raise NotReady(
|
|
1812
|
+
f"The rollout of {where} did not complete within {timeout} ({detail}). Pass a longer "
|
|
1813
|
+
"--timeout to wait more; the diagnostics above show why the pods are not ready."
|
|
1814
|
+
)
|
|
1815
|
+
|
|
1816
|
+
|
|
1817
|
+
def _restart(
|
|
1818
|
+
settings: DeploySettings,
|
|
1819
|
+
env: str,
|
|
1820
|
+
target: Target,
|
|
1821
|
+
*,
|
|
1822
|
+
timeout: str,
|
|
1823
|
+
dry_run: bool,
|
|
1824
|
+
console: Console,
|
|
1825
|
+
) -> None:
|
|
1826
|
+
"""``kubectl rollout restart``, then wait (bounded) until the new pods are ready."""
|
|
1827
|
+
if settings.cd == _modes.ARGOCD:
|
|
1828
|
+
console.print(
|
|
1829
|
+
" Warning: this environment is reconciled by Argo CD with self-heal; the restart "
|
|
1830
|
+
"annotation may be reverted. Prefer an Argo resource action (restart) on the Deployment.",
|
|
1831
|
+
style="yellow",
|
|
1832
|
+
)
|
|
1833
|
+
where = f"deployment/{settings.release} in {target.namespace}"
|
|
1834
|
+
_kube.kubectl(
|
|
1835
|
+
["rollout", "restart", f"deployment/{settings.release}"],
|
|
1836
|
+
target,
|
|
1837
|
+
capture=False,
|
|
1838
|
+
dry_run=dry_run,
|
|
1839
|
+
console=console,
|
|
1840
|
+
)
|
|
1841
|
+
if dry_run:
|
|
1842
|
+
_kube.echo_cmd(
|
|
1843
|
+
_rollout_status_cmd(settings, target, timeout), dry_run=True, console=console
|
|
1844
|
+
)
|
|
1845
|
+
console.print(f"Would restart {where} and wait up to {timeout} for the new pods.")
|
|
1846
|
+
return
|
|
1847
|
+
started = _now()
|
|
1848
|
+
state, detail = _wait_for_rollout(settings, target, timeout=timeout, console=console)
|
|
1849
|
+
_print_workload_summary(settings, target, console=console)
|
|
1850
|
+
if state == _ROLLOUT_OK:
|
|
1851
|
+
console.print(f"Restarted {where}: the new pods are ready.", style="green", markup=False)
|
|
1852
|
+
return
|
|
1853
|
+
_print_rollout_diagnostics(
|
|
1854
|
+
settings,
|
|
1855
|
+
target,
|
|
1856
|
+
since=started,
|
|
1857
|
+
headline=f"The restart of {where} did not finish within {timeout}; diagnostics:",
|
|
1858
|
+
console=console,
|
|
1859
|
+
)
|
|
1860
|
+
raise _kube.ToolFailed(
|
|
1861
|
+
f"The restarted pods of {where} did not become ready within {timeout} ({detail}).\n"
|
|
1862
|
+
" The pods that were running keep serving until new ones are ready. Fix the cause "
|
|
1863
|
+
f"(often a Secret value: `graph-agents-cli secrets status --env {env}`) and restart "
|
|
1864
|
+
f"again, or go back to the previous pods with `kubectl rollout undo "
|
|
1865
|
+
f"deployment/{settings.release} -n {target.namespace}`."
|
|
1866
|
+
)
|