cpflow 5.2.0 → 6.0.0.rc.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. checksums.yaml +4 -4
  2. data/.agents/agent-workflow.yml +26 -2
  3. data/.agents/bin/README.md +16 -4
  4. data/.agents/bin/docs +1 -1
  5. data/.agents/bin/lint +1 -1
  6. data/.agents/bin/setup +1 -1
  7. data/.agents/bin/test +1 -1
  8. data/.agents/bin/validate +2 -1
  9. data/.coderabbit.yaml +13 -0
  10. data/.github/actions/cpflow-delete-control-plane-app/action.yml +2 -1
  11. data/.github/actions/cpflow-setup-environment/action.yml +9 -7
  12. data/.github/actions/cpflow-wait-for-health/action.yml +87 -15
  13. data/.github/dependabot.yml +10 -0
  14. data/.github/pull_request_template.md +18 -0
  15. data/.github/workflows/check_cpln_links.yml +6 -1
  16. data/.github/workflows/claude-code-review.yml +5 -2
  17. data/.github/workflows/claude.yml +97 -4
  18. data/.github/workflows/coderabbit-lifecycle-review.yml +29 -0
  19. data/.github/workflows/command_docs.yml +7 -2
  20. data/.github/workflows/cpflow-cleanup-stale-review-apps.yml +12 -5
  21. data/.github/workflows/cpflow-delete-review-app.yml +638 -43
  22. data/.github/workflows/cpflow-deploy-review-app.yml +665 -36
  23. data/.github/workflows/cpflow-deploy-staging.yml +16 -10
  24. data/.github/workflows/cpflow-help-command.yml +3 -3
  25. data/.github/workflows/cpflow-promote-staging-to-production.yml +7 -7
  26. data/.github/workflows/cpflow-review-app-help.yml +6 -14
  27. data/.github/workflows/rspec-shared.yml +34 -26
  28. data/.github/workflows/rspec-specific.yml +10 -2
  29. data/.github/workflows/rspec.yml +71 -4
  30. data/.github/workflows/rubocop.yml +9 -2
  31. data/.github/workflows/trigger-docs-site.yml +2 -0
  32. data/.rubocop.yml +6 -1
  33. data/AGENTS.md +6 -49
  34. data/CHANGELOG.md +56 -1
  35. data/CONTRIBUTING.md +48 -3
  36. data/Gemfile.lock +1 -1
  37. data/README.md +6 -1
  38. data/cpflow.gemspec +3 -15
  39. data/docs/ai-github-flow-prompt.md +2 -2
  40. data/docs/ci-automation.md +216 -99
  41. data/docs/commands.md +36 -9
  42. data/docs/rds-private-networking.md +12 -12
  43. data/docs/releasing.md +15 -4
  44. data/docs/secrets-and-env-values.md +8 -0
  45. data/docs/tips.md +18 -0
  46. data/examples/controlplane.yml +5 -0
  47. data/lib/command/ai_github_flow_prompt.rb +1 -1
  48. data/lib/command/apply_template.rb +104 -2
  49. data/lib/command/base.rb +52 -3
  50. data/lib/command/cleanup_stale_apps.rb +28 -4
  51. data/lib/command/copy_image_from_upstream.rb +3 -0
  52. data/lib/command/deploy_image.rb +54 -3
  53. data/lib/command/generate_github_actions.rb +36 -12
  54. data/lib/command/run.rb +357 -44
  55. data/lib/command/setup_app.rb +10 -5
  56. data/lib/command/update_github_actions.rb +7 -6
  57. data/lib/core/controlplane.rb +179 -79
  58. data/lib/core/controlplane_api.rb +8 -0
  59. data/lib/core/controlplane_api_direct.rb +257 -63
  60. data/lib/core/shell.rb +39 -7
  61. data/lib/core/timed_command.rb +111 -0
  62. data/lib/cpflow/version.rb +1 -1
  63. data/lib/cpflow.rb +1 -1
  64. data/lib/github_flow_templates/.github/cpflow-help.md +24 -10
  65. data/lib/github_flow_templates/.github/workflows/cpflow-delete-review-app.yml +10 -0
  66. data/lib/github_flow_templates/.github/workflows/cpflow-deploy-review-app.yml +9 -0
  67. data/lib/github_flow_templates/.github/workflows/cpflow-promote-staging-to-production.yml +7 -7
  68. data/lib/github_flow_templates/bin/test-cpflow-github-flow +63 -3
  69. data/lib/patches/hash.rb +2 -2
  70. data/rakelib/create_release.rake +14 -14
  71. data/script/check_shell_scripts +66 -0
  72. metadata +12 -18
@@ -6,13 +6,14 @@ path — not the public internet.
6
6
 
7
7
  This guide covers the recommended setup: **CPLN Cloud Wormhole via an Agent**.
8
8
 
9
- > **Sourcing note — verify field casing before you apply.** Field names, schema, and limits in this guide
10
- > are sourced from the public Control Plane documentation at <https://shakadocs.controlplane.com> as of
11
- > May 2026 and have **not** been end-to-end verified against a live org. This matters because a casing
12
- > mismatch **fails silently**: `cpln apply` accepts the file, ignores the unrecognized field, and the
13
- > workload then can't reach the database — with no error pointing back to the YAML. Before applying in
14
- > production, diff your edited files against a fresh `cpln identity get <name> -o yaml-slim` (and
15
- > `cpln agent get <name> -o yaml`) export to confirm the exact field names, and consult
9
+ > **Field casing verified.** Sanitized `cpln identity get <name> -o yaml-slim` exports from live orgs
10
+ > confirmed the `networkResources` nesting and the common `name`, `agentLink`, `FQDN`, and `ports` field casing.
11
+ > The current Control Plane [Identity reference](https://shakadocs.controlplane.com/reference/identity) and
12
+ > [Identity API](https://shakadocs.controlplane.com/api-reference/identity/get-an-identity-by-gvc-and-name)
13
+ > confirm the alternative `IPs` form and optional `resolverIP` spelling; neither appeared in the inspected
14
+ > live resources. Field names are case-sensitive: `cpln apply` can accept an unrecognized field while omitting
15
+ > the intended route, without pointing back to the identity YAML. Before applying in production, check the new
16
+ > block against these casing sources, diff your edited file against a fresh identity export, and consult
16
17
  > `cpln <command> --help` for the latest CLI flags.
17
18
 
18
19
  ## Why private networking
@@ -234,11 +235,10 @@ cpln identity get my-app-production-identity \
234
235
  Edit `identity-db.yaml` and **add** the `networkResources` block — keep every other apply-safe field
235
236
  exactly as exported:
236
237
 
237
- > ⚠️ **Verify field casing against your org before applying.** The field names below (`FQDN`, `IPs`,
238
- > `agentLink`, `resolverIP`) are sourced from public CPLN docs. If the live API uses different casing,
239
- > `cpln apply` will accept the file but silently ignore the resource — workloads will hit
240
- > `could not translate host name` with no obvious link to the identity YAML. Diff your edited file
241
- > against the original `yaml-slim` export from `cpln identity get` to confirm.
238
+ > **Apply-safe editing:** start from a fresh `yaml-slim` export and diff your edited file against the
239
+ > original before applying. This helps detect unintended changes to exported fields, but it does not validate
240
+ > the newly added `networkResources` block. Check every key in that block, including `agentLink`, `FQDN`/`IPs`,
241
+ > `resolverIP`, and `ports`, against the verified casing sources above.
242
242
 
243
243
  ```yaml
244
244
  # identity-db.yaml (after edit — abbreviated, your file will have more fields)
data/docs/releasing.md CHANGED
@@ -23,6 +23,11 @@ Always update `CHANGELOG.md` before running the release task.
23
23
  - Fixes, improvements, deprecations, removals, or security updates: patch
24
24
  4. Merge the changelog PR before releasing.
25
25
 
26
+ A minimum Ruby version change is a breaking change. Merge the gemspec floor,
27
+ CI coverage for each supported Ruby minor, and an `Unreleased` breaking-change
28
+ entry together. Keep `Cpflow::VERSION` unchanged in that implementation PR.
29
+ The release task applies the required major version when the change ships.
30
+
26
31
  If updating the changelog manually, move the relevant `Unreleased` entries into
27
32
  a versioned header:
28
33
 
@@ -49,9 +54,13 @@ With no arguments, `rake release`:
49
54
 
50
55
  1. Reads the first versioned `CHANGELOG.md` header, such as `## [4.2.0]`.
51
56
  2. Uses that version when it is newer than the current gem version.
52
- 3. Uses the current version if the changelog version matches the gem version
53
- but has not been tagged yet.
54
- 4. Falls back to a patch bump if no new changelog version is found.
57
+ 3. Keeps using that version when it matches the current gem version, including
58
+ on a retry after a release attempt created the version commit or tag.
59
+ 4. Lets the existing-tag policy stop an incomplete retry instead of silently
60
+ selecting a different version.
61
+ 5. Falls back to a patch bump only from a stable current version when the
62
+ changelog does not name the current or a newer version. Prereleases require
63
+ an explicit target if the changelog cannot supply one.
55
64
 
56
65
  Other supported forms:
57
66
 
@@ -152,4 +161,6 @@ bundle exec rake "sync_github_release[4.2.0]"
152
161
  ```
153
162
 
154
163
  If the tag was pushed but the gem was not published, delete or correct the tag
155
- and version commit intentionally before trying again.
164
+ and version commit intentionally before trying again. A no-argument retry keeps
165
+ the changelog version authoritative and refuses to reinterpret a prerelease as
166
+ its stable version.
@@ -97,6 +97,14 @@ to every configured shared policy. `cpflow deploy-image` repairs missing shared
97
97
  updated, which helps existing review apps recover after the config is added. `cpflow delete` and `cpflow cleanup-stale-apps`
98
98
  remove those shared policy bindings when a review app is deleted.
99
99
 
100
+ The generated Postgres template ships with `the_password` as a password placeholder. When a `shared_secret_grants`
101
+ target still has that exact value in its `password` field, `cpflow setup-app` and `cpflow deploy-image` warn before
102
+ release or deployment work because review apps will fail authentication until the value is replaced. The diagnostic
103
+ does not print secret values. It makes one bounded, non-retried reveal attempt with the current Control Plane token and
104
+ skips the check without blocking setup or deployment when that token cannot reveal the shared secret. Do not validate the replacement only by connecting to PostgreSQL through `127.0.0.1`:
105
+ a loopback `trust` rule in `pg_hba.conf` can allow that connection without checking the password and produce a false
106
+ pass. Verify the credential through a non-loopback path that uses the same authentication route as the review app.
107
+
100
108
  For shared databases, keep runtime data isolated by using a per-review-app database name, schema, or tenant key. A common
101
109
  pattern is to keep the host, user, and password in the shared secret, then have `hooks.post_creation` create the
102
110
  PR-specific database/schema. Avoid a generic `hooks.pre_deletion` that drops the database: `cpflow delete` runs the
data/docs/tips.md CHANGED
@@ -18,6 +18,7 @@
18
18
  13. [Minimizing Non-Production App Costs](#minimizing-non-production-app-costs)
19
19
  - [Share One Control Plane Postgres for Staging and Review Apps](#share-one-control-plane-postgres-for-staging-and-review-apps)
20
20
  - [Enable Capacity AI for Demo and Starter Staging Apps](#enable-capacity-ai-for-demo-and-starter-staging-apps)
21
+ - [Use an Always-Available Landing Page for a Serverless App](#use-an-always-available-landing-page-for-a-serverless-app)
21
22
  - [Delete or Pause Abandoned Apps with `cleanup-stale-apps`](#delete-or-pause-abandoned-apps-with-cleanup-stale-apps)
22
23
  - [Pause and Resume with `ps:stop` / `ps:start`](#pause-and-resume-with-psstop--psstart)
23
24
  14. [Right-Sizing Non-Production Workloads](#right-sizing-non-production-workloads)
@@ -616,6 +617,23 @@ migration and can interrupt traffic.
616
617
  > **Warning:** Treat a `standard` to `serverless` conversion as an operational migration because deleting a running
617
618
  > workload can interrupt traffic.
618
619
 
620
+ ### Use an Always-Available Landing Page for a Serverless App
621
+
622
+ For a public demo, review app, or staging app where the first request's cold start would be a poor first impression,
623
+ serve a lightweight landing page independently of the app that scales to zero. Route visitors to that landing page
624
+ first, then have a button on that page request the separate serverless app. The page can immediately explain that the
625
+ app is starting while the serverless workload wakes; it does not remove the cold start, but it keeps that wait out of
626
+ the initial page render.
627
+
628
+ ```text
629
+ always-available landing page -> Open app request -> serverless app (minScale: 0) -> cold start -> app response
630
+ ```
631
+
632
+ The landing page, its hosting, and any DNS, proxy, rewrite, or redirect rules are application infrastructure you
633
+ choose and operate. `cpflow` does not create that routing infrastructure. Keep the app as a separate serverless
634
+ workload from its first deployment (or perform the planned delete/recreate migration above); it cannot convert an
635
+ existing standard workload in place. The wake-up path also requires the HTTP autoscaling configuration shown above.
636
+
619
637
  > **Note:** if you later suspend the app with `cpflow ps:stop`, Control Plane will not auto-wake it on the next
620
638
  > request. Run `cpflow ps:start` explicitly first. See
621
639
  > [Pause and Resume](#pause-and-resume-with-psstop--psstart).
@@ -104,6 +104,11 @@ aliases:
104
104
  # If not specified, defaults to 21600 (6 hours).
105
105
  runner_job_timeout: 1000
106
106
 
107
+ # Sets how long `cpflow run` waits for Control Plane to reconcile a cron job status
108
+ # after the non-interactive command prints its completion marker.
109
+ # If not specified, defaults to 1200 (20 minutes).
110
+ runner_job_status_reconciliation_timeout: 1200
111
+
107
112
  # Apps with a deployed image created before this amount of days will be listed for deletion
108
113
  # when running the command `cpflow cleanup-stale-apps`.
109
114
  stale_app_image_deployed_days: 5
@@ -32,7 +32,7 @@ module Command
32
32
  <<~PROMPT
33
33
  Set up Control Plane GitHub Flow for this repo. First make sure the `cpflow` CLI is available: use the repo's existing `bundle exec cpflow` if present, otherwise install the published `cpflow` Ruby gem with `gem install cpflow`; if neither is possible, stop and report that blocker. Use the same `cpflow` invocation for the rest of the rollout. Start with `cpflow github-flow-readiness` and stop on any reported blockers. The repo must be deployable from a clean clone: published package versions, complete runtime scaffold, and a production Dockerfile that can build the app. If any package version is unpublished, inaccessible from CI, or requires credentials that are not already modeled in the repo or GitHub settings, stop and report the blocker instead of generating workflow files. If the repo is a legacy sample pinned to an obsolete Ruby or Bundler toolchain, if it does not even have a production Dockerfile yet, or if it is a monorepo without an already-decided single app boundary for this flow, stop and report that as a prerequisite instead of forcing the rollout.
34
34
 
35
- If `.controlplane/` is missing, run `cpflow generate`. Treat the generated app names as the repo-name default (`#{inferred_app_prefix}`) and rename them only if the project needs a different prefix. Then run `cpflow generate-github-actions` (or `cpflow generate-github-actions --staging-branch BRANCH` when staging should deploy from a branch other than `main`/`master`), keep review apps opt-in via `+review-app-deploy`, make sure any `STAGING_APP_BRANCH` repository variable is also present in the generated staging workflow's `on.push.branches` filter, and list the GitHub secrets and variables that must be configured. Do not hand-edit duplicated upstream refs into the generated wrappers: the only downstream Control Plane Flow pin should be the reusable workflow `uses: ...@vX.Y.Z` value generated from the installed `cpflow` gem version, and upstream workflows load their matching shared actions automatically. When bumping the `cpflow` gem in a downstream repo, run `cpflow update-github-actions` (or `bundle exec cpflow update-github-actions`) and validate with `bin/test-cpflow-github-flow` in the same PR so the checked-in wrappers move to the matching release tag. Keep the normal generated review-app setup simple: review apps require only `CPLN_TOKEN_STAGING` when the generated review app config can be inferred. For public demos, starter staging apps, and long-lived review apps, keep the app workload `type: standard` with one warm replica, set its autoscaling metric to `disabled`, and enable `capacityAI: true` so Control Plane can right-size CPU and memory allocation at that fixed replica count. Shared Postgres and other stateful workloads are the usual exceptions and should stay manually sized; Capacity AI is for supported stateless app/service workloads. If true idle scale-to-zero is explicitly required, create a separate `serverless` workload before first deploy or plan a delete/recreate migration because Control Plane will not change an existing `standard` workload to `serverless` in place. For shared review-app resources such as one staging database, use `shared_secret_grants` and `{{SHARED_SECRET_DATABASE}}` placeholders instead of hardcoding the base app secret name; this keeps review-app policy binding and cleanup automatic while avoiding per-PR database cost. Document the one-time Control Plane bootstrap command for persistent staging and production apps with `cpflow setup-app --skip-post-creation-hook`; for existing apps or later template updates, document `cpflow apply-template` and the need for the app identity to have `reveal` on the app secret policy. Do not imply the staging deploy or promotion workflows create those persistent GVCs. For production promotion, document a protected `production` GitHub Environment with required reviewers, prevent self-review, and `CPLN_TOKEN_PRODUCTION` stored as an environment secret, not as a repository or organization secret.
35
+ If `.controlplane/` is missing, run `cpflow generate`. Treat the generated app names as the repo-name default (`#{inferred_app_prefix}`) and rename them only if the project needs a different prefix. Then run `cpflow generate-github-actions` (or `cpflow generate-github-actions --staging-branch BRANCH` when staging should deploy from a branch other than `main`/`master`), keep review apps opt-in via `+review-app-deploy`, make sure any `STAGING_APP_BRANCH` repository variable is also present in the generated staging workflow's `on.push.branches` filter, and list the GitHub secrets and variables that must be configured. Do not hand-edit duplicated upstream refs into the generated wrappers. Keep every generated Control Plane Flow source pin synchronized at the same `vX.Y.Z`: each reusable-workflow `uses: ...@vX.Y.Z` ref, the production promotion `.cpflow` checkout ref, and its `control_plane_flow_ref` setup input. By default, `cpflow generate-github-actions` derives these values from the installed `cpflow` gem version; temporary explicit ref overrides are for unreleased testing and must keep all three values synchronized. Keep the generated `.github/actions/cpflow-*` files from that same gem unchanged; upstream reusable workflows invoke those checked-in local actions and separately check out the matching runtime source at `.cpflow`. When bumping the `cpflow` gem in a downstream repo, run `cpflow update-github-actions` (or `bundle exec cpflow update-github-actions`) and validate with `bin/test-cpflow-github-flow` in the same PR so the checked-in workflows and local actions move together. Keep the normal generated review-app setup simple: review apps require only `CPLN_TOKEN_STAGING` when the generated review app config can be inferred. For public demos, starter staging apps, and long-lived review apps, keep the app workload `type: standard` with one warm replica, set its autoscaling metric to `disabled`, and enable `capacityAI: true` so Control Plane can right-size CPU and memory allocation at that fixed replica count. Shared Postgres and other stateful workloads are the usual exceptions and should stay manually sized; Capacity AI is for supported stateless app/service workloads. If true idle scale-to-zero is explicitly required, create a separate `serverless` workload before first deploy or plan a delete/recreate migration because Control Plane will not change an existing `standard` workload to `serverless` in place. For shared review-app resources such as one staging database, use `shared_secret_grants` and `{{SHARED_SECRET_DATABASE}}` placeholders instead of hardcoding the base app secret name; this keeps review-app policy binding and cleanup automatic while avoiding per-PR database cost. Document the one-time Control Plane bootstrap command for persistent staging and production apps with `cpflow setup-app --skip-post-creation-hook`; for existing apps or later template updates, document `cpflow apply-template` and the need for the app identity to have `reveal` on the app secret policy. Do not imply the staging deploy or promotion workflows create those persistent GVCs. For production promotion, document a protected `production` GitHub Environment with required reviewers, prevent self-review, and `CPLN_TOKEN_PRODUCTION` stored as an environment secret, not as a repository or organization secret.
36
36
 
37
37
  Keep Node available in the final image if asset compilation or SSR depends on ExecJS, Yarn, `pnpm`, or npm after the main install layer. Make sure the generated Dockerfile uses a Ruby base image compatible with the app's declared Ruby requirement. Preserve repo-defined frontend build hooks: if `config/shakapacker.yml` defines a `precompile_hook`, or React on Rails enables `config.auto_load_bundle = true`, confirm the generated Dockerfile runs that codegen step before `rails assets:precompile`. If `config/database.yml` shows SQLite in production, confirm that the generated scaffold uses persistent `db` and `storage` volumes plus a release script that runs `rails db:prepare`; otherwise keep the default Postgres workload. If the public workload is not named `rails`, set `PRIMARY_WORKLOAD` or adjust the generated workflows. Inspect the Dockerfile and package sources for private GitHub dependencies or `RUN --mount=type=ssh`; if present, wire `DOCKER_BUILD_SSH_KEY`, optionally set `DOCKER_BUILD_SSH_KNOWN_HOSTS` for non-GitHub SSH hosts, and keep `DOCKER_BUILD_EXTRA_ARGS` to newline-delimited single tokens such as `--build-arg=FOO=bar`.
38
38
 
@@ -9,7 +9,8 @@ module Command
9
9
  app_option(required: true),
10
10
  location_option,
11
11
  skip_confirm_option,
12
- add_app_identity_option
12
+ add_app_identity_option,
13
+ preserve_existing_runtime_option
13
14
  ].freeze
14
15
  DESCRIPTION = "Applies application-specific configs from templates"
15
16
  LONG_DESCRIPTION = <<~DESC
@@ -17,6 +18,8 @@ module Command
17
18
  - Publishes (creates or updates) those at Control Plane infrastructure
18
19
  - Picks templates from the `.controlplane/templates` directory
19
20
  - Templates are ordinary Control Plane templates but with variable preprocessing
21
+ - Use `--preserve-existing-runtime` to retain each workload container's configured app image, even when the workload is unready, and skip existing secret resources entirely while applying other template changes
22
+ - Missing or invalid workload images use only an unambiguous app image from ready workloads; refresh fails before applying templates when no safe fallback exists
20
23
 
21
24
  **Preprocessed template variables:**
22
25
 
@@ -58,6 +61,7 @@ module Command
58
61
  @skipped_templates = []
59
62
 
60
63
  templates = @template_parser.parse(@names_to_filenames.values)
64
+ templates = preserve_existing_runtime(templates) if config.options[:preserve_existing_runtime]
61
65
  pending_templates = confirm_templates(templates)
62
66
  add_app_identity_template(pending_templates) if config.options[:add_app_identity]
63
67
  pending_templates.each do |template|
@@ -117,7 +121,7 @@ module Command
117
121
  end
118
122
 
119
123
  def confirm_workload(template) # rubocop:disable Naming/PredicateMethod
120
- workload = cp.fetch_workload(template["name"])
124
+ workload = fetch_workload(template["name"])
121
125
  return true unless workload
122
126
 
123
127
  confirmed = confirm_apply("Workload '#{template['name']}' already exists, do you want to re-create it?")
@@ -146,6 +150,104 @@ module Command
146
150
  pending_templates
147
151
  end
148
152
 
153
+ def preserve_existing_runtime(templates)
154
+ cache_existing_workloads(templates)
155
+ ready_fallback_image = unambiguous_ready_app_image
156
+
157
+ templates.filter_map do |template|
158
+ if template["kind"] == "secret" && cp.fetch_secret(template["name"])
159
+ report_skipped(template)
160
+ next
161
+ end
162
+
163
+ preserve_workload_images(template, ready_fallback_image) if template["kind"] == "workload"
164
+ template
165
+ end
166
+ end
167
+
168
+ def cache_existing_workloads(templates)
169
+ template_names = templates.filter_map { |template| template["name"] if template["kind"] == "workload" }
170
+ workloads = cp.fetch_workloads
171
+ cp.fetch_gvc! unless workloads
172
+ @existing_workloads = Array(workloads.fetch("items")).to_h do |workload|
173
+ [workload.fetch("name"), workload]
174
+ end
175
+ template_names.each { |name| @existing_workloads[name] = nil unless @existing_workloads.key?(name) }
176
+ end
177
+
178
+ def fetch_workload(name)
179
+ return @existing_workloads[name] if @existing_workloads&.key?(name)
180
+
181
+ cp.fetch_workload(name)
182
+ end
183
+
184
+ def unambiguous_ready_app_image
185
+ app_images = ready_existing_app_workloads.flat_map do |workload|
186
+ Array(workload.dig("spec", "containers")).filter_map do |container|
187
+ container["image"] if deployable_app_image?(container["image"])
188
+ end
189
+ end
190
+ image_prefix = "/org/#{config.org}/image/"
191
+ canonical_images = app_images.map { |image| image.delete_prefix(image_prefix) }
192
+ app_images.first if canonical_images.uniq.one?
193
+ end
194
+
195
+ def ready_existing_app_workloads
196
+ @existing_workloads.values.compact.filter_map do |workload|
197
+ next unless Array(workload.dig("spec", "containers")).any? do |container|
198
+ deployable_app_image?(container["image"])
199
+ end
200
+
201
+ workload = workload_with_readiness(workload)
202
+ workload if workload_ready?(workload)
203
+ end
204
+ end
205
+
206
+ def preserve_workload_images(template, ready_fallback_image) # rubocop:disable Metrics/MethodLength
207
+ existing_workload = fetch_workload(template["name"])
208
+ existing_containers = Array(existing_workload&.dig("spec", "containers"))
209
+ .to_h { |container| [container["name"], container] }
210
+
211
+ Array(template.dig("spec", "containers")).each do |container|
212
+ next unless app_image?(container["image"])
213
+
214
+ existing_image = existing_containers.dig(container["name"], "image")
215
+ preserved_image = preserved_image(existing_image, ready_fallback_image)
216
+ unless preserved_image
217
+ raise "Cannot safely refresh app image for workload '#{template['name']}' " \
218
+ "without a valid existing workload image or an unambiguous ready fallback image."
219
+ end
220
+
221
+ container["image"] = preserved_image
222
+ end
223
+ end
224
+
225
+ def preserved_image(existing_image, ready_fallback_image)
226
+ return existing_image if deployable_app_image?(existing_image)
227
+
228
+ ready_fallback_image
229
+ end
230
+
231
+ def workload_ready?(workload)
232
+ workload&.dig("status", "readyLatest") == true
233
+ end
234
+
235
+ def workload_with_readiness(workload)
236
+ return workload unless workload.dig("status", "readyLatest").nil?
237
+
238
+ cp.fetch_workload_with_status(workload.fetch("name")) || workload
239
+ end
240
+
241
+ def app_image?(image)
242
+ image.to_s.match?(
243
+ %r{\A(?:/org/#{Regexp.escape(config.org)}/image/)?#{Regexp.escape(config.app)}[:@]}
244
+ )
245
+ end
246
+
247
+ def deployable_app_image?(image)
248
+ app_image?(image) && !image.to_s.end_with?(Controlplane::NO_IMAGE_AVAILABLE)
249
+ end
250
+
149
251
  def add_app_identity_template(templates)
150
252
  app_template_index = templates.index { |template| template["name"] == config.app }
151
253
  app_identity_template_index = templates.index { |template| template["name"] == config.identity }
data/lib/command/base.rb CHANGED
@@ -1,5 +1,6 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "shellwords"
3
4
  require_relative "../core/helpers"
4
5
 
5
6
  module Command
@@ -11,6 +12,7 @@ module Command
11
12
  VALIDATIONS_WITHOUT_ADDITIONAL_OPTIONS = %w[config].freeze
12
13
  VALIDATIONS_WITH_ADDITIONAL_OPTIONS = %w[templates].freeze
13
14
  ALL_VALIDATIONS = VALIDATIONS_WITHOUT_ADDITIONAL_OPTIONS + VALIDATIONS_WITH_ADDITIONAL_OPTIONS
15
+ GENERATED_POSTGRES_PASSWORD_PLACEHOLDER = "the_password"
14
16
 
15
17
  # Used to call the command (`cpflow SUBCOMMAND_NAME NAME`)
16
18
  SUBCOMMAND_NAME = nil
@@ -459,6 +461,28 @@ module Command
459
461
  }
460
462
  end
461
463
 
464
+ def self.refresh_templates_option(required: false)
465
+ {
466
+ name: :refresh_templates,
467
+ params: {
468
+ desc: "Refreshes configured templates for an existing app without running post-creation hooks",
469
+ type: :boolean,
470
+ required: required
471
+ }
472
+ }
473
+ end
474
+
475
+ def self.preserve_existing_runtime_option(required: false)
476
+ {
477
+ name: :preserve_existing_runtime,
478
+ params: {
479
+ desc: "Preserves deployed app images and skips existing secret resources entirely while applying templates",
480
+ type: :boolean,
481
+ required: required
482
+ }
483
+ }
484
+ end
485
+
462
486
  def self.skip_pre_deletion_hook_option(required: false)
463
487
  {
464
488
  name: :skip_pre_deletion_hook,
@@ -528,10 +552,12 @@ module Command
528
552
  end
529
553
  end
530
554
 
531
- # NOTE: use simplified variant atm, as shelljoin do different escaping
532
- # TODO: most probably need better logic for escaping various quotes
533
555
  def args_join(args)
534
- args.join(" ")
556
+ # A single CLI argument is an intentional shell program (for example, an env assignment or pipeline).
557
+ # Multiple CLI arguments are argv elements and must be escaped before entering the remote shell script.
558
+ return args.first if args.size == 1
559
+
560
+ Shellwords.join(args)
535
561
  end
536
562
 
537
563
  def progress
@@ -619,9 +645,32 @@ module Command
619
645
  raise shared_secret_policy_missing_message(grant) if policy.nil?
620
646
 
621
647
  ensure_shared_secret_policy_targets_secret!(grant, policy)
648
+ warn_if_shared_secret_uses_generated_password_placeholder(grant)
622
649
  policy
623
650
  end
624
651
 
652
+ def warn_if_shared_secret_uses_generated_password_placeholder(grant)
653
+ secret_name = grant.fetch(:secret_name)
654
+ secret = cp.reveal_secret(secret_name)
655
+ return unless secret&.dig("data", "password") == GENERATED_POSTGRES_PASSWORD_PLACEHOLDER
656
+
657
+ Shell.warn(
658
+ "Shared secret grant '#{grant.fetch(:name)}' targets secret '#{secret_name}', whose password " \
659
+ "is still the generated placeholder. Review apps will fail authentication until it is replaced."
660
+ )
661
+ # This is a best-effort warning. API, transport, and response-shape failures
662
+ # must not turn an optional diagnostic into a deployment blocker.
663
+ rescue StandardError
664
+ debug_shared_secret_placeholder_check_failure(secret_name)
665
+ end
666
+
667
+ def debug_shared_secret_placeholder_check_failure(secret_name)
668
+ Shell.debug(
669
+ "WARN",
670
+ "Could not inspect shared secret '#{secret_name}'; continuing without the optional placeholder diagnostic."
671
+ )
672
+ end
673
+
625
674
  def bind_shared_secret_policy_grant(grant, policy)
626
675
  policy_name = grant.fetch(:policy_name)
627
676
  return if identity_bound_to_policy_with_reveal?(policy)
@@ -1,7 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Command
4
- class CleanupStaleApps < Base
4
+ class CleanupStaleApps < Base # rubocop:disable Metrics/ClassLength
5
5
  CLEANUP_MODE_OPTION = {
6
6
  name: :mode,
7
7
  params: {
@@ -24,7 +24,7 @@ module Command
24
24
  LONG_DESCRIPTION = <<~DESC
25
25
  - Acts on stale apps based on the creation date of the latest image, or the GVC if no images exist
26
26
  - With `--mode=delete` (default): deletes the whole app (GVC with all workloads, all volumesets and all images), and unbinds the app from the secrets policy and any configured `shared_secret_grants` policies as long as both the identity and each policy exist (and are bound)
27
- - With `--mode=stop`: suspends all workloads via `cpflow ps:stop` — no GVC, volumeset, or image is removed; resume with `cpflow ps:start`
27
+ - With `--mode=stop`: suspends configured workloads that exist in the live GVC through Control Plane — no GVC, volumeset, or image is removed; resume with `cpflow ps:start`
28
28
  - `--mode=stop` only suspends workloads listed in `app_workloads` + `additional_workloads`; workloads present in the live GVC but missing from the config are skipped silently
29
29
  - `--mode=stop` returns once each workload is marked suspended; it does not wait for the workload to reach a not-ready state
30
30
  - Specify the amount of days after an app should be considered stale through `stale_app_image_deployed_days` in the `.controlplane/controlplane.yml` file
@@ -101,14 +101,38 @@ module Command
101
101
  def process_app(app)
102
102
  if mode == "stop"
103
103
  progress.puts("Stopping app '#{app}'")
104
- run_cpflow_command("ps:stop", "-a", app)
104
+ stop_configured_live_workloads(app)
105
105
  else
106
106
  run_cpflow_command("delete", "-a", app, "--yes")
107
107
  end
108
108
  end
109
109
 
110
+ def stop_configured_live_workloads(app)
111
+ app_config = config.find_app_config(app)
112
+ raise "Can't find config for stale app '#{app}'." unless app_config
113
+
114
+ configured_workloads = required_app_option(app_config, app, :app_workloads) +
115
+ required_app_option(app_config, app, :additional_workloads)
116
+ live_workloads = (cp.fetch_workloads(app)&.fetch("items", []) || []).map { |workload| workload.fetch("name") }
117
+
118
+ (configured_workloads & live_workloads).each do |workload|
119
+ step("Stopping workload '#{workload}'") do
120
+ cp.set_workload_suspend(workload, true, app, missing_ok: true)
121
+ end
122
+ end
123
+ end
124
+
125
+ def required_app_option(app_config, app, option)
126
+ raise "Can't find option '#{option}' for app '#{app}' in 'controlplane.yml'." unless app_config.key?(option)
127
+
128
+ value = app_config.fetch(option)
129
+ raise "Option '#{option}' for app '#{app}' in 'controlplane.yml' must be an array." unless value.is_a?(Array)
130
+
131
+ value
132
+ end
133
+
110
134
  def action_description
111
- mode == "stop" ? "suspend all workloads in" : "delete"
135
+ mode == "stop" ? "suspend configured workloads in" : "delete"
112
136
  end
113
137
 
114
138
  def mode
@@ -80,6 +80,9 @@ module Command
80
80
  upstream_image = cp.latest_image(@upstream, @upstream_org) if !upstream_image || upstream_image == "latest"
81
81
  @commit = cp.extract_image_commit(upstream_image)
82
82
  @upstream_image_url = "#{@upstream_org}.registry.cpln.io/#{upstream_image}"
83
+ rescue ControlplaneApiDirect::ForbiddenError => e
84
+ Shell.write_to_tmp_stderr(e.message)
85
+ false
83
86
  end
84
87
  end
85
88
 
@@ -4,6 +4,12 @@ require "resolv"
4
4
 
5
5
  module Command
6
6
  class DeployImage < Base # rubocop:disable Metrics/ClassLength
7
+ WORKLOAD_IMAGE_UPDATE_MAX_ATTEMPTS = 30
8
+ # Preserve the general failure cap; only optimistic-concurrency conflicts get the longer convergence window.
9
+ WORKLOAD_IMAGE_UPDATE_CONFLICT_MAX_ATTEMPTS = 120
10
+ WORKLOAD_IMAGE_UPDATE_CONFLICT_PATTERN =
11
+ /(?:\b409\b|(?:\A|\n)[ \t]*(?:error:[ \t]*)?conflict[ \t]*(?:\r?\n|\z))/i
12
+
7
13
  NAME = "deploy-image"
8
14
  OPTIONS = [
9
15
  app_option(required: true),
@@ -107,17 +113,26 @@ module Command
107
113
  @requested_workload_names ||= Array(config.options[:workload]).map(&:to_s).uniq
108
114
  end
109
115
 
116
+ def app_image?(image)
117
+ image.to_s.match?(
118
+ %r{\A(?:/org/#{Regexp.escape(config.org)}/image/)?#{Regexp.escape(config.app)}[:@]}
119
+ )
120
+ end
121
+
110
122
  def deploy_image_to_workloads(image, workload_data_by_name) # rubocop:disable Metrics/MethodLength
111
123
  deployed_endpoints = {}
112
124
 
113
125
  workload_data_by_name.each do |workload, workload_data|
114
126
  workload_data.dig("spec", "containers").each do |container|
115
- next unless container["image"].match?(%r{^/org/#{config.org}/image/#{config.app}[:@]})
127
+ next unless app_image?(container["image"])
116
128
 
117
129
  container_name = container["name"]
118
130
  step("Deploying image '#{image}' for workload '#{workload}'") do
119
- cp.workload_set_image_ref(workload, container: container_name, image: image)
131
+ update_workload_image_ref(workload, container_name, image)
132
+
120
133
  deployed_endpoints[workload] = endpoint_for_workload(workload_data)
134
+ # A missing public endpoint is valid; the image update still completed successfully.
135
+ true
121
136
  end
122
137
  # Deploy the first matching app-image container per workload; CPLN workloads
123
138
  # are expected to have a single container that runs the app image.
@@ -128,10 +143,40 @@ module Command
128
143
  deployed_endpoints
129
144
  end
130
145
 
146
+ def update_workload_image_ref(workload, container, image)
147
+ attempts = 0
148
+
149
+ loop do
150
+ attempts += 1
151
+ result = cp.workload_set_image_ref(workload, container: container, image: image)
152
+ return true if result.fetch(:success)
153
+
154
+ output = result.fetch(:output).to_s
155
+ raise workload_image_update_error(output) if attempts >= workload_image_update_limit(output)
156
+
157
+ wait_before_workload_image_update_retry
158
+ end
159
+ end
160
+
161
+ def workload_image_update_limit(output)
162
+ return WORKLOAD_IMAGE_UPDATE_CONFLICT_MAX_ATTEMPTS if output.match?(WORKLOAD_IMAGE_UPDATE_CONFLICT_PATTERN)
163
+
164
+ WORKLOAD_IMAGE_UPDATE_MAX_ATTEMPTS
165
+ end
166
+
167
+ def workload_image_update_error(output)
168
+ output.strip.empty? ? "Command exited with non-zero status." : output
169
+ end
170
+
171
+ def wait_before_workload_image_update_retry
172
+ progress.print(".")
173
+ Kernel.sleep(1)
174
+ end
175
+
131
176
  def print_deployed_endpoints(deployed_endpoints)
132
177
  progress.puts("\nDeployed endpoints:")
133
178
  deployed_endpoints.each do |workload, endpoint|
134
- progress.puts(" - #{workload}: #{endpoint}")
179
+ progress.puts(" - #{workload}: #{endpoint || '(no public endpoint)'}")
135
180
  end
136
181
  end
137
182
 
@@ -172,9 +217,15 @@ module Command
172
217
 
173
218
  def endpoint_for_workload(workload_data)
174
219
  endpoint = workload_data.dig("status", "endpoint")
220
+ return fallback_endpoint_for_workload(workload_data) unless endpoint
221
+
175
222
  Resolv.getaddress(endpoint.split("/").last)
176
223
  endpoint
177
224
  rescue Resolv::ResolvError
225
+ fallback_endpoint_for_workload(workload_data)
226
+ end
227
+
228
+ def fallback_endpoint_for_workload(workload_data)
178
229
  deployments = cp.fetch_workload_deployments(workload_data["name"])
179
230
  deployments.dig("items", 0, "status", "endpoint")
180
231
  end