log10x-mcp 1.30.9 → 1.30.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (126) hide show
  1. package/build/lib/advisor/compaction-support.d.ts +5 -6
  2. package/build/lib/advisor/compaction-support.js +9 -10
  3. package/build/lib/advisor/compaction-support.js.map +1 -1
  4. package/build/lib/advisor/reporter-forwarders.d.ts +5 -4
  5. package/build/lib/advisor/reporter-forwarders.js.map +1 -1
  6. package/build/lib/advisor/reporter.d.ts +9 -8
  7. package/build/lib/advisor/reporter.js +8 -7
  8. package/build/lib/advisor/reporter.js.map +1 -1
  9. package/build/lib/advisor/retriever.d.ts +131 -8
  10. package/build/lib/advisor/retriever.js +609 -124
  11. package/build/lib/advisor/retriever.js.map +1 -1
  12. package/build/lib/cost.d.ts +154 -15
  13. package/build/lib/cost.js +192 -18
  14. package/build/lib/cost.js.map +1 -1
  15. package/build/lib/offload-recipes.js +1 -1
  16. package/build/lib/plan-solver.d.ts +15 -0
  17. package/build/lib/plan-solver.js +18 -0
  18. package/build/lib/plan-solver.js.map +1 -1
  19. package/build/lib/poc-envelope-v2.d.ts +9 -1
  20. package/build/lib/poc-envelope-v2.js +72 -16
  21. package/build/lib/poc-envelope-v2.js.map +1 -1
  22. package/build/lib/poc-report-renderer.d.ts +1 -1
  23. package/build/lib/poc-report-renderer.js +43 -8
  24. package/build/lib/poc-report-renderer.js.map +1 -1
  25. package/build/lib/rate-resolution.js +11 -4
  26. package/build/lib/rate-resolution.js.map +1 -1
  27. package/build/lib/server-instructions.js +5 -4
  28. package/build/lib/server-instructions.js.map +1 -1
  29. package/build/lib/siem/clickhouse.js +6 -0
  30. package/build/lib/siem/clickhouse.js.map +1 -1
  31. package/build/lib/siem/pricing.js +7 -3
  32. package/build/lib/siem/pricing.js.map +1 -1
  33. package/build/product-kb/docs/api/github.md +0 -3
  34. package/build/product-kb/docs/apps/compiler/run.md +62 -38
  35. package/build/product-kb/docs/apps/mcp/run.md +8 -3
  36. package/build/product-kb/docs/apps/receiver/compact/elasticsearch.md +2 -2
  37. package/build/product-kb/docs/apps/receiver/compact/index.md +6 -6
  38. package/build/product-kb/docs/apps/receiver/compact/splunk.md +2 -2
  39. package/build/product-kb/docs/apps/receiver/faq.md +40 -65
  40. package/build/product-kb/docs/apps/receiver/index.md +2 -2
  41. package/build/product-kb/docs/apps/receiver/run.md +288 -70
  42. package/build/product-kb/docs/apps/reporter/faq.md +3 -3
  43. package/build/product-kb/docs/apps/reporter/run.md +255 -47
  44. package/build/product-kb/docs/apps/retriever/deploy/azure.md +7 -1
  45. package/build/product-kb/docs/apps/retriever/deploy/index.md +11 -5
  46. package/build/product-kb/docs/apps/retriever/faq.md +8 -8
  47. package/build/product-kb/docs/apps/retriever/run.md +268 -48
  48. package/build/product-kb/docs/compile/bootstrap/index.md +12 -5
  49. package/build/product-kb/docs/compile/link/index.md +1 -1
  50. package/build/product-kb/docs/compile/monitor/index.md +1 -1
  51. package/build/product-kb/docs/compile/scan/index.md +1 -1
  52. package/build/product-kb/docs/compile/scanner/executable/index.md +2 -2
  53. package/build/product-kb/docs/doc/bootstrap/index.md +28 -14
  54. package/build/product-kb/docs/doc/monitor/index.md +1 -1
  55. package/build/product-kb/docs/engine/launcher/lambda.md +1 -1
  56. package/build/product-kb/docs/faq/apps/receiver.md +40 -65
  57. package/build/product-kb/docs/faq/apps/reporter.md +3 -3
  58. package/build/product-kb/docs/faq/apps/retriever.md +5 -5
  59. package/build/product-kb/docs/faq/general.md +8 -8
  60. package/build/product-kb/docs/faq/pricing/overview.md +1 -1
  61. package/build/product-kb/docs/faq/stacks/datadog/comparisons.md +1 -1
  62. package/build/product-kb/docs/faq/stacks/datadog/optimization.md +1 -1
  63. package/build/product-kb/docs/faq/stacks/elasticsearch/comparisons.md +3 -3
  64. package/build/product-kb/docs/faq/stacks/elasticsearch/index.md +1 -1
  65. package/build/product-kb/docs/faq/stacks/index.md +1 -1
  66. package/build/product-kb/docs/faq/stacks/splunk/comparisons.md +4 -4
  67. package/build/product-kb/docs/faq/stacks/splunk/index.md +3 -3
  68. package/build/product-kb/docs/faq/stacks/splunk/optimization.md +1 -1
  69. package/build/product-kb/docs/install/datadog.md +99 -0
  70. package/build/product-kb/docs/run/bootstrap/index.md +42 -21
  71. package/build/product-kb/docs/run/index.md +1 -1
  72. package/build/product-kb/docs/run/initialize/message/index.md +29 -8
  73. package/build/product-kb/docs/run/input/analyzer/cloudwatchLogs/index.md +1 -1
  74. package/build/product-kb/docs/run/input/analyzer/datadogLogs/index.md +1 -1
  75. package/build/product-kb/docs/run/input/analyzer/elasticsearch/index.md +1 -1
  76. package/build/product-kb/docs/run/input/analyzer/index.md +1 -1
  77. package/build/product-kb/docs/run/input/analyzer/splunk/index.md +1 -1
  78. package/build/product-kb/docs/run/input/forwarder/datadogAgent/index.md +19 -7
  79. package/build/product-kb/docs/run/input/forwarder/filebeat/index.md +16 -2
  80. package/build/product-kb/docs/run/input/forwarder/fluentbit/index.md +18 -7
  81. package/build/product-kb/docs/run/input/forwarder/fluentd/index.md +18 -7
  82. package/build/product-kb/docs/run/input/forwarder/logstash/index.md +15 -5
  83. package/build/product-kb/docs/run/input/forwarder/otel-collector/index.md +43 -9
  84. package/build/product-kb/docs/run/input/forwarder/vector/index.md +18 -7
  85. package/build/product-kb/docs/run/input/objectStorage/delete/index.md +16 -6
  86. package/build/product-kb/docs/run/input/objectStorage/index.md +2 -2
  87. package/build/product-kb/docs/run/input/objectStorage/input/index.md +14 -6
  88. package/build/product-kb/docs/run/input/objectStorage/query/index.md +11 -5
  89. package/build/product-kb/docs/run/input/stream/index.md +1 -1
  90. package/build/product-kb/docs/run/monitor/index.md +1 -1
  91. package/build/product-kb/docs/run/output/event/dev/index.md +2 -2
  92. package/build/product-kb/docs/run/output/event/outputStream/index.md +1 -1
  93. package/build/product-kb/docs/run/output/metric/index.md +8 -8
  94. package/build/product-kb/docs/run/output/metric/log10x/index.md +291 -1
  95. package/build/product-kb/docs/run/output/metric/prometheus/remote-write/index.md +1 -1
  96. package/build/product-kb/docs/run/output/stream/index.md +2 -2
  97. package/build/product-kb/docs/run/receive/compact/index.md +7 -7
  98. package/build/product-kb/docs/run/receive/rate/index.md +67 -33
  99. package/build/product-kb/docs/run/transform/index.md +1 -1
  100. package/build/product-kb/docs/summary.md +1 -1
  101. package/build/tools/advise-install.d.ts +4 -4
  102. package/build/tools/advise-install.js +73 -15
  103. package/build/tools/advise-install.js.map +1 -1
  104. package/build/tools/advise-retriever.d.ts +11 -0
  105. package/build/tools/advise-retriever.js +185 -30
  106. package/build/tools/advise-retriever.js.map +1 -1
  107. package/build/tools/baseline.d.ts +15 -0
  108. package/build/tools/baseline.js +20 -3
  109. package/build/tools/baseline.js.map +1 -1
  110. package/build/tools/configure-engine.d.ts +9 -4
  111. package/build/tools/configure-engine.js +9 -4
  112. package/build/tools/configure-engine.js.map +1 -1
  113. package/build/tools/cost-options.js +24 -10
  114. package/build/tools/cost-options.js.map +1 -1
  115. package/build/tools/estimate-savings.d.ts +57 -1
  116. package/build/tools/estimate-savings.js +103 -8
  117. package/build/tools/estimate-savings.js.map +1 -1
  118. package/build/tools/explain-mode.js +11 -9
  119. package/build/tools/explain-mode.js.map +1 -1
  120. package/build/tools/measure-compaction.js +38 -3
  121. package/build/tools/measure-compaction.js.map +1 -1
  122. package/build/tools/pattern-mitigate.js +1 -1
  123. package/build/tools/pattern-mitigate.js.map +1 -1
  124. package/default-manifest.json +1 -1
  125. package/package.json +1 -1
  126. package/build/product-kb/docs/apps/receiver/compact/clickhouse.md +0 -583
@@ -11,12 +11,19 @@
11
11
  * The advisor's job is to:
12
12
  * - Surface the AWS infra the Retriever expects (S3 input bucket,
13
13
  * index bucket, 4 SQS queues, IRSA role).
14
- * - Preflight-fail if any of the required AWS resources is missing
15
- * from the discovery snapshot.
14
+ * - List in `blockers` every input a complete plan is still missing,
15
+ * and emit no steps while that list is non-empty. The preflight
16
+ * table is the state report beside it: a `fail` row there is
17
+ * reported (and counted in the envelope's `preflight_summary`),
18
+ * not a gate, because the conditions it reads (kubectl unusable,
19
+ * a release already installed) are not answered by re-invoking
20
+ * with a different argument.
16
21
  * - Emit a values.yaml that wires the infra into the chart.
17
- * - Provide verify probes that prove indexing + querying work.
18
- * - Provide teardown (helm uninstall only — leaves AWS infra alone;
19
- * infra lifecycle is a Terraform concern).
22
+ * - Provide verify probes that prove indexing + querying work,
23
+ * each gated on the storage provider they belong to.
24
+ * - Provide teardown. On AWS, helm uninstall only: infra lifecycle is
25
+ * a Terraform concern. On Azure, the provisioning script's own
26
+ * `--destroy`, which deletes the resource group it created.
20
27
  *
21
28
  * Two storage providers. `aws` is the historical path: S3 buckets, four SQS
22
29
  * queues, an IRSA role. `azure` targets AKS with Azure Blob Storage and Azure
@@ -46,6 +53,22 @@ function mapForwarderToOffload(forwarders) {
46
53
  }
47
54
  return null;
48
55
  }
56
+ /**
57
+ * Blob containers the provisioning script creates, and the names it creates
58
+ * them under when the plan passes neither `--input-container` nor
59
+ * `--index-container` (which it does not). Read off
60
+ * `scripts/azure/provision-retriever.sh` in chart 1.0.24: the defaults are
61
+ * `logs` and `tenx-index`, and the BlobCreated event subscription the same run
62
+ * creates is filtered to `--subject-begins-with
63
+ * /blobServices/default/containers/<input container>/`.
64
+ *
65
+ * Both facts matter to the caller: a container name other than `logs` names
66
+ * something the script never created, and even once created by hand it carries
67
+ * no event subscription, so an upload into it raises no BlobCreated event and
68
+ * the indexer never hears about the blob.
69
+ */
70
+ export const AZURE_SCRIPT_INPUT_CONTAINER = 'logs';
71
+ export const AZURE_SCRIPT_INDEX_CONTAINER = 'tenx-index';
49
72
  /** Default Azure Storage Queue names, matching the provisioning script's own. */
50
73
  const AZURE_DEFAULT_QUEUES = {
51
74
  index: 'tenx-index',
@@ -66,8 +89,28 @@ export const RETRIEVER_IMAGE_TAG = '1.1.78';
66
89
  * Node size for a cluster the script creates. The Azure CLI default
67
90
  * (`Standard_D4d_v4`) is refused on subscriptions that do not carry that
68
91
  * family, which stops a first install dead, so the size is always passed.
92
+ *
93
+ * The value matches the provisioning script's own default. `Standard_D2s_v5`
94
+ * was refused on the subscription the Azure path was proved against, and the
95
+ * script moved to v7; an advisor that keeps passing v5 overrides the working
96
+ * default with the refused one.
69
97
  */
70
- export const AKS_NODE_SIZE = 'Standard_D2s_v5';
98
+ export const AKS_NODE_SIZE = 'Standard_D2s_v7';
99
+ /**
100
+ * What the chart labels a retriever pod, and what it names the container.
101
+ *
102
+ * From `retriever-10x` 1.0.24: `templates/deployment.yaml` stamps
103
+ * `app: {{ chart name }}` and `cluster: {{ cluster.name }}` on the pod, and
104
+ * names the container `{{ chart name }}-{{ cluster.name }}`. The default
105
+ * cluster in `values.yaml` is `all-in-one`. Nothing in the chart sets
106
+ * `app.kubernetes.io/instance`, so a selector on that key matches no pod and
107
+ * every probe built on it reports "No resources found" instead of the state
108
+ * it was asked about.
109
+ */
110
+ export const RETRIEVER_POD_SELECTOR = 'app=retriever-10x';
111
+ export const RETRIEVER_CONTAINER = 'retriever-10x-all-in-one';
112
+ /** Cluster entry the chart ships, and the suffix on every per-cluster object. */
113
+ export const RETRIEVER_CLUSTER_NAME = 'all-in-one';
71
114
  /**
72
115
  * The provisioning script that ships with the retriever chart. One run creates
73
116
  * the account, the two containers, the four queues, the managed identity and
@@ -86,8 +129,22 @@ export const AZURE_PROVISION_SCRIPT = `${RETRIEVER_CHART_NAME}/scripts/azure/pro
86
129
  * the script's `--index-path` (default `tenx`); the literal `tenx` segment
87
130
  * after it is the engine's own, and `<app>` is the first path segment of the
88
131
  * indexed blob, which is also what the query's `name` field must equal.
132
+ *
133
+ * One level below the queryId comes a slice segment, `<sliceFromMs>_<sliceToMs>`,
134
+ * because each scan task writes under the time slice it was dispatched for
135
+ * (`IndexObjectQueryResultsWriter`: `{queryId}/{sliceFrom}_{sliceTo}/{worker}.jsonl`).
136
+ * A listing that stops at the queryId prefix sees folders rather than objects,
137
+ * so every list in this plan is recursive.
89
138
  */
90
- export const AZURE_RESULT_PATH = '<index-container>/<index-path>/tenx/<app>/qr/<queryId>/*.jsonl';
139
+ export const AZURE_RESULT_PATH = '<index-container>/<index-path>/tenx/<app>/qr/<queryId>/<sliceFromMs>_<sliceToMs>/<hash>.jsonl';
140
+ /**
141
+ * How long a bounded poll of the results prefix runs before the answer comes
142
+ * from `_DONE.json` instead. Ten polls fifteen seconds apart is two and a half
143
+ * minutes, which covers a one-hour window sliced a minute at a time on a
144
+ * single-node cluster.
145
+ */
146
+ export const AZURE_RESULT_POLL_ATTEMPTS = 10;
147
+ export const AZURE_RESULT_POLL_INTERVAL_SEC = 15;
91
148
  /**
92
149
  * Two facts a first install needs and neither the chart nor the script states:
93
150
  * the operator's own data-plane access, and what `_DONE.json` is not.
@@ -96,18 +153,111 @@ export const AZURE_OPERATOR_ROLES_NOTE = 'The script grants "Storage Blob Data C
96
153
  'identity AND to the operator running it, so `az storage blob upload` and `az storage message put` work ' +
97
154
  'with `--auth-mode login` from the same shell. On a subscription where role assignment is not yours to ' +
98
155
  'make, pass `--account-key` on those commands instead.';
99
- export const AZURE_RESULTS_NOTE = `Results land as JSONL under \`${AZURE_RESULT_PATH}\`. ` +
100
- '`_DONE.json` is a dispatch marker, not a completion signal: it is written before the workers finish, and ' +
101
- 'its scanned / matched / streamRequests counters read 0 on a run that goes on to write results. Poll the ' +
102
- '`qr/<queryId>/` prefix for objects, and treat an empty prefix as "not yet", never as "no matches".';
156
+ export const AZURE_RESULTS_NOTE = `Results land as JSONL under \`${AZURE_RESULT_PATH}\`. Poll the \`qr/<queryId>/\` prefix at most ` +
157
+ `${AZURE_RESULT_POLL_ATTEMPTS} times, ${AZURE_RESULT_POLL_INTERVAL_SEC} seconds apart, and then stop. ` +
158
+ 'A prefix still empty at the end of those polls is answered by `_DONE.json` under that same prefix, which ' +
159
+ 'the coordinator writes seconds after the query message is picked up, once every scan task has gone out. ' +
160
+ 'Its fields are `queryId`, `completedAt`, `elapsedMs`, `reason`, `scanned`, `matched`, `skippedSearch`, ' +
161
+ '`skippedTemplate`, `streamRequests`, `streamBlobs`, `submittedTasks` and `expectedMarkers`. ' +
162
+ '`reason: "empty-range"` with `submittedTasks: 0` says the coordinator saw no index objects in the window ' +
163
+ 'and dispatched nothing. `reason: "dispatched"` with `submittedTasks` above zero says that many scan tasks ' +
164
+ 'reached the queue, so an empty prefix at the end of the polls above is this dispatch reporting no matches. ' +
165
+ 'A marker that is still absent puts the question on the query handler and the query queue rather than on ' +
166
+ 'the result: the message has yet to be picked up. ' +
167
+ 'The `scanned`, `matched`, `streamRequests` and `expectedMarkers` counters read 0 in the marker on this ' +
168
+ 'path whatever the workers go on to write, because the scan and stream workers run in processes of their ' +
169
+ 'own and the coordinator exits after dispatch. The `.jsonl` objects under the prefix are what carry the ' +
170
+ 'matches. ' +
171
+ 'Running the query again mints a NEW queryId and a new prefix. The first prefix stays as it was, so a ' +
172
+ 'second attempt means listing the new id.';
173
+ /**
174
+ * P1 from the second acceptance round. Indexing keys on the timestamp parsed
175
+ * out of the event, and the sample query asks for `now("-1h")` to `now()`, so
176
+ * a sample line stamped with a fixed hour matches its own query only during
177
+ * that hour. The line is therefore generated by the command, at the moment the
178
+ * operator runs it.
179
+ */
180
+ export const AZURE_SAMPLE_LOG_COMMAND = `printf '%s\\n' "$(date -u +%Y-%m-%dT%H:%M:%SZ) ERROR checkout failed for order ORD-DEMO-1" > ./test.log`;
181
+ export const AZURE_EVENT_TIME_NOTE = 'The query window is evaluated against the timestamp parsed out of the log line rather than the time the ' +
182
+ 'blob was written, so the sample line the upload step writes carries the current time at the moment the ' +
183
+ 'command runs. A line stamped with a fixed hour falls outside the `now("-1h")` window the sample query ' +
184
+ 'asks for once that hour has passed, and the query then dispatches its scan tasks and writes no results.';
185
+ /**
186
+ * P7 from the second acceptance round. Every index run logs a 403 that reads
187
+ * as a failure and is the expected state on this path.
188
+ */
189
+ export const AZURE_FLAT_NAMESPACE_403_NOTE = 'Every index run logs `could not read account information for <account>, status 403; assuming a flat ' +
190
+ 'namespace`. The managed identity holds the two data-plane roles and no management-plane read, so the ' +
191
+ 'engine has no way to ask the account whether hierarchical namespace is on and falls back to the ' +
192
+ 'flat-namespace assumption. The provisioning script creates flat-namespace accounts, so the assumption ' +
193
+ 'holds and the line belongs to a healthy install.';
194
+ /**
195
+ * P6 from the second acceptance round. Storage account names live in one
196
+ * global namespace, which neither the question nor its example said.
197
+ */
198
+ export const AZURE_STORAGE_ACCOUNT_UNIQUE_NOTE = 'Storage account names are globally unique across Azure, so a short generic name is usually taken by ' +
199
+ 'another subscription already and account creation stops with a name-unavailable error. Pick one carrying ' +
200
+ 'a suffix of this tenant\'s own, such as `tenxlogs7f3a`, and test a candidate first with ' +
201
+ '`az storage account check-name --name <name>`. 3 to 24 characters, lowercase letters and digits only.';
202
+ /**
203
+ * P5 from the second acceptance round. `log10x_discover_env` probes kubectl
204
+ * and AWS. On the machine the acceptance run used it enumerated an unrelated
205
+ * AWS estate and stamped `estate=serverless` into a snapshot that then backed
206
+ * an Azure plan.
207
+ */
208
+ export const AZURE_SNAPSHOT_SCOPE_NOTE = 'The discovery snapshot behind this plan covers kubectl and AWS. `log10x_discover_env` runs no Azure ' +
209
+ 'probes today, so its buckets, queues, roles and estate verdict describe an AWS account reachable from ' +
210
+ 'the machine that ran discovery and say nothing about the Azure subscription this plan installs into. ' +
211
+ 'Every Azure value below came from the wizard answers or from the provisioning script, and no ' +
212
+ 'AWS-derived snapshot field is read on this path.';
103
213
  export const AZURE_API_KEY_NOTE = '`log10xApiKey` is optional. Left empty, the engine runs on its built-in evaluation licence and says so in ' +
104
214
  'the pod log. Set it only when you hold a Log10x licence key.';
215
+ /**
216
+ * A licence the caller did not hand over is never written into an emitted
217
+ * values file. The file stays on the operator's disk and the plan tells them
218
+ * to keep it, so a key put there without being asked for is a key leaked into
219
+ * a file nobody agreed to hold.
220
+ */
221
+ export function licenseNotEmittedNote(storageProvider) {
222
+ // The two paths write the key into different value slots, so the flag that
223
+ // replaces it differs as well.
224
+ const flag = storageProvider === 'azure'
225
+ ? '`--set-string log10xApiKey="$LOG10X_API_KEY"`'
226
+ : '`--set-string tenx.apiKey="$LOG10X_API_KEY"`';
227
+ return ('No licence key is written into the values file. The engine runs on its built-in evaluation licence until ' +
228
+ `a key is supplied, and a key is supplied at install time with ${flag} so it stays out of any file on disk.`);
229
+ }
230
+ /** Node-size refusals stop a first install dead, so the retry path is stated up front. */
231
+ export const AZURE_NODE_SIZE_NOTE = `The provisioning command passes \`--node-size ${AKS_NODE_SIZE}\`. Subscriptions differ in which VM sizes they ` +
232
+ 'allow. On a refusal the script prints the query that lists the sizes this subscription and region do allow ' +
233
+ '(`az vm list-skus --location <location> --resource-type virtualMachines`) and the exact re-run carrying ' +
234
+ '`--node-size <SIZE>`. Nothing is left half created: the re-run converges on what already exists.';
235
+ /**
236
+ * Azure teardown. The resource group holds the storage account, the queues,
237
+ * the managed identity, the Event Grid subscription and the AKS cluster, and
238
+ * the script's `--destroy` deletes the group and everything in it. No
239
+ * Terraform state exists on this path.
240
+ */
241
+ export function buildAzureTeardownCommand(resourceGroup) {
242
+ return `bash ${AZURE_PROVISION_SCRIPT} --destroy --resource-group ${resourceGroup}`;
243
+ }
244
+ /**
245
+ * The query body the provisioning script prints in its own runbook. `name`
246
+ * has to equal the first path segment of the uploaded blob, which the upload
247
+ * step below makes `app`.
248
+ */
249
+ export const AZURE_SAMPLE_QUERY_BODY = '{"name":"app","from":"now(\\"-1h\\")","to":"now()","search":"severity_level==\\"ERROR\\"","writeResults":true}';
105
250
  /**
106
251
  * The provisioning commands, in the order a customer runs them: add the repo,
107
252
  * pull and untar the chart, then run the script from the untarred directory.
108
253
  * `helm repo add` on a repo that is already present skips without refreshing
109
254
  * the index, so `helm repo update` runs before the pull or `--version` can
110
255
  * miss a freshly published chart.
256
+ *
257
+ * The script is invoked through `bash`. `helm package` writes every file in a
258
+ * chart tarball as mode 0644 whatever its mode in git, so the copy that comes
259
+ * out of `helm pull --untar` carries no exec bit and a direct invocation is
260
+ * refused with "permission denied".
111
261
  */
112
262
  export function buildAzureProvisionCommands(opts) {
113
263
  return [
@@ -115,7 +265,7 @@ export function buildAzureProvisionCommands(opts) {
115
265
  'helm repo update',
116
266
  `helm pull ${RETRIEVER_CHART_REF} --version ${RETRIEVER_CHART_VERSION} --untar`,
117
267
  [
118
- `${AZURE_PROVISION_SCRIPT} \\`,
268
+ `bash ${AZURE_PROVISION_SCRIPT} \\`,
119
269
  ` --resource-group ${opts.resourceGroup} \\`,
120
270
  ` --location ${opts.location} \\`,
121
271
  ` --account ${opts.account} \\`,
@@ -139,32 +289,53 @@ export async function buildRetrieverPlan(args) {
139
289
  (installedNamespace
140
290
  ? installedNamespace // actual pod namespace from discover_env
141
291
  : snapshot.recommendations.suggestedNamespace ?? 'logging');
292
+ const storageProvider = args.storageProvider ?? 'aws';
293
+ const isAzure = storageProvider === 'azure';
142
294
  // Infra: prefer caller-supplied values; fall back to snapshot-derived.
143
295
  // Fix 82: for verify, also try to resolve the bucket from the installed
144
296
  // Helm release values if installedComponentsDetail.retriever is present.
145
- const installedBucket = installedNamespace
297
+ //
298
+ // P5, second acceptance round: `log10x_discover_env` probes kubectl and AWS
299
+ // and has no Azure side. On the acceptance machine it enumerated an
300
+ // unrelated AWS account's S3 buckets and SQS queues and stamped
301
+ // `estate=serverless` into the snapshot an Azure plan was then built from.
302
+ // Every snapshot field below that describes AWS resources is therefore read
303
+ // only on an AWS plan; an Azure plan takes its values from the wizard
304
+ // answers and from the provisioning script, and says so in its notes.
305
+ const installedBucket = !isAzure && installedNamespace
146
306
  ? await resolveInstalledBucket(releaseName, installedNamespace)
147
307
  : undefined;
148
- const inputBucket = args.inputBucket ?? installedBucket ?? snapshot.recommendations.retrieverS3Bucket;
308
+ // `snapshot.recommendations.retrieverS3Bucket` is an S3 bucket name that
309
+ // discovery pattern-matched out of the AWS estate. On an Azure plan it is
310
+ // not a blob container and never belongs in one: pasted into `az storage
311
+ // blob list -c`, it points every storage probe at a name that does not
312
+ // exist on the account. Only what the caller passed is used there.
313
+ const inputBucket = isAzure
314
+ ? args.inputBucket
315
+ : args.inputBucket ?? installedBucket ?? snapshot.recommendations.retrieverS3Bucket;
149
316
  const indexBucket = args.indexBucket ??
150
- (args.storageProvider === 'azure'
317
+ (isAzure
151
318
  ? args.storageAccount
152
319
  ? `${args.storageAccount}/tenx-index/tenx`
153
320
  : undefined
154
321
  : inputBucket
155
322
  ? `${inputBucket}/indexing-results/`
156
323
  : undefined);
157
- const irsaRoleArn = args.irsaRoleArn ??
158
- snapshot.kubectl.serviceAccountIrsa.find((sa) => sa.name.toLowerCase().includes('retriever') || sa.name.toLowerCase().includes('tenx-retriever'))?.roleArn;
159
- const detectedQueues = snapshot.recommendations.retrieverSqsUrls ?? {};
160
- const sqsUrls = {
161
- index: args.sqsUrls?.index ?? detectedQueues.index,
162
- query: args.sqsUrls?.query ?? detectedQueues.query,
163
- subquery: args.sqsUrls?.subquery ?? detectedQueues.subquery,
164
- stream: args.sqsUrls?.stream ?? detectedQueues.stream,
165
- };
166
- const storageProvider = args.storageProvider ?? 'aws';
167
- const isAzure = storageProvider === 'azure';
324
+ const irsaRoleArn = isAzure
325
+ ? undefined
326
+ : args.irsaRoleArn ??
327
+ snapshot.kubectl.serviceAccountIrsa.find((sa) => sa.name.toLowerCase().includes('retriever') || sa.name.toLowerCase().includes('tenx-retriever'))?.roleArn;
328
+ // An AKS install has no SQS queue. Reading the four detected URLs on an
329
+ // Azure plan put an unrelated account's queue behind a "pass" in round one.
330
+ const detectedQueues = isAzure ? {} : snapshot.recommendations.retrieverSqsUrls ?? {};
331
+ const sqsUrls = isAzure
332
+ ? { index: undefined, query: undefined, subquery: undefined, stream: undefined }
333
+ : {
334
+ index: args.sqsUrls?.index ?? detectedQueues.index,
335
+ query: args.sqsUrls?.query ?? detectedQueues.query,
336
+ subquery: args.sqsUrls?.subquery ?? detectedQueues.subquery,
337
+ stream: args.sqsUrls?.stream ?? detectedQueues.stream,
338
+ };
168
339
  const azureQueues = {
169
340
  index: args.azureQueues?.index ?? AZURE_DEFAULT_QUEUES.index,
170
341
  query: args.azureQueues?.query ?? AZURE_DEFAULT_QUEUES.query,
@@ -216,10 +387,12 @@ export async function buildRetrieverPlan(args) {
216
387
  });
217
388
  const notes = [];
218
389
  if (snapshot.recommendations.alreadyInstalled.retriever) {
219
- notes.push(`A Retriever is already installed in namespace \`${snapshot.recommendations.alreadyInstalled.retriever}\`. Installing a second release requires a separate set of SQS queues + IRSA role — running two retrievers against the same queues will race.`);
390
+ notes.push(isAzure
391
+ ? `A Retriever is already installed in namespace \`${snapshot.recommendations.alreadyInstalled.retriever}\`. A second release needs its own four Storage Queues and its own managed identity: two releases polling one set of queues race each other for every message.`
392
+ : `A Retriever is already installed in namespace \`${snapshot.recommendations.alreadyInstalled.retriever}\`. Installing a second release requires a separate set of SQS queues + IRSA role, since running two retrievers against the same queues will race.`);
220
393
  }
221
394
  // Fix 81/82: audit trail for namespace + bucket resolution source.
222
- if (installedNamespace) {
395
+ if (installedNamespace && !isAzure) {
223
396
  const bucketSource = args.inputBucket
224
397
  ? 'caller-supplied'
225
398
  : installedBucket
@@ -232,7 +405,11 @@ export async function buildRetrieverPlan(args) {
232
405
  ? `Retriever infra on Azure (storage account, blob containers, Storage Queues, managed identity and its blob/queue data roles, the Event Grid BlobCreated subscription, and the federated credential) is provisioned by the chart's own script, NOT by this advisor. The script ships inside the chart tarball at \`${AZURE_PROVISION_SCRIPT}\`, reached with \`helm pull ${RETRIEVER_CHART_REF} --version ${RETRIEVER_CHART_VERSION} --untar\`. Step 1 below does both.`
233
406
  : 'Retriever infra (S3 buckets, SQS queues, IAM role + IRSA binding, CloudWatch log groups) is provisioned via the Terraform module, NOT by this advisor. The plan below assumes infra already exists.');
234
407
  if (isAzure) {
408
+ notes.push(AZURE_SNAPSHOT_SCOPE_NOTE);
235
409
  notes.push(AZURE_RESULTS_NOTE);
410
+ notes.push(AZURE_EVENT_TIME_NOTE);
411
+ notes.push(AZURE_FLAT_NAMESPACE_403_NOTE);
412
+ notes.push(AZURE_STORAGE_ACCOUNT_UNIQUE_NOTE);
236
413
  notes.push(AZURE_OPERATOR_ROLES_NOTE);
237
414
  notes.push(AZURE_API_KEY_NOTE);
238
415
  notes.push(`The plan pins engine image tag \`${RETRIEVER_IMAGE_TAG}\`. The chart's own appVersion trails the released engine, so an unpinned install runs an older image than the one this path is tested against.`);
@@ -245,7 +422,12 @@ export async function buildRetrieverPlan(args) {
245
422
  // nothing to index. The note appears above the preflight table so users
246
423
  // see it before running any commands. The verify probe (receiver-offload-
247
424
  // capability) will confirm the state after the Receiver config is updated.
248
- const receiverDetail = snapshot.recommendations.installedComponentsDetail?.receiver;
425
+ // Gated off the Azure path: the recipe routes bytes to S3, and log10x
426
+ // emits no forwarder offload recipe for Blob today. The Azure note above
427
+ // states that instead.
428
+ const receiverDetail = isAzure
429
+ ? undefined
430
+ : snapshot.recommendations.installedComponentsDetail?.receiver;
249
431
  if (receiverDetail) {
250
432
  notes.push(`**Receiver config update required for S3 offload.** ` +
251
433
  `The Receiver is installed in namespace \`${receiverDetail.namespace}\` but may be ` +
@@ -261,12 +443,16 @@ export async function buildRetrieverPlan(args) {
261
443
  const install = [];
262
444
  const verify = [];
263
445
  const teardown = [];
446
+ // A licence the caller handed over is wired into the values file. One the
447
+ // wizard minted on their behalf is not: see LICENSE_NOT_EMITTED_NOTE.
448
+ const licenseSupplied = args.licenseSupplied !== false && !!args.licenseJwt;
264
449
  if (!args.skipInstall && blockers.length === 0) {
265
450
  install.push(...(isAzure
266
451
  ? buildAzureInstallSteps({
267
452
  releaseName,
268
453
  namespace,
269
- licenseJwt: args.licenseJwt,
454
+ licenseSupplied,
455
+ ...(licenseSupplied ? { licenseJwt: args.licenseJwt } : {}),
270
456
  storageAccount: args.storageAccount,
271
457
  inputContainer: inputBucket,
272
458
  indexContainer: indexBucket,
@@ -275,25 +461,42 @@ export async function buildRetrieverPlan(args) {
275
461
  queues: azureQueues,
276
462
  ...(args.resourceGroup !== undefined ? { resourceGroup: args.resourceGroup } : {}),
277
463
  ...(args.location !== undefined ? { location: args.location } : {}),
464
+ ...(args.aksCluster !== undefined ? { aksCluster: args.aksCluster } : {}),
278
465
  })
279
466
  : buildInstallSteps({
280
467
  releaseName,
281
468
  namespace,
282
- licenseJwt: args.licenseJwt,
469
+ licenseSupplied,
470
+ ...(licenseSupplied ? { licenseJwt: args.licenseJwt } : {}),
283
471
  inputBucket: inputBucket,
284
472
  indexBucket: indexBucket,
285
473
  irsaRoleArn: irsaRoleArn,
286
474
  sqsUrls: sqsUrls,
287
475
  })));
288
476
  }
477
+ if (!args.skipInstall && blockers.length === 0 && !licenseSupplied) {
478
+ notes.push(licenseNotEmittedNote(storageProvider));
479
+ }
289
480
  if (!args.skipVerify) {
290
481
  // Pass the Receiver namespace so the offload-capability probe can inspect
291
482
  // whether the Receiver ConfigMap has outputOffload wired (Fix 88).
292
- const receiverNamespace = snapshot.recommendations.installedComponentsDetail?.receiver?.namespace;
293
- verify.push(...buildVerifyProbes(releaseName, namespace, inputBucket, sqsUrls.index, receiverNamespace));
483
+ const receiverNamespace = isAzure
484
+ ? undefined
485
+ : snapshot.recommendations.installedComponentsDetail?.receiver?.namespace;
486
+ verify.push(...buildVerifyProbes({
487
+ releaseName,
488
+ namespace,
489
+ storageProvider,
490
+ ...(inputBucket !== undefined ? { inputBucket } : {}),
491
+ ...(sqsUrls.index !== undefined ? { indexQueueUrl: sqsUrls.index } : {}),
492
+ ...(receiverNamespace !== undefined ? { receiverNamespace } : {}),
493
+ ...(args.storageAccount !== undefined ? { storageAccount: args.storageAccount } : {}),
494
+ ...(indexBucket !== undefined ? { indexContainer: indexBucket } : {}),
495
+ azureQueues,
496
+ }));
294
497
  }
295
498
  if (!args.skipTeardown) {
296
- teardown.push(...buildTeardownSteps(releaseName, namespace));
499
+ teardown.push(...buildTeardownSteps(releaseName, namespace, storageProvider, args.resourceGroup));
297
500
  }
298
501
  // Forwarder offload section: how to route the routeState="drop" slice to the
299
502
  // customer's own S3 (the bucket the Retriever reads) + SIEM down-tier
@@ -320,7 +523,7 @@ export async function buildRetrieverPlan(args) {
320
523
  // absence is a reliable signal of running outside the cluster.
321
524
  const runningInsideCluster = process.env['KUBERNETES_SERVICE_HOST'] !== undefined;
322
525
  const retrieverAccessMarkdown = !runningInsideCluster
323
- ? buildRetrieverExternalAccessMarkdown(releaseName, namespace)
526
+ ? buildRetrieverExternalAccessMarkdown(releaseName, namespace, storageProvider)
324
527
  : undefined;
325
528
  return {
326
529
  app: 'retriever',
@@ -348,8 +551,37 @@ export async function buildRetrieverPlan(args) {
348
551
  * (ephemeral dev/test), Service type LoadBalancer (persistent prod), and
349
552
  * deploying the MCP server inside the cluster (cleanest for shared teams).
350
553
  */
351
- function buildRetrieverExternalAccessMarkdown(releaseName, namespace) {
352
- const svcName = `${releaseName}-retriever-10x-all-in-one`;
554
+ function buildRetrieverExternalAccessMarkdown(releaseName, namespace, storageProvider = 'aws') {
555
+ const isAzure = storageProvider === 'azure';
556
+ // The chart names every per-cluster object `<fullname>-<cluster>`. The Azure
557
+ // values file sets `fullnameOverride: <release>` so the ServiceAccount name
558
+ // matches the federated credential subject, which also shortens the Service
559
+ // to `<release>-all-in-one`. Without the override the fullname is
560
+ // `<release>-retriever-10x`, which is what the AWS path still gets.
561
+ const svcName = isAzure
562
+ ? `${releaseName}-${RETRIEVER_CLUSTER_NAME}`
563
+ : `${releaseName}-${RETRIEVER_CHART_NAME}-${RETRIEVER_CLUSTER_NAME}`;
564
+ const loadBalancerValues = isAzure
565
+ ? [
566
+ '```yaml',
567
+ '# retriever-values-lb.yaml',
568
+ 'service:',
569
+ ' type: LoadBalancer',
570
+ '```',
571
+ ]
572
+ : [
573
+ '```yaml',
574
+ '# retriever-values-lb.yaml',
575
+ 'retriever:',
576
+ ' service:',
577
+ ' type: LoadBalancer',
578
+ ' annotations:',
579
+ ' service.beta.kubernetes.io/aws-load-balancer-type: "nlb"',
580
+ '```',
581
+ ];
582
+ const loadBalancerIntro = isAzure
583
+ ? 'Add to the values file and upgrade:'
584
+ : 'Add to your Terraform module call or values file:';
353
585
  return [
354
586
  '## How to query the Retriever from outside the cluster',
355
587
  '',
@@ -376,21 +608,16 @@ function buildRetrieverExternalAccessMarkdown(releaseName, namespace) {
376
608
  '',
377
609
  '### Option B — Migrate the Service to LoadBalancer (persistent)',
378
610
  '',
379
- 'Add to your Terraform module call or values file:',
611
+ loadBalancerIntro,
380
612
  '',
381
- '```yaml',
382
- '# retriever-values-lb.yaml',
383
- 'retriever:',
384
- ' service:',
385
- ' type: LoadBalancer',
386
- ' annotations:',
387
- ' service.beta.kubernetes.io/aws-load-balancer-type: "nlb"',
388
- '```',
613
+ ...loadBalancerValues,
389
614
  '',
390
615
  'Apply with:',
391
616
  '',
392
617
  '```bash',
393
- `helm upgrade ${releaseName} ${RETRIEVER_CHART_REF} -n ${namespace} -f retriever-values-lb.yaml`,
618
+ isAzure
619
+ ? `helm upgrade ${releaseName} ${RETRIEVER_CHART_REF} --version ${RETRIEVER_CHART_VERSION} \\\n -n ${namespace} \\\n -f ${releaseName}-azure-values.yaml \\\n -f ${releaseName}-azure-provisioned.yaml \\\n -f retriever-values-lb.yaml`
620
+ : `helm upgrade ${releaseName} ${RETRIEVER_CHART_REF} -n ${namespace} -f retriever-values-lb.yaml`,
394
621
  '```',
395
622
  '',
396
623
  'After the LoadBalancer is provisioned, run `kubectl -n ' +
@@ -641,9 +868,16 @@ function renderRetrieverValues(opts) {
641
868
  // `tenx.apiKey` slot rather than the Reporter chart's `log10xLicenseJwt`
642
869
  // convention, so the license JWT goes into that slot. The engine
643
870
  // validates the JWT regardless of which value-key it arrives through.
871
+ //
872
+ // The slot is filled only when the caller handed a licence over. Otherwise
873
+ // the line is a comment carrying the install-time flag, so no key lands in
874
+ // a file the plan tells the operator to keep.
875
+ const apiKeyLine = opts.licenseSupplied && opts.licenseJwt
876
+ ? ` apiKey: "${opts.licenseJwt}"`
877
+ : ' # apiKey: pass at install time with --set-string tenx.apiKey="$LOG10X_API_KEY"';
644
878
  return `tenx:
645
879
  enabled: true
646
- apiKey: "${opts.licenseJwt}"
880
+ ${apiKeyLine}
647
881
  runtimeName: "${opts.releaseName}"
648
882
  gitToken: "public-repo-no-token-needed"
649
883
  config:
@@ -665,40 +899,79 @@ subQueryQueueUrl: "${opts.sqsUrls.subquery}"
665
899
  streamQueueUrl: "${opts.sqsUrls.stream}"
666
900
  `;
667
901
  }
902
+ /**
903
+ * `indexContainer` arrives as `<account>/<container>/<index-path>`, the shape
904
+ * the chart's `storage.azure.indexContainer` takes and the shape the script
905
+ * writes. The results prefix is the index path, then the engine's own `tenx`
906
+ * segment, then the app.
907
+ */
908
+ function splitAzureIndexContainer(indexContainer) {
909
+ const parts = indexContainer.split('/');
910
+ return { container: parts[1] ?? 'tenx-index', path: parts[2] ?? 'tenx' };
911
+ }
668
912
  /**
669
913
  * AKS + Azure Blob install steps.
670
914
  *
671
915
  * Step 1 pulls the chart and runs the provisioning script that ships inside
672
916
  * it, which creates every Azure resource the Retriever needs and emits a
673
- * values file. Steps 2 to 5 mirror the AWS path: create the namespace, write
674
- * the values, install, wait. The values block is the chart's
675
- * `storage.provider: azure` shape, so the emitted file and the script's own
676
- * output describe the same install.
917
+ * values file. Step 2 points kubectl at the cluster, because every step after
918
+ * it is a kubectl or helm call and a fresh shell has no context for a cluster
919
+ * the script just created. Steps 3 to 6 create the namespace, write the
920
+ * values, install and wait, and steps 7 to 9 are the loop the provisioning
921
+ * script prints in its own runbook: upload a blob, put a query on the query
922
+ * queue, read the JSONL the workers wrote.
923
+ *
924
+ * The values file this step writes and the one the script writes carry
925
+ * different names, so neither overwrites the other. The install passes both,
926
+ * the script's second, so any key the script wrote from what it actually
927
+ * created wins over a default carried here.
677
928
  */
678
929
  function buildAzureInstallSteps(opts) {
679
930
  const steps = [];
931
+ // Two names, two files. The script owns `<release>-azure-provisioned.yaml`
932
+ // through `--values-out`; this plan owns `<release>-azure-values.yaml`.
933
+ const provisionedValuesFile = `${opts.releaseName}-azure-provisioned.yaml`;
680
934
  const valuesFile = `${opts.releaseName}-azure-values.yaml`;
681
935
  const rg = opts.resourceGroup ?? '<resource-group>';
682
936
  const loc = opts.location ?? '<location>';
937
+ const aks = opts.aksCluster ?? '<aks-cluster-name>';
938
+ const kubeconfigFile = `${aks}.kubeconfig`;
939
+ const { container: indexContainerName, path: indexPath } = splitAzureIndexContainer(opts.indexContainer);
940
+ const resultsPrefix = `${indexPath}/tenx/app/qr/`;
683
941
  steps.push({
684
942
  title: 'Pull the chart and provision the Azure resources',
685
943
  rationale: `The script lives inside the chart tarball, at \`${AZURE_PROVISION_SCRIPT}\`, so the pull comes first. ` +
686
- 'One run creates the storage account (flat namespace), the input and index containers, the four Storage ' +
944
+ `One run creates the storage account (flat namespace), the two blob containers ` +
945
+ `\`${AZURE_SCRIPT_INPUT_CONTAINER}\` for the source logs and \`${AZURE_SCRIPT_INDEX_CONTAINER}\` for the ` +
946
+ 'index, the four Storage ' +
687
947
  'Queues, the user-assigned managed identity with "Storage Blob Data Contributor" and "Storage Queue Data ' +
688
948
  'Contributor" on the account, the Event Grid system topic with a BlobCreated subscription onto the index ' +
689
949
  'queue, and the federated credential binding the identity to this release\'s ServiceAccount. It ends by ' +
690
- `writing a values file pinning image tag \`${RETRIEVER_IMAGE_TAG}\`. ${AZURE_OPERATOR_ROLES_NOTE}`,
950
+ `writing \`${provisionedValuesFile}\`, pinning image tag \`${RETRIEVER_IMAGE_TAG}\`. ` +
951
+ `${AZURE_OPERATOR_ROLES_NOTE} ${AZURE_NODE_SIZE_NOTE}`,
691
952
  commands: buildAzureProvisionCommands({
692
953
  resourceGroup: rg,
693
954
  location: loc,
694
955
  account: opts.storageAccount,
695
- aksCluster: '<aks-cluster-name>',
956
+ aksCluster: aks,
696
957
  namespace: opts.namespace,
697
958
  releaseName: opts.releaseName,
698
- valuesOut: valuesFile,
959
+ valuesOut: provisionedValuesFile,
699
960
  }),
700
961
  expectDurationSec: 900,
701
962
  });
963
+ steps.push({
964
+ title: 'Point kubectl at the cluster',
965
+ rationale: `Every step below runs kubectl or helm against \`${aks}\`. A shell that has not fetched the credentials ` +
966
+ 'has no context for a cluster the previous step just created, and `--overwrite-existing` keeps a stale ' +
967
+ 'entry of the same name from winning. The write goes to a file of its own rather than into the default ' +
968
+ 'kubeconfig, so nothing already in `~/.kube/config` is touched.',
969
+ commands: [
970
+ `az aks get-credentials --resource-group ${rg} --name ${aks} \\\n --file ./${kubeconfigFile} --overwrite-existing`,
971
+ `export KUBECONFIG="$PWD/${kubeconfigFile}"`,
972
+ 'kubectl get nodes',
973
+ ],
974
+ });
702
975
  steps.push({
703
976
  title: 'Create target namespace',
704
977
  rationale: `The Retriever installs into \`${opts.namespace}\`.`,
@@ -708,9 +981,13 @@ function buildAzureInstallSteps(opts) {
708
981
  });
709
982
  steps.push({
710
983
  title: 'Write Helm values',
711
- rationale: 'Wires the tenx block and the chart\'s `storage.provider: azure` block: the account, both containers, ' +
712
- 'the four Storage Queue names, and workload-identity auth. Compare against the file the provisioning ' +
713
- 'script wrote and keep whichever the operator edited.',
984
+ rationale: 'Carries the chart\'s `storage.provider: azure` block: the account, both containers, the four Storage ' +
985
+ 'Queue names, and workload-identity auth. `fullnameOverride` is the release name, which is what makes the ' +
986
+ `ServiceAccount \`${opts.releaseName}\` rather than \`${opts.releaseName}-${RETRIEVER_CHART_NAME}\`, the ` +
987
+ `name the federated credential subject \`system:serviceaccount:${opts.namespace}:${opts.releaseName}\` ` +
988
+ 'binds to. Without it the pod reaches 2/2 Running and every queue poll comes back AADSTS700213, no ' +
989
+ 'matching federated identity record for the presented subject. ' +
990
+ `The file sits beside \`${provisionedValuesFile}\` rather than on top of it.`,
714
991
  file: {
715
992
  path: valuesFile,
716
993
  contents: renderAzureRetrieverValues(opts),
@@ -720,33 +997,70 @@ function buildAzureInstallSteps(opts) {
720
997
  });
721
998
  steps.push({
722
999
  title: 'Install via Helm',
723
- rationale: 'Deploys the indexer + query-handler + stream-worker against Blob and the Storage Queues. ' +
1000
+ rationale: 'Deploys the indexer, the query handler and the stream worker against Blob and the Storage Queues. ' +
1001
+ `Both values files are passed, \`${provisionedValuesFile}\` second, so what the script recorded about ` +
1002
+ `what it created wins over any default in \`${valuesFile}\`. ` +
724
1003
  AZURE_API_KEY_NOTE,
725
1004
  commands: [
726
- `helm upgrade --install ${opts.releaseName} ${RETRIEVER_CHART_REF} \\\n --version ${RETRIEVER_CHART_VERSION} \\\n -n ${opts.namespace} --create-namespace \\\n -f ${valuesFile}`,
1005
+ `helm upgrade --install ${opts.releaseName} ${RETRIEVER_CHART_REF} \\\n --version ${RETRIEVER_CHART_VERSION} \\\n -n ${opts.namespace} --create-namespace \\\n -f ${valuesFile} \\\n -f ${provisionedValuesFile}`,
727
1006
  ],
728
1007
  });
729
1008
  steps.push({
730
1009
  title: 'Wait for rollout',
731
- rationale: 'Blocks until indexer + query-handler + stream-worker report Ready.',
1010
+ rationale: `The chart labels the pod \`${RETRIEVER_POD_SELECTOR}\` and names the container ` +
1011
+ `\`${RETRIEVER_CONTAINER}\`. A selector on \`app.kubernetes.io/instance\` matches nothing here, so it ` +
1012
+ 'reports "No resources found" whatever the pod is doing.',
732
1013
  commands: [
733
- `kubectl -n ${opts.namespace} rollout status deployment -l app.kubernetes.io/instance=${opts.releaseName} --timeout=10m || true`,
734
- `kubectl -n ${opts.namespace} logs -l app.kubernetes.io/instance=${opts.releaseName} --tail=50`,
1014
+ `kubectl -n ${opts.namespace} rollout status deployment/${opts.releaseName}-${RETRIEVER_CLUSTER_NAME} --timeout=10m || true`,
1015
+ `kubectl -n ${opts.namespace} logs -l ${RETRIEVER_POD_SELECTOR} -c ${RETRIEVER_CONTAINER} --tail=50`,
735
1016
  ],
736
1017
  expectDurationSec: 600,
737
1018
  });
738
- // `indexContainer` arrives as `<account>/<container>/<index-path>`, the shape
739
- // the chart's `storage.azure.indexContainer` takes and the shape the script
740
- // writes. The results prefix is the index path, then the engine's own `tenx`
741
- // segment, then the app.
742
- const indexParts = opts.indexContainer.split('/');
743
- const indexContainerName = indexParts[1] ?? 'tenx-index';
744
- const indexPath = indexParts[2] ?? 'tenx';
1019
+ // The upload lands in whatever container the wizard was told about. The
1020
+ // provisioning script created `logs` and filtered the BlobCreated
1021
+ // subscription to it, so any other name needs both the container and the
1022
+ // subscription before this step can index anything.
1023
+ const inputContainerIsScriptDefault = opts.inputContainer === AZURE_SCRIPT_INPUT_CONTAINER;
1024
+ const uploadContainerNote = inputContainerIsScriptDefault
1025
+ ? `\`${AZURE_SCRIPT_INPUT_CONTAINER}\` is the container step 1 created, and the BlobCreated subscription ` +
1026
+ 'the same run created is filtered to it.'
1027
+ : `\`${opts.inputContainer}\` is not the container step 1 created. The script creates ` +
1028
+ `\`${AZURE_SCRIPT_INPUT_CONTAINER}\` and filters the BlobCreated subscription to it, so this upload ` +
1029
+ `raises no event the indexer hears. Re-run the provisioning command in step 1 with ` +
1030
+ `\`--input-container ${opts.inputContainer}\` before this step, or upload into ` +
1031
+ `\`${AZURE_SCRIPT_INPUT_CONTAINER}\` instead.`;
1032
+ steps.push({
1033
+ title: 'Upload a log to the input container',
1034
+ rationale: `The first path segment of the blob name is the application name, so \`app/test.log\` indexes under ` +
1035
+ '`app` and the query below has to carry the same name. Event Grid delivers the BlobCreated event to ' +
1036
+ `\`${opts.queues.index}\` and the pod writes the index. A query naming anything else returns silence ` +
1037
+ `rather than an error. ${uploadContainerNote} ${AZURE_EVENT_TIME_NOTE}`,
1038
+ commands: [
1039
+ AZURE_SAMPLE_LOG_COMMAND,
1040
+ `az storage blob upload \\\n --account-name ${opts.storageAccount} \\\n --auth-mode login \\\n -c ${opts.inputContainer} \\\n -n app/test.log \\\n -f ./test.log \\\n -o none`,
1041
+ ],
1042
+ });
1043
+ steps.push({
1044
+ title: 'Put a query on the query queue',
1045
+ rationale: `The query body names the app \`app\`, matching the blob uploaded above, and sets \`writeResults\` so the ` +
1046
+ `workers write JSONL under \`${indexContainerName}/${resultsPrefix}<queryId>/<sliceFromMs>_<sliceToMs>/\`. ` +
1047
+ 'The window is `now("-1h")` to `now()`, evaluated against the timestamp inside each indexed line, which ' +
1048
+ 'is why the step above stamps the sample with the current time. The second command lists the results ' +
1049
+ 'prefix recursively, which is how the queryId becomes known: the engine mints it, rather than this plan.',
1050
+ commands: [
1051
+ `az storage message put \\\n --account-name ${opts.storageAccount} \\\n --auth-mode login \\\n --queue-name ${opts.queues.query} \\\n --content '${AZURE_SAMPLE_QUERY_BODY}' \\\n -o none`,
1052
+ `az storage blob list \\\n --account-name ${opts.storageAccount} \\\n --auth-mode login \\\n -c ${indexContainerName} \\\n --prefix ${resultsPrefix} \\\n --query "[].name" -o tsv`,
1053
+ ],
1054
+ });
745
1055
  steps.push({
746
1056
  title: 'Read the results',
747
- rationale: AZURE_RESULTS_NOTE,
1057
+ rationale: 'The first command picks the most recently written `.jsonl` under the results prefix, the second ' +
1058
+ `downloads it, the third prints it. One matched event per line. ${AZURE_RESULTS_NOTE}`,
1059
+ expectDurationSec: AZURE_RESULT_POLL_ATTEMPTS * AZURE_RESULT_POLL_INTERVAL_SEC,
748
1060
  commands: [
749
- `az storage blob list --account-name ${opts.storageAccount} \\\n --container-name ${indexContainerName} \\\n --prefix "${indexPath}/tenx/<app>/qr/<queryId>/" \\\n --auth-mode login -o table`,
1061
+ `blob="$(az storage blob list \\\n --account-name ${opts.storageAccount} \\\n --auth-mode login \\\n -c ${indexContainerName} \\\n --prefix ${resultsPrefix} \\\n --query "sort_by([?ends_with(name, '.jsonl')], &properties.lastModified)[-1].name" \\\n -o tsv)"`,
1062
+ `az storage blob download \\\n --account-name ${opts.storageAccount} \\\n --auth-mode login \\\n -c ${indexContainerName} \\\n -n "$blob" \\\n -f ./results.jsonl \\\n -o none`,
1063
+ 'cat ./results.jsonl',
750
1064
  ],
751
1065
  });
752
1066
  return steps;
@@ -754,27 +1068,36 @@ function buildAzureInstallSteps(opts) {
754
1068
  function renderAzureRetrieverValues(opts) {
755
1069
  // `invoke: queue` is the Azure equivalent of the AWS `sqs` fan-out: the
756
1070
  // pipeline hands the next stage to an Azure Storage Queue. `scheduledQueries`
757
- // is off because the CronJob shells `aws sqs send-message` from an aws-cli
758
- // image, which has no Azure equivalent in the chart today.
1071
+ // is off because the CronJob shells an aws-cli image, which has no Azure
1072
+ // equivalent in the chart today.
759
1073
  //
760
1074
  // `image.tag` is pinned rather than left to the chart's appVersion, which
761
1075
  // trails the released engine.
1076
+ //
1077
+ // `fullnameOverride` is the release name. The chart names the ServiceAccount
1078
+ // after the fullname, and the federated credential the provisioning script
1079
+ // creates binds `system:serviceaccount:<namespace>:<release>`. Left out, the
1080
+ // chart names it `<release>-retriever-10x`, the subject no longer matches,
1081
+ // and every queue poll returns AADSTS700213 from a pod that is otherwise
1082
+ // healthy.
1083
+ //
1084
+ // No `tenx:` block: the published chart carries no such key, so everything
1085
+ // under it configures nothing. Its `apiKey` slot was the second copy of the
1086
+ // licence in this file.
1087
+ const apiKeyLines = opts.licenseSupplied && opts.licenseJwt
1088
+ ? [`log10xApiKey: "${opts.licenseJwt}"`]
1089
+ : [
1090
+ '# Supplied at install time so no key lands in this file:',
1091
+ '# --set-string log10xApiKey="$LOG10X_API_KEY"',
1092
+ ];
762
1093
  return `# log10xApiKey is optional: empty means the built-in evaluation licence.
763
- log10xApiKey: "${opts.licenseJwt}"
1094
+ ${apiKeyLines.join('\n')}
1095
+
1096
+ fullnameOverride: "${opts.releaseName}"
764
1097
 
765
1098
  image:
766
1099
  tag: "${RETRIEVER_IMAGE_TAG}"
767
1100
 
768
- tenx:
769
- enabled: true
770
- apiKey: "${opts.licenseJwt}"
771
- runtimeName: "${opts.releaseName}"
772
- gitToken: "public-repo-no-token-needed"
773
- config:
774
- git:
775
- enabled: true
776
- url: "https://github.com/log-10x/config.git"
777
-
778
1101
  storage:
779
1102
  provider: azure
780
1103
  azure:
@@ -798,33 +1121,120 @@ scheduledQueries:
798
1121
  enabled: false
799
1122
  `;
800
1123
  }
801
- function buildVerifyProbes(releaseName, namespace, inputBucket, indexQueueUrl,
802
- /** Namespace where the Receiver DaemonSet runs (if installed). Used to probe outputOffload config. */
803
- receiverNamespace) {
1124
+ /**
1125
+ * Verify probes.
1126
+ *
1127
+ * Every AWS probe is gated on `storageProvider === 'aws'`. An Azure install
1128
+ * has no S3 bucket and no SQS queue, and the AWS-shaped probes did not fail
1129
+ * loudly on one: probe 5 pasted the blob container name into an `s3://` URL
1130
+ * and returned AccessDenied, and the queue-depth probe polled an SQS URL
1131
+ * scraped from an unrelated AWS estate and reported zero messages, which
1132
+ * reads as a pass.
1133
+ */
1134
+ function buildVerifyProbes(opts) {
1135
+ const { releaseName, namespace, storageProvider } = opts;
1136
+ const isAzure = storageProvider === 'azure';
804
1137
  const probes = [];
1138
+ // On Azure the values file sets `fullnameOverride`, so the per-cluster
1139
+ // objects are `<release>-all-in-one`. On AWS the chart derives the fullname
1140
+ // and they are `<release>-retriever-10x-all-in-one`.
1141
+ const workloadName = isAzure
1142
+ ? `${releaseName}-${RETRIEVER_CLUSTER_NAME}`
1143
+ : `${releaseName}-${RETRIEVER_CHART_NAME}-${RETRIEVER_CLUSTER_NAME}`;
1144
+ // `app.kubernetes.io/instance` is set by nothing in the chart. The pod
1145
+ // labels are `app=retriever-10x` and `cluster=all-in-one`.
1146
+ const podSelector = isAzure ? RETRIEVER_POD_SELECTOR : `app.kubernetes.io/instance=${releaseName}`;
1147
+ const containerFlag = isAzure ? ` -c ${RETRIEVER_CONTAINER}` : '';
805
1148
  probes.push({
806
1149
  name: 'pods-ready',
807
1150
  question: 'Are indexer + query-handler + stream-worker pods Ready?',
808
1151
  commands: [
809
- `kubectl -n ${namespace} wait --for=condition=Ready pod -l app.kubernetes.io/instance=${releaseName} --timeout=10m`,
1152
+ `kubectl -n ${namespace} wait --for=condition=Ready pod -l ${podSelector} --timeout=10m`,
810
1153
  ],
811
1154
  expectOutput: 'condition met',
812
1155
  timeoutSec: 600,
813
1156
  });
814
- probes.push({
815
- name: 'indexer-healthy',
816
- question: 'Is the indexer processing messages from the index queue?',
817
- commands: [
818
- `kubectl -n ${namespace} logs -l app.kubernetes.io/instance=${releaseName},app.kubernetes.io/component=indexer --tail=200 | grep -iE 'index|processed|bloom' | head -20`,
819
- ],
820
- timeoutSec: 120,
821
- });
1157
+ if (isAzure) {
1158
+ // Two failures, two probes.
1159
+ //
1160
+ // 1. `grep -iE 'index'` matched class names such as `IndexQueryWriter`, so
1161
+ // the probe printed lines whether or not a single blob had been
1162
+ // indexed. `index written` is the line the indexer logs per index
1163
+ // object.
1164
+ // 2. `--tail=200` then hid that line. The marker is written once per index
1165
+ // object, and the query the operator ran to prove the install pushed it
1166
+ // out of the tail: a live pod held 1164 lines with exactly one `index
1167
+ // written` at line 120, so the probe printed nothing and reported a
1168
+ // healthy indexer as one that had indexed nothing. `--tail=-1` is
1169
+ // kubectl's "every retained line", and it has to be passed explicitly
1170
+ // because a label selector drops the default to 10. The bound that
1171
+ // matters for output size is `head -20`, which stays. `--since` was the
1172
+ // other candidate and carries the same defect on a different axis: an
1173
+ // install verified an hour after the upload has the marker outside any
1174
+ // window short enough to be worth writing down.
1175
+ //
1176
+ // The `AADSTS` half moves to a probe of its own. One grep over both
1177
+ // patterns answers two questions at once, and `expectOutput` cannot say
1178
+ // "the first pattern, not the second". Split, the index probe grades on
1179
+ // `expectOutput`, so an empty run fails rather than passing on the exit 0
1180
+ // that `head` hands back whatever grep matched.
1181
+ probes.push({
1182
+ name: 'indexer-healthy',
1183
+ question: 'Has the indexer written an index object? A healthy run prints one `index written` line per index ' +
1184
+ 'object. The command reads every retained line of the pod log, so empty output means the pod has ' +
1185
+ 'written no index object since its log last rotated.',
1186
+ commands: [
1187
+ `kubectl -n ${namespace} logs -l ${podSelector}${containerFlag} --tail=-1 | grep -E 'index written' | head -20`,
1188
+ ],
1189
+ expectOutput: 'index written',
1190
+ timeoutSec: 120,
1191
+ });
1192
+ probes.push({
1193
+ name: 'indexer-token-refusals',
1194
+ question: 'How many AADSTS token refusals does the pod log carry? The count runs over every retained line ' +
1195
+ 'rather than a bounded tail, which reports zero once the refusals scroll past the window. A healthy ' +
1196
+ 'install answers 0.',
1197
+ commands: [
1198
+ `kubectl -n ${namespace} logs -l ${podSelector}${containerFlag} --tail=-1 | grep -c AADSTS || true`,
1199
+ ],
1200
+ expectOutput: '^0$',
1201
+ timeoutSec: 120,
1202
+ });
1203
+ }
1204
+ else {
1205
+ probes.push({
1206
+ name: 'indexer-healthy',
1207
+ question: 'Is the indexer processing messages from the index queue?',
1208
+ commands: [
1209
+ `kubectl -n ${namespace} logs -l ${podSelector},app.kubernetes.io/component=indexer --tail=200 | grep -iE 'index|processed|bloom' | head -20`,
1210
+ ],
1211
+ timeoutSec: 120,
1212
+ });
1213
+ }
1214
+ if (isAzure) {
1215
+ // The failure this probe exists for: the ServiceAccount name has to equal
1216
+ // the subject of the federated credential, or the pod runs and every call
1217
+ // to Blob and the queues comes back AADSTS700213.
1218
+ probes.push({
1219
+ name: 'workload-identity-binding',
1220
+ question: `Does the ServiceAccount \`${releaseName}\` carry the managed-identity client id, and how many ` +
1221
+ 'AADSTS700213 refusals does the pod log carry? The first command prints the client id, the second ' +
1222
+ 'counts the refusals over every retained line, and a healthy install answers with an id and a zero.',
1223
+ commands: [
1224
+ `kubectl -n ${namespace} get sa ${releaseName} -o jsonpath='{.metadata.annotations.azure\\.workload\\.identity/client-id}{"\\n"}'`,
1225
+ // Same bounded-tail defect as the indexer probe, pointing the other
1226
+ // way: a `--tail=200` count reports 0 once the refusals scroll past
1227
+ // the window, which reads as a pass.
1228
+ `kubectl -n ${namespace} logs -l ${podSelector}${containerFlag} --tail=-1 | grep -c AADSTS700213 || true`,
1229
+ ],
1230
+ });
1231
+ }
822
1232
  probes.push({
823
1233
  name: 'query-endpoint-healthy',
824
1234
  question: 'Is the query endpoint responding?',
825
- commands: [
826
- `kubectl -n ${namespace} get ingress,svc -l app.kubernetes.io/instance=${releaseName}`,
827
- ],
1235
+ commands: isAzure
1236
+ ? [`kubectl -n ${namespace} get svc ${workloadName}`]
1237
+ : [`kubectl -n ${namespace} get ingress,svc -l app.kubernetes.io/instance=${releaseName}`],
828
1238
  });
829
1239
  // Retriever Service external-access probe.
830
1240
  // The helm chart defaults to ClusterIP. The MCP server (and CLI) run on
@@ -843,16 +1253,58 @@ receiverNamespace) {
843
1253
  probes.push({
844
1254
  name: 'retriever-service-accessibility',
845
1255
  question: accessQuestion,
846
- commands: [
847
- // Use jq when available; fall back to plain kubectl wide output.
848
- // The agent reads the type= field from the output to determine if
849
- // external access guidance applies.
850
- `kubectl -n ${namespace} get svc -l app.kubernetes.io/instance=${releaseName} -o json 2>/dev/null` +
851
- ` | jq -r '.items[] | "Service \\(.metadata.name): type=\\(.spec.type) port=\\(.spec.ports[0].port // "?")"'` +
852
- ` 2>/dev/null` +
853
- ` || kubectl -n ${namespace} get svc -l app.kubernetes.io/instance=${releaseName} -o wide`,
854
- ],
1256
+ commands: isAzure
1257
+ ? [
1258
+ // The chart puts no `app` label on the Service object itself, only
1259
+ // on the pods the Service selects, so the Service is addressed by
1260
+ // name here rather than by label.
1261
+ `kubectl -n ${namespace} get svc ${workloadName} -o json 2>/dev/null` +
1262
+ ` | jq -r '"Service \\(.metadata.name): type=\\(.spec.type) port=\\(.spec.ports[0].port // "?")"'` +
1263
+ ` 2>/dev/null` +
1264
+ ` || kubectl -n ${namespace} get svc ${workloadName} -o wide`,
1265
+ ]
1266
+ : [
1267
+ // Use jq when available; fall back to plain kubectl wide output.
1268
+ // The agent reads the type= field from the output to determine if
1269
+ // external access guidance applies.
1270
+ `kubectl -n ${namespace} get svc -l app.kubernetes.io/instance=${releaseName} -o json 2>/dev/null` +
1271
+ ` | jq -r '.items[] | "Service \\(.metadata.name): type=\\(.spec.type) port=\\(.spec.ports[0].port // "?")"'` +
1272
+ ` 2>/dev/null` +
1273
+ ` || kubectl -n ${namespace} get svc -l app.kubernetes.io/instance=${releaseName} -o wide`,
1274
+ ],
855
1275
  });
1276
+ if (isAzure) {
1277
+ const account = opts.storageAccount ?? '<storage-account>';
1278
+ const { container: indexContainerName, path: indexPath } = splitAzureIndexContainer(opts.indexContainer ?? `${account}/tenx-index/tenx`);
1279
+ if (opts.inputBucket) {
1280
+ probes.push({
1281
+ name: 'blob-input',
1282
+ question: 'Does the input container hold blobs under the app prefix?',
1283
+ commands: [
1284
+ `az storage blob list \\\n --account-name ${account} \\\n --auth-mode login \\\n -c ${opts.inputBucket} \\\n --prefix app/ \\\n --query "[].name" -o tsv | head -5`,
1285
+ ],
1286
+ });
1287
+ }
1288
+ probes.push({
1289
+ name: 'blob-index-written',
1290
+ question: 'Is the indexer writing the index into the index container?',
1291
+ commands: [
1292
+ `az storage blob list \\\n --account-name ${account} \\\n --auth-mode login \\\n -c ${indexContainerName} \\\n --prefix ${indexPath}/ \\\n --query "[].name" -o tsv | head -5`,
1293
+ ],
1294
+ });
1295
+ if (opts.azureQueues) {
1296
+ probes.push({
1297
+ name: 'storage-queue-drainage',
1298
+ question: 'Is the index queue being drained (messages not piling up)?',
1299
+ commands: [
1300
+ `az storage message peek \\\n --account-name ${account} \\\n --auth-mode login \\\n --queue-name ${opts.azureQueues.index} \\\n --num-messages 32 \\\n --query "length(@)" -o tsv`,
1301
+ ],
1302
+ });
1303
+ }
1304
+ }
1305
+ const inputBucket = isAzure ? undefined : opts.inputBucket;
1306
+ const indexQueueUrl = isAzure ? undefined : opts.indexQueueUrl;
1307
+ const receiverNamespace = isAzure ? undefined : opts.receiverNamespace;
856
1308
  if (inputBucket) {
857
1309
  // Write side of the loop, checked FIRST: is the forwarder actually
858
1310
  // offloading the dropped slice into the source bucket? Without this, an
@@ -883,7 +1335,9 @@ receiverNamespace) {
883
1335
  ],
884
1336
  });
885
1337
  }
886
- // Fix 88 — Receiver outputOffload capability probe.
1338
+ // Fix 88 — Receiver outputOffload capability probe. AWS only: the recipe it
1339
+ // points at writes to S3, and offload delivery into Blob has no recipe, so
1340
+ // on Azure the probe would ask for a state no config can reach.
887
1341
  // If a Receiver is installed, the user may be running it with the rate
888
1342
  // regulator only (soft-drop / sample) rather than the outputOffload mode
889
1343
  // that actually routes bytes to S3 for the Retriever to index. Without
@@ -901,36 +1355,67 @@ receiverNamespace) {
901
1355
  }
902
1356
  return probes;
903
1357
  }
904
- function buildTeardownSteps(releaseName, namespace) {
905
- return [
1358
+ function buildTeardownSteps(releaseName, namespace, storageProvider = 'aws', resourceGroup) {
1359
+ const isAzure = storageProvider === 'azure';
1360
+ const selector = isAzure
1361
+ ? RETRIEVER_POD_SELECTOR
1362
+ : `app.kubernetes.io/instance=${releaseName}`;
1363
+ const workloadName = isAzure
1364
+ ? `${releaseName}-${RETRIEVER_CLUSTER_NAME}`
1365
+ : `${releaseName}-${RETRIEVER_CHART_NAME}-${RETRIEVER_CLUSTER_NAME}`;
1366
+ const steps = [
906
1367
  {
907
1368
  title: 'Uninstall the Helm release',
908
- rationale: 'Removes indexer, query-handler, stream-worker deployments, filter CronJobs, ConfigMaps, and the chart-created ServiceAccount. LEAVES AWS infra (S3, SQS, IAM role) intact — that lifecycle belongs to Terraform.',
1369
+ rationale: isAzure
1370
+ ? 'Removes the Deployment, the Service, the ConfigMaps and the chart-created ServiceAccount. The ' +
1371
+ 'storage account, the containers, the queues, the managed identity and the AKS cluster stay: the ' +
1372
+ 'last step below is what deletes those.'
1373
+ : 'Removes indexer, query-handler, stream-worker deployments, filter CronJobs, ConfigMaps, and the chart-created ServiceAccount. LEAVES AWS infra (S3, SQS, IAM role) intact — that lifecycle belongs to Terraform.',
909
1374
  commands: [`helm -n ${namespace} uninstall ${releaseName}`],
910
1375
  },
911
1376
  {
912
1377
  title: 'Clean up derived resources',
913
1378
  rationale: 'Helm does not reap PVCs or Secrets created outside the release.',
914
- commands: [
915
- `kubectl -n ${namespace} delete pvc -l app.kubernetes.io/instance=${releaseName} --ignore-not-found`,
916
- ],
1379
+ commands: [`kubectl -n ${namespace} delete pvc -l ${selector} --ignore-not-found`],
917
1380
  },
918
1381
  {
919
1382
  title: 'Verify nothing remains',
920
- rationale: 'Confirm no workloads are lingering under the release label.',
921
- commands: [
922
- `kubectl -n ${namespace} get all,configmap,secret,pvc -l app.kubernetes.io/instance=${releaseName}`,
923
- ],
1383
+ rationale: isAzure
1384
+ ? 'Confirms no workload is lingering. The pod label and the Service name are checked separately: the ' +
1385
+ 'chart labels the pods and names the Service, and the Service object carries no `app` label.'
1386
+ : 'Confirm no workloads are lingering under the release label.',
1387
+ commands: isAzure
1388
+ ? [
1389
+ `kubectl -n ${namespace} get all,configmap,secret,pvc -l ${selector}`,
1390
+ `kubectl -n ${namespace} get svc ${workloadName} --ignore-not-found`,
1391
+ ]
1392
+ : [`kubectl -n ${namespace} get all,configmap,secret,pvc -l ${selector}`],
924
1393
  },
925
- {
1394
+ ];
1395
+ if (isAzure) {
1396
+ // No Terraform state exists on this path. The provisioning script created
1397
+ // the resource group and its own `--destroy` deletes it, which is the only
1398
+ // command that stops the AKS cluster and the storage account billing.
1399
+ steps.push({
1400
+ title: 'Delete the Azure resources',
1401
+ rationale: 'One command deletes the resource group and everything the provisioning script put in it: the storage ' +
1402
+ 'account with both containers, the four Storage Queues, the Event Grid subscription, the managed ' +
1403
+ 'identity and the AKS cluster. Skipping it leaves a running AKS cluster and a storage account billing.',
1404
+ commands: [buildAzureTeardownCommand(resourceGroup ?? '<resource-group>')],
1405
+ expectDurationSec: 600,
1406
+ });
1407
+ }
1408
+ else {
1409
+ steps.push({
926
1410
  title: '(Optional) teardown AWS infra',
927
1411
  rationale: 'If you\'re fully removing the Retriever, tear down the Terraform module that created the S3 buckets, SQS queues, and IAM role. Skipping this leaves empty AWS infra behind (zero-cost for SQS idle, pennies for S3 storage).',
928
1412
  commands: [
929
1413
  `# From your terraform directory:`,
930
1414
  `# terraform destroy -target=module.retriever_aws_infra`,
931
1415
  ],
932
- },
933
- ];
1416
+ });
1417
+ }
1418
+ return steps;
934
1419
  }
935
1420
  // ── Fix 82: resolve the input bucket from the installed Helm release values ──
936
1421
  /**