openmates 0.24.0-alpha.0 → 0.24.0-alpha.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{chunk-2KYUZC7W.js → chunk-VEQ3R7MK.js} +81 -8
- package/dist/cli.js +1 -1
- package/dist/index.js +1 -1
- package/package.json +1 -1
- package/templates/core/docker-compose.selfhost.yml +2 -2
- package/templates/core/monitoring/prometheus/alert_rules.yml +164 -0
- package/templates/core/monitoring/prometheus/prometheus.yml +96 -0
- package/templates/upload/docker-compose.yml +4 -1
|
@@ -32710,6 +32710,7 @@ var RUNTIME_CHECKS = {
|
|
|
32710
32710
|
{ id: "http.role_health", required: true, timeoutSeconds: 10 },
|
|
32711
32711
|
{ id: "core.database", required: true, timeoutSeconds: 10 },
|
|
32712
32712
|
{ id: "core.cache", required: true, timeoutSeconds: 10 },
|
|
32713
|
+
{ id: "core.cms_cache_consistency", required: true, timeoutSeconds: 10 },
|
|
32713
32714
|
{ id: "core.vault", required: true, timeoutSeconds: 10 },
|
|
32714
32715
|
{ id: "core.worker_queue", required: true, timeoutSeconds: CELERY_PROBE_CHECK_TIMEOUT_SECONDS },
|
|
32715
32716
|
{ id: "core.scheduler_freshness", required: true, timeoutSeconds: CELERY_PROBE_CHECK_TIMEOUT_SECONDS },
|
|
@@ -32981,6 +32982,19 @@ function planRestore(input) {
|
|
|
32981
32982
|
steps: input.yes === true ? ["stop", "restore", "start", "health-check"] : ["confirm", "stop", "restore", "start", "health-check"]
|
|
32982
32983
|
};
|
|
32983
32984
|
}
|
|
32985
|
+
function resolveTemplateSource(input) {
|
|
32986
|
+
const role = parseServerRole(input.role);
|
|
32987
|
+
const definition = ROLE_DEFINITIONS[role];
|
|
32988
|
+
if (input.templateUrl) return { type: "url", url: input.templateUrl };
|
|
32989
|
+
if (input.packagedTemplateExists && input.packageVersion && input.imageTag === `v${input.packageVersion}`) {
|
|
32990
|
+
return { type: "packaged", path: definition.templatePath };
|
|
32991
|
+
}
|
|
32992
|
+
return {
|
|
32993
|
+
type: "github-raw",
|
|
32994
|
+
ref: input.templateRef ?? "dev",
|
|
32995
|
+
path: `frontend/packages/openmates-cli/${definition.templatePath}`
|
|
32996
|
+
};
|
|
32997
|
+
}
|
|
32984
32998
|
function parseSecretEnvKey(envKey) {
|
|
32985
32999
|
if (!envKey.startsWith("SECRET__")) return null;
|
|
32986
33000
|
const parts = envKey.slice("SECRET__".length).split("__", 2);
|
|
@@ -33991,6 +34005,18 @@ async function deliverRuntimeNotification(config, payload) {
|
|
|
33991
34005
|
}
|
|
33992
34006
|
return Promise.all(deliveries);
|
|
33993
34007
|
}
|
|
34008
|
+
var CMS_CACHE_INSPECT_FORMAT = '{{range .Config.Env}}{{if eq . "CACHE_SKIP_ALLOWED=true"}}CACHE_SKIP_ALLOWED=true,{{end}}{{if eq . "CACHE_AUTO_PURGE=true"}}CACHE_AUTO_PURGE=true,{{end}}{{end}}';
|
|
34009
|
+
function evaluateCmsCacheConsistency(input) {
|
|
34010
|
+
const result = { id: "core.cms_cache_consistency", required: true, duration_ms: 0 };
|
|
34011
|
+
if (!input.containerFound || !input.inspectionSucceeded) {
|
|
34012
|
+
return { ...result, status: "failed", failureClass: "configuration", sanitized_reason: "cms_cache_inspection_unavailable" };
|
|
34013
|
+
}
|
|
34014
|
+
const flags = new Set(input.filteredEnvironment.split(",").map((value) => value.trim()));
|
|
34015
|
+
if (!flags.has("CACHE_SKIP_ALLOWED=true") || !flags.has("CACHE_AUTO_PURGE=true")) {
|
|
34016
|
+
return { ...result, status: "failed", failureClass: "configuration", sanitized_reason: "cms_cache_flags_missing_or_disabled" };
|
|
34017
|
+
}
|
|
34018
|
+
return { ...result, status: "passed" };
|
|
34019
|
+
}
|
|
33994
34020
|
function applyRuntimeCheckResults(state, results, timestamp) {
|
|
33995
34021
|
const current = state ?? initialRuntimeIncidentState();
|
|
33996
34022
|
const checks = { ...current.checks ?? {} };
|
|
@@ -34369,6 +34395,7 @@ var ROLE_TEMPLATE_FILES = {
|
|
|
34369
34395
|
var CORE_NO_WEBAPP_TEMPLATE_FILE = join15("core", "docker-compose.no-webapp.yml");
|
|
34370
34396
|
var CORE_PROMTAIL_CONFIG_FILE = join15("backend", "core", "monitoring", "promtail", "promtail-config.yaml");
|
|
34371
34397
|
var CORE_ALERTMANAGER_CONFIG_FILE = join15("backend", "core", "monitoring", "alertmanager", "alertmanager.yml");
|
|
34398
|
+
var CORE_PROMETHEUS_CONFIG_DIR = join15("backend", "core", "monitoring", "prometheus");
|
|
34372
34399
|
var COMPOSE_OVERRIDE2 = join15("backend", "core", "docker-compose.override.yml");
|
|
34373
34400
|
var DEFAULT_INSTALL_PATH = join15(homedir6(), "openmates");
|
|
34374
34401
|
var REPO_URL = "https://github.com/glowingkitty/OpenMates.git";
|
|
@@ -34931,6 +34958,19 @@ function packagedCaddyTemplatePath(role) {
|
|
|
34931
34958
|
function packagedCoreAlertmanagerTemplatePath() {
|
|
34932
34959
|
return join15(dirname8(new URL(import.meta.url).pathname), "..", "templates", "core", "monitoring", "alertmanager", "alertmanager.yml");
|
|
34933
34960
|
}
|
|
34961
|
+
function ensureCorePrometheusRuntimeFiles(installPath) {
|
|
34962
|
+
const packagedDir = join15(dirname8(new URL(import.meta.url).pathname), "..", "templates", "core", "monitoring", "prometheus");
|
|
34963
|
+
const runtimeDir = join15(installPath, CORE_PROMETHEUS_CONFIG_DIR);
|
|
34964
|
+
for (const fileName of ["prometheus.yml", "alert_rules.yml"]) {
|
|
34965
|
+
const packagedPath = join15(packagedDir, fileName);
|
|
34966
|
+
if (!existsSync13(packagedPath)) throw new Error(`Packaged Prometheus file not found: ${packagedPath}`);
|
|
34967
|
+
}
|
|
34968
|
+
mkdirSync13(runtimeDir, { recursive: true });
|
|
34969
|
+
for (const fileName of ["prometheus.yml", "alert_rules.yml"]) {
|
|
34970
|
+
const runtimePath = join15(runtimeDir, fileName);
|
|
34971
|
+
if (!existsSync13(runtimePath)) copyFileSync2(join15(packagedDir, fileName), runtimePath);
|
|
34972
|
+
}
|
|
34973
|
+
}
|
|
34934
34974
|
function readOfficialCloudNoWebappComposeTemplate() {
|
|
34935
34975
|
const packaged = packagedNoWebappTemplatePath();
|
|
34936
34976
|
if (existsSync13(packaged)) return readFileSync14(packaged, "utf-8");
|
|
@@ -34957,29 +34997,35 @@ function fileHash(path2) {
|
|
|
34957
34997
|
if (!existsSync13(path2)) return null;
|
|
34958
34998
|
return createHash18("sha256").update(readFileSync14(path2)).digest("hex");
|
|
34959
34999
|
}
|
|
34960
|
-
async function loadSelfHostComposeTemplate(templateRef, role) {
|
|
35000
|
+
async function loadSelfHostComposeTemplate(templateRef, role, imageTag, packageVersion = getPackageVersion()) {
|
|
34961
35001
|
const templateDir = process.env.OPENMATES_SELFHOST_TEMPLATE_DIR;
|
|
34962
35002
|
if (templateDir) {
|
|
34963
35003
|
return readFileSync14(join15(resolve7(templateDir), ROLE_TEMPLATE_FILES[role]), "utf-8");
|
|
34964
35004
|
}
|
|
34965
|
-
const overrideUrl = process.env.OPENMATES_SELFHOST_COMPOSE_URL;
|
|
34966
|
-
if (overrideUrl) {
|
|
34967
|
-
return fetchText(overrideUrl);
|
|
34968
|
-
}
|
|
34969
35005
|
const packaged = packagedTemplatePath(role);
|
|
34970
|
-
|
|
34971
|
-
|
|
35006
|
+
const source = resolveTemplateSource({
|
|
35007
|
+
role,
|
|
35008
|
+
packagedTemplateExists: existsSync13(packaged),
|
|
35009
|
+
templateUrl: process.env.OPENMATES_SELFHOST_COMPOSE_URL,
|
|
35010
|
+
templateRef,
|
|
35011
|
+
imageTag,
|
|
35012
|
+
packageVersion
|
|
35013
|
+
});
|
|
35014
|
+
if (source.type === "packaged") return readFileSync14(packaged, "utf-8");
|
|
35015
|
+
if (source.type === "url") return fetchText(source.url);
|
|
35016
|
+
return fetchText(`https://raw.githubusercontent.com/glowingkitty/OpenMates/${source.ref}/${source.path}`);
|
|
34972
35017
|
}
|
|
34973
35018
|
async function writeImageModeRuntimeFiles(installPath, imageTag, role) {
|
|
34974
35019
|
const roleDir = join15(installPath, "backend", role === "core" ? "core" : role);
|
|
34975
35020
|
const vaultConfigDir = join15(roleDir, "vault", "config");
|
|
34976
35021
|
mkdirSync13(vaultConfigDir, { recursive: true });
|
|
34977
35022
|
mkdirSync13(join15(installPath, "config", "providers"), { recursive: true });
|
|
34978
|
-
writeFileSync8(join15(installPath, ROLE_IMAGE_COMPOSE_FILES[role]), await loadSelfHostComposeTemplate(templateRefForImageTag(imageTag, getPackageVersion()), role));
|
|
35023
|
+
writeFileSync8(join15(installPath, ROLE_IMAGE_COMPOSE_FILES[role]), await loadSelfHostComposeTemplate(templateRefForImageTag(imageTag, getPackageVersion()), role, imageTag));
|
|
34979
35024
|
if (role === "core") {
|
|
34980
35025
|
writeFileSync8(join15(installPath, OFFICIAL_CLOUD_NO_WEBAPP_COMPOSE_FILE), readOfficialCloudNoWebappComposeTemplate());
|
|
34981
35026
|
}
|
|
34982
35027
|
if (role === "core") {
|
|
35028
|
+
ensureCorePrometheusRuntimeFiles(installPath);
|
|
34983
35029
|
const promtailConfigPath = join15(installPath, CORE_PROMTAIL_CONFIG_FILE);
|
|
34984
35030
|
mkdirSync13(dirname8(promtailConfigPath), { recursive: true });
|
|
34985
35031
|
writeFileSync8(promtailConfigPath, SELFHOST_PROMTAIL_CONFIG_TEMPLATE);
|
|
@@ -36259,6 +36305,28 @@ async function installContinuousUpdateService(flags) {
|
|
|
36259
36305
|
}
|
|
36260
36306
|
console.log(`Installed ${plan.timerName}.`);
|
|
36261
36307
|
}
|
|
36308
|
+
function inspectCmsCacheConsistency(installPath, withOverrides, installMode) {
|
|
36309
|
+
const composeResult = spawnSync(
|
|
36310
|
+
"docker",
|
|
36311
|
+
[...composeArgs(installPath, withOverrides, installMode, "core"), "ps", "-q", "cms"],
|
|
36312
|
+
{ cwd: installPath, encoding: "utf-8", timeout: 5e3 }
|
|
36313
|
+
);
|
|
36314
|
+
const containerIds = composeResult.status === 0 ? composeResult.stdout.trim().split(/\s+/).filter(Boolean) : [];
|
|
36315
|
+
const containerId = containerIds.length === 1 ? containerIds[0] : void 0;
|
|
36316
|
+
if (!containerId) {
|
|
36317
|
+
return evaluateCmsCacheConsistency({ containerFound: false, inspectionSucceeded: false, filteredEnvironment: "" });
|
|
36318
|
+
}
|
|
36319
|
+
const inspectResult = spawnSync(
|
|
36320
|
+
"docker",
|
|
36321
|
+
["inspect", "--format", CMS_CACHE_INSPECT_FORMAT, containerId],
|
|
36322
|
+
{ cwd: installPath, encoding: "utf-8", timeout: 5e3 }
|
|
36323
|
+
);
|
|
36324
|
+
return evaluateCmsCacheConsistency({
|
|
36325
|
+
containerFound: true,
|
|
36326
|
+
inspectionSucceeded: inspectResult.status === 0,
|
|
36327
|
+
filteredEnvironment: inspectResult.status === 0 ? inspectResult.stdout : ""
|
|
36328
|
+
});
|
|
36329
|
+
}
|
|
36262
36330
|
function runRuntimeVerification(installPath, role, config) {
|
|
36263
36331
|
const envText = existsSync13(join15(installPath, ".env")) ? readFileSync14(join15(installPath, ".env"), "utf-8") : "";
|
|
36264
36332
|
const mode = resolveRuntimeDeploymentMode({
|
|
@@ -36285,6 +36353,11 @@ function runRuntimeVerification(installPath, role, config) {
|
|
|
36285
36353
|
throw new Error(`Runtime verifier returned invalid output (${result.status ?? "unknown"}).`);
|
|
36286
36354
|
}
|
|
36287
36355
|
output.checks = output.checks.map((check) => ({ ...check, failureClass: check.failureClass ?? check.failure_class }));
|
|
36356
|
+
if (role === "core") {
|
|
36357
|
+
const cmsCacheCheck = inspectCmsCacheConsistency(installPath, withOverrides, installMode);
|
|
36358
|
+
output.checks.push(cmsCacheCheck);
|
|
36359
|
+
if (cmsCacheCheck.status !== "passed") output.status = "failed";
|
|
36360
|
+
}
|
|
36288
36361
|
if (mode.effectiveMode === "official_cloud") {
|
|
36289
36362
|
const destinations = runtimeNotificationConfig(installPath);
|
|
36290
36363
|
const configuredCount = [destinations.email, destinations.discordWebhookUrl, destinations.genericWebhook].filter(Boolean).length;
|
package/dist/cli.js
CHANGED
package/dist/index.js
CHANGED
package/package.json
CHANGED
|
@@ -232,12 +232,12 @@ services:
|
|
|
232
232
|
mem_limit: ${CORE_WORKER_MEMORY_LIMIT:-3g}
|
|
233
233
|
environment:
|
|
234
234
|
<<: *openmates-worker-env
|
|
235
|
-
CELERY_QUEUES: persistence,health_check,server_stats,demo,e2e_tests,push
|
|
235
|
+
CELERY_QUEUES: persistence,health_check,server_stats,demo,e2e_tests,push,leaderboard
|
|
236
236
|
CELERY_AUTOSCALE_MAX: ${CORE_WORKER_CONCURRENCY:-${CELERY_AUTOSCALE_MAX:-3}}
|
|
237
237
|
CELERY_AUTOSCALE_MIN: ${CELERY_AUTOSCALE_MIN:-1}
|
|
238
238
|
CELERY_METRICS_PORT: "9109"
|
|
239
239
|
command: >
|
|
240
|
-
sh -c "chown -R celeryuser:celeryuser /vault-data && gosu celeryuser python -m celery -A backend.core.api.app.tasks.celery_config worker --loglevel=info --queues=persistence,health_check,server_stats,demo,e2e_tests,push --concurrency=$${CELERY_AUTOSCALE_MAX} --max-tasks-per-child=50 --max-memory-per-child=600000 --prefetch-multiplier=1"
|
|
240
|
+
sh -c "chown -R celeryuser:celeryuser /vault-data && gosu celeryuser python -m celery -A backend.core.api.app.tasks.celery_config worker --loglevel=info --queues=persistence,health_check,server_stats,demo,e2e_tests,push,leaderboard --concurrency=$${CELERY_AUTOSCALE_MAX} --max-tasks-per-child=50 --max-memory-per-child=600000 --prefetch-multiplier=1"
|
|
241
241
|
|
|
242
242
|
user-tasks-worker:
|
|
243
243
|
<<: *openmates-worker-base
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
# Prometheus alert rules for OpenMates.
|
|
2
|
+
# These fire when monitored conditions breach thresholds, and Alertmanager
|
|
3
|
+
# routes the resulting notifications (with throttling to prevent email floods).
|
|
4
|
+
#
|
|
5
|
+
# Metric names used here must match what the API actually exposes.
|
|
6
|
+
# API metrics are defined in backend/core/api/app/services/metrics.py:
|
|
7
|
+
# - api_requests_total (Counter, labels: method, endpoint, status_code)
|
|
8
|
+
# - api_request_duration_seconds (Histogram, labels: method, endpoint)
|
|
9
|
+
# - celery_task_failures_total (Counter, labels: task_name, queue) — added in celery_config.py
|
|
10
|
+
#
|
|
11
|
+
# Architecture context: See docs/architecture/logging-and-monitoring.md
|
|
12
|
+
|
|
13
|
+
groups:
|
|
14
|
+
# -------------------------------------------------------------------------
|
|
15
|
+
# API health
|
|
16
|
+
# -------------------------------------------------------------------------
|
|
17
|
+
- name: api_alerts
|
|
18
|
+
rules:
|
|
19
|
+
- alert: HighErrorRate
|
|
20
|
+
expr: |
|
|
21
|
+
(
|
|
22
|
+
sum(rate(api_requests_total{status_code=~"5.."}[5m]))
|
|
23
|
+
/
|
|
24
|
+
sum(rate(api_requests_total[5m]))
|
|
25
|
+
) > 0.05
|
|
26
|
+
for: 5m
|
|
27
|
+
labels:
|
|
28
|
+
severity: critical
|
|
29
|
+
annotations:
|
|
30
|
+
summary: "API 5xx error rate above 5%"
|
|
31
|
+
description: >-
|
|
32
|
+
Over the last 5 minutes, {{ $value | humanizePercentage }} of
|
|
33
|
+
requests returned 5xx errors.
|
|
34
|
+
|
|
35
|
+
- alert: HighRequestLatency
|
|
36
|
+
expr: |
|
|
37
|
+
histogram_quantile(0.95,
|
|
38
|
+
sum(rate(api_request_duration_seconds_bucket[5m])) by (le)
|
|
39
|
+
) > 5
|
|
40
|
+
for: 10m
|
|
41
|
+
labels:
|
|
42
|
+
severity: warning
|
|
43
|
+
annotations:
|
|
44
|
+
summary: "P95 request latency above 5 s"
|
|
45
|
+
description: >-
|
|
46
|
+
P95 latency is {{ $value | humanizeDuration }}.
|
|
47
|
+
|
|
48
|
+
- alert: APIDown
|
|
49
|
+
expr: up{job="api"} == 0
|
|
50
|
+
for: 2m
|
|
51
|
+
labels:
|
|
52
|
+
severity: critical
|
|
53
|
+
annotations:
|
|
54
|
+
summary: "API container is down"
|
|
55
|
+
description: "Prometheus cannot scrape the API metrics endpoint."
|
|
56
|
+
|
|
57
|
+
# -------------------------------------------------------------------------
|
|
58
|
+
# Celery task health
|
|
59
|
+
# Requires celery_task_failures_total counter — added in celery_config.py
|
|
60
|
+
# worker_process_init signal starts a prometheus_client HTTP server per worker.
|
|
61
|
+
# -------------------------------------------------------------------------
|
|
62
|
+
- name: celery_alerts
|
|
63
|
+
rules:
|
|
64
|
+
- alert: TaskFailureSpike
|
|
65
|
+
expr: |
|
|
66
|
+
sum(increase(celery_task_failures_total[15m])) > 10
|
|
67
|
+
for: 5m
|
|
68
|
+
labels:
|
|
69
|
+
severity: warning
|
|
70
|
+
annotations:
|
|
71
|
+
summary: "More than 10 Celery task failures in 15 min"
|
|
72
|
+
description: >-
|
|
73
|
+
{{ $value }} task failures detected in the last 15 minutes.
|
|
74
|
+
|
|
75
|
+
# -------------------------------------------------------------------------
|
|
76
|
+
# Container / resource health (cAdvisor)
|
|
77
|
+
# -------------------------------------------------------------------------
|
|
78
|
+
- name: container_alerts
|
|
79
|
+
rules:
|
|
80
|
+
- alert: ContainerRestarting
|
|
81
|
+
expr: |
|
|
82
|
+
increase(container_restart_count[30m]) > 3
|
|
83
|
+
for: 5m
|
|
84
|
+
labels:
|
|
85
|
+
severity: warning
|
|
86
|
+
annotations:
|
|
87
|
+
summary: "Container {{ $labels.name }} restarted >3 times in 30 min"
|
|
88
|
+
description: >-
|
|
89
|
+
Container {{ $labels.name }} has restarted {{ $value }} times.
|
|
90
|
+
|
|
91
|
+
- alert: HighMemoryUsage
|
|
92
|
+
expr: |
|
|
93
|
+
(
|
|
94
|
+
container_memory_usage_bytes{name!="",container_label_com_docker_compose_project="openmates"}
|
|
95
|
+
/
|
|
96
|
+
container_spec_memory_limit_bytes{name!="",container_label_com_docker_compose_project="openmates"}
|
|
97
|
+
) > 0.90
|
|
98
|
+
and on(instance, id, name, environment, container_label_com_docker_compose_project)
|
|
99
|
+
container_spec_memory_limit_bytes{name!="",container_label_com_docker_compose_project="openmates"} > 0
|
|
100
|
+
for: 10m
|
|
101
|
+
labels:
|
|
102
|
+
severity: warning
|
|
103
|
+
annotations:
|
|
104
|
+
summary: "Container {{ $labels.name }} memory above 90%"
|
|
105
|
+
description: >-
|
|
106
|
+
Memory usage is at {{ $value | humanizePercentage }}.
|
|
107
|
+
|
|
108
|
+
- name: host_resource_alerts
|
|
109
|
+
rules:
|
|
110
|
+
- alert: HostDiskSpaceWarning
|
|
111
|
+
expr: |
|
|
112
|
+
node_filesystem_avail_bytes{mountpoint="/",fstype!~"tmpfs|overlay|squashfs"}
|
|
113
|
+
/
|
|
114
|
+
node_filesystem_size_bytes{mountpoint="/",fstype!~"tmpfs|overlay|squashfs"}
|
|
115
|
+
< 0.15
|
|
116
|
+
for: 15m
|
|
117
|
+
labels:
|
|
118
|
+
severity: warning
|
|
119
|
+
annotations:
|
|
120
|
+
summary: "Host disk usage above 85% on {{ $labels.mountpoint }}"
|
|
121
|
+
description: "Disk usage is above 85% for {{ $labels.mountpoint }}."
|
|
122
|
+
|
|
123
|
+
- alert: HostDiskSpaceCritical
|
|
124
|
+
expr: |
|
|
125
|
+
node_filesystem_avail_bytes{mountpoint="/",fstype!~"tmpfs|overlay|squashfs"}
|
|
126
|
+
/
|
|
127
|
+
node_filesystem_size_bytes{mountpoint="/",fstype!~"tmpfs|overlay|squashfs"}
|
|
128
|
+
< 0.05
|
|
129
|
+
for: 5m
|
|
130
|
+
labels:
|
|
131
|
+
severity: critical
|
|
132
|
+
annotations:
|
|
133
|
+
summary: "Host disk usage above 95% on {{ $labels.mountpoint }}"
|
|
134
|
+
description: "Disk usage is above 95% for {{ $labels.mountpoint }}."
|
|
135
|
+
|
|
136
|
+
- alert: OperationalReportStale
|
|
137
|
+
expr: |
|
|
138
|
+
(time() - operational_report_last_success_timestamp_seconds > 93600)
|
|
139
|
+
or
|
|
140
|
+
(
|
|
141
|
+
(time() - operational_report_monitoring_started_timestamp_seconds > 93600)
|
|
142
|
+
unless on(environment) operational_report_last_success_timestamp_seconds
|
|
143
|
+
)
|
|
144
|
+
for: 5m
|
|
145
|
+
labels:
|
|
146
|
+
severity: critical
|
|
147
|
+
annotations:
|
|
148
|
+
summary: "{{ $labels.environment }} operational report is stale"
|
|
149
|
+
description: "No accepted operational report has been recorded for more than 26 hours."
|
|
150
|
+
|
|
151
|
+
# -------------------------------------------------------------------------
|
|
152
|
+
# Prometheus self-monitoring
|
|
153
|
+
# -------------------------------------------------------------------------
|
|
154
|
+
- name: prometheus_self
|
|
155
|
+
rules:
|
|
156
|
+
- alert: PrometheusTargetDown
|
|
157
|
+
expr: up == 0
|
|
158
|
+
for: 5m
|
|
159
|
+
labels:
|
|
160
|
+
severity: critical
|
|
161
|
+
annotations:
|
|
162
|
+
summary: "Scrape target {{ $labels.job }} is down"
|
|
163
|
+
description: >-
|
|
164
|
+
Prometheus cannot reach {{ $labels.instance }} (job={{ $labels.job }}).
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
global:
|
|
2
|
+
scrape_interval: 30s
|
|
3
|
+
scrape_timeout: 10s
|
|
4
|
+
evaluation_interval: 30s
|
|
5
|
+
external_labels:
|
|
6
|
+
environment: "${SERVER_ENVIRONMENT}"
|
|
7
|
+
|
|
8
|
+
alerting:
|
|
9
|
+
alertmanagers:
|
|
10
|
+
- static_configs:
|
|
11
|
+
- targets: ["alertmanager:9093"]
|
|
12
|
+
scheme: http
|
|
13
|
+
timeout: 10s
|
|
14
|
+
|
|
15
|
+
rule_files:
|
|
16
|
+
- "/etc/prometheus/alert_rules.yml"
|
|
17
|
+
|
|
18
|
+
# Self-host image mode keeps metrics locally; remote write requires separately provisioned credentials.
|
|
19
|
+
scrape_configs:
|
|
20
|
+
- job_name: "prometheus"
|
|
21
|
+
static_configs:
|
|
22
|
+
- targets: ["localhost:9090"]
|
|
23
|
+
|
|
24
|
+
- job_name: "api"
|
|
25
|
+
metrics_path: /metrics
|
|
26
|
+
static_configs:
|
|
27
|
+
- targets: ["api:8000"]
|
|
28
|
+
labels:
|
|
29
|
+
environment: "${SERVER_ENVIRONMENT}"
|
|
30
|
+
|
|
31
|
+
- job_name: "cadvisor"
|
|
32
|
+
scrape_interval: 60s
|
|
33
|
+
scrape_timeout: 10s
|
|
34
|
+
static_configs:
|
|
35
|
+
- targets: ["cadvisor:8080"]
|
|
36
|
+
labels:
|
|
37
|
+
environment: "${SERVER_ENVIRONMENT}"
|
|
38
|
+
|
|
39
|
+
- job_name: "node"
|
|
40
|
+
scrape_interval: 60s
|
|
41
|
+
static_configs:
|
|
42
|
+
- targets: ["node-exporter:9100"]
|
|
43
|
+
labels:
|
|
44
|
+
environment: "${SERVER_ENVIRONMENT}"
|
|
45
|
+
|
|
46
|
+
# Celery worker Prometheus metrics servers.
|
|
47
|
+
# Each worker starts a prometheus_client HTTP server on CELERY_METRICS_PORT
|
|
48
|
+
# via the worker_process_init signal in celery_config.py.
|
|
49
|
+
# Port assignments (must match CELERY_METRICS_PORT in docker-compose.yml):
|
|
50
|
+
# task-worker: 9101
|
|
51
|
+
# user-init-worker: 9110
|
|
52
|
+
# core-worker: 9109
|
|
53
|
+
# user-tasks-worker: 9112
|
|
54
|
+
# reminder-worker: 9111
|
|
55
|
+
# app-ai-worker: 9102
|
|
56
|
+
# app-images-worker: 9103
|
|
57
|
+
# app-pdf-worker: 9104
|
|
58
|
+
- job_name: "celery-task-worker"
|
|
59
|
+
static_configs:
|
|
60
|
+
- targets: ["task-worker:9101"]
|
|
61
|
+
labels: {environment: "${SERVER_ENVIRONMENT}"}
|
|
62
|
+
|
|
63
|
+
- job_name: "celery-user-init-worker"
|
|
64
|
+
static_configs:
|
|
65
|
+
- targets: ["user-init-worker:9110"]
|
|
66
|
+
labels: {environment: "${SERVER_ENVIRONMENT}"}
|
|
67
|
+
|
|
68
|
+
- job_name: "celery-core-worker"
|
|
69
|
+
static_configs:
|
|
70
|
+
- targets: ["core-worker:9109"]
|
|
71
|
+
labels: {environment: "${SERVER_ENVIRONMENT}"}
|
|
72
|
+
|
|
73
|
+
- job_name: "celery-user-tasks-worker"
|
|
74
|
+
static_configs:
|
|
75
|
+
- targets: ["user-tasks-worker:9112"]
|
|
76
|
+
labels: {environment: "${SERVER_ENVIRONMENT}"}
|
|
77
|
+
|
|
78
|
+
- job_name: "celery-reminder-worker"
|
|
79
|
+
static_configs:
|
|
80
|
+
- targets: ["reminder-worker:9111"]
|
|
81
|
+
labels: {environment: "${SERVER_ENVIRONMENT}"}
|
|
82
|
+
|
|
83
|
+
- job_name: "celery-app-ai-worker"
|
|
84
|
+
static_configs:
|
|
85
|
+
- targets: ["app-ai-worker:9102"]
|
|
86
|
+
labels: {environment: "${SERVER_ENVIRONMENT}"}
|
|
87
|
+
|
|
88
|
+
- job_name: "celery-app-images-worker"
|
|
89
|
+
static_configs:
|
|
90
|
+
- targets: ["app-images-worker:9103"]
|
|
91
|
+
labels: {environment: "${SERVER_ENVIRONMENT}"}
|
|
92
|
+
|
|
93
|
+
- job_name: "celery-app-pdf-worker"
|
|
94
|
+
static_configs:
|
|
95
|
+
- targets: ["app-pdf-worker:9104"]
|
|
96
|
+
labels: {environment: "${SERVER_ENVIRONMENT}"}
|
|
@@ -21,6 +21,7 @@ services:
|
|
|
21
21
|
VAULT_AUTO_UNSEAL: "true"
|
|
22
22
|
volumes:
|
|
23
23
|
- vault-setup-data:/app/data
|
|
24
|
+
- vault-app-data:/app/app-data
|
|
24
25
|
networks: [uploads]
|
|
25
26
|
depends_on: [vault]
|
|
26
27
|
restart: on-failure
|
|
@@ -45,7 +46,7 @@ services:
|
|
|
45
46
|
CLAMAV_HOST: clamav
|
|
46
47
|
CLAMAV_PORT: 3310
|
|
47
48
|
volumes:
|
|
48
|
-
- vault-
|
|
49
|
+
- vault-app-data:/vault-data:ro
|
|
49
50
|
networks: [uploads]
|
|
50
51
|
depends_on: [clamav, vault-setup]
|
|
51
52
|
healthcheck:
|
|
@@ -81,3 +82,5 @@ volumes:
|
|
|
81
82
|
name: openmates-uploads-vault-data
|
|
82
83
|
vault-setup-data:
|
|
83
84
|
name: openmates-uploads-vault-setup-data
|
|
85
|
+
vault-app-data:
|
|
86
|
+
name: openmates-uploads-vault-app-data
|