openmates 0.24.0-alpha.1 → 0.24.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -34395,6 +34395,7 @@ var ROLE_TEMPLATE_FILES = {
34395
34395
  var CORE_NO_WEBAPP_TEMPLATE_FILE = join15("core", "docker-compose.no-webapp.yml");
34396
34396
  var CORE_PROMTAIL_CONFIG_FILE = join15("backend", "core", "monitoring", "promtail", "promtail-config.yaml");
34397
34397
  var CORE_ALERTMANAGER_CONFIG_FILE = join15("backend", "core", "monitoring", "alertmanager", "alertmanager.yml");
34398
+ var CORE_PROMETHEUS_CONFIG_DIR = join15("backend", "core", "monitoring", "prometheus");
34398
34399
  var COMPOSE_OVERRIDE2 = join15("backend", "core", "docker-compose.override.yml");
34399
34400
  var DEFAULT_INSTALL_PATH = join15(homedir6(), "openmates");
34400
34401
  var REPO_URL = "https://github.com/glowingkitty/OpenMates.git";
@@ -34957,6 +34958,19 @@ function packagedCaddyTemplatePath(role) {
34957
34958
  function packagedCoreAlertmanagerTemplatePath() {
34958
34959
  return join15(dirname8(new URL(import.meta.url).pathname), "..", "templates", "core", "monitoring", "alertmanager", "alertmanager.yml");
34959
34960
  }
34961
+ function ensureCorePrometheusRuntimeFiles(installPath) {
34962
+ const packagedDir = join15(dirname8(new URL(import.meta.url).pathname), "..", "templates", "core", "monitoring", "prometheus");
34963
+ const runtimeDir = join15(installPath, CORE_PROMETHEUS_CONFIG_DIR);
34964
+ for (const fileName of ["prometheus.yml", "alert_rules.yml"]) {
34965
+ const packagedPath = join15(packagedDir, fileName);
34966
+ if (!existsSync13(packagedPath)) throw new Error(`Packaged Prometheus file not found: ${packagedPath}`);
34967
+ }
34968
+ mkdirSync13(runtimeDir, { recursive: true });
34969
+ for (const fileName of ["prometheus.yml", "alert_rules.yml"]) {
34970
+ const runtimePath = join15(runtimeDir, fileName);
34971
+ if (!existsSync13(runtimePath)) copyFileSync2(join15(packagedDir, fileName), runtimePath);
34972
+ }
34973
+ }
34960
34974
  function readOfficialCloudNoWebappComposeTemplate() {
34961
34975
  const packaged = packagedNoWebappTemplatePath();
34962
34976
  if (existsSync13(packaged)) return readFileSync14(packaged, "utf-8");
@@ -35011,6 +35025,7 @@ async function writeImageModeRuntimeFiles(installPath, imageTag, role) {
35011
35025
  writeFileSync8(join15(installPath, OFFICIAL_CLOUD_NO_WEBAPP_COMPOSE_FILE), readOfficialCloudNoWebappComposeTemplate());
35012
35026
  }
35013
35027
  if (role === "core") {
35028
+ ensureCorePrometheusRuntimeFiles(installPath);
35014
35029
  const promtailConfigPath = join15(installPath, CORE_PROMTAIL_CONFIG_FILE);
35015
35030
  mkdirSync13(dirname8(promtailConfigPath), { recursive: true });
35016
35031
  writeFileSync8(promtailConfigPath, SELFHOST_PROMTAIL_CONFIG_TEMPLATE);
package/dist/cli.js CHANGED
@@ -19,7 +19,7 @@ import {
19
19
  serializeToYaml,
20
20
  shouldRequireTrustedAccountGuard,
21
21
  waitForWorkflowRun
22
- } from "./chunk-ETF6SSYS.js";
22
+ } from "./chunk-VEQ3R7MK.js";
23
23
  import "./chunk-IWKP55ZB.js";
24
24
  import "./chunk-BBKJZ23O.js";
25
25
  export {
package/dist/index.js CHANGED
@@ -43,7 +43,7 @@ import {
43
43
  selectAssistantMessagesForSpeech,
44
44
  serializeToYaml,
45
45
  summarizeAssistantSpeech
46
- } from "./chunk-ETF6SSYS.js";
46
+ } from "./chunk-VEQ3R7MK.js";
47
47
  import "./chunk-IWKP55ZB.js";
48
48
  import "./chunk-BBKJZ23O.js";
49
49
  export {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openmates",
3
- "version": "0.24.0-alpha.1",
3
+ "version": "0.24.0",
4
4
  "description": "OpenMates CLI and SDK",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",
@@ -232,12 +232,12 @@ services:
232
232
  mem_limit: ${CORE_WORKER_MEMORY_LIMIT:-3g}
233
233
  environment:
234
234
  <<: *openmates-worker-env
235
- CELERY_QUEUES: persistence,health_check,server_stats,demo,e2e_tests,push
235
+ CELERY_QUEUES: persistence,health_check,server_stats,demo,e2e_tests,push,leaderboard
236
236
  CELERY_AUTOSCALE_MAX: ${CORE_WORKER_CONCURRENCY:-${CELERY_AUTOSCALE_MAX:-3}}
237
237
  CELERY_AUTOSCALE_MIN: ${CELERY_AUTOSCALE_MIN:-1}
238
238
  CELERY_METRICS_PORT: "9109"
239
239
  command: >
240
- sh -c "chown -R celeryuser:celeryuser /vault-data && gosu celeryuser python -m celery -A backend.core.api.app.tasks.celery_config worker --loglevel=info --queues=persistence,health_check,server_stats,demo,e2e_tests,push --concurrency=$${CELERY_AUTOSCALE_MAX} --max-tasks-per-child=50 --max-memory-per-child=600000 --prefetch-multiplier=1"
240
+ sh -c "chown -R celeryuser:celeryuser /vault-data && gosu celeryuser python -m celery -A backend.core.api.app.tasks.celery_config worker --loglevel=info --queues=persistence,health_check,server_stats,demo,e2e_tests,push,leaderboard --concurrency=$${CELERY_AUTOSCALE_MAX} --max-tasks-per-child=50 --max-memory-per-child=600000 --prefetch-multiplier=1"
241
241
 
242
242
  user-tasks-worker:
243
243
  <<: *openmates-worker-base
@@ -0,0 +1,164 @@
1
+ # Prometheus alert rules for OpenMates.
2
+ # These fire when monitored conditions breach thresholds, and Alertmanager
3
+ # routes the resulting notifications (with throttling to prevent email floods).
4
+ #
5
+ # Metric names used here must match what the API actually exposes.
6
+ # API metrics are defined in backend/core/api/app/services/metrics.py:
7
+ # - api_requests_total (Counter, labels: method, endpoint, status_code)
8
+ # - api_request_duration_seconds (Histogram, labels: method, endpoint)
9
+ # - celery_task_failures_total (Counter, labels: task_name, queue) — added in celery_config.py
10
+ #
11
+ # Architecture context: See docs/architecture/logging-and-monitoring.md
12
+
13
+ groups:
14
+ # -------------------------------------------------------------------------
15
+ # API health
16
+ # -------------------------------------------------------------------------
17
+ - name: api_alerts
18
+ rules:
19
+ - alert: HighErrorRate
20
+ expr: |
21
+ (
22
+ sum(rate(api_requests_total{status_code=~"5.."}[5m]))
23
+ /
24
+ sum(rate(api_requests_total[5m]))
25
+ ) > 0.05
26
+ for: 5m
27
+ labels:
28
+ severity: critical
29
+ annotations:
30
+ summary: "API 5xx error rate above 5%"
31
+ description: >-
32
+ Over the last 5 minutes, {{ $value | humanizePercentage }} of
33
+ requests returned 5xx errors.
34
+
35
+ - alert: HighRequestLatency
36
+ expr: |
37
+ histogram_quantile(0.95,
38
+ sum(rate(api_request_duration_seconds_bucket[5m])) by (le)
39
+ ) > 5
40
+ for: 10m
41
+ labels:
42
+ severity: warning
43
+ annotations:
44
+ summary: "P95 request latency above 5 s"
45
+ description: >-
46
+ P95 latency is {{ $value | humanizeDuration }}.
47
+
48
+ - alert: APIDown
49
+ expr: up{job="api"} == 0
50
+ for: 2m
51
+ labels:
52
+ severity: critical
53
+ annotations:
54
+ summary: "API container is down"
55
+ description: "Prometheus cannot scrape the API metrics endpoint."
56
+
57
+ # -------------------------------------------------------------------------
58
+ # Celery task health
59
+ # Requires celery_task_failures_total counter — added in celery_config.py
60
+ # worker_process_init signal starts a prometheus_client HTTP server per worker.
61
+ # -------------------------------------------------------------------------
62
+ - name: celery_alerts
63
+ rules:
64
+ - alert: TaskFailureSpike
65
+ expr: |
66
+ sum(increase(celery_task_failures_total[15m])) > 10
67
+ for: 5m
68
+ labels:
69
+ severity: warning
70
+ annotations:
71
+ summary: "More than 10 Celery task failures in 15 min"
72
+ description: >-
73
+ {{ $value }} task failures detected in the last 15 minutes.
74
+
75
+ # -------------------------------------------------------------------------
76
+ # Container / resource health (cAdvisor)
77
+ # -------------------------------------------------------------------------
78
+ - name: container_alerts
79
+ rules:
80
+ - alert: ContainerRestarting
81
+ expr: |
82
+ increase(container_restart_count[30m]) > 3
83
+ for: 5m
84
+ labels:
85
+ severity: warning
86
+ annotations:
87
+ summary: "Container {{ $labels.name }} restarted >3 times in 30 min"
88
+ description: >-
89
+ Container {{ $labels.name }} has restarted {{ $value }} times.
90
+
91
+ - alert: HighMemoryUsage
92
+ expr: |
93
+ (
94
+ container_memory_usage_bytes{name!="",container_label_com_docker_compose_project="openmates"}
95
+ /
96
+ container_spec_memory_limit_bytes{name!="",container_label_com_docker_compose_project="openmates"}
97
+ ) > 0.90
98
+ and on(instance, id, name, environment, container_label_com_docker_compose_project)
99
+ container_spec_memory_limit_bytes{name!="",container_label_com_docker_compose_project="openmates"} > 0
100
+ for: 10m
101
+ labels:
102
+ severity: warning
103
+ annotations:
104
+ summary: "Container {{ $labels.name }} memory above 90%"
105
+ description: >-
106
+ Memory usage is at {{ $value | humanizePercentage }}.
107
+
108
+ - name: host_resource_alerts
109
+ rules:
110
+ - alert: HostDiskSpaceWarning
111
+ expr: |
112
+ node_filesystem_avail_bytes{mountpoint="/",fstype!~"tmpfs|overlay|squashfs"}
113
+ /
114
+ node_filesystem_size_bytes{mountpoint="/",fstype!~"tmpfs|overlay|squashfs"}
115
+ < 0.15
116
+ for: 15m
117
+ labels:
118
+ severity: warning
119
+ annotations:
120
+ summary: "Host disk usage above 85% on {{ $labels.mountpoint }}"
121
+ description: "Disk usage is above 85% for {{ $labels.mountpoint }}."
122
+
123
+ - alert: HostDiskSpaceCritical
124
+ expr: |
125
+ node_filesystem_avail_bytes{mountpoint="/",fstype!~"tmpfs|overlay|squashfs"}
126
+ /
127
+ node_filesystem_size_bytes{mountpoint="/",fstype!~"tmpfs|overlay|squashfs"}
128
+ < 0.05
129
+ for: 5m
130
+ labels:
131
+ severity: critical
132
+ annotations:
133
+ summary: "Host disk usage above 95% on {{ $labels.mountpoint }}"
134
+ description: "Disk usage is above 95% for {{ $labels.mountpoint }}."
135
+
136
+ - alert: OperationalReportStale
137
+ expr: |
138
+ (time() - operational_report_last_success_timestamp_seconds > 93600)
139
+ or
140
+ (
141
+ (time() - operational_report_monitoring_started_timestamp_seconds > 93600)
142
+ unless on(environment) operational_report_last_success_timestamp_seconds
143
+ )
144
+ for: 5m
145
+ labels:
146
+ severity: critical
147
+ annotations:
148
+ summary: "{{ $labels.environment }} operational report is stale"
149
+ description: "No accepted operational report has been recorded for more than 26 hours."
150
+
151
+ # -------------------------------------------------------------------------
152
+ # Prometheus self-monitoring
153
+ # -------------------------------------------------------------------------
154
+ - name: prometheus_self
155
+ rules:
156
+ - alert: PrometheusTargetDown
157
+ expr: up == 0
158
+ for: 5m
159
+ labels:
160
+ severity: critical
161
+ annotations:
162
+ summary: "Scrape target {{ $labels.job }} is down"
163
+ description: >-
164
+ Prometheus cannot reach {{ $labels.instance }} (job={{ $labels.job }}).
@@ -0,0 +1,96 @@
1
+ global:
2
+ scrape_interval: 30s
3
+ scrape_timeout: 10s
4
+ evaluation_interval: 30s
5
+ external_labels:
6
+ environment: "${SERVER_ENVIRONMENT}"
7
+
8
+ alerting:
9
+ alertmanagers:
10
+ - static_configs:
11
+ - targets: ["alertmanager:9093"]
12
+ scheme: http
13
+ timeout: 10s
14
+
15
+ rule_files:
16
+ - "/etc/prometheus/alert_rules.yml"
17
+
18
+ # Self-host image mode keeps metrics locally; remote write requires separately provisioned credentials.
19
+ scrape_configs:
20
+ - job_name: "prometheus"
21
+ static_configs:
22
+ - targets: ["localhost:9090"]
23
+
24
+ - job_name: "api"
25
+ metrics_path: /metrics
26
+ static_configs:
27
+ - targets: ["api:8000"]
28
+ labels:
29
+ environment: "${SERVER_ENVIRONMENT}"
30
+
31
+ - job_name: "cadvisor"
32
+ scrape_interval: 60s
33
+ scrape_timeout: 10s
34
+ static_configs:
35
+ - targets: ["cadvisor:8080"]
36
+ labels:
37
+ environment: "${SERVER_ENVIRONMENT}"
38
+
39
+ - job_name: "node"
40
+ scrape_interval: 60s
41
+ static_configs:
42
+ - targets: ["node-exporter:9100"]
43
+ labels:
44
+ environment: "${SERVER_ENVIRONMENT}"
45
+
46
+ # Celery worker Prometheus metrics servers.
47
+ # Each worker starts a prometheus_client HTTP server on CELERY_METRICS_PORT
48
+ # via the worker_process_init signal in celery_config.py.
49
+ # Port assignments (must match CELERY_METRICS_PORT in docker-compose.yml):
50
+ # task-worker: 9101
51
+ # user-init-worker: 9110
52
+ # core-worker: 9109
53
+ # user-tasks-worker: 9112
54
+ # reminder-worker: 9111
55
+ # app-ai-worker: 9102
56
+ # app-images-worker: 9103
57
+ # app-pdf-worker: 9104
58
+ - job_name: "celery-task-worker"
59
+ static_configs:
60
+ - targets: ["task-worker:9101"]
61
+ labels: {environment: "${SERVER_ENVIRONMENT}"}
62
+
63
+ - job_name: "celery-user-init-worker"
64
+ static_configs:
65
+ - targets: ["user-init-worker:9110"]
66
+ labels: {environment: "${SERVER_ENVIRONMENT}"}
67
+
68
+ - job_name: "celery-core-worker"
69
+ static_configs:
70
+ - targets: ["core-worker:9109"]
71
+ labels: {environment: "${SERVER_ENVIRONMENT}"}
72
+
73
+ - job_name: "celery-user-tasks-worker"
74
+ static_configs:
75
+ - targets: ["user-tasks-worker:9112"]
76
+ labels: {environment: "${SERVER_ENVIRONMENT}"}
77
+
78
+ - job_name: "celery-reminder-worker"
79
+ static_configs:
80
+ - targets: ["reminder-worker:9111"]
81
+ labels: {environment: "${SERVER_ENVIRONMENT}"}
82
+
83
+ - job_name: "celery-app-ai-worker"
84
+ static_configs:
85
+ - targets: ["app-ai-worker:9102"]
86
+ labels: {environment: "${SERVER_ENVIRONMENT}"}
87
+
88
+ - job_name: "celery-app-images-worker"
89
+ static_configs:
90
+ - targets: ["app-images-worker:9103"]
91
+ labels: {environment: "${SERVER_ENVIRONMENT}"}
92
+
93
+ - job_name: "celery-app-pdf-worker"
94
+ static_configs:
95
+ - targets: ["app-pdf-worker:9104"]
96
+ labels: {environment: "${SERVER_ENVIRONMENT}"}