@ferrox-node/observability 1.1.1 → 1.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/metrics.d.ts CHANGED
@@ -1,44 +1,84 @@
1
1
  import * as client from 'prom-client';
2
+ /**
3
+ * Enterprise Global Metrics Engine for Ferrox-Node Observability.
4
+ *
5
+ * Provides a centralized singleton manager for exposing system telemetry,
6
+ * business KPIs, and infrastructure health checks to Prometheus and Grafana.
7
+ * Automatically tracks CPU, Memory Heap, Event Loop Lag, and Application Panics.
8
+ *
9
+ * Features:
10
+ * - Prometheus Exporter (`prom-client`) integration
11
+ * - Automated sampling of V8 Engine internals (Event Loop, Garbage Collection)
12
+ * - Built-in threshold alerting for critical bottlenecks
13
+ * - Panic hooks for uncaught exceptions tracing
14
+ *
15
+ * @example
16
+ * ```typescript
17
+ * GlobalMetricsEngine.init();
18
+ * const metricsStr = await GlobalMetricsEngine.getMetricsString();
19
+ * ```
20
+ */
2
21
  export declare class GlobalMetricsEngine {
3
22
  private static isInitialized;
4
23
  private static samplingTimer;
5
24
  /**
6
- * Gauge metric that tracks the Node.js Event Loop Lag.
7
- * Crucial for detecting if synchronous code is blocking the main thread.
25
+ * Gauge metric that tracks the Node.js Event Loop Lag in milliseconds.
26
+ * Crucial for detecting if synchronous code is blocking the main thread (CPU starvation).
27
+ * @type {client.Gauge<string>}
8
28
  */
9
29
  static eventLoopLag: client.Gauge<string>;
10
30
  /**
11
31
  * Gauge metric tracking the number of active connections in the database pool.
32
+ * Useful to detect connection leaks or database starvation.
33
+ * @type {client.Gauge<string>}
12
34
  */
13
35
  static activeDatabaseConnections: client.Gauge<string>;
14
36
  /**
15
37
  * Counter tracking the total number of HTTP 5xx Server Errors (Crashes/Panics).
38
+ * Monitored by the panic hooks to alert on system degradation.
39
+ * @type {client.Counter<string>}
16
40
  */
17
41
  static http5xxErrorRate: client.Counter<string>;
18
42
  /**
19
43
  * Gauge metric for the current V8 Memory Heap Used in bytes.
44
+ * Automatically alerts if the heap approaches the V8 max limit (e.g. 1.5GB default).
45
+ * @type {client.Gauge<string>}
20
46
  */
21
47
  static processMemoryHeapUsed: client.Gauge<string>;
22
48
  /**
23
- * Initializes default global system metrics and alerts
49
+ * Initializes default global system metrics, registers Prometheus collectors,
50
+ * and starts the background sampling interval.
51
+ *
52
+ * This method is idempotent and will safely return if called multiple times.
24
53
  */
25
54
  static init(): void;
26
55
  /**
27
56
  * Periodically sample non-event-driven metrics.
28
- * Starts a detached setInterval that samples event loop lag and memory usage.
29
- * Includes built-in alerting thresholds (e.g. 100ms lag, 1.5GB memory).
57
+ * Uses `setImmediate` and `hrtime` to calculate precise Event Loop delay.
58
+ * Includes built-in hardcoded alerting thresholds (e.g., > 100ms lag, > 1.5GB memory).
30
59
  */
31
60
  static sampleMetrics(): void;
61
+ /**
62
+ * Starts a detached background interval for metric sampling.
63
+ * The interval is unreferenced (`unref()`) to prevent it from keeping the Node process alive.
64
+ * @private
65
+ */
32
66
  private static startPeriodicSampling;
67
+ /**
68
+ * Gracefully tears down the metrics engine.
69
+ * Stops the sampling timer and clears the Prometheus registry.
70
+ */
33
71
  static destroy(): void;
34
72
  /**
35
- * Hook into process-level crash events to record them before the process dies.
36
- * Captures uncaught exceptions and unhandled promise rejections, increments
73
+ * Hooks into process-level crash events to record them before the process dies.
74
+ * Captures `uncaughtException` and `unhandledRejection`, increments
37
75
  * the 5xx error rate metric, and logs the critical failure.
76
+ * @private
38
77
  */
39
78
  private static setupPanicHooks;
40
79
  /**
41
- * Generates the Prometheus Metrics text to be exposed on an endpoint (e.g., /metrics).
80
+ * Generates the Prometheus Metrics text to be exposed on an endpoint (e.g., `/metrics`).
81
+ * Fetches all registered metrics from the prom-client global registry.
42
82
  *
43
83
  * @returns {Promise<string>} A string containing all metrics formatted for Prometheus scraping.
44
84
  */
package/dist/metrics.js CHANGED
@@ -37,28 +37,57 @@ exports.GlobalMetricsEngine = void 0;
37
37
  const client = __importStar(require("prom-client"));
38
38
  const logger_1 = require("@node-yalc/logger");
39
39
  const logger = (0, logger_1.AppLoggerFactory)('GlobalMetricsEngine');
40
+ /**
41
+ * Enterprise Global Metrics Engine for Ferrox-Node Observability.
42
+ *
43
+ * Provides a centralized singleton manager for exposing system telemetry,
44
+ * business KPIs, and infrastructure health checks to Prometheus and Grafana.
45
+ * Automatically tracks CPU, Memory Heap, Event Loop Lag, and Application Panics.
46
+ *
47
+ * Features:
48
+ * - Prometheus Exporter (`prom-client`) integration
49
+ * - Automated sampling of V8 Engine internals (Event Loop, Garbage Collection)
50
+ * - Built-in threshold alerting for critical bottlenecks
51
+ * - Panic hooks for uncaught exceptions tracing
52
+ *
53
+ * @example
54
+ * ```typescript
55
+ * GlobalMetricsEngine.init();
56
+ * const metricsStr = await GlobalMetricsEngine.getMetricsString();
57
+ * ```
58
+ */
40
59
  class GlobalMetricsEngine {
41
60
  static isInitialized = false;
42
61
  static samplingTimer = null;
43
62
  /**
44
- * Gauge metric that tracks the Node.js Event Loop Lag.
45
- * Crucial for detecting if synchronous code is blocking the main thread.
63
+ * Gauge metric that tracks the Node.js Event Loop Lag in milliseconds.
64
+ * Crucial for detecting if synchronous code is blocking the main thread (CPU starvation).
65
+ * @type {client.Gauge<string>}
46
66
  */
47
67
  static eventLoopLag;
48
68
  /**
49
69
  * Gauge metric tracking the number of active connections in the database pool.
70
+ * Useful to detect connection leaks or database starvation.
71
+ * @type {client.Gauge<string>}
50
72
  */
51
73
  static activeDatabaseConnections;
52
74
  /**
53
75
  * Counter tracking the total number of HTTP 5xx Server Errors (Crashes/Panics).
76
+ * Monitored by the panic hooks to alert on system degradation.
77
+ * @type {client.Counter<string>}
54
78
  */
55
79
  static http5xxErrorRate;
56
80
  /**
57
81
  * Gauge metric for the current V8 Memory Heap Used in bytes.
82
+ * Automatically alerts if the heap approaches the V8 max limit (e.g. 1.5GB default).
83
+ * @type {client.Gauge<string>}
58
84
  */
59
85
  static processMemoryHeapUsed;
60
86
  /**
61
- * Initializes default global system metrics and alerts
87
+ * Initializes default global system metrics, registers Prometheus collectors,
88
+ * and starts the background sampling interval.
89
+ *
90
+ * This method is idempotent and will safely return if called multiple times.
62
91
  */
63
92
  static init() {
64
93
  if (this.isInitialized)
@@ -90,8 +119,8 @@ class GlobalMetricsEngine {
90
119
  }
91
120
  /**
92
121
  * Periodically sample non-event-driven metrics.
93
- * Starts a detached setInterval that samples event loop lag and memory usage.
94
- * Includes built-in alerting thresholds (e.g. 100ms lag, 1.5GB memory).
122
+ * Uses `setImmediate` and `hrtime` to calculate precise Event Loop delay.
123
+ * Includes built-in hardcoded alerting thresholds (e.g., > 100ms lag, > 1.5GB memory).
95
124
  */
96
125
  static sampleMetrics() {
97
126
  // Monitor Event Loop Lag
@@ -113,12 +142,21 @@ class GlobalMetricsEngine {
113
142
  logger.error(`[ALERT] CRITICAL MEMORY USAGE! Heap is at ${(memUsage.heapUsed / 1024 / 1024).toFixed(2)} MB`);
114
143
  }
115
144
  }
145
+ /**
146
+ * Starts a detached background interval for metric sampling.
147
+ * The interval is unreferenced (`unref()`) to prevent it from keeping the Node process alive.
148
+ * @private
149
+ */
116
150
  static startPeriodicSampling() {
117
151
  this.samplingTimer = setInterval(() => {
118
152
  this.sampleMetrics();
119
153
  }, 5000);
120
154
  this.samplingTimer.unref(); // unref so it doesn't prevent Node from exiting
121
155
  }
156
+ /**
157
+ * Gracefully tears down the metrics engine.
158
+ * Stops the sampling timer and clears the Prometheus registry.
159
+ */
122
160
  static destroy() {
123
161
  if (this.samplingTimer) {
124
162
  clearInterval(this.samplingTimer);
@@ -128,9 +166,10 @@ class GlobalMetricsEngine {
128
166
  this.isInitialized = false;
129
167
  }
130
168
  /**
131
- * Hook into process-level crash events to record them before the process dies.
132
- * Captures uncaught exceptions and unhandled promise rejections, increments
169
+ * Hooks into process-level crash events to record them before the process dies.
170
+ * Captures `uncaughtException` and `unhandledRejection`, increments
133
171
  * the 5xx error rate metric, and logs the critical failure.
172
+ * @private
134
173
  */
135
174
  static setupPanicHooks() {
136
175
  process.on('uncaughtException', (err) => {
@@ -143,7 +182,8 @@ class GlobalMetricsEngine {
143
182
  });
144
183
  }
145
184
  /**
146
- * Generates the Prometheus Metrics text to be exposed on an endpoint (e.g., /metrics).
185
+ * Generates the Prometheus Metrics text to be exposed on an endpoint (e.g., `/metrics`).
186
+ * Fetches all registered metrics from the prom-client global registry.
147
187
  *
148
188
  * @returns {Promise<string>} A string containing all metrics formatted for Prometheus scraping.
149
189
  */
package/package.json CHANGED
@@ -1,16 +1,17 @@
1
- {
2
- "name": "@ferrox-node/observability",
3
- "version": "1.1.1",
4
- "main": "dist/index.js",
5
- "types": "dist/index.d.ts",
6
- "scripts": {
7
- "build": "tsc"
8
- },
9
- "dependencies": {
10
- "@ferrox-node/core": "*",
11
- "prom-client": "^15.1.0"
12
- },
13
- "devDependencies": {
14
- "typescript": "^5.0.0"
15
- }
16
- }
1
+ {
2
+ "name": "@ferrox-node/observability",
3
+ "version": "1.1.2",
4
+ "main": "dist/index.js",
5
+ "types": "dist/index.d.ts",
6
+ "scripts": {
7
+ "build": "tsc"
8
+ },
9
+ "dependencies": {
10
+ "@ferrox-node/core": "*",
11
+ "prom-client": "^15.1.0"
12
+ },
13
+ "devDependencies": {
14
+ "typescript": "^5.0.0"
15
+ },
16
+ "type": "module"
17
+ }
package/src/metrics.ts CHANGED
@@ -4,35 +4,64 @@ import * as perf_hooks from 'perf_hooks';
4
4
 
5
5
  const logger = AppLoggerFactory('GlobalMetricsEngine');
6
6
 
7
+ /**
8
+ * Enterprise Global Metrics Engine for Ferrox-Node Observability.
9
+ *
10
+ * Provides a centralized singleton manager for exposing system telemetry,
11
+ * business KPIs, and infrastructure health checks to Prometheus and Grafana.
12
+ * Automatically tracks CPU, Memory Heap, Event Loop Lag, and Application Panics.
13
+ *
14
+ * Features:
15
+ * - Prometheus Exporter (`prom-client`) integration
16
+ * - Automated sampling of V8 Engine internals (Event Loop, Garbage Collection)
17
+ * - Built-in threshold alerting for critical bottlenecks
18
+ * - Panic hooks for uncaught exceptions tracing
19
+ *
20
+ * @example
21
+ * ```typescript
22
+ * GlobalMetricsEngine.init();
23
+ * const metricsStr = await GlobalMetricsEngine.getMetricsString();
24
+ * ```
25
+ */
7
26
  export class GlobalMetricsEngine {
8
27
  private static isInitialized = false;
9
28
  private static samplingTimer: NodeJS.Timeout | null = null;
10
29
 
11
30
  /**
12
- * Gauge metric that tracks the Node.js Event Loop Lag.
13
- * Crucial for detecting if synchronous code is blocking the main thread.
31
+ * Gauge metric that tracks the Node.js Event Loop Lag in milliseconds.
32
+ * Crucial for detecting if synchronous code is blocking the main thread (CPU starvation).
33
+ * @type {client.Gauge<string>}
14
34
  */
15
35
  public static eventLoopLag: client.Gauge<string>;
16
36
 
17
37
  /**
18
38
  * Gauge metric tracking the number of active connections in the database pool.
39
+ * Useful to detect connection leaks or database starvation.
40
+ * @type {client.Gauge<string>}
19
41
  */
20
42
  public static activeDatabaseConnections: client.Gauge<string>;
21
43
 
22
44
  /**
23
45
  * Counter tracking the total number of HTTP 5xx Server Errors (Crashes/Panics).
46
+ * Monitored by the panic hooks to alert on system degradation.
47
+ * @type {client.Counter<string>}
24
48
  */
25
49
  public static http5xxErrorRate: client.Counter<string>;
26
50
 
27
51
  /**
28
52
  * Gauge metric for the current V8 Memory Heap Used in bytes.
53
+ * Automatically alerts if the heap approaches the V8 max limit (e.g. 1.5GB default).
54
+ * @type {client.Gauge<string>}
29
55
  */
30
56
  public static processMemoryHeapUsed: client.Gauge<string>;
31
57
 
32
58
  /**
33
- * Initializes default global system metrics and alerts
59
+ * Initializes default global system metrics, registers Prometheus collectors,
60
+ * and starts the background sampling interval.
61
+ *
62
+ * This method is idempotent and will safely return if called multiple times.
34
63
  */
35
- public static init() {
64
+ public static init(): void {
36
65
  if (this.isInitialized) return;
37
66
  this.isInitialized = true;
38
67
 
@@ -69,10 +98,10 @@ export class GlobalMetricsEngine {
69
98
 
70
99
  /**
71
100
  * Periodically sample non-event-driven metrics.
72
- * Starts a detached setInterval that samples event loop lag and memory usage.
73
- * Includes built-in alerting thresholds (e.g. 100ms lag, 1.5GB memory).
101
+ * Uses `setImmediate` and `hrtime` to calculate precise Event Loop delay.
102
+ * Includes built-in hardcoded alerting thresholds (e.g., > 100ms lag, > 1.5GB memory).
74
103
  */
75
- public static sampleMetrics() {
104
+ public static sampleMetrics(): void {
76
105
  // Monitor Event Loop Lag
77
106
  const start = process.hrtime.bigint();
78
107
  setImmediate(() => {
@@ -96,14 +125,23 @@ export class GlobalMetricsEngine {
96
125
  }
97
126
  }
98
127
 
99
- private static startPeriodicSampling() {
128
+ /**
129
+ * Starts a detached background interval for metric sampling.
130
+ * The interval is unreferenced (`unref()`) to prevent it from keeping the Node process alive.
131
+ * @private
132
+ */
133
+ private static startPeriodicSampling(): void {
100
134
  this.samplingTimer = setInterval(() => {
101
135
  this.sampleMetrics();
102
136
  }, 5000);
103
137
  this.samplingTimer.unref(); // unref so it doesn't prevent Node from exiting
104
138
  }
105
139
 
106
- public static destroy() {
140
+ /**
141
+ * Gracefully tears down the metrics engine.
142
+ * Stops the sampling timer and clears the Prometheus registry.
143
+ */
144
+ public static destroy(): void {
107
145
  if (this.samplingTimer) {
108
146
  clearInterval(this.samplingTimer);
109
147
  this.samplingTimer = null;
@@ -113,11 +151,12 @@ export class GlobalMetricsEngine {
113
151
  }
114
152
 
115
153
  /**
116
- * Hook into process-level crash events to record them before the process dies.
117
- * Captures uncaught exceptions and unhandled promise rejections, increments
154
+ * Hooks into process-level crash events to record them before the process dies.
155
+ * Captures `uncaughtException` and `unhandledRejection`, increments
118
156
  * the 5xx error rate metric, and logs the critical failure.
157
+ * @private
119
158
  */
120
- private static setupPanicHooks() {
159
+ private static setupPanicHooks(): void {
121
160
  process.on('uncaughtException', (err) => {
122
161
  logger.error(`[ALERT] Uncaught Exception (Process Crash imminent): ${err.message}`);
123
162
  this.http5xxErrorRate.inc(); // Increment crash counter
@@ -130,7 +169,8 @@ export class GlobalMetricsEngine {
130
169
  }
131
170
 
132
171
  /**
133
- * Generates the Prometheus Metrics text to be exposed on an endpoint (e.g., /metrics).
172
+ * Generates the Prometheus Metrics text to be exposed on an endpoint (e.g., `/metrics`).
173
+ * Fetches all registered metrics from the prom-client global registry.
134
174
  *
135
175
  * @returns {Promise<string>} A string containing all metrics formatted for Prometheus scraping.
136
176
  */