@ferrox-node/observability 1.1.1 → 1.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/metrics.d.ts +48 -8
- package/dist/metrics.js +48 -8
- package/package.json +17 -16
- package/src/metrics.ts +53 -13
package/dist/metrics.d.ts
CHANGED
|
@@ -1,44 +1,84 @@
|
|
|
1
1
|
import * as client from 'prom-client';
|
|
2
|
+
/**
|
|
3
|
+
* Enterprise Global Metrics Engine for Ferrox-Node Observability.
|
|
4
|
+
*
|
|
5
|
+
* Provides a centralized singleton manager for exposing system telemetry,
|
|
6
|
+
* business KPIs, and infrastructure health checks to Prometheus and Grafana.
|
|
7
|
+
* Automatically tracks CPU, Memory Heap, Event Loop Lag, and Application Panics.
|
|
8
|
+
*
|
|
9
|
+
* Features:
|
|
10
|
+
* - Prometheus Exporter (`prom-client`) integration
|
|
11
|
+
* - Automated sampling of V8 Engine internals (Event Loop, Garbage Collection)
|
|
12
|
+
* - Built-in threshold alerting for critical bottlenecks
|
|
13
|
+
* - Panic hooks for uncaught exceptions tracing
|
|
14
|
+
*
|
|
15
|
+
* @example
|
|
16
|
+
* ```typescript
|
|
17
|
+
* GlobalMetricsEngine.init();
|
|
18
|
+
* const metricsStr = await GlobalMetricsEngine.getMetricsString();
|
|
19
|
+
* ```
|
|
20
|
+
*/
|
|
2
21
|
export declare class GlobalMetricsEngine {
|
|
3
22
|
private static isInitialized;
|
|
4
23
|
private static samplingTimer;
|
|
5
24
|
/**
|
|
6
|
-
* Gauge metric that tracks the Node.js Event Loop Lag.
|
|
7
|
-
* Crucial for detecting if synchronous code is blocking the main thread.
|
|
25
|
+
* Gauge metric that tracks the Node.js Event Loop Lag in milliseconds.
|
|
26
|
+
* Crucial for detecting if synchronous code is blocking the main thread (CPU starvation).
|
|
27
|
+
* @type {client.Gauge<string>}
|
|
8
28
|
*/
|
|
9
29
|
static eventLoopLag: client.Gauge<string>;
|
|
10
30
|
/**
|
|
11
31
|
* Gauge metric tracking the number of active connections in the database pool.
|
|
32
|
+
* Useful to detect connection leaks or database starvation.
|
|
33
|
+
* @type {client.Gauge<string>}
|
|
12
34
|
*/
|
|
13
35
|
static activeDatabaseConnections: client.Gauge<string>;
|
|
14
36
|
/**
|
|
15
37
|
* Counter tracking the total number of HTTP 5xx Server Errors (Crashes/Panics).
|
|
38
|
+
* Monitored by the panic hooks to alert on system degradation.
|
|
39
|
+
* @type {client.Counter<string>}
|
|
16
40
|
*/
|
|
17
41
|
static http5xxErrorRate: client.Counter<string>;
|
|
18
42
|
/**
|
|
19
43
|
* Gauge metric for the current V8 Memory Heap Used in bytes.
|
|
44
|
+
* Automatically alerts if the heap approaches the V8 max limit (e.g. 1.5GB default).
|
|
45
|
+
* @type {client.Gauge<string>}
|
|
20
46
|
*/
|
|
21
47
|
static processMemoryHeapUsed: client.Gauge<string>;
|
|
22
48
|
/**
|
|
23
|
-
* Initializes default global system metrics
|
|
49
|
+
* Initializes default global system metrics, registers Prometheus collectors,
|
|
50
|
+
* and starts the background sampling interval.
|
|
51
|
+
*
|
|
52
|
+
* This method is idempotent and will safely return if called multiple times.
|
|
24
53
|
*/
|
|
25
54
|
static init(): void;
|
|
26
55
|
/**
|
|
27
56
|
* Periodically sample non-event-driven metrics.
|
|
28
|
-
*
|
|
29
|
-
* Includes built-in alerting thresholds (e.g
|
|
57
|
+
* Uses `setImmediate` and `hrtime` to calculate precise Event Loop delay.
|
|
58
|
+
* Includes built-in hardcoded alerting thresholds (e.g., > 100ms lag, > 1.5GB memory).
|
|
30
59
|
*/
|
|
31
60
|
static sampleMetrics(): void;
|
|
61
|
+
/**
|
|
62
|
+
* Starts a detached background interval for metric sampling.
|
|
63
|
+
* The interval is unreferenced (`unref()`) to prevent it from keeping the Node process alive.
|
|
64
|
+
* @private
|
|
65
|
+
*/
|
|
32
66
|
private static startPeriodicSampling;
|
|
67
|
+
/**
|
|
68
|
+
* Gracefully tears down the metrics engine.
|
|
69
|
+
* Stops the sampling timer and clears the Prometheus registry.
|
|
70
|
+
*/
|
|
33
71
|
static destroy(): void;
|
|
34
72
|
/**
|
|
35
|
-
*
|
|
36
|
-
* Captures
|
|
73
|
+
* Hooks into process-level crash events to record them before the process dies.
|
|
74
|
+
* Captures `uncaughtException` and `unhandledRejection`, increments
|
|
37
75
|
* the 5xx error rate metric, and logs the critical failure.
|
|
76
|
+
* @private
|
|
38
77
|
*/
|
|
39
78
|
private static setupPanicHooks;
|
|
40
79
|
/**
|
|
41
|
-
* Generates the Prometheus Metrics text to be exposed on an endpoint (e.g.,
|
|
80
|
+
* Generates the Prometheus Metrics text to be exposed on an endpoint (e.g., `/metrics`).
|
|
81
|
+
* Fetches all registered metrics from the prom-client global registry.
|
|
42
82
|
*
|
|
43
83
|
* @returns {Promise<string>} A string containing all metrics formatted for Prometheus scraping.
|
|
44
84
|
*/
|
package/dist/metrics.js
CHANGED
|
@@ -37,28 +37,57 @@ exports.GlobalMetricsEngine = void 0;
|
|
|
37
37
|
const client = __importStar(require("prom-client"));
|
|
38
38
|
const logger_1 = require("@node-yalc/logger");
|
|
39
39
|
const logger = (0, logger_1.AppLoggerFactory)('GlobalMetricsEngine');
|
|
40
|
+
/**
|
|
41
|
+
* Enterprise Global Metrics Engine for Ferrox-Node Observability.
|
|
42
|
+
*
|
|
43
|
+
* Provides a centralized singleton manager for exposing system telemetry,
|
|
44
|
+
* business KPIs, and infrastructure health checks to Prometheus and Grafana.
|
|
45
|
+
* Automatically tracks CPU, Memory Heap, Event Loop Lag, and Application Panics.
|
|
46
|
+
*
|
|
47
|
+
* Features:
|
|
48
|
+
* - Prometheus Exporter (`prom-client`) integration
|
|
49
|
+
* - Automated sampling of V8 Engine internals (Event Loop, Garbage Collection)
|
|
50
|
+
* - Built-in threshold alerting for critical bottlenecks
|
|
51
|
+
* - Panic hooks for uncaught exceptions tracing
|
|
52
|
+
*
|
|
53
|
+
* @example
|
|
54
|
+
* ```typescript
|
|
55
|
+
* GlobalMetricsEngine.init();
|
|
56
|
+
* const metricsStr = await GlobalMetricsEngine.getMetricsString();
|
|
57
|
+
* ```
|
|
58
|
+
*/
|
|
40
59
|
class GlobalMetricsEngine {
|
|
41
60
|
static isInitialized = false;
|
|
42
61
|
static samplingTimer = null;
|
|
43
62
|
/**
|
|
44
|
-
* Gauge metric that tracks the Node.js Event Loop Lag.
|
|
45
|
-
* Crucial for detecting if synchronous code is blocking the main thread.
|
|
63
|
+
* Gauge metric that tracks the Node.js Event Loop Lag in milliseconds.
|
|
64
|
+
* Crucial for detecting if synchronous code is blocking the main thread (CPU starvation).
|
|
65
|
+
* @type {client.Gauge<string>}
|
|
46
66
|
*/
|
|
47
67
|
static eventLoopLag;
|
|
48
68
|
/**
|
|
49
69
|
* Gauge metric tracking the number of active connections in the database pool.
|
|
70
|
+
* Useful to detect connection leaks or database starvation.
|
|
71
|
+
* @type {client.Gauge<string>}
|
|
50
72
|
*/
|
|
51
73
|
static activeDatabaseConnections;
|
|
52
74
|
/**
|
|
53
75
|
* Counter tracking the total number of HTTP 5xx Server Errors (Crashes/Panics).
|
|
76
|
+
* Monitored by the panic hooks to alert on system degradation.
|
|
77
|
+
* @type {client.Counter<string>}
|
|
54
78
|
*/
|
|
55
79
|
static http5xxErrorRate;
|
|
56
80
|
/**
|
|
57
81
|
* Gauge metric for the current V8 Memory Heap Used in bytes.
|
|
82
|
+
* Automatically alerts if the heap approaches the V8 max limit (e.g. 1.5GB default).
|
|
83
|
+
* @type {client.Gauge<string>}
|
|
58
84
|
*/
|
|
59
85
|
static processMemoryHeapUsed;
|
|
60
86
|
/**
|
|
61
|
-
* Initializes default global system metrics
|
|
87
|
+
* Initializes default global system metrics, registers Prometheus collectors,
|
|
88
|
+
* and starts the background sampling interval.
|
|
89
|
+
*
|
|
90
|
+
* This method is idempotent and will safely return if called multiple times.
|
|
62
91
|
*/
|
|
63
92
|
static init() {
|
|
64
93
|
if (this.isInitialized)
|
|
@@ -90,8 +119,8 @@ class GlobalMetricsEngine {
|
|
|
90
119
|
}
|
|
91
120
|
/**
|
|
92
121
|
* Periodically sample non-event-driven metrics.
|
|
93
|
-
*
|
|
94
|
-
* Includes built-in alerting thresholds (e.g
|
|
122
|
+
* Uses `setImmediate` and `hrtime` to calculate precise Event Loop delay.
|
|
123
|
+
* Includes built-in hardcoded alerting thresholds (e.g., > 100ms lag, > 1.5GB memory).
|
|
95
124
|
*/
|
|
96
125
|
static sampleMetrics() {
|
|
97
126
|
// Monitor Event Loop Lag
|
|
@@ -113,12 +142,21 @@ class GlobalMetricsEngine {
|
|
|
113
142
|
logger.error(`[ALERT] CRITICAL MEMORY USAGE! Heap is at ${(memUsage.heapUsed / 1024 / 1024).toFixed(2)} MB`);
|
|
114
143
|
}
|
|
115
144
|
}
|
|
145
|
+
/**
|
|
146
|
+
* Starts a detached background interval for metric sampling.
|
|
147
|
+
* The interval is unreferenced (`unref()`) to prevent it from keeping the Node process alive.
|
|
148
|
+
* @private
|
|
149
|
+
*/
|
|
116
150
|
static startPeriodicSampling() {
|
|
117
151
|
this.samplingTimer = setInterval(() => {
|
|
118
152
|
this.sampleMetrics();
|
|
119
153
|
}, 5000);
|
|
120
154
|
this.samplingTimer.unref(); // unref so it doesn't prevent Node from exiting
|
|
121
155
|
}
|
|
156
|
+
/**
|
|
157
|
+
* Gracefully tears down the metrics engine.
|
|
158
|
+
* Stops the sampling timer and clears the Prometheus registry.
|
|
159
|
+
*/
|
|
122
160
|
static destroy() {
|
|
123
161
|
if (this.samplingTimer) {
|
|
124
162
|
clearInterval(this.samplingTimer);
|
|
@@ -128,9 +166,10 @@ class GlobalMetricsEngine {
|
|
|
128
166
|
this.isInitialized = false;
|
|
129
167
|
}
|
|
130
168
|
/**
|
|
131
|
-
*
|
|
132
|
-
* Captures
|
|
169
|
+
* Hooks into process-level crash events to record them before the process dies.
|
|
170
|
+
* Captures `uncaughtException` and `unhandledRejection`, increments
|
|
133
171
|
* the 5xx error rate metric, and logs the critical failure.
|
|
172
|
+
* @private
|
|
134
173
|
*/
|
|
135
174
|
static setupPanicHooks() {
|
|
136
175
|
process.on('uncaughtException', (err) => {
|
|
@@ -143,7 +182,8 @@ class GlobalMetricsEngine {
|
|
|
143
182
|
});
|
|
144
183
|
}
|
|
145
184
|
/**
|
|
146
|
-
* Generates the Prometheus Metrics text to be exposed on an endpoint (e.g.,
|
|
185
|
+
* Generates the Prometheus Metrics text to be exposed on an endpoint (e.g., `/metrics`).
|
|
186
|
+
* Fetches all registered metrics from the prom-client global registry.
|
|
147
187
|
*
|
|
148
188
|
* @returns {Promise<string>} A string containing all metrics formatted for Prometheus scraping.
|
|
149
189
|
*/
|
package/package.json
CHANGED
|
@@ -1,16 +1,17 @@
|
|
|
1
|
-
{
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
1
|
+
{
|
|
2
|
+
"name": "@ferrox-node/observability",
|
|
3
|
+
"version": "1.1.2",
|
|
4
|
+
"main": "dist/index.js",
|
|
5
|
+
"types": "dist/index.d.ts",
|
|
6
|
+
"scripts": {
|
|
7
|
+
"build": "tsc"
|
|
8
|
+
},
|
|
9
|
+
"dependencies": {
|
|
10
|
+
"@ferrox-node/core": "*",
|
|
11
|
+
"prom-client": "^15.1.0"
|
|
12
|
+
},
|
|
13
|
+
"devDependencies": {
|
|
14
|
+
"typescript": "^5.0.0"
|
|
15
|
+
},
|
|
16
|
+
"type": "module"
|
|
17
|
+
}
|
package/src/metrics.ts
CHANGED
|
@@ -4,35 +4,64 @@ import * as perf_hooks from 'perf_hooks';
|
|
|
4
4
|
|
|
5
5
|
const logger = AppLoggerFactory('GlobalMetricsEngine');
|
|
6
6
|
|
|
7
|
+
/**
|
|
8
|
+
* Enterprise Global Metrics Engine for Ferrox-Node Observability.
|
|
9
|
+
*
|
|
10
|
+
* Provides a centralized singleton manager for exposing system telemetry,
|
|
11
|
+
* business KPIs, and infrastructure health checks to Prometheus and Grafana.
|
|
12
|
+
* Automatically tracks CPU, Memory Heap, Event Loop Lag, and Application Panics.
|
|
13
|
+
*
|
|
14
|
+
* Features:
|
|
15
|
+
* - Prometheus Exporter (`prom-client`) integration
|
|
16
|
+
* - Automated sampling of V8 Engine internals (Event Loop, Garbage Collection)
|
|
17
|
+
* - Built-in threshold alerting for critical bottlenecks
|
|
18
|
+
* - Panic hooks for uncaught exceptions tracing
|
|
19
|
+
*
|
|
20
|
+
* @example
|
|
21
|
+
* ```typescript
|
|
22
|
+
* GlobalMetricsEngine.init();
|
|
23
|
+
* const metricsStr = await GlobalMetricsEngine.getMetricsString();
|
|
24
|
+
* ```
|
|
25
|
+
*/
|
|
7
26
|
export class GlobalMetricsEngine {
|
|
8
27
|
private static isInitialized = false;
|
|
9
28
|
private static samplingTimer: NodeJS.Timeout | null = null;
|
|
10
29
|
|
|
11
30
|
/**
|
|
12
|
-
* Gauge metric that tracks the Node.js Event Loop Lag.
|
|
13
|
-
* Crucial for detecting if synchronous code is blocking the main thread.
|
|
31
|
+
* Gauge metric that tracks the Node.js Event Loop Lag in milliseconds.
|
|
32
|
+
* Crucial for detecting if synchronous code is blocking the main thread (CPU starvation).
|
|
33
|
+
* @type {client.Gauge<string>}
|
|
14
34
|
*/
|
|
15
35
|
public static eventLoopLag: client.Gauge<string>;
|
|
16
36
|
|
|
17
37
|
/**
|
|
18
38
|
* Gauge metric tracking the number of active connections in the database pool.
|
|
39
|
+
* Useful to detect connection leaks or database starvation.
|
|
40
|
+
* @type {client.Gauge<string>}
|
|
19
41
|
*/
|
|
20
42
|
public static activeDatabaseConnections: client.Gauge<string>;
|
|
21
43
|
|
|
22
44
|
/**
|
|
23
45
|
* Counter tracking the total number of HTTP 5xx Server Errors (Crashes/Panics).
|
|
46
|
+
* Monitored by the panic hooks to alert on system degradation.
|
|
47
|
+
* @type {client.Counter<string>}
|
|
24
48
|
*/
|
|
25
49
|
public static http5xxErrorRate: client.Counter<string>;
|
|
26
50
|
|
|
27
51
|
/**
|
|
28
52
|
* Gauge metric for the current V8 Memory Heap Used in bytes.
|
|
53
|
+
* Automatically alerts if the heap approaches the V8 max limit (e.g. 1.5GB default).
|
|
54
|
+
* @type {client.Gauge<string>}
|
|
29
55
|
*/
|
|
30
56
|
public static processMemoryHeapUsed: client.Gauge<string>;
|
|
31
57
|
|
|
32
58
|
/**
|
|
33
|
-
* Initializes default global system metrics
|
|
59
|
+
* Initializes default global system metrics, registers Prometheus collectors,
|
|
60
|
+
* and starts the background sampling interval.
|
|
61
|
+
*
|
|
62
|
+
* This method is idempotent and will safely return if called multiple times.
|
|
34
63
|
*/
|
|
35
|
-
public static init() {
|
|
64
|
+
public static init(): void {
|
|
36
65
|
if (this.isInitialized) return;
|
|
37
66
|
this.isInitialized = true;
|
|
38
67
|
|
|
@@ -69,10 +98,10 @@ export class GlobalMetricsEngine {
|
|
|
69
98
|
|
|
70
99
|
/**
|
|
71
100
|
* Periodically sample non-event-driven metrics.
|
|
72
|
-
*
|
|
73
|
-
* Includes built-in alerting thresholds (e.g
|
|
101
|
+
* Uses `setImmediate` and `hrtime` to calculate precise Event Loop delay.
|
|
102
|
+
* Includes built-in hardcoded alerting thresholds (e.g., > 100ms lag, > 1.5GB memory).
|
|
74
103
|
*/
|
|
75
|
-
public static sampleMetrics() {
|
|
104
|
+
public static sampleMetrics(): void {
|
|
76
105
|
// Monitor Event Loop Lag
|
|
77
106
|
const start = process.hrtime.bigint();
|
|
78
107
|
setImmediate(() => {
|
|
@@ -96,14 +125,23 @@ export class GlobalMetricsEngine {
|
|
|
96
125
|
}
|
|
97
126
|
}
|
|
98
127
|
|
|
99
|
-
|
|
128
|
+
/**
|
|
129
|
+
* Starts a detached background interval for metric sampling.
|
|
130
|
+
* The interval is unreferenced (`unref()`) to prevent it from keeping the Node process alive.
|
|
131
|
+
* @private
|
|
132
|
+
*/
|
|
133
|
+
private static startPeriodicSampling(): void {
|
|
100
134
|
this.samplingTimer = setInterval(() => {
|
|
101
135
|
this.sampleMetrics();
|
|
102
136
|
}, 5000);
|
|
103
137
|
this.samplingTimer.unref(); // unref so it doesn't prevent Node from exiting
|
|
104
138
|
}
|
|
105
139
|
|
|
106
|
-
|
|
140
|
+
/**
|
|
141
|
+
* Gracefully tears down the metrics engine.
|
|
142
|
+
* Stops the sampling timer and clears the Prometheus registry.
|
|
143
|
+
*/
|
|
144
|
+
public static destroy(): void {
|
|
107
145
|
if (this.samplingTimer) {
|
|
108
146
|
clearInterval(this.samplingTimer);
|
|
109
147
|
this.samplingTimer = null;
|
|
@@ -113,11 +151,12 @@ export class GlobalMetricsEngine {
|
|
|
113
151
|
}
|
|
114
152
|
|
|
115
153
|
/**
|
|
116
|
-
*
|
|
117
|
-
* Captures
|
|
154
|
+
* Hooks into process-level crash events to record them before the process dies.
|
|
155
|
+
* Captures `uncaughtException` and `unhandledRejection`, increments
|
|
118
156
|
* the 5xx error rate metric, and logs the critical failure.
|
|
157
|
+
* @private
|
|
119
158
|
*/
|
|
120
|
-
private static setupPanicHooks() {
|
|
159
|
+
private static setupPanicHooks(): void {
|
|
121
160
|
process.on('uncaughtException', (err) => {
|
|
122
161
|
logger.error(`[ALERT] Uncaught Exception (Process Crash imminent): ${err.message}`);
|
|
123
162
|
this.http5xxErrorRate.inc(); // Increment crash counter
|
|
@@ -130,7 +169,8 @@ export class GlobalMetricsEngine {
|
|
|
130
169
|
}
|
|
131
170
|
|
|
132
171
|
/**
|
|
133
|
-
* Generates the Prometheus Metrics text to be exposed on an endpoint (e.g.,
|
|
172
|
+
* Generates the Prometheus Metrics text to be exposed on an endpoint (e.g., `/metrics`).
|
|
173
|
+
* Fetches all registered metrics from the prom-client global registry.
|
|
134
174
|
*
|
|
135
175
|
* @returns {Promise<string>} A string containing all metrics formatted for Prometheus scraping.
|
|
136
176
|
*/
|